{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.feature_selection import mutual_info_regression\nfrom sklearn.preprocessing import QuantileTransformer, MinMaxScaler, StandardScaler, OrdinalEncoder","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T09:17:03.785075Z","iopub.execute_input":"2024-12-30T09:17:03.786088Z","iopub.status.idle":"2024-12-30T09:17:16.904842Z","shell.execute_reply.started":"2024-12-30T09:17:03.786036Z","shell.execute_reply":"2024-12-30T09:17:16.904073Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Working on the DataPipe**\n--","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T09:17:16.906188Z","iopub.execute_input":"2024-12-30T09:17:16.906707Z","iopub.status.idle":"2024-12-30T09:17:21.889787Z","shell.execute_reply.started":"2024-12-30T09:17:16.906677Z","shell.execute_reply":"2024-12-30T09:17:21.888973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T09:17:25.909915Z","iopub.execute_input":"2024-12-30T09:17:25.910268Z","iopub.status.idle":"2024-12-30T09:17:26.467892Z","shell.execute_reply.started":"2024-12-30T09:17:25.910232Z","shell.execute_reply":"2024-12-30T09:17:26.466983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.nunique().sort_values()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T09:17:26.54202Z","iopub.execute_input":"2024-12-30T09:17:26.542328Z","iopub.status.idle":"2024-12-30T09:17:27.583862Z","shell.execute_reply.started":"2024-12-30T09:17:26.5423Z","shell.execute_reply":"2024-12-30T09:17:27.583002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(train_df.isna().sum() / len(train_df)) * 100","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T09:17:27.991819Z","iopub.execute_input":"2024-12-30T09:17:27.992166Z","iopub.status.idle":"2024-12-30T09:17:28.534163Z","shell.execute_reply.started":"2024-12-30T09:17:27.992136Z","shell.execute_reply":"2024-12-30T09:17:28.533298Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Inferences from EDA**\n--","metadata":{}},{"cell_type":"markdown","source":"- Normal Distributions\n    - Premium Amount **(Target)**\n    - Credit Score\n    - Health Score\n    - Annual Income\n- Uniform Distributions\n    - Age\n    - Vehicle Age\n    - Insurance Duration\n    - All the categorical variables demonstrate a certain uniform distribution \n- Trends\n    - Much of the insurance policies were started after 2019.\n    - Throughout the year there is a uniform distribution of the number of insurances provided per month and per day of the month.\n    - All of the dataset and its joint distributions with the target variable and other variables is mostly uniformly distributed.\n\n**Important**\n- After the initial cleaning provided by [notebook](https://www.kaggle.com/code/khsamaha/lgbm-optuna-shap-insurance-policy-py) the dataset has little to no corelation with the target.\n- However there are a couple of mild trends in the corelation among the features.","metadata":{}},{"cell_type":"markdown","source":"**Feature Analysis**\n--","metadata":{}},{"cell_type":"code","source":"no_na_train = train_df.dropna()\nno_na_train = no_na_train.drop_duplicates()\n\nfor column in no_na_train.select_dtypes(\"object\"):\n    no_na_train[column], _ = no_na_train[column].factorize()\n\ncorr_mat = no_na_train.corr()\n\nsns.heatmap(\n    corr_mat, cmap=\"jet\"\n)\nplt.title(\"Correlation Scores\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T09:17:32.240165Z","iopub.execute_input":"2024-12-30T09:17:32.240983Z","iopub.status.idle":"2024-12-30T09:17:34.667138Z","shell.execute_reply.started":"2024-12-30T09:17:32.240948Z","shell.execute_reply":"2024-12-30T09:17:34.666273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_mi_scores(X, y, discrete_features):\n    mi_scores = mutual_info_regression(X, y, discrete_features=discrete_features)\n    mi_scores = pd.Series(mi_scores, name=\"Mutual Information Scores\", index=X.columns)\n    mi_scores = mi_scores.sort_values(ascending=False)\n    return mi_scores\n\ndef plot_mi_scores(mi_scores):\n    scores = mi_scores.sort_values(ascending=True)\n    width = np.arange(len(scores))\n    ticks = list(scores.index)\n    plt.barh(width, scores)\n    plt.yticks(width, ticks)\n    plt.title(\"Mutual Information Scores\")\n\n# sample_df = no_na_train.sample(100000)\n# X_mi = sample_df.drop([\"id\", \"Premium Amount\"], axis=1)\n# y_mi = sample_df[\"Premium Amount\"]\n# discrete_features = X_mi.dtypes == int\n\n# mi_scores = get_mi_scores(X_mi, y_mi, discrete_features)\n# print(mi_scores)\n# plot_mi_scores(mi_scores)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T09:17:34.668815Z","iopub.execute_input":"2024-12-30T09:17:34.669151Z","iopub.status.idle":"2024-12-30T09:17:34.675797Z","shell.execute_reply.started":"2024-12-30T09:17:34.669113Z","shell.execute_reply":"2024-12-30T09:17:34.674959Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Data Cleaning, Visualisation and Preprocessing**\n--","metadata":{}},{"cell_type":"markdown","source":"## Categorical Features\n\n**More Analysis**\n- Occupation (check scope for imputing since 30% of the data is being dropped by na values)\n- **Will need more thinking later on, currently dropped since even the MI score is 0.**\n\n**Current Analysis**\n- Marital Status (impute 1.5% missing data)\n- Customer Feedback (impute 6.5% missing data)\n- **Mostly balanced so can be imputed using clustering to classify the category of each missing values.**","metadata":{}},{"cell_type":"code","source":"# All the categorical features\ncat_columns = [\n    \"Gender\", \"Marital Status\", \"Education Level\", \"Occupation\", \"Location\",\n    \"Policy Type\", \"Customer Feedback\", \"Smoking Status\", \"Exercise Frequency\",\n    \"Property Type\"\n]\n\n# Columns to be dropped or imputed for later\ndrop_cols = [\"Occupation\"]\n\n# A dataframe of all the categorical data\ncat_df = train_df[[\"id\"] + cat_columns].copy()\ncat_df = cat_df.drop(drop_cols, axis=1)\n\ncat_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T09:17:36.469946Z","iopub.execute_input":"2024-12-30T09:17:36.470288Z","iopub.status.idle":"2024-12-30T09:17:37.245946Z","shell.execute_reply.started":"2024-12-30T09:17:36.470259Z","shell.execute_reply":"2024-12-30T09:17:37.24505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_ids = cat_df.pop(\"id\")\n\nord_encoder = OrdinalEncoder()\ncat_tfx = ord_encoder.fit_transform(cat_df)\ncat_df_tfx = pd.DataFrame(cat_tfx, columns=cat_df.columns)\ncat_df_tfx.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T09:17:39.055721Z","iopub.execute_input":"2024-12-30T09:17:39.056067Z","iopub.status.idle":"2024-12-30T09:17:41.266429Z","shell.execute_reply.started":"2024-12-30T09:17:39.056037Z","shell.execute_reply":"2024-12-30T09:17:41.265581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"frequency_impute = SimpleImputer(strategy=\"most_frequent\")\ncat_full_tfx = frequency_impute.fit_transform(cat_df_tfx)\ncat_full_df = pd.DataFrame(cat_full_tfx, columns=cat_df_tfx.columns)\ncat_full_df[\"id\"] = cat_ids\ncat_full_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T09:17:56.022064Z","iopub.execute_input":"2024-12-30T09:17:56.022846Z","iopub.status.idle":"2024-12-30T09:17:56.369402Z","shell.execute_reply.started":"2024-12-30T09:17:56.022812Z","shell.execute_reply":"2024-12-30T09:17:56.368405Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Numerical Features\n\n**More Analysis**\n- Previous Claims (30% of missing values, however needs to be imputed due to promise in MI and Corr)\n\n**Current Analysis**\n- Age (1.5% missing values)\n    - Could use imputing to capture the missing values since shown promise in MI and Corr\n- Annual Income (3.7% missing values)\n    - Could use quantile transformer to scale the data\n- No of Dependants (9.2% missing values) \n- Vehicle Age and Insurance duration\n    - (Miniscule missing values can be dropped)\n- **Health Score important feature with most promise (6.2% missing values)**\n    - Could use MM or Std scalers to scale the data depending on model interpretability\n- **Credit Score important feature with most promise (11.5% missing values)**\n    - Could use MM or Std scalers to scale the data depending on model interpretability","metadata":{}},{"cell_type":"code","source":"# List of Numerical Features\nnum_cols = [\n    \"Age\", \"Annual Income\", \"Number of Dependents\", \"Health Score\", \"Previous Claims\",\n    \"Vehicle Age\", \"Credit Score\", \"Insurance Duration\", \"Policy Start Date\"\n]\n\n# List of dropped features\ndrop_cols_num = [\"Previous Claims\", \"Policy Start Date\"]\n\n# Numerical Features DataFrame\nnum_df = train_df[[\"id\"] + num_cols].copy()\n\n# Splitting the Policy Start Date Attribute\ndatetimes = num_df[[\"Policy Start Date\"]]\ndates = datetimes.map(lambda x: x.split()[0])\nnum_df[\"Policy Year\"] = dates.map(lambda x: int(x.split(\"-\")[0]))\nnum_df[\"Policy Month\"] = dates.map(lambda x: int(x.split(\"-\")[1]))\nnum_df[\"Policy Date\"] = dates.map(lambda x: int(x.split(\"-\")[2]))\n\n# Final Preprocessing\nnum_df = num_df.drop(drop_cols_num, axis=1)\nnum_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T10:02:20.743433Z","iopub.execute_input":"2024-12-30T10:02:20.743801Z","iopub.status.idle":"2024-12-30T10:02:23.291889Z","shell.execute_reply.started":"2024-12-30T10:02:20.74377Z","shell.execute_reply":"2024-12-30T10:02:23.291024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.histplot(\n    num_df, x=\"Annual Income\"\n)\n\n# mm_scaler = MinMaxScaler()\n# sns.histplot(\n#     x=np.log1p(mm_scaler.fit_transform(num_df[[\"Annual Income\"]])).ravel()\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T10:02:23.293309Z","iopub.execute_input":"2024-12-30T10:02:23.29361Z","iopub.status.idle":"2024-12-30T10:02:24.110386Z","shell.execute_reply.started":"2024-12-30T10:02:23.293582Z","shell.execute_reply":"2024-12-30T10:02:24.109395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"quantile_factor = 0.25\n\ndistribution_cache = {}\n\nfor column in num_df.columns:\n    mean_val = num_df[column].mean()\n    non_null_vals = num_df[column].dropna()\n    q1, q3 = non_null_vals.quantile([0.25, 0.75])\n    mean_val_revised = mean_val + quantile_factor * (q3 - q1)\n    num_df[column] = num_df[column].fillna(mean_val_revised)\n\n    if column not in distribution_cache:\n        distribution_cache[column] = mean_val_revised\n\nnum_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T10:02:24.111428Z","iopub.execute_input":"2024-12-30T10:02:24.111713Z","iopub.status.idle":"2024-12-30T10:02:24.530055Z","shell.execute_reply.started":"2024-12-30T10:02:24.111686Z","shell.execute_reply":"2024-12-30T10:02:24.529058Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Additional Features","metadata":{}},{"cell_type":"code","source":"num_df[\"Income Age Ratio\"] = num_df[\"Annual Income\"] / num_df[\"Age\"]\nnum_df[\"Health Score Age Ratio\"] = num_df[\"Health Score\"] / num_df[\"Age\"]\nnum_df[\"Credit Score Age Ratio\"] = num_df[\"Credit Score\"] / num_df[\"Age\"]\nnum_df[\"Insurance Duration Age Ratio\"] = num_df[\"Insurance Duration\"] / num_df[\"Age\"]\n\nnum_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T10:02:29.136897Z","iopub.execute_input":"2024-12-30T10:02:29.137669Z","iopub.status.idle":"2024-12-30T10:02:29.168912Z","shell.execute_reply.started":"2024-12-30T10:02:29.137635Z","shell.execute_reply":"2024-12-30T10:02:29.167866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ids for Reattachment\nnum_ids = num_df.pop(\"id\")\n\n# Initialisation\nstd_scaler = StandardScaler()\n\n# MinMaxing all the features but the policy start date\nnum_tfx = std_scaler.fit_transform(num_df)\n\n# Creating a new dataframe for the transformed data\nnum_full_df = pd.DataFrame(\n    num_tfx, columns=num_df.columns\n)\n\n# Reducing the skewness in the Annual Income attribute\n# num_full_df[\"Annual Income\"] = np.log1p(num_full_df[\"Annual Income\"])\nnum_full_df[\"id\"] = num_ids\n# num_full_df = num_full_df.dropna()\nnum_full_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T10:02:30.999647Z","iopub.execute_input":"2024-12-30T10:02:31.000007Z","iopub.status.idle":"2024-12-30T10:02:31.348025Z","shell.execute_reply.started":"2024-12-30T10:02:30.999977Z","shell.execute_reply":"2024-12-30T10:02:31.347072Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Target Scaling","metadata":{}},{"cell_type":"code","source":"# premium_amt = train_df[[\"Premium Amount\"]]\n# mm_scaler_target = MinMaxScaler()\n# premium_amt_tfx = mm_scaler_target.fit_transform(premium_amt)\n\nnum_full_df[\"Premium Amount\"] = np.log1p(train_df[\"Premium Amount\"])\nnum_full_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T10:02:38.997079Z","iopub.execute_input":"2024-12-30T10:02:38.997762Z","iopub.status.idle":"2024-12-30T10:02:39.040065Z","shell.execute_reply.started":"2024-12-30T10:02:38.99773Z","shell.execute_reply":"2024-12-30T10:02:39.039176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# target_min = mm_scaler_target.data_min_\n# target_max = mm_scaler_target.data_max_\n# target_range = mm_scaler_target.data_range_\n\n# print(\"Minimum: \", target_min)\n# print(\"Maximum: \", target_max)\n# print(\"Range: \", target_range)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T10:02:41.72996Z","iopub.execute_input":"2024-12-30T10:02:41.730653Z","iopub.status.idle":"2024-12-30T10:02:41.734299Z","shell.execute_reply.started":"2024-12-30T10:02:41.730617Z","shell.execute_reply":"2024-12-30T10:02:41.733396Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Combined Dataset","metadata":{}},{"cell_type":"code","source":"combined_dataset = pd.merge(\n    left=cat_full_df, right=num_full_df, on=\"id\", how=\"outer\"\n)\ncombined_dataset = combined_dataset.dropna()\ncombined_dataset = combined_dataset.drop(\"id\", axis=1)\ncombined_dataset.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T10:02:49.505457Z","iopub.execute_input":"2024-12-30T10:02:49.506321Z","iopub.status.idle":"2024-12-30T10:02:50.189865Z","shell.execute_reply.started":"2024-12-30T10:02:49.506288Z","shell.execute_reply":"2024-12-30T10:02:50.188942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"combined_dataset.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T10:02:54.550785Z","iopub.execute_input":"2024-12-30T10:02:54.551479Z","iopub.status.idle":"2024-12-30T10:02:54.572213Z","shell.execute_reply.started":"2024-12-30T10:02:54.551445Z","shell.execute_reply":"2024-12-30T10:02:54.571279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.displot(\n    combined_dataset, x=\"Credit Score\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T10:02:58.376615Z","iopub.execute_input":"2024-12-30T10:02:58.377285Z","iopub.status.idle":"2024-12-30T10:02:59.237374Z","shell.execute_reply.started":"2024-12-30T10:02:58.37725Z","shell.execute_reply":"2024-12-30T10:02:59.236602Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## New Baseline Performance before Feature Engineering and Imputing","metadata":{}},{"cell_type":"code","source":"X = combined_dataset.copy()\ny = X.pop(\"Premium Amount\")\n\nX_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.05\n)\n\nX_train, X_valid = X_train[:-100000], X_train[-100000:]\ny_train, y_valid = y_train[:-100000], y_train[-100000:]\n\nprint(X_train.shape)\nprint(y_train.shape)\nprint(X_valid.shape)\nprint(y_valid.shape)\nprint(X_test.shape)\nprint(y_test.shape)\n\ntf_train_set = tf.data.Dataset.from_tensor_slices(\n    (X_train, y_train), name=\"train_set\"\n).shuffle(500000).batch(128).prefetch(1)\n\ntf_valid_set = tf.data.Dataset.from_tensor_slices(\n    (X_valid, y_valid), name=\"valid_set\"\n).shuffle(50000).batch(128).prefetch(1)\n\ntf_test_set = tf.data.Dataset.from_tensor_slices(\n    (X_test, y_test), name=\"test_set\"\n).batch(128).cache()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T10:05:48.382331Z","iopub.execute_input":"2024-12-30T10:05:48.383179Z","iopub.status.idle":"2024-12-30T10:05:49.729637Z","shell.execute_reply.started":"2024-12-30T10:05:48.383141Z","shell.execute_reply":"2024-12-30T10:05:49.728647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.keras.backend.clear_session()\n\nsimple_reg = tf.keras.Sequential([\n    tf.keras.layers.Input(shape=X.shape[1:]),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.Dense(32, activation=\"elu\", kernel_initializer=\"he_normal\"),\n    tf.keras.layers.Dense(64, activation=\"elu\", kernel_initializer=\"he_normal\"),\n    tf.keras.layers.Dropout(0.2),\n    tf.keras.layers.Dense(128, activation=\"elu\", kernel_initializer=\"he_normal\"),\n    tf.keras.layers.Dropout(0.2),\n    tf.keras.layers.Dense(64, activation=\"elu\", kernel_initializer=\"he_normal\"),\n    tf.keras.layers.Dense(32, activation=\"elu\", kernel_initializer=\"he_normal\"),\n    tf.keras.layers.Dropout(0.2),\n    tf.keras.layers.Dense(1)\n])\n\nrmse = tf.keras.metrics.RootMeanSquaredError()\nmae = tf.keras.losses.MeanAbsoluteError()\nearly_stop = tf.keras.callbacks.EarlyStopping(\n    monitor=\"val_root_mean_squared_error\", patience=5, mode=\"min\", restore_best_weights=True\n)\n\nsimple_reg.compile(\n    loss=tf.keras.losses.Huber(),\n    metrics=[rmse, mae],\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-5)\n)\n\nhistory = simple_reg.fit(\n    tf_train_set, validation_data=tf_valid_set, epochs=100, callbacks=[early_stop]\n)\n\npd.DataFrame(history.history).plot()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T10:38:46.595043Z","iopub.execute_input":"2024-12-30T10:38:46.59538Z","iopub.status.idle":"2024-12-30T10:51:46.175378Z","shell.execute_reply.started":"2024-12-30T10:38:46.595352Z","shell.execute_reply":"2024-12-30T10:51:46.174509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"simple_reg.evaluate(tf_test_set)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T10:51:46.177287Z","iopub.execute_input":"2024-12-30T10:51:46.177964Z","iopub.status.idle":"2024-12-30T10:51:46.859978Z","shell.execute_reply.started":"2024-12-30T10:51:46.177922Z","shell.execute_reply":"2024-12-30T10:51:46.859188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = simple_reg.predict(X_test)\nerror = mean_squared_error(y_test, predictions)\nprint(np.sqrt(error))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T11:17:03.544177Z","iopub.execute_input":"2024-12-30T11:17:03.544578Z","iopub.status.idle":"2024-12-30T11:17:06.366392Z","shell.execute_reply.started":"2024-12-30T11:17:03.544546Z","shell.execute_reply":"2024-12-30T11:17:06.365468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"combined_dataset.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T11:17:49.678454Z","iopub.execute_input":"2024-12-30T11:17:49.679218Z","iopub.status.idle":"2024-12-30T11:17:49.704659Z","shell.execute_reply.started":"2024-12-30T11:17:49.679182Z","shell.execute_reply":"2024-12-30T11:17:49.703621Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Realtime Testing","metadata":{}},{"cell_type":"code","source":"full_test_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\nfull_test_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T11:17:53.015812Z","iopub.execute_input":"2024-12-30T11:17:53.016416Z","iopub.status.idle":"2024-12-30T11:17:55.373106Z","shell.execute_reply.started":"2024-12-30T11:17:53.016383Z","shell.execute_reply":"2024-12-30T11:17:55.372213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"full_test_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T11:17:55.374347Z","iopub.execute_input":"2024-12-30T11:17:55.374659Z","iopub.status.idle":"2024-12-30T11:17:55.392115Z","shell.execute_reply.started":"2024-12-30T11:17:55.374632Z","shell.execute_reply":"2024-12-30T11:17:55.391399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def batch_transform(transformer, set_, batch_size=10000):\n    # List of all the transformed samples\n    transformed_set = []\n\n    # Looping through all the samples in the dataset\n    for i in range(0, len(set_), batch_size):\n        batch = set_.iloc[i:i+batch_size]\n        transformed_batch = transformer.transform(batch)\n        transformed_set.append(transformed_batch)\n\n    # Transformed Dataset\n    transformed_set = np.vstack(transformed_set)\n    transformed_df = pd.DataFrame(\n        transformed_set, columns=set_.columns\n    )\n    \n    return transformed_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T11:17:58.68613Z","iopub.execute_input":"2024-12-30T11:17:58.68688Z","iopub.status.idle":"2024-12-30T11:17:58.691948Z","shell.execute_reply.started":"2024-12-30T11:17:58.686846Z","shell.execute_reply":"2024-12-30T11:17:58.691017Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_test = full_test_df[[\"id\"] + cat_columns].copy()\ncat_test = cat_test.drop(drop_cols, axis=1)\ncat_test_ids = cat_test.pop(\"id\")\n\ncat_test_tfx = batch_transform(ord_encoder, cat_test)\ncat_test_impute = batch_transform(frequency_impute, cat_test_tfx)\ncat_test_df = pd.DataFrame(\n    cat_test_impute, columns=cat_test.columns\n)\ncat_test_df[\"id\"] = cat_test_ids\n\ncat_test_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T11:18:03.368135Z","iopub.execute_input":"2024-12-30T11:18:03.368633Z","iopub.status.idle":"2024-12-30T11:18:05.079922Z","shell.execute_reply.started":"2024-12-30T11:18:03.368594Z","shell.execute_reply":"2024-12-30T11:18:05.07902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_test = full_test_df[[\"id\"] + num_cols].copy()\n\ndatetimes = num_test[[\"Policy Start Date\"]]\ndates = datetimes.map(lambda x: x.split()[0])\nnum_test[\"Policy Year\"] = dates.map(lambda x: int(x.split(\"-\")[0]))\nnum_test[\"Policy Month\"] = dates.map(lambda x: int(x.split(\"-\")[1]))\nnum_test[\"Policy Date\"] = dates.map(lambda x: int(x.split(\"-\")[2]))\n\nnum_test = num_test.drop(drop_cols_num, axis=1)\nnum_test_ids = num_test.pop(\"id\")\n\nfor col in num_test.columns:\n    num_test[col] = num_test[col].fillna(distribution_cache[col])\n\nnum_test[\"Income Age Ratio\"] = num_test[\"Annual Income\"] / num_test[\"Age\"]\nnum_test[\"Health Score Age Ratio\"] = num_test[\"Health Score\"] / num_test[\"Age\"]\nnum_test[\"Credit Score Age Ratio\"] = num_test[\"Credit Score\"] / num_test[\"Age\"]\nnum_test[\"Insurance Duration Age Ratio\"] = num_test[\"Insurance Duration\"] / num_test[\"Age\"]\n\nnum_test_tfx = batch_transform(std_scaler, num_test)\nnum_test_df = pd.DataFrame(\n    num_test_tfx, columns=num_test.columns\n)\nnum_test_df[\"id\"] = num_test_ids\n\nnum_test_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T11:19:37.134762Z","iopub.execute_input":"2024-12-30T11:19:37.135117Z","iopub.status.idle":"2024-12-30T11:19:39.123613Z","shell.execute_reply.started":"2024-12-30T11:19:37.135088Z","shell.execute_reply":"2024-12-30T11:19:39.122679Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"combined_test = pd.merge(\n    left=cat_test_df, right=num_test_df, on=\"id\", how=\"outer\"\n)\ntest_ids = combined_test.pop(\"id\")\n\ncombined_test.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T11:19:43.665632Z","iopub.execute_input":"2024-12-30T11:19:43.666468Z","iopub.status.idle":"2024-12-30T11:19:43.823095Z","shell.execute_reply.started":"2024-12-30T11:19:43.666437Z","shell.execute_reply":"2024-12-30T11:19:43.822193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"real_predictions = simple_reg.predict(combined_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T11:19:51.154886Z","iopub.execute_input":"2024-12-30T11:19:51.155217Z","iopub.status.idle":"2024-12-30T11:20:33.069365Z","shell.execute_reply.started":"2024-12-30T11:19:51.155187Z","shell.execute_reply":"2024-12-30T11:20:33.068344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"true = np.expm1(real_predictions[:100])\nfor i in range(100):\n    print(f\"Prediction: {real_predictions[i]} \\t True: {true[i]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T11:23:34.018909Z","iopub.execute_input":"2024-12-30T11:23:34.019262Z","iopub.status.idle":"2024-12-30T11:23:34.04014Z","shell.execute_reply.started":"2024-12-30T11:23:34.019232Z","shell.execute_reply":"2024-12-30T11:23:34.039173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.displot(\n    x=np.expm1(real_predictions).ravel()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T11:23:40.459645Z","iopub.execute_input":"2024-12-30T11:23:40.46052Z","iopub.status.idle":"2024-12-30T11:23:41.484345Z","shell.execute_reply.started":"2024-12-30T11:23:40.460484Z","shell.execute_reply":"2024-12-30T11:23:41.483455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"real_scaled_anti_log = np.expm1(real_predictions)\npredictions_data = np.c_[test_ids, np.round(real_scaled_anti_log, 3)]\n\npredictions_df = pd.DataFrame(\n    predictions_data, columns=[\"id\", \"Premium Amount\"]\n)\npredictions_df[\"id\"] = predictions_df[\"id\"].astype(int)\npredictions_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T11:23:44.684821Z","iopub.execute_input":"2024-12-30T11:23:44.68547Z","iopub.status.idle":"2024-12-30T11:23:44.708601Z","shell.execute_reply.started":"2024-12-30T11:23:44.685438Z","shell.execute_reply":"2024-12-30T11:23:44.70774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions_df.head(50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T11:23:47.201456Z","iopub.execute_input":"2024-12-30T11:23:47.202176Z","iopub.status.idle":"2024-12-30T11:23:47.212495Z","shell.execute_reply.started":"2024-12-30T11:23:47.202143Z","shell.execute_reply":"2024-12-30T11:23:47.211696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions_df.to_csv(\"SUBMISSION.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T11:23:50.713726Z","iopub.execute_input":"2024-12-30T11:23:50.714337Z","iopub.status.idle":"2024-12-30T11:23:52.068228Z","shell.execute_reply.started":"2024-12-30T11:23:50.714304Z","shell.execute_reply":"2024-12-30T11:23:52.067255Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Feature Importances and Correlations**\n--","metadata":{}},{"cell_type":"code","source":"corr_combined = combined_dataset.corr()\n\nsns.heatmap(\n    corr_combined, cmap=\"jet\"\n)\nplt.title(\"Updated Correlation Scores\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T11:23:54.239355Z","iopub.execute_input":"2024-12-30T11:23:54.239718Z","iopub.status.idle":"2024-12-30T11:23:56.412339Z","shell.execute_reply.started":"2024-12-30T11:23:54.239686Z","shell.execute_reply":"2024-12-30T11:23:56.411513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_mi = combined_dataset.drop([\"Premium Amount\"], axis=1)\ny_mi = combined_dataset[\"Premium Amount\"]\ndiscrete_features = X_mi.dtypes == int\n\nmi_scores = get_mi_scores(X_mi, y_mi, discrete_features)\nprint(mi_scores)\nplot_mi_scores(mi_scores)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T10:35:26.226272Z","iopub.execute_input":"2024-12-30T10:35:26.226668Z","iopub.status.idle":"2024-12-30T10:37:58.746636Z","shell.execute_reply.started":"2024-12-30T10:35:26.226634Z","shell.execute_reply":"2024-12-30T10:37:58.745393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}