{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30823,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Insuarance Premium Amount\n## 1. Preparation \n### 1.1. Importing libraries and data","metadata":{}},{"cell_type":"code","source":"# import necessary libaries\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom sklearn.model_selection import cross_val_score\nimport lightgbm as lgb","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T02:03:19.395095Z","iopub.execute_input":"2024-12-26T02:03:19.395549Z","iopub.status.idle":"2024-12-26T02:03:22.941164Z","shell.execute_reply.started":"2024-12-26T02:03:19.395515Z","shell.execute_reply":"2024-12-26T02:03:22.940262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import df\ndf = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T02:03:22.942362Z","iopub.execute_input":"2024-12-26T02:03:22.943094Z","iopub.status.idle":"2024-12-26T02:03:34.116724Z","shell.execute_reply.started":"2024-12-26T02:03:22.943065Z","shell.execute_reply":"2024-12-26T02:03:34.115637Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1.2. Functions","metadata":{}},{"cell_type":"code","source":"# create functions for later use\n# creat table for missing data\ndef missing_data_table(df):\n    total = df.isnull().sum().sort_values(ascending=False)\n    percent = (df.isnull().sum()/df_test.isnull().count()).sort_values(ascending=False)\n    missing_data = pd.concat([total, percent], axis=1, keys=[\"Total\", \"Percent\"])\n    return missing_data\n\n# ordinal encoding\ndef ordinal_encode(df, col, levels):\n    levels_dict = {}\n    for i, value in enumerate(levels):\n        levels_dict[value] = i\n    new_col = col + \" Ord\"\n    df[new_col] = df[col].map(levels_dict)\n    df.drop(col, axis=1, inplace=True)\n\ndef nominal_encode(df, col, values):\n    levels_dict = {}\n    for i, value in enumerate(values):\n        levels_dict[value] = i\n    new_col = col + \" Nominal\"\n    df[new_col] = df[col].map(levels_dict)\n    df.drop(col, axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T02:03:34.11842Z","iopub.execute_input":"2024-12-26T02:03:34.118713Z","iopub.status.idle":"2024-12-26T02:03:34.12637Z","shell.execute_reply.started":"2024-12-26T02:03:34.118688Z","shell.execute_reply":"2024-12-26T02:03:34.124933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def exp_log_mean(x):\n    return np.expm1(np.log1p(x).mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T02:14:45.393082Z","iopub.execute_input":"2024-12-26T02:14:45.393625Z","iopub.status.idle":"2024-12-26T02:14:45.39984Z","shell.execute_reply.started":"2024-12-26T02:14:45.393582Z","shell.execute_reply":"2024-12-26T02:14:45.398225Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1.3. Overview","metadata":{}},{"cell_type":"code","source":"df.describe(include=\"all\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T02:03:34.130421Z","iopub.execute_input":"2024-12-26T02:03:34.130728Z","iopub.status.idle":"2024-12-26T02:03:36.714393Z","shell.execute_reply.started":"2024-12-26T02:03:34.130692Z","shell.execute_reply":"2024-12-26T02:03:36.713366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.drop(\"id\", axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T02:03:36.715411Z","iopub.execute_input":"2024-12-26T02:03:36.715908Z","iopub.status.idle":"2024-12-26T02:03:36.932217Z","shell.execute_reply.started":"2024-12-26T02:03:36.715879Z","shell.execute_reply":"2024-12-26T02:03:36.931308Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1.4. Missing data","metadata":{}},{"cell_type":"code","source":"missing_data_table(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T02:17:20.732546Z","iopub.execute_input":"2024-12-26T02:17:20.732915Z","iopub.status.idle":"2024-12-26T02:17:22.380089Z","shell.execute_reply.started":"2024-12-26T02:17:20.732889Z","shell.execute_reply":"2024-12-26T02:17:22.379096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in [\"Insurance Duration\", \"Vehicle Age\", \"Age\", \"Annual Income\", \"Health Score\", \"Number of Dependents\", \"Previous Claims\"]:\n    df[col] = df[col].fillna(exp_log_mean(df[col]))\n\nfor col in [\"Marital Status\", \"Customer Feedback\"]:\n    df[col] = df[col].fillna(df[col].value_counts().idxmax())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T02:17:14.318422Z","iopub.execute_input":"2024-12-26T02:17:14.31882Z","iopub.status.idle":"2024-12-26T02:17:14.903656Z","shell.execute_reply.started":"2024-12-26T02:17:14.31879Z","shell.execute_reply":"2024-12-26T02:17:14.902457Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1.5. Encoding\n#### 1.5.1. Nominal encoding\nSince LGBM can handle categorical data but in integers only.","metadata":{}},{"cell_type":"code","source":"for i in [\"Property Type\", \"Smoking Status\", \"Gender\", \"Location\", \"Marital Status\"]:\n    nominal_encode(df, i, df[i].unique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T02:17:30.702461Z","iopub.execute_input":"2024-12-26T02:17:30.70286Z","iopub.status.idle":"2024-12-26T02:17:32.36651Z","shell.execute_reply.started":"2024-12-26T02:17:30.702829Z","shell.execute_reply":"2024-12-26T02:17:32.365551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"Post COVID 19\"] = 0\ndf['Year'] = df['Policy Start Date'].str[:4].astype(int)\ndf['Month'] = df['Policy Start Date'].str[5:7].astype(int)\ndf = df.drop(\"Policy Start Date\", axis=1)\ndf['Post COVID 19'] = ((df['Year'] > 2022) & (df['Month'] > 4)).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:40:18.424719Z","iopub.execute_input":"2024-12-23T04:40:18.425013Z","iopub.status.idle":"2024-12-23T04:40:19.547834Z","shell.execute_reply.started":"2024-12-23T04:40:18.424987Z","shell.execute_reply":"2024-12-23T04:40:19.546683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.drop([\"Year\", \"Month\"], axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:40:19.54891Z","iopub.execute_input":"2024-12-23T04:40:19.549207Z","iopub.status.idle":"2024-12-23T04:40:19.653436Z","shell.execute_reply.started":"2024-12-23T04:40:19.549182Z","shell.execute_reply":"2024-12-23T04:40:19.651977Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### 1.5.2. Ordinal encoding","metadata":{}},{"cell_type":"code","source":"ordinal_encode(df, \"Education Level\", levels=[\"High School\", \"Bachelor's\", \"Master's\", \"PhD\"])\nordinal_encode(df, \"Policy Type\", levels=[\"Basic\", \"Comprehensive\", \"Premium\"])\nordinal_encode(df, \"Exercise Frequency\", levels=[\"Rarely\", \"Monthly\", \"Weekly\", \"Daily\"])\nordinal_encode(df, \"Customer Feedback\", levels=[\"Poor\", \"Average\", \"Good\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:40:19.654729Z","iopub.execute_input":"2024-12-23T04:40:19.655068Z","iopub.status.idle":"2024-12-23T04:40:20.322619Z","shell.execute_reply.started":"2024-12-23T04:40:19.655037Z","shell.execute_reply":"2024-12-23T04:40:20.321256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:40:20.324061Z","iopub.execute_input":"2024-12-23T04:40:20.324512Z","iopub.status.idle":"2024-12-23T04:40:20.345289Z","shell.execute_reply.started":"2024-12-23T04:40:20.324471Z","shell.execute_reply":"2024-12-23T04:40:20.344276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12,10))\nsns.heatmap(df.drop([\"Occupation\"], axis=1).corr(), annot=True, fmt=\".2f\", cmap=\"YlGnBu\", cbar_kws={\"shrink\": .8})\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:40:20.346323Z","iopub.execute_input":"2024-12-23T04:40:20.346598Z","iopub.status.idle":"2024-12-23T04:40:23.060608Z","shell.execute_reply.started":"2024-12-23T04:40:20.346577Z","shell.execute_reply":"2024-12-23T04:40:23.059485Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1.6. Imputation\n#### 1.6.1. Credit Score\nImpute mean credit scores based on annual income groups.","metadata":{}},{"cell_type":"code","source":"bins = [0, 15000, 25000, 35000, df[\"Annual Income\"].max()]\ndf[\"Annual Income Binned\"] = pd.cut(df[\"Annual Income\"], bins=bins, labels=[\"Low\", \"Middle\", \"High\", \"Higher\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:40:23.061651Z","iopub.execute_input":"2024-12-23T04:40:23.061957Z","iopub.status.idle":"2024-12-23T04:40:23.105895Z","shell.execute_reply.started":"2024-12-23T04:40:23.061929Z","shell.execute_reply":"2024-12-23T04:40:23.104647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.set_theme()\nsns.lineplot(x=df[\"Annual Income Binned\"], y=df[\"Credit Score\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:40:23.107288Z","iopub.execute_input":"2024-12-23T04:40:23.108002Z","iopub.status.idle":"2024-12-23T04:40:34.148506Z","shell.execute_reply.started":"2024-12-23T04:40:23.10796Z","shell.execute_reply":"2024-12-23T04:40:34.147097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def impute(df, group_by, impute_col):\n    # Calculate the means \n    means = df.groupby(group_by, observed=False)[impute_col].mean()\n\n    # Convert means to float to ensure compatibility\n    map_means = means.to_dict()\n    map_means = {k: float(v) for k, v in map_means.items()}\n\n    # Fill missing values using a different approach\n    df[impute_col] = df.apply(\n        lambda row: map_means[row[group_by]] \n        if pd.isna(row[impute_col]) \n        else row[impute_col],\n        axis=1\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:40:34.149815Z","iopub.execute_input":"2024-12-23T04:40:34.150265Z","iopub.status.idle":"2024-12-23T04:40:34.15684Z","shell.execute_reply.started":"2024-12-23T04:40:34.15022Z","shell.execute_reply":"2024-12-23T04:40:34.155705Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"impute(df, \"Annual Income Binned\", \"Credit Score\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:40:34.157867Z","iopub.execute_input":"2024-12-23T04:40:34.158316Z","iopub.status.idle":"2024-12-23T04:40:45.75414Z","shell.execute_reply.started":"2024-12-23T04:40:34.158275Z","shell.execute_reply":"2024-12-23T04:40:45.753057Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### 1.6.2. Occupation","metadata":{}},{"cell_type":"code","source":"freq_occupation = df.groupby('Annual Income Binned', observed=False)['Occupation'].agg(lambda x: x.value_counts().idxmax())\ndf[\"Occupation\"] = df.apply(\n        lambda row: freq_occupation.to_dict()[row[\"Annual Income Binned\"]] \n        if pd.isna(row[\"Occupation\"]) \n        else row[\"Occupation\"],\n        axis=1\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:41:33.213573Z","iopub.execute_input":"2024-12-23T04:41:33.213905Z","iopub.status.idle":"2024-12-23T04:41:58.523396Z","shell.execute_reply.started":"2024-12-23T04:41:33.213871Z","shell.execute_reply":"2024-12-23T04:41:58.522228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nominal_encode(df, \"Occupation\", df[\"Occupation\"].unique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:41:58.528557Z","iopub.execute_input":"2024-12-23T04:41:58.528941Z","iopub.status.idle":"2024-12-23T04:41:58.809957Z","shell.execute_reply.started":"2024-12-23T04:41:58.528904Z","shell.execute_reply":"2024-12-23T04:41:58.808862Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1.7. Ensure normality ","metadata":{}},{"cell_type":"code","source":"from scipy import stats\n\n# check normality of data\ndef normal(df, x):\n    sns.histplot(df[x], kde=True)\n    fig = plt.figure()\n    res = stats.probplot(df[x], plot=plt)\n    plt.show()\n\n# yeo jonhson transformation\ndef yj(df, x):\n    new_x = x + \" Transformed\"\n    df[new_x], lambda_value = stats.yeojohnson(df[x])\n    df.drop(x, axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:42:10.716117Z","iopub.execute_input":"2024-12-23T04:42:10.716548Z","iopub.status.idle":"2024-12-23T04:42:10.722621Z","shell.execute_reply.started":"2024-12-23T04:42:10.716507Z","shell.execute_reply":"2024-12-23T04:42:10.721283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"normal(df, \"Credit Score\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:42:10.723755Z","iopub.execute_input":"2024-12-23T04:42:10.724096Z","iopub.status.idle":"2024-12-23T04:42:19.230779Z","shell.execute_reply.started":"2024-12-23T04:42:10.724058Z","shell.execute_reply":"2024-12-23T04:42:19.229716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"yj(df, \"Credit Score\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:42:19.231867Z","iopub.execute_input":"2024-12-23T04:42:19.232254Z","iopub.status.idle":"2024-12-23T04:42:20.943807Z","shell.execute_reply.started":"2024-12-23T04:42:19.232216Z","shell.execute_reply":"2024-12-23T04:42:20.942693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"normal(df, \"Credit Score Transformed\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:42:20.945056Z","iopub.execute_input":"2024-12-23T04:42:20.945409Z","iopub.status.idle":"2024-12-23T04:42:29.749891Z","shell.execute_reply.started":"2024-12-23T04:42:20.945372Z","shell.execute_reply":"2024-12-23T04:42:29.748684Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Only improves slightly.","metadata":{}},{"cell_type":"code","source":"normal(df, \"Annual Income\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:42:29.750898Z","iopub.execute_input":"2024-12-23T04:42:29.751273Z","iopub.status.idle":"2024-12-23T04:42:38.91207Z","shell.execute_reply.started":"2024-12-23T04:42:29.751243Z","shell.execute_reply":"2024-12-23T04:42:38.910989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"yj(df, \"Annual Income\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:42:38.913065Z","iopub.execute_input":"2024-12-23T04:42:38.913419Z","iopub.status.idle":"2024-12-23T04:42:41.48714Z","shell.execute_reply.started":"2024-12-23T04:42:38.91339Z","shell.execute_reply":"2024-12-23T04:42:41.486194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"normal(df, \"Annual Income Transformed\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:42:41.488099Z","iopub.execute_input":"2024-12-23T04:42:41.488385Z","iopub.status.idle":"2024-12-23T04:42:50.409963Z","shell.execute_reply.started":"2024-12-23T04:42:41.488361Z","shell.execute_reply":"2024-12-23T04:42:50.408809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"Premium Amount Transformed\"] = np.log(df[\"Premium Amount\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:42:50.411366Z","iopub.execute_input":"2024-12-23T04:42:50.411659Z","iopub.status.idle":"2024-12-23T04:42:50.429956Z","shell.execute_reply.started":"2024-12-23T04:42:50.411635Z","shell.execute_reply":"2024-12-23T04:42:50.428779Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Model Implementation\n### 2.1. Train model","metadata":{}},{"cell_type":"code","source":"df = df.drop([\"Annual Income Binned\"], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:42:50.431085Z","iopub.execute_input":"2024-12-23T04:42:50.431597Z","iopub.status.idle":"2024-12-23T04:42:50.567148Z","shell.execute_reply.started":"2024-12-23T04:42:50.431564Z","shell.execute_reply":"2024-12-23T04:42:50.566009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x = df.drop([\"Premium Amount Transformed\", \"Premium Amount\"], axis=1)\ny = df[\"Premium Amount Transformed\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:42:50.568268Z","iopub.execute_input":"2024-12-23T04:42:50.568599Z","iopub.status.idle":"2024-12-23T04:42:50.646472Z","shell.execute_reply.started":"2024-12-23T04:42:50.568565Z","shell.execute_reply":"2024-12-23T04:42:50.645287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x.columns = x.columns.str.replace(\" \", \"_\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:42:50.647559Z","iopub.execute_input":"2024-12-23T04:42:50.64788Z","iopub.status.idle":"2024-12-23T04:42:50.653114Z","shell.execute_reply.started":"2024-12-23T04:42:50.647851Z","shell.execute_reply":"2024-12-23T04:42:50.651761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = lgb.LGBMRegressor()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:42:50.654366Z","iopub.execute_input":"2024-12-23T04:42:50.65493Z","iopub.status.idle":"2024-12-23T04:42:50.670029Z","shell.execute_reply.started":"2024-12-23T04:42:50.654829Z","shell.execute_reply":"2024-12-23T04:42:50.6687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.fit(x, y, categorical_feature=['Property_Type_Nominal', 'Smoking_Status_Nominal',\n       'Gender_Nominal', 'Location_Nominal', 'Marital_Status_Nominal', 'Occupation_Nominal'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:42:50.671197Z","iopub.execute_input":"2024-12-23T04:42:50.67155Z","iopub.status.idle":"2024-12-23T04:42:58.024328Z","shell.execute_reply.started":"2024-12-23T04:42:50.671512Z","shell.execute_reply":"2024-12-23T04:42:58.022847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.sqrt(np.abs(cross_val_score(model, x, y, scoring=\"neg_mean_squared_log_error\",cv=4 ).mean()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:46:28.568499Z","iopub.execute_input":"2024-12-23T04:46:28.569002Z","iopub.status.idle":"2024-12-23T04:46:54.513921Z","shell.execute_reply.started":"2024-12-23T04:46:28.568963Z","shell.execute_reply":"2024-12-23T04:46:54.512601Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 2.2. Transform Test Data","metadata":{}},{"cell_type":"code","source":"# filling in missing values\nfor col in [\"Insurance Duration\", \"Vehicle Age\", \"Age\", \"Annual Income\", \"Health Score\", \"Number of Dependents\", \"Previous Claims\"]:\n    df_test[col] = df_test[col].fillna(exp_log_mean(df_test[col]))\n\nfor col in [\"Marital Status\", \"Customer Feedback\"]:\n    df_test[col] = df_test[col].fillna(df_test[col].value_counts().idxmax())\n\ndf_test[\"Post COVID 19\"] = 0\ndf_test['Year'] = df_test['Policy Start Date'].str[:4].astype(int)\ndf_test['Month'] = df_test['Policy Start Date'].str[5:7].astype(int)\ndf_test = df_test.drop(\"Policy Start Date\", axis=1)\ndf_test['Post COVID 19'] = ((df_test['Year'] > 2022) & (df_test['Month'] > 4)).astype(int)\ndf_test.drop([\"Year\", \"Month\"], axis=1, inplace=True)\n\n# ordinal encoding\nordinal_encode(df_test, \"Education Level\", levels=[\"High School\", \"Bachelor's\", \"Master's\", \"PhD\"])\nordinal_encode(df_test, \"Policy Type\", levels=[\"Basic\", \"Comprehensive\", \"Premium\"])\nordinal_encode(df_test, \"Exercise Frequency\", levels=[\"Rarely\", \"Monthly\", \"Weekly\", \"Daily\"])\nordinal_encode(df_test, \"Customer Feedback\", levels=[\"Poor\", \"Average\", \"Good\"])\n\n# imputing missing values based on Annual Income groups\nbins = [0, 15000, 25000, 35000, df_test[\"Annual Income\"].max()]\ndf_test[\"Annual Income Binned\"] = pd.cut(df_test[\"Annual Income\"], bins=bins, labels=[\"Low\", \"Middle\", \"High\", \"Higher\"])\nimpute(df_test, \"Annual Income Binned\", \"Credit Score\")\n\nfreq_occupation = df_test.groupby('Annual Income Binned', observed=False)['Occupation'].agg(lambda x: x.value_counts().idxmax())\ndf_test[\"Occupation\"] = df_test.apply(\n        lambda row: freq_occupation.to_dict()[row[\"Annual Income Binned\"]] \n        if pd.isna(row[\"Occupation\"]) \n        else row[\"Occupation\"],\n        axis=1\n    )\nnominal_encode(df_test, \"Occupation\", df_test[\"Occupation\"].unique())\nimpute(df_test, \"Annual Income Binned\", \"Previous Claims\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:43:24.800643Z","iopub.execute_input":"2024-12-23T04:43:24.800913Z","iopub.status.idle":"2024-12-23T04:44:15.007058Z","shell.execute_reply.started":"2024-12-23T04:43:24.800891Z","shell.execute_reply":"2024-12-23T04:44:15.006001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in [\"Property Type\", \"Smoking Status\", \"Gender\", \"Location\", \"Marital Status\"]:\n    nominal_encode(df_test, i, df_test[i].unique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:44:15.008374Z","iopub.execute_input":"2024-12-23T04:44:15.008785Z","iopub.status.idle":"2024-12-23T04:44:15.761649Z","shell.execute_reply.started":"2024-12-23T04:44:15.008731Z","shell.execute_reply":"2024-12-23T04:44:15.760519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# yeo jonhson transformation on continuous data\nyj(df_test, \"Credit Score\")\nyj(df_test, \"Annual Income\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:44:15.762639Z","iopub.execute_input":"2024-12-23T04:44:15.762919Z","iopub.status.idle":"2024-12-23T04:44:18.326228Z","shell.execute_reply.started":"2024-12-23T04:44:15.762895Z","shell.execute_reply":"2024-12-23T04:44:18.325137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test = df_test.drop([\"Annual Income Binned\"], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:44:18.327054Z","iopub.execute_input":"2024-12-23T04:44:18.327626Z","iopub.status.idle":"2024-12-23T04:44:18.385088Z","shell.execute_reply.started":"2024-12-23T04:44:18.327582Z","shell.execute_reply":"2024-12-23T04:44:18.383919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_test = df_test.drop([\"id\"], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:44:18.386246Z","iopub.execute_input":"2024-12-23T04:44:18.386591Z","iopub.status.idle":"2024-12-23T04:44:18.442131Z","shell.execute_reply.started":"2024-12-23T04:44:18.386562Z","shell.execute_reply":"2024-12-23T04:44:18.44094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_test.columns = x_test.columns.str.replace(\" \", \"_\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:44:18.443301Z","iopub.execute_input":"2024-12-23T04:44:18.443758Z","iopub.status.idle":"2024-12-23T04:44:18.449338Z","shell.execute_reply.started":"2024-12-23T04:44:18.443727Z","shell.execute_reply":"2024-12-23T04:44:18.448125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_predict = model.predict(x_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:44:18.450592Z","iopub.execute_input":"2024-12-23T04:44:18.450999Z","iopub.status.idle":"2024-12-23T04:44:20.854888Z","shell.execute_reply.started":"2024-12-23T04:44:18.450953Z","shell.execute_reply":"2024-12-23T04:44:20.853681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_predict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:44:20.856015Z","iopub.execute_input":"2024-12-23T04:44:20.856355Z","iopub.status.idle":"2024-12-23T04:44:20.863149Z","shell.execute_reply.started":"2024-12-23T04:44:20.856325Z","shell.execute_reply":"2024-12-23T04:44:20.861958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_predict_inverted = np.exp(y_predict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:44:20.864148Z","iopub.execute_input":"2024-12-23T04:44:20.864497Z","iopub.status.idle":"2024-12-23T04:44:20.886456Z","shell.execute_reply.started":"2024-12-23T04:44:20.864468Z","shell.execute_reply":"2024-12-23T04:44:20.885138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_predict_inverted_df = pd.DataFrame(y_predict_inverted, columns=[\"Premium Amount\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:44:20.887723Z","iopub.execute_input":"2024-12-23T04:44:20.88811Z","iopub.status.idle":"2024-12-23T04:44:20.894364Z","shell.execute_reply.started":"2024-12-23T04:44:20.888073Z","shell.execute_reply":"2024-12-23T04:44:20.893245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test = df_test.reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:44:20.895293Z","iopub.execute_input":"2024-12-23T04:44:20.895582Z","iopub.status.idle":"2024-12-23T04:44:21.015818Z","shell.execute_reply.started":"2024-12-23T04:44:20.895552Z","shell.execute_reply":"2024-12-23T04:44:21.01469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = df_test[[\"id\"]].merge(y_predict_inverted_df, left_index=True, right_index=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:44:21.016917Z","iopub.execute_input":"2024-12-23T04:44:21.017295Z","iopub.status.idle":"2024-12-23T04:44:21.027901Z","shell.execute_reply.started":"2024-12-23T04:44:21.017255Z","shell.execute_reply":"2024-12-23T04:44:21.026753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:44:21.02901Z","iopub.execute_input":"2024-12-23T04:44:21.029351Z","iopub.status.idle":"2024-12-23T04:44:23.02059Z","shell.execute_reply.started":"2024-12-23T04:44:21.029311Z","shell.execute_reply":"2024-12-23T04:44:23.019321Z"}},"outputs":[],"execution_count":null}]}