{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30823,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:16.299823Z","iopub.execute_input":"2024-12-28T12:18:16.300088Z","iopub.status.idle":"2024-12-28T12:18:16.306865Z","shell.execute_reply.started":"2024-12-28T12:18:16.300067Z","shell.execute_reply":"2024-12-28T12:18:16.305813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nplt.style.use('ggplot')\nimport seaborn as sns\nimport scipy.stats as stats\nfrom scipy.stats import shapiro\nimport optuna\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.cluster import KMeans\nfrom lightgbm import LGBMRegressor\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.model_selection import GridSearchCV\nimport shap\n\nimport warnings\nwarnings.simplefilter(action='ignore', category=UserWarning)\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:16.307902Z","iopub.execute_input":"2024-12-28T12:18:16.308098Z","iopub.status.idle":"2024-12-28T12:18:30.91685Z","shell.execute_reply.started":"2024-12-28T12:18:16.308082Z","shell.execute_reply":"2024-12-28T12:18:30.916077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_=pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest=pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\nsubmission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:30.91812Z","iopub.execute_input":"2024-12-28T12:18:30.918669Z","iopub.status.idle":"2024-12-28T12:18:40.211189Z","shell.execute_reply.started":"2024-12-28T12:18:30.918646Z","shell.execute_reply":"2024-12-28T12:18:40.210555Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Have a peek of the dataset","metadata":{}},{"cell_type":"code","source":"print(train_.info())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:40.212202Z","iopub.execute_input":"2024-12-28T12:18:40.212433Z","iopub.status.idle":"2024-12-28T12:18:40.768262Z","shell.execute_reply.started":"2024-12-28T12:18:40.212414Z","shell.execute_reply":"2024-12-28T12:18:40.767484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test.info())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:40.769068Z","iopub.execute_input":"2024-12-28T12:18:40.769394Z","iopub.status.idle":"2024-12-28T12:18:41.125938Z","shell.execute_reply.started":"2024-12-28T12:18:40.769344Z","shell.execute_reply":"2024-12-28T12:18:41.125072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:41.126798Z","iopub.execute_input":"2024-12-28T12:18:41.127098Z","iopub.status.idle":"2024-12-28T12:18:41.131274Z","shell.execute_reply.started":"2024-12-28T12:18:41.127068Z","shell.execute_reply":"2024-12-28T12:18:41.130635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:41.131997Z","iopub.execute_input":"2024-12-28T12:18:41.132192Z","iopub.status.idle":"2024-12-28T12:18:41.147527Z","shell.execute_reply.started":"2024-12-28T12:18:41.132175Z","shell.execute_reply":"2024-12-28T12:18:41.146868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:41.149619Z","iopub.execute_input":"2024-12-28T12:18:41.149854Z","iopub.status.idle":"2024-12-28T12:18:41.162619Z","shell.execute_reply.started":"2024-12-28T12:18:41.149835Z","shell.execute_reply":"2024-12-28T12:18:41.161769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:41.163951Z","iopub.execute_input":"2024-12-28T12:18:41.164134Z","iopub.status.idle":"2024-12-28T12:18:41.177198Z","shell.execute_reply.started":"2024-12-28T12:18:41.164118Z","shell.execute_reply":"2024-12-28T12:18:41.176456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:41.177964Z","iopub.execute_input":"2024-12-28T12:18:41.178141Z","iopub.status.idle":"2024-12-28T12:18:41.357551Z","shell.execute_reply.started":"2024-12-28T12:18:41.178126Z","shell.execute_reply":"2024-12-28T12:18:41.356808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:41.358209Z","iopub.execute_input":"2024-12-28T12:18:41.358459Z","iopub.status.idle":"2024-12-28T12:18:41.374456Z","shell.execute_reply.started":"2024-12-28T12:18:41.358439Z","shell.execute_reply":"2024-12-28T12:18:41.37383Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Summary Statistics","metadata":{}},{"cell_type":"code","source":"# summarize categorical variable distributions\nobj_cols_tr=[var for var in train_.columns if train_[var].dtype in ['object']]\ntrain_[obj_cols_tr].describe().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:41.375215Z","iopub.execute_input":"2024-12-28T12:18:41.375528Z","iopub.status.idle":"2024-12-28T12:18:43.010654Z","shell.execute_reply.started":"2024-12-28T12:18:41.375498Z","shell.execute_reply":"2024-12-28T12:18:43.009782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# summarize categorical variable distributions\nobj_cols_te=[var for var in test.columns if test[var].dtype in ['object']]\ntest[obj_cols_te].describe().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:43.011579Z","iopub.execute_input":"2024-12-28T12:18:43.011884Z","iopub.status.idle":"2024-12-28T12:18:44.122152Z","shell.execute_reply.started":"2024-12-28T12:18:43.011854Z","shell.execute_reply":"2024-12-28T12:18:44.121403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# summarize numerical variable distributions\nnum_cols_tr=[var for var in train_.columns if train_[var].dtype in ('float','int')]\ntrain_[num_cols_tr].describe().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:44.122996Z","iopub.execute_input":"2024-12-28T12:18:44.123319Z","iopub.status.idle":"2024-12-28T12:18:44.707323Z","shell.execute_reply.started":"2024-12-28T12:18:44.123275Z","shell.execute_reply":"2024-12-28T12:18:44.706606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# summarize numerical variable distributions\nnum_cols_te=[var for var in test.columns if test[var].dtype in ('float','int')]\ntest[num_cols_te].describe().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:44.708211Z","iopub.execute_input":"2024-12-28T12:18:44.70857Z","iopub.status.idle":"2024-12-28T12:18:45.050391Z","shell.execute_reply.started":"2024-12-28T12:18:44.708532Z","shell.execute_reply":"2024-12-28T12:18:45.049695Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Cleaning","metadata":{}},{"cell_type":"code","source":"train_.drop_duplicates(inplace=True)\nprint(train_.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:45.051076Z","iopub.execute_input":"2024-12-28T12:18:45.051277Z","iopub.status.idle":"2024-12-28T12:18:46.484028Z","shell.execute_reply.started":"2024-12-28T12:18:45.051261Z","shell.execute_reply":"2024-12-28T12:18:46.483272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.drop_duplicates(inplace=True)\nprint(test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:46.484782Z","iopub.execute_input":"2024-12-28T12:18:46.484991Z","iopub.status.idle":"2024-12-28T12:18:47.449469Z","shell.execute_reply.started":"2024-12-28T12:18:46.484973Z","shell.execute_reply":"2024-12-28T12:18:47.448706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Detect and remove outliers in target variable\n# IQR\n# Calculate the upper and lower limits\nprint(\"Old Shape: \", train_.shape)\nQ1=train_['Premium Amount'].quantile(0.25)\nQ3=train_['Premium Amount'].quantile(0.75)\nIQR=Q3-Q1\nlower=Q1-1.5*IQR\nupper=Q3+1.5*IQR\n\n# Create arrays of Boolean values indicating the outlier rows\nupper_array = train_[train_['Premium Amount'] > upper].index\nlower_array = train_[train_['Premium Amount'] < lower].index\n\n# Removing the outliers by the target\ntrain=train_.drop(index=upper_array, axis=1)\ntrain=train.drop(index=lower_array, axis=1)\n\n# Print the new shape of the DataFrame\nprint(\"New_Shape: \", train.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:47.45013Z","iopub.execute_input":"2024-12-28T12:18:47.450344Z","iopub.status.idle":"2024-12-28T12:18:47.949253Z","shell.execute_reply.started":"2024-12-28T12:18:47.450327Z","shell.execute_reply":"2024-12-28T12:18:47.948533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_tr=train.isnull().sum()/train.shape[0]*100\nmissing_tr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:47.950043Z","iopub.execute_input":"2024-12-28T12:18:47.950337Z","iopub.status.idle":"2024-12-28T12:18:48.455974Z","shell.execute_reply.started":"2024-12-28T12:18:47.950304Z","shell.execute_reply":"2024-12-28T12:18:48.45506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_te=test.isnull().sum()/test.shape[0]*100\nmissing_te","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:48.456748Z","iopub.execute_input":"2024-12-28T12:18:48.456974Z","iopub.status.idle":"2024-12-28T12:18:48.804153Z","shell.execute_reply.started":"2024-12-28T12:18:48.456954Z","shell.execute_reply":"2024-12-28T12:18:48.80345Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Input missing values with median or mode depending of features class\ntrain['Marital Status'].fillna(train['Marital Status'].mode()[0], inplace=True)\ntrain['Marital Status'] = train['Marital Status'].astype(object)\ntrain['Customer Feedback'].fillna(train['Customer Feedback'].mode()[0], inplace=True)\ntrain['Customer Feedback'] = train['Customer Feedback'].astype(object)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:48.804926Z","iopub.execute_input":"2024-12-28T12:18:48.805147Z","iopub.status.idle":"2024-12-28T12:18:49.067644Z","shell.execute_reply.started":"2024-12-28T12:18:48.805119Z","shell.execute_reply":"2024-12-28T12:18:49.066975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Input missing values with median or mode depending of features class\ntest['Marital Status'].fillna(test['Marital Status'].mode()[0], inplace=True)\ntest['Marital Status'] = test['Marital Status'].astype(object)\ntest['Customer Feedback'].fillna(test['Customer Feedback'].mode()[0], inplace=True)\ntest['Customer Feedback'] = test['Customer Feedback'].astype(object)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:49.068365Z","iopub.execute_input":"2024-12-28T12:18:49.068684Z","iopub.status.idle":"2024-12-28T12:18:49.253079Z","shell.execute_reply.started":"2024-12-28T12:18:49.068654Z","shell.execute_reply":"2024-12-28T12:18:49.252422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Input missing values with median or mode depending of features class\ntrain['Age'].fillna(train['Age'].median(), inplace=True)\ntrain['Annual Income'].fillna(train['Annual Income'].median(), inplace=True)\ntrain['Number of Dependents'].fillna(train['Number of Dependents'].median(), inplace=True)\ntrain['Health Score'].fillna(train['Health Score'].median(), inplace=True)\ntrain['Credit Score'].fillna(train['Credit Score'].median(), inplace=True)\ntrain['Insurance Duration'].fillna(train['Insurance Duration'].median(), inplace=True)\ntrain['Vehicle Age'].fillna(train['Vehicle Age'].median(), inplace=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:49.257237Z","iopub.execute_input":"2024-12-28T12:18:49.257461Z","iopub.status.idle":"2024-12-28T12:18:49.431495Z","shell.execute_reply.started":"2024-12-28T12:18:49.257443Z","shell.execute_reply":"2024-12-28T12:18:49.430777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Input missing values with median or mode depending of features class\ntest['Age'].fillna(test['Age'].median(), inplace=True)\ntest['Annual Income'].fillna(test['Annual Income'].median(), inplace=True)\ntest['Number of Dependents'].fillna(test['Number of Dependents'].median(), inplace=True)\ntest['Health Score'].fillna(test['Health Score'].median(), inplace=True)\ntest['Credit Score'].fillna(test['Credit Score'].median(), inplace=True)\ntest['Insurance Duration'].fillna(test['Insurance Duration'].median(), inplace=True)\ntest['Vehicle Age'].fillna(test['Vehicle Age'].median(), inplace=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:49.433359Z","iopub.execute_input":"2024-12-28T12:18:49.433585Z","iopub.status.idle":"2024-12-28T12:18:49.551059Z","shell.execute_reply.started":"2024-12-28T12:18:49.433567Z","shell.execute_reply":"2024-12-28T12:18:49.550412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Input missing values with mode or median when is missing and add a boolean feature \ntrain['Occupation_Missing'] = train['Occupation'].isnull().astype(int)\ntrain['Previous_Claims_Missing'] = train['Previous Claims'].isnull().astype(int)\ntrain['Occupation'].fillna(train['Occupation'].mode()[0], inplace=True)\ntrain['Occupation']=train['Occupation'].astype('object')\ntrain['Previous Claims'].fillna(train['Previous Claims'].median(), inplace=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:49.551864Z","iopub.execute_input":"2024-12-28T12:18:49.552193Z","iopub.status.idle":"2024-12-28T12:18:49.765046Z","shell.execute_reply.started":"2024-12-28T12:18:49.552159Z","shell.execute_reply":"2024-12-28T12:18:49.764399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Input missing values with mode or median when is missing and add a boolean feature \ntest['Occupation_Missing'] = test['Occupation'].isnull().astype(int)\ntest['Previous_Claims_Missing'] = test['Previous Claims'].isnull().astype(int)\ntest['Occupation'].fillna(test['Occupation'].mode()[0], inplace=True)\ntest['Occupation']=test['Occupation'].astype('object')\ntest['Previous Claims'].fillna(test['Previous Claims'].median(), inplace=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:49.765826Z","iopub.execute_input":"2024-12-28T12:18:49.766028Z","iopub.status.idle":"2024-12-28T12:18:49.914694Z","shell.execute_reply.started":"2024-12-28T12:18:49.766011Z","shell.execute_reply":"2024-12-28T12:18:49.913995Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Functions","metadata":{}},{"cell_type":"code","source":"def plot_target(data, var):\n    plt.rcParams['figure.figsize']=(15,5)\n    plt.suptitle('Premium Amount')\n    plt.subplot(1,3,1)\n    x=data[var]\n    plt.hist(x, color='green',edgecolor='black')\n    plt.title('{} histogram'.format(var))\n    plt.yticks(rotation=45, fontsize=15)\n    plt.xticks(rotation=45, fontsize=15)\n\n    plt.subplot(1,3,2)\n    x=data[var]\n    sns.boxplot(x, color='orange')\n    plt.title('{} boxplot'.format(var))\n    plt.yticks(rotation=45, fontsize=15)\n    plt.xticks(rotation=45, fontsize=15)\n\n    plt.subplot(1,3,3)\n    res=stats.probplot(data[var],plot=plt)\n    plt.title('{} Q-Q plot'.format(var))\n    plt.yticks(rotation=45, fontsize=15)\n    plt.xticks(rotation=45, fontsize=15)\n\n    plt.show()\n    \n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:49.915343Z","iopub.execute_input":"2024-12-28T12:18:49.915557Z","iopub.status.idle":"2024-12-28T12:18:49.921346Z","shell.execute_reply.started":"2024-12-28T12:18:49.915539Z","shell.execute_reply":"2024-12-28T12:18:49.920512Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define function to plot the distribution of the numerical variables\ndef plot_num(data, var):\n    plt.rcParams['figure.figsize']=(15,5)\n    plt.subplot(1,3,1)\n    x=data[var]\n    plt.hist(x,color='green',edgecolor='black')\n    plt.title('{} histogram'.format(var))\n    plt.yticks(rotation=0, fontsize=15)\n    plt.xticks(rotation=45, fontsize=15)\n\n\n    plt.subplot(1,3,2)\n    x=data[var]\n    sns.boxplot(x, color=\"orange\")\n    plt.title('{} boxplot'.format(var))\n    plt.yticks(rotation=0, fontsize=15)\n    plt.xticks(rotation=45, fontsize=15)\n\n\n    plt.subplot(1,3,3)\n    res = stats.probplot(data[var], plot=plt)\n    plt.title('{} Q-Q plot'.format(var))\n    plt.yticks(rotation=0, fontsize=15)\n    plt.xticks(rotation=45, fontsize=15)\n\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:49.922321Z","iopub.execute_input":"2024-12-28T12:18:49.92262Z","iopub.status.idle":"2024-12-28T12:18:49.937153Z","shell.execute_reply.started":"2024-12-28T12:18:49.92259Z","shell.execute_reply":"2024-12-28T12:18:49.936572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define function to plot scatterplot between the target and numerical variables\ndef plot_scatterplot(data, var):\n    plt.rcParams['figure.figsize']=(15,5)\n    sns.scatterplot(data=data, x=var, y='Premium Amount')\n    plt.suptitle('Premium Amount pr {}'.format(var), fontsize=10)\n    plt.xlabel('{}'.format(var), fontsize=15)\n    plt.ylabel('Premium Amount', fontsize=15)\n    plt.yticks(rotation=45, fontsize=15)\n    plt.xticks(rotation=45, fontsize=15)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:49.937798Z","iopub.execute_input":"2024-12-28T12:18:49.937985Z","iopub.status.idle":"2024-12-28T12:18:49.952977Z","shell.execute_reply.started":"2024-12-28T12:18:49.937969Z","shell.execute_reply":"2024-12-28T12:18:49.952385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define function to plot scatterplot between the target and numerical variables\ndef plot_scatterplot(data,var):\n    plt.rcParams['figure.figsize']=(10,5)\n    sns.scatterplot(data=data, x=var, y='Premium Amount')\n    plt.suptitle('Premium Amount Distribution per {}'.format(var),fontsize=10)\n    plt.xlabel('{}'.format(var), fontsize=15)\n    plt.ylabel('premium Amount', fontsize=15)\n    plt.yticks(rotation=0,fontsize=15)\n    plt.xticks(rotation=45, fontsize=15)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:49.953609Z","iopub.execute_input":"2024-12-28T12:18:49.953859Z","iopub.status.idle":"2024-12-28T12:18:49.972913Z","shell.execute_reply.started":"2024-12-28T12:18:49.95384Z","shell.execute_reply":"2024-12-28T12:18:49.972228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define function to plot the distribution of the categorical variables\ndef plot_cat(data, var):\n    plt.rcParams['figure.figsize']=(15,5)\n    sns.countplot(x=data[var], data=data).set_title(\"Barplot {} Variable Distribution\".format(var))\n    plt.yticks(rotation=0, fontsize=5)\n    plt.xticks(rotation=90, fontsize=10)\n    plt.show()\n     ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:49.97363Z","iopub.execute_input":"2024-12-28T12:18:49.973832Z","iopub.status.idle":"2024-12-28T12:18:49.9883Z","shell.execute_reply.started":"2024-12-28T12:18:49.973816Z","shell.execute_reply":"2024-12-28T12:18:49.987557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define function to plot boxplot between the target and categorical variables\ndef plot_boxplot(data, var):\n    plt.rcParams['figure.figsize']=(20,10)\n    sns.boxplot(x=data[var], y='Premium Amount', linewidth=2, palette=\"Set1\", data=data)\n    plt.suptitle('Premium Amount Distribution per {}'.format(var),fontsize=10)\n    plt.xlabel('{}'.format(var), fontsize=15)\n    plt.ylabel('Premium Amount', fontsize=15)\n    plt.yticks(rotation=0,fontsize=15)\n    plt.xticks(rotation=90, fontsize=15)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:49.98911Z","iopub.execute_input":"2024-12-28T12:18:49.989411Z","iopub.status.idle":"2024-12-28T12:18:50.001467Z","shell.execute_reply.started":"2024-12-28T12:18:49.989357Z","shell.execute_reply":"2024-12-28T12:18:50.000699Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Target Analysis","metadata":{}},{"cell_type":"code","source":"# split train and target variable\nX_tr=train.copy()\ny=X_tr['Premium Amount']\nX_tr.drop(['Premium Amount', 'id'], axis=1, inplace=True)\ntest.drop(['id'], axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:50.002168Z","iopub.execute_input":"2024-12-28T12:18:50.002445Z","iopub.status.idle":"2024-12-28T12:18:50.860531Z","shell.execute_reply.started":"2024-12-28T12:18:50.002425Z","shell.execute_reply":"2024-12-28T12:18:50.859619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Univariate analysis looking at Mean, Variance, Standard Deviation, Skewness and Kurtosis\nprint(y.name,\n      '\\nMean :', np.mean(y),\n      '\\nVariance :', np.var(y),\n      '\\nStandard Deviation :', np.var(y)**0.5,\n      '\\nSkewness :', stats.skew(y),\n      '\\nKurtosis :', stats.kurtosis(y))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:50.861478Z","iopub.execute_input":"2024-12-28T12:18:50.861784Z","iopub.status.idle":"2024-12-28T12:18:50.910003Z","shell.execute_reply.started":"2024-12-28T12:18:50.861755Z","shell.execute_reply":"2024-12-28T12:18:50.909249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_target(train, var='Premium Amount')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:50.910825Z","iopub.execute_input":"2024-12-28T12:18:50.91108Z","iopub.status.idle":"2024-12-28T12:18:53.752267Z","shell.execute_reply.started":"2024-12-28T12:18:50.911047Z","shell.execute_reply":"2024-12-28T12:18:53.751472Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Numerical Variable Analysis","metadata":{}},{"cell_type":"code","source":"# Select numerical features\nnumerical_cols=[var for var in X_tr.columns if X_tr[var].dtype in ['int64','float64']]\n# Subset with numerical columns\nnum_tr=X_tr[numerical_cols]\nnum_te=test[numerical_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:53.753179Z","iopub.execute_input":"2024-12-28T12:18:53.753433Z","iopub.status.idle":"2024-12-28T12:18:53.807889Z","shell.execute_reply.started":"2024-12-28T12:18:53.753403Z","shell.execute_reply":"2024-12-28T12:18:53.807198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot distributions of numerical features\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:53.808709Z","iopub.execute_input":"2024-12-28T12:18:53.809006Z","iopub.status.idle":"2024-12-28T12:18:53.812469Z","shell.execute_reply.started":"2024-12-28T12:18:53.808977Z","shell.execute_reply":"2024-12-28T12:18:53.811707Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_num(num_tr, var='Age')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:53.813298Z","iopub.execute_input":"2024-12-28T12:18:53.81359Z","iopub.status.idle":"2024-12-28T12:18:56.509263Z","shell.execute_reply.started":"2024-12-28T12:18:53.81357Z","shell.execute_reply":"2024-12-28T12:18:56.508446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_num(num_te, var='Age')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:56.510118Z","iopub.execute_input":"2024-12-28T12:18:56.510433Z","iopub.status.idle":"2024-12-28T12:18:58.549301Z","shell.execute_reply.started":"2024-12-28T12:18:56.510398Z","shell.execute_reply":"2024-12-28T12:18:58.548436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_num(num_tr, var='Annual Income')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:18:58.550174Z","iopub.execute_input":"2024-12-28T12:18:58.550488Z","iopub.status.idle":"2024-12-28T12:19:01.401039Z","shell.execute_reply.started":"2024-12-28T12:18:58.550449Z","shell.execute_reply":"2024-12-28T12:19:01.400203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_num(num_te, var='Annual Income')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:01.401976Z","iopub.execute_input":"2024-12-28T12:19:01.402312Z","iopub.status.idle":"2024-12-28T12:19:03.533164Z","shell.execute_reply.started":"2024-12-28T12:19:01.402273Z","shell.execute_reply":"2024-12-28T12:19:03.532319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_num(num_tr, var='Number of Dependents')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:03.534102Z","iopub.execute_input":"2024-12-28T12:19:03.534442Z","iopub.status.idle":"2024-12-28T12:19:06.184075Z","shell.execute_reply.started":"2024-12-28T12:19:03.534409Z","shell.execute_reply":"2024-12-28T12:19:06.183224Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_num(num_tr, var='Number of Dependents')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:06.184791Z","iopub.execute_input":"2024-12-28T12:19:06.185023Z","iopub.status.idle":"2024-12-28T12:19:08.907018Z","shell.execute_reply.started":"2024-12-28T12:19:06.185003Z","shell.execute_reply":"2024-12-28T12:19:08.906162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_num(num_tr, var='Health Score')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:08.907802Z","iopub.execute_input":"2024-12-28T12:19:08.908027Z","iopub.status.idle":"2024-12-28T12:19:11.630132Z","shell.execute_reply.started":"2024-12-28T12:19:08.908007Z","shell.execute_reply":"2024-12-28T12:19:11.629341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_num(num_te, var='Health Score')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:11.631014Z","iopub.execute_input":"2024-12-28T12:19:11.63134Z","iopub.status.idle":"2024-12-28T12:19:13.680742Z","shell.execute_reply.started":"2024-12-28T12:19:11.631308Z","shell.execute_reply":"2024-12-28T12:19:13.679863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_num(num_tr, var='Previous Claims')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:13.681452Z","iopub.execute_input":"2024-12-28T12:19:13.681686Z","iopub.status.idle":"2024-12-28T12:19:16.456223Z","shell.execute_reply.started":"2024-12-28T12:19:13.681666Z","shell.execute_reply":"2024-12-28T12:19:16.455395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_num(num_te, var='Previous Claims')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:16.4571Z","iopub.execute_input":"2024-12-28T12:19:16.457362Z","iopub.status.idle":"2024-12-28T12:19:18.705218Z","shell.execute_reply.started":"2024-12-28T12:19:16.457339Z","shell.execute_reply":"2024-12-28T12:19:18.704386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_num(num_tr, var='Vehicle Age')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:18.706158Z","iopub.execute_input":"2024-12-28T12:19:18.706509Z","iopub.status.idle":"2024-12-28T12:19:21.400189Z","shell.execute_reply.started":"2024-12-28T12:19:18.706472Z","shell.execute_reply":"2024-12-28T12:19:21.399311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_num(num_te, var='Vehicle Age')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:21.401039Z","iopub.execute_input":"2024-12-28T12:19:21.401289Z","iopub.status.idle":"2024-12-28T12:19:23.343718Z","shell.execute_reply.started":"2024-12-28T12:19:21.40125Z","shell.execute_reply":"2024-12-28T12:19:23.342846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_num(num_tr, var='Credit Score')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:23.344578Z","iopub.execute_input":"2024-12-28T12:19:23.344834Z","iopub.status.idle":"2024-12-28T12:19:25.944326Z","shell.execute_reply.started":"2024-12-28T12:19:23.344812Z","shell.execute_reply":"2024-12-28T12:19:25.943479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_num(num_te, var='Credit Score')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:25.945073Z","iopub.execute_input":"2024-12-28T12:19:25.945313Z","iopub.status.idle":"2024-12-28T12:19:27.87409Z","shell.execute_reply.started":"2024-12-28T12:19:25.94529Z","shell.execute_reply":"2024-12-28T12:19:27.873274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_num(num_tr, var='Insurance Duration')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:27.874906Z","iopub.execute_input":"2024-12-28T12:19:27.875178Z","iopub.status.idle":"2024-12-28T12:19:30.519582Z","shell.execute_reply.started":"2024-12-28T12:19:27.875156Z","shell.execute_reply":"2024-12-28T12:19:30.518664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_num(num_te, var='Insurance Duration')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:30.520541Z","iopub.execute_input":"2024-12-28T12:19:30.520861Z","iopub.status.idle":"2024-12-28T12:19:32.50221Z","shell.execute_reply.started":"2024-12-28T12:19:30.520812Z","shell.execute_reply":"2024-12-28T12:19:32.501414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Combine target variable with numerical features for scatter plots\nnum2= pd.concat([y,num_tr], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:32.503075Z","iopub.execute_input":"2024-12-28T12:19:32.503335Z","iopub.status.idle":"2024-12-28T12:19:32.527998Z","shell.execute_reply.started":"2024-12-28T12:19:32.503295Z","shell.execute_reply":"2024-12-28T12:19:32.527438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot scatterplots between target and numerical features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:32.528647Z","iopub.execute_input":"2024-12-28T12:19:32.528832Z","iopub.status.idle":"2024-12-28T12:19:32.532141Z","shell.execute_reply.started":"2024-12-28T12:19:32.528815Z","shell.execute_reply":"2024-12-28T12:19:32.531335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_scatterplot(num2, var='Age')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:32.533106Z","iopub.execute_input":"2024-12-28T12:19:32.533351Z","iopub.status.idle":"2024-12-28T12:19:34.724119Z","shell.execute_reply.started":"2024-12-28T12:19:32.53333Z","shell.execute_reply":"2024-12-28T12:19:34.723245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_scatterplot(num2, var='Annual Income')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:34.725002Z","iopub.execute_input":"2024-12-28T12:19:34.725248Z","iopub.status.idle":"2024-12-28T12:19:36.982568Z","shell.execute_reply.started":"2024-12-28T12:19:34.725226Z","shell.execute_reply":"2024-12-28T12:19:36.9816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_scatterplot(num2, var='Number of Dependents')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:36.983515Z","iopub.execute_input":"2024-12-28T12:19:36.983765Z","iopub.status.idle":"2024-12-28T12:19:39.138827Z","shell.execute_reply.started":"2024-12-28T12:19:36.983743Z","shell.execute_reply":"2024-12-28T12:19:39.137927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_scatterplot(num2, var='Health Score')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:39.139826Z","iopub.execute_input":"2024-12-28T12:19:39.140157Z","iopub.status.idle":"2024-12-28T12:19:41.347963Z","shell.execute_reply.started":"2024-12-28T12:19:39.140123Z","shell.execute_reply":"2024-12-28T12:19:41.347061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_scatterplot(num2, var='Previous Claims')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:41.348956Z","iopub.execute_input":"2024-12-28T12:19:41.349249Z","iopub.status.idle":"2024-12-28T12:19:43.48013Z","shell.execute_reply.started":"2024-12-28T12:19:41.349223Z","shell.execute_reply":"2024-12-28T12:19:43.479086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_scatterplot(num2, var='Vehicle Age')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:43.481108Z","iopub.execute_input":"2024-12-28T12:19:43.481437Z","iopub.status.idle":"2024-12-28T12:19:45.734611Z","shell.execute_reply.started":"2024-12-28T12:19:43.4814Z","shell.execute_reply":"2024-12-28T12:19:45.733813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_scatterplot(num2, var='Credit Score')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:45.735702Z","iopub.execute_input":"2024-12-28T12:19:45.736032Z","iopub.status.idle":"2024-12-28T12:19:47.971859Z","shell.execute_reply.started":"2024-12-28T12:19:45.735997Z","shell.execute_reply":"2024-12-28T12:19:47.970962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_scatterplot(num2, var='Insurance Duration')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:47.972713Z","iopub.execute_input":"2024-12-28T12:19:47.972949Z","iopub.status.idle":"2024-12-28T12:19:50.065552Z","shell.execute_reply.started":"2024-12-28T12:19:47.972928Z","shell.execute_reply":"2024-12-28T12:19:50.06461Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Categorical Variable Analysis","metadata":{}},{"cell_type":"code","source":"\n# let's have a look at how many labels for categorical features\nfor col in X_tr.columns:\n    if X_tr[col].dtype ==\"object\":\n        print(col, ': ', len(X_tr[col].unique()), ' labels')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:50.066255Z","iopub.execute_input":"2024-12-28T12:19:50.066512Z","iopub.status.idle":"2024-12-28T12:19:50.761875Z","shell.execute_reply.started":"2024-12-28T12:19:50.066491Z","shell.execute_reply":"2024-12-28T12:19:50.761159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select categorical columns with relatively low cardinality (convenient but arbitrary)\ncategorical_cols = [cname for cname in X_tr.columns if\n                    X_tr[cname].nunique() <= 15 and\n                    X_tr[cname].dtype == \"object\"]\ncat_tr=X_tr[categorical_cols]\ncat_te=test[categorical_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:50.762652Z","iopub.execute_input":"2024-12-28T12:19:50.762928Z","iopub.status.idle":"2024-12-28T12:19:51.704251Z","shell.execute_reply.started":"2024-12-28T12:19:50.762891Z","shell.execute_reply":"2024-12-28T12:19:51.703617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot distributions of categorical features\n     ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:51.704966Z","iopub.execute_input":"2024-12-28T12:19:51.705181Z","iopub.status.idle":"2024-12-28T12:19:51.708282Z","shell.execute_reply.started":"2024-12-28T12:19:51.705163Z","shell.execute_reply":"2024-12-28T12:19:51.707675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_tr, var = 'Gender')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:51.708991Z","iopub.execute_input":"2024-12-28T12:19:51.709186Z","iopub.status.idle":"2024-12-28T12:19:52.600698Z","shell.execute_reply.started":"2024-12-28T12:19:51.709168Z","shell.execute_reply":"2024-12-28T12:19:52.599839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_te, var = 'Gender')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:52.608935Z","iopub.execute_input":"2024-12-28T12:19:52.609182Z","iopub.status.idle":"2024-12-28T12:19:53.269116Z","shell.execute_reply.started":"2024-12-28T12:19:52.60916Z","shell.execute_reply":"2024-12-28T12:19:53.268342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_tr, var = 'Marital Status')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:53.271735Z","iopub.execute_input":"2024-12-28T12:19:53.271971Z","iopub.status.idle":"2024-12-28T12:19:54.401986Z","shell.execute_reply.started":"2024-12-28T12:19:53.27195Z","shell.execute_reply":"2024-12-28T12:19:54.401093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_te, var = 'Marital Status')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:54.402785Z","iopub.execute_input":"2024-12-28T12:19:54.403054Z","iopub.status.idle":"2024-12-28T12:19:55.075807Z","shell.execute_reply.started":"2024-12-28T12:19:54.403032Z","shell.execute_reply":"2024-12-28T12:19:55.074582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_tr, var = 'Education Level')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:55.07721Z","iopub.execute_input":"2024-12-28T12:19:55.077676Z","iopub.status.idle":"2024-12-28T12:19:56.038585Z","shell.execute_reply.started":"2024-12-28T12:19:55.077635Z","shell.execute_reply":"2024-12-28T12:19:56.037795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_te, var = 'Education Level')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:56.039236Z","iopub.execute_input":"2024-12-28T12:19:56.039508Z","iopub.status.idle":"2024-12-28T12:19:56.78493Z","shell.execute_reply.started":"2024-12-28T12:19:56.039487Z","shell.execute_reply":"2024-12-28T12:19:56.784059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_tr, var = 'Occupation')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:56.785779Z","iopub.execute_input":"2024-12-28T12:19:56.78602Z","iopub.status.idle":"2024-12-28T12:19:57.647836Z","shell.execute_reply.started":"2024-12-28T12:19:56.785998Z","shell.execute_reply":"2024-12-28T12:19:57.647169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_te, var = 'Occupation')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:57.648554Z","iopub.execute_input":"2024-12-28T12:19:57.64875Z","iopub.status.idle":"2024-12-28T12:19:58.251294Z","shell.execute_reply.started":"2024-12-28T12:19:57.648733Z","shell.execute_reply":"2024-12-28T12:19:58.250469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_tr, var = 'Location')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:58.252104Z","iopub.execute_input":"2024-12-28T12:19:58.252327Z","iopub.status.idle":"2024-12-28T12:19:59.154356Z","shell.execute_reply.started":"2024-12-28T12:19:58.252302Z","shell.execute_reply":"2024-12-28T12:19:59.153149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_te, var = 'Location')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:59.155897Z","iopub.execute_input":"2024-12-28T12:19:59.156255Z","iopub.status.idle":"2024-12-28T12:19:59.888677Z","shell.execute_reply.started":"2024-12-28T12:19:59.156227Z","shell.execute_reply":"2024-12-28T12:19:59.887905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_tr, var = 'Policy Type')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:19:59.889403Z","iopub.execute_input":"2024-12-28T12:19:59.889609Z","iopub.status.idle":"2024-12-28T12:20:00.858292Z","shell.execute_reply.started":"2024-12-28T12:19:59.889592Z","shell.execute_reply":"2024-12-28T12:20:00.857481Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_te, var = 'Policy Type')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:00.859098Z","iopub.execute_input":"2024-12-28T12:20:00.859341Z","iopub.status.idle":"2024-12-28T12:20:01.533447Z","shell.execute_reply.started":"2024-12-28T12:20:00.85932Z","shell.execute_reply":"2024-12-28T12:20:01.532678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_tr, var = 'Customer Feedback')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:01.534159Z","iopub.execute_input":"2024-12-28T12:20:01.534407Z","iopub.status.idle":"2024-12-28T12:20:02.367234Z","shell.execute_reply.started":"2024-12-28T12:20:01.534385Z","shell.execute_reply":"2024-12-28T12:20:02.366436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_te, var = 'Customer Feedback')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:02.367968Z","iopub.execute_input":"2024-12-28T12:20:02.368171Z","iopub.status.idle":"2024-12-28T12:20:03.020985Z","shell.execute_reply.started":"2024-12-28T12:20:02.368153Z","shell.execute_reply":"2024-12-28T12:20:03.020204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_tr, var = 'Smoking Status')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:03.021795Z","iopub.execute_input":"2024-12-28T12:20:03.022103Z","iopub.status.idle":"2024-12-28T12:20:03.888983Z","shell.execute_reply.started":"2024-12-28T12:20:03.02207Z","shell.execute_reply":"2024-12-28T12:20:03.888263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_te, var = 'Smoking Status')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:03.889767Z","iopub.execute_input":"2024-12-28T12:20:03.889979Z","iopub.status.idle":"2024-12-28T12:20:04.47583Z","shell.execute_reply.started":"2024-12-28T12:20:03.889961Z","shell.execute_reply":"2024-12-28T12:20:04.474847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_tr, var = 'Exercise Frequency')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:04.476797Z","iopub.execute_input":"2024-12-28T12:20:04.477105Z","iopub.status.idle":"2024-12-28T12:20:05.454733Z","shell.execute_reply.started":"2024-12-28T12:20:04.477075Z","shell.execute_reply":"2024-12-28T12:20:05.453782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_te, var = 'Exercise Frequency')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:05.455564Z","iopub.execute_input":"2024-12-28T12:20:05.455852Z","iopub.status.idle":"2024-12-28T12:20:06.190283Z","shell.execute_reply.started":"2024-12-28T12:20:05.455826Z","shell.execute_reply":"2024-12-28T12:20:06.189478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_tr, var = 'Property Type')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:06.191139Z","iopub.execute_input":"2024-12-28T12:20:06.191443Z","iopub.status.idle":"2024-12-28T12:20:07.127795Z","shell.execute_reply.started":"2024-12-28T12:20:06.191408Z","shell.execute_reply":"2024-12-28T12:20:07.126951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_cat(cat_te, var = 'Property Type')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:07.128543Z","iopub.execute_input":"2024-12-28T12:20:07.128768Z","iopub.status.idle":"2024-12-28T12:20:07.812825Z","shell.execute_reply.started":"2024-12-28T12:20:07.128747Z","shell.execute_reply":"2024-12-28T12:20:07.8118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Combine target variable with categorical features for box plots\ncat2=pd.concat([y,cat_tr], axis=1)\n     ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:07.813748Z","iopub.execute_input":"2024-12-28T12:20:07.814047Z","iopub.status.idle":"2024-12-28T12:20:07.868775Z","shell.execute_reply.started":"2024-12-28T12:20:07.814023Z","shell.execute_reply":"2024-12-28T12:20:07.867844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Plot boxplots between target and categorical features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:07.869654Z","iopub.execute_input":"2024-12-28T12:20:07.869882Z","iopub.status.idle":"2024-12-28T12:20:07.87308Z","shell.execute_reply.started":"2024-12-28T12:20:07.869863Z","shell.execute_reply":"2024-12-28T12:20:07.872439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_boxplot(cat2, var='Gender')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:07.87386Z","iopub.execute_input":"2024-12-28T12:20:07.87405Z","iopub.status.idle":"2024-12-28T12:20:08.599185Z","shell.execute_reply.started":"2024-12-28T12:20:07.874033Z","shell.execute_reply":"2024-12-28T12:20:08.598445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_boxplot(cat2, var='Marital Status')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:08.60006Z","iopub.execute_input":"2024-12-28T12:20:08.600378Z","iopub.status.idle":"2024-12-28T12:20:09.383664Z","shell.execute_reply.started":"2024-12-28T12:20:08.600344Z","shell.execute_reply":"2024-12-28T12:20:09.38278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_boxplot(cat2, var='Education Level')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:09.384673Z","iopub.execute_input":"2024-12-28T12:20:09.384955Z","iopub.status.idle":"2024-12-28T12:20:10.209427Z","shell.execute_reply.started":"2024-12-28T12:20:09.384921Z","shell.execute_reply":"2024-12-28T12:20:10.208469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_boxplot(cat2, var='Occupation')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:10.210262Z","iopub.execute_input":"2024-12-28T12:20:10.210617Z","iopub.status.idle":"2024-12-28T12:20:10.942064Z","shell.execute_reply.started":"2024-12-28T12:20:10.210582Z","shell.execute_reply":"2024-12-28T12:20:10.941181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_boxplot(cat2, var='Location')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:10.94294Z","iopub.execute_input":"2024-12-28T12:20:10.94325Z","iopub.status.idle":"2024-12-28T12:20:11.758226Z","shell.execute_reply.started":"2024-12-28T12:20:10.943218Z","shell.execute_reply":"2024-12-28T12:20:11.757348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_boxplot(cat2, var='Policy Type')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:11.759041Z","iopub.execute_input":"2024-12-28T12:20:11.759316Z","iopub.status.idle":"2024-12-28T12:20:12.540588Z","shell.execute_reply.started":"2024-12-28T12:20:11.759294Z","shell.execute_reply":"2024-12-28T12:20:12.539737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_boxplot(cat2, var='Customer Feedback')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:12.541599Z","iopub.execute_input":"2024-12-28T12:20:12.541918Z","iopub.status.idle":"2024-12-28T12:20:13.373103Z","shell.execute_reply.started":"2024-12-28T12:20:12.541886Z","shell.execute_reply":"2024-12-28T12:20:13.372276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_boxplot(cat2, var='Smoking Status')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:13.373849Z","iopub.execute_input":"2024-12-28T12:20:13.374133Z","iopub.status.idle":"2024-12-28T12:20:14.157081Z","shell.execute_reply.started":"2024-12-28T12:20:13.374111Z","shell.execute_reply":"2024-12-28T12:20:14.156254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_boxplot(cat2, var='Exercise Frequency')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:14.157889Z","iopub.execute_input":"2024-12-28T12:20:14.158201Z","iopub.status.idle":"2024-12-28T12:20:14.959013Z","shell.execute_reply.started":"2024-12-28T12:20:14.15817Z","shell.execute_reply":"2024-12-28T12:20:14.95815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_boxplot(cat2, var='Property Type')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:14.959819Z","iopub.execute_input":"2024-12-28T12:20:14.960044Z","iopub.status.idle":"2024-12-28T12:20:15.803343Z","shell.execute_reply.started":"2024-12-28T12:20:14.960025Z","shell.execute_reply":"2024-12-28T12:20:15.802456Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Outliers Management","metadata":{}},{"cell_type":"code","source":"# cap residual outliers on numerical feature\ni = 'Age'\nq75, q25 = np.percentile(num_tr[i].dropna(), [75 ,25])\niqr = q75 - q25\nmin_val = q25 - (iqr*1.5)\nmax_val = q75 + (iqr*1.5)\n\n# Create a copy of the DataFrame to avoid SettingWithCopyWarning\nnum_out_tr = num_tr.copy()\n\n# Use .loc to set the values within the IQR range\nnum_out_tr.loc[num_out_tr[i] < min_val, i] = min_val\nnum_out_tr.loc[num_out_tr[i] > max_val, i] = max_val\n\n# Create a copy of the DataFrame to avoid SettingWithCopyWarning\nnum_out_te = num_te.copy()\n\n# Use .loc to set the values within the IQR range\nnum_out_te.loc[num_out_te[i] < min_val, i] = min_val\nnum_out_te.loc[num_out_te[i] > max_val, i] = max_val\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:15.804456Z","iopub.execute_input":"2024-12-28T12:20:15.804777Z","iopub.status.idle":"2024-12-28T12:20:15.879213Z","shell.execute_reply.started":"2024-12-28T12:20:15.804742Z","shell.execute_reply":"2024-12-28T12:20:15.878517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cap residual outliers on numerical feature\ni = 'Annual Income'\nq75, q25 = np.percentile(num_tr[i].dropna(), [75 ,25])\niqr = q75 - q25\nmin_val = q25 - (iqr*1.5)\nmax_val = q75 + (iqr*1.5)\n\n# Create a copy of the DataFrame to avoid SettingWithCopyWarning\nnum_out_tr = num_tr.copy()\n\n# Use .loc to set the values within the IQR range\nnum_out_tr.loc[num_out_tr[i] < min_val, i] = min_val\nnum_out_tr.loc[num_out_tr[i] > max_val, i] = max_val\n\n# Create a copy of the DataFrame to avoid SettingWithCopyWarning\nnum_out_te = num_te.copy()\n\n# Use .loc to set the values within the IQR range\nnum_out_te.loc[num_out_te[i] < min_val, i] = min_val\nnum_out_te.loc[num_out_te[i] > max_val, i] = max_val\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:15.880205Z","iopub.execute_input":"2024-12-28T12:20:15.88059Z","iopub.status.idle":"2024-12-28T12:20:15.958527Z","shell.execute_reply.started":"2024-12-28T12:20:15.880551Z","shell.execute_reply":"2024-12-28T12:20:15.957814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cap residual outliers on numerical feature\ni = 'Number of Dependents'\nq75, q25 = np.percentile(num_tr[i].dropna(), [75 ,25])\niqr = q75 - q25\nmin_val = q25 - (iqr*1.5)\nmax_val = q75 + (iqr*1.5)\n\n# Create a copy of the DataFrame to avoid SettingWithCopyWarning\nnum_out_tr = num_tr.copy()\n\n# Use .loc to set the values within the IQR range\nnum_out_tr.loc[num_out_tr[i] < min_val, i] = min_val\nnum_out_tr.loc[num_out_tr[i] > max_val, i] = max_val\n\n# Create a copy of the DataFrame to avoid SettingWithCopyWarning\nnum_out_te = num_te.copy()\n\n# Use .loc to set the values within the IQR range\nnum_out_te.loc[num_out_te[i] < min_val, i] = min_val\nnum_out_te.loc[num_out_te[i] > max_val, i] = max_val\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:15.959278Z","iopub.execute_input":"2024-12-28T12:20:15.959597Z","iopub.status.idle":"2024-12-28T12:20:16.030148Z","shell.execute_reply.started":"2024-12-28T12:20:15.959565Z","shell.execute_reply":"2024-12-28T12:20:16.029529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cap residual outliers on numerical feature\ni = 'Health Score'\nq75, q25 = np.percentile(num_tr[i].dropna(), [75 ,25])\niqr = q75 - q25\nmin_val = q25 - (iqr*1.5)\nmax_val = q75 + (iqr*1.5)\n\n# Create a copy of the DataFrame to avoid SettingWithCopyWarning\nnum_out_tr = num_tr.copy()\n\n# Use .loc to set the values within the IQR range\nnum_out_tr.loc[num_out_tr[i] < min_val, i] = min_val\nnum_out_tr.loc[num_out_tr[i] > max_val, i] = max_val\n\n# Create a copy of the DataFrame to avoid SettingWithCopyWarning\nnum_out_te = num_te.copy()\n\n# Use .loc to set the values within the IQR range\nnum_out_te.loc[num_out_te[i] < min_val, i] = min_val\nnum_out_te.loc[num_out_te[i] > max_val, i] = max_val\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:16.030892Z","iopub.execute_input":"2024-12-28T12:20:16.031176Z","iopub.status.idle":"2024-12-28T12:20:16.100841Z","shell.execute_reply.started":"2024-12-28T12:20:16.031148Z","shell.execute_reply":"2024-12-28T12:20:16.100145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cap residual outliers on numerical feature\ni = 'Previous Claims'\nq75, q25 = np.percentile(num_tr[i].dropna(), [75 ,25])\niqr = q75 - q25\nmin_val = q25 - (iqr*1.5)\nmax_val = q75 + (iqr*1.5)\n\n# Create a copy of the DataFrame to avoid SettingWithCopyWarning\nnum_out_tr = num_tr.copy()\n\n# Use .loc to set the values within the IQR range\nnum_out_tr.loc[num_out_tr[i] < min_val, i] = min_val\nnum_out_tr.loc[num_out_tr[i] > max_val, i] = max_val\n\n# Create a copy of the DataFrame to avoid SettingWithCopyWarning\nnum_out_te = num_te.copy()\n\n# Use .loc to set the values within the IQR range\nnum_out_te.loc[num_out_te[i] < min_val, i] = min_val\nnum_out_te.loc[num_out_te[i] > max_val, i] = max_val\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:16.101573Z","iopub.execute_input":"2024-12-28T12:20:16.101801Z","iopub.status.idle":"2024-12-28T12:20:16.172061Z","shell.execute_reply.started":"2024-12-28T12:20:16.101781Z","shell.execute_reply":"2024-12-28T12:20:16.171422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cap residual outliers on numerical feature\ni = 'Vehicle Age'\nq75, q25 = np.percentile(num_tr[i].dropna(), [75 ,25])\niqr = q75 - q25\nmin_val = q25 - (iqr*1.5)\nmax_val = q75 + (iqr*1.5)\n\n# Create a copy of the DataFrame to avoid SettingWithCopyWarning\nnum_out_tr = num_tr.copy()\n\n# Use .loc to set the values within the IQR range\nnum_out_tr.loc[num_out_tr[i] < min_val, i] = min_val\nnum_out_tr.loc[num_out_tr[i] > max_val, i] = max_val\n\n# Create a copy of the DataFrame to avoid SettingWithCopyWarning\nnum_out_te = num_te.copy()\n\n# Use .loc to set the values within the IQR range\nnum_out_te.loc[num_out_te[i] < min_val, i] = min_val\nnum_out_te.loc[num_out_te[i] > max_val, i] = max_val\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:16.172859Z","iopub.execute_input":"2024-12-28T12:20:16.173094Z","iopub.status.idle":"2024-12-28T12:20:16.25097Z","shell.execute_reply.started":"2024-12-28T12:20:16.173073Z","shell.execute_reply":"2024-12-28T12:20:16.249948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cap residual outliers on numerical feature\ni = 'Credit Score'\nq75, q25 = np.percentile(num_tr[i].dropna(), [75 ,25])\niqr = q75 - q25\nmin_val = q25 - (iqr*1.5)\nmax_val = q75 + (iqr*1.5)\n\n# Create a copy of the DataFrame to avoid SettingWithCopyWarning\nnum_out_tr = num_tr.copy()\n\n# Use .loc to set the values within the IQR range\nnum_out_tr.loc[num_out_tr[i] < min_val, i] = min_val\nnum_out_tr.loc[num_out_tr[i] > max_val, i] = max_val\n\n# Create a copy of the DataFrame to avoid SettingWithCopyWarning\nnum_out_te = num_te.copy()\n\n# Use .loc to set the values within the IQR range\nnum_out_te.loc[num_out_te[i] < min_val, i] = min_val\nnum_out_te.loc[num_out_te[i] > max_val, i] = max_val\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:16.251891Z","iopub.execute_input":"2024-12-28T12:20:16.252156Z","iopub.status.idle":"2024-12-28T12:20:16.325517Z","shell.execute_reply.started":"2024-12-28T12:20:16.252135Z","shell.execute_reply":"2024-12-28T12:20:16.324607Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cap residual outliers on numerical feature\ni = 'Insurance Duration'\nq75, q25 = np.percentile(num_tr[i].dropna(), [75 ,25])\niqr = q75 - q25\nmin_val = q25 - (iqr*1.5)\nmax_val = q75 + (iqr*1.5)\n\n# Create a copy of the DataFrame to avoid SettingWithCopyWarning\nnum_out_tr = num_tr.copy()\n\n# Use .loc to set the values within the IQR range\nnum_out_tr.loc[num_out_tr[i] < min_val, i] = min_val\nnum_out_tr.loc[num_out_tr[i] > max_val, i] = max_val\n\n# Create a copy of the DataFrame to avoid SettingWithCopyWarning\nnum_out_te = num_te.copy()\n\n# Use .loc to set the values within the IQR range\nnum_out_te.loc[num_out_te[i] < min_val, i] = min_val\nnum_out_te.loc[num_out_te[i] > max_val, i] = max_val\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:16.326444Z","iopub.execute_input":"2024-12-28T12:20:16.3267Z","iopub.status.idle":"2024-12-28T12:20:16.396224Z","shell.execute_reply.started":"2024-12-28T12:20:16.326668Z","shell.execute_reply":"2024-12-28T12:20:16.395309Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Encoding Categorical Variables","metadata":{}},{"cell_type":"code","source":"encoder = OrdinalEncoder()\ncat_encoded_tr=cat_tr.copy()\ncat_encoded_tr.loc[:,'Gender_encoded'] = encoder.fit_transform(cat_encoded_tr[['Gender']])\ncat_encoded_tr.loc[:,'Marital_Status_encoded'] = encoder.fit_transform(cat_encoded_tr[['Marital Status']])\ncat_encoded_tr.loc[:,'Education_Level_encoded'] = encoder.fit_transform(cat_encoded_tr[['Education Level']])\ncat_encoded_tr.loc[:,'Occupation_encoded'] = encoder.fit_transform(cat_encoded_tr[['Occupation']])\ncat_encoded_tr.loc[:,'Location_encoded'] = encoder.fit_transform(cat_encoded_tr[['Location']])\ncat_encoded_tr.loc[:,'Policy_Type_encoded'] = encoder.fit_transform(cat_encoded_tr[['Policy Type']])\ncat_encoded_tr.loc[:,'Customer_Feedback_encoded'] = encoder.fit_transform(cat_encoded_tr[['Customer Feedback']])\ncat_encoded_tr.loc[:,'Smoking_Status_encoded'] = encoder.fit_transform(cat_encoded_tr[['Smoking Status']])\ncat_encoded_tr.loc[:,'Exercise_Frequency_encoded'] = encoder.fit_transform(cat_encoded_tr[['Exercise Frequency']])\ncat_encoded_tr.loc[:,'Property_Type_encoded'] = encoder.fit_transform(cat_encoded_tr[['Property Type']])\ncat_encoded_tr.drop(['Gender','Marital Status','Education Level','Occupation','Location','Policy Type',\n                 'Customer Feedback','Smoking Status','Exercise Frequency','Property Type'], axis=1, inplace=True)\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:16.397013Z","iopub.execute_input":"2024-12-28T12:20:16.397225Z","iopub.status.idle":"2024-12-28T12:20:18.665791Z","shell.execute_reply.started":"2024-12-28T12:20:16.397208Z","shell.execute_reply":"2024-12-28T12:20:18.665084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_encoded_te=cat_te.copy()\ncat_encoded_te.loc[:,'Gender_encoded'] = encoder.fit_transform(cat_encoded_te[['Gender']])\ncat_encoded_te.loc[:,'Marital_Status_encoded'] = encoder.fit_transform(cat_encoded_te[['Marital Status']])\ncat_encoded_te.loc[:,'Education_Level_encoded'] = encoder.fit_transform(cat_encoded_te[['Education Level']])\ncat_encoded_te.loc[:,'Occupation_encoded'] = encoder.fit_transform(cat_encoded_te[['Occupation']])\ncat_encoded_te.loc[:,'Location_encoded'] = encoder.fit_transform(cat_encoded_te[['Location']])\ncat_encoded_te.loc[:,'Policy_Type_encoded'] = encoder.fit_transform(cat_encoded_te[['Policy Type']])\ncat_encoded_te.loc[:,'Customer_Feedback_encoded'] = encoder.fit_transform(cat_encoded_te[['Customer Feedback']])\ncat_encoded_te.loc[:,'Smoking_Status_encoded'] = encoder.fit_transform(cat_encoded_te[['Smoking Status']])\ncat_encoded_te.loc[:,'Exercise_Frequency_encoded'] = encoder.fit_transform(cat_encoded_te[['Exercise Frequency']])\ncat_encoded_te.loc[:,'Property_Type_encoded'] = encoder.fit_transform(cat_encoded_te[['Property Type']])\n\ncat_encoded_te.drop(['Gender','Marital Status','Education Level','Occupation','Location','Policy Type',\n                 'Customer Feedback','Smoking Status','Exercise Frequency','Property Type'], axis=1, inplace=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:18.666552Z","iopub.execute_input":"2024-12-28T12:20:18.666782Z","iopub.status.idle":"2024-12-28T12:20:20.303787Z","shell.execute_reply.started":"2024-12-28T12:20:18.666763Z","shell.execute_reply":"2024-12-28T12:20:20.302857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Combine all processed categorical and numerical features\nX_all_tr=pd.concat([cat_encoded_tr, num_out_tr,train[['Policy Start Date']]], axis=1)\nX_all_te=pd.concat([cat_encoded_te, num_out_te,test[['Policy Start Date']]], axis=1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:20.304657Z","iopub.execute_input":"2024-12-28T12:20:20.30487Z","iopub.status.idle":"2024-12-28T12:20:20.491845Z","shell.execute_reply.started":"2024-12-28T12:20:20.304852Z","shell.execute_reply":"2024-12-28T12:20:20.49115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert 'Policy Start Date' to datetime format and add Policy and date\nX_all_tr['Policy Start Date'] = pd.to_datetime(X_all_tr['Policy Start Date'])\n# Extract year, month and day\nX_all_tr['Year_Start'] = X_all_tr['Policy Start Date'].dt.year\nX_all_tr['Month_Start'] = X_all_tr['Policy Start Date'].dt.month\nX_all_tr['Day_Start'] = X_all_tr['Policy Start Date'].dt.day\n# Add duration to policy start date\n# Convert 'Insurance Duration' to timedelta\nX_all_tr['Duration Timedelta'] = pd.to_timedelta(X_all_tr['Insurance Duration'], unit='D')\nX_all_tr['Policy End Date'] = X_all_tr['Policy Start Date']+X_all_tr['Duration Timedelta']\nX_all_tr['Year_End'] = X_all_tr['Policy Start Date'].dt.year\nX_all_tr['Month_End'] = X_all_tr['Policy Start Date'].dt.month\nX_all_tr['Day_End'] = X_all_tr['Policy Start Date'].dt.day\n# drop policy start date and end date\nX_all_tr.drop(['Policy Start Date','Policy End Date','Duration Timedelta'], axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:20.492603Z","iopub.execute_input":"2024-12-28T12:20:20.4929Z","iopub.status.idle":"2024-12-28T12:20:21.189667Z","shell.execute_reply.started":"2024-12-28T12:20:20.492871Z","shell.execute_reply":"2024-12-28T12:20:21.188984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert 'Policy Start Date' to datetime format\nX_all_te['Policy Start Date'] = pd.to_datetime(X_all_te['Policy Start Date'])\n# Extract year, month and day\nX_all_te['Yea_Start'] = X_all_te['Policy Start Date'].dt.year\nX_all_te['Month_Start'] = X_all_te['Policy Start Date'].dt.month\nX_all_te['Day_Start'] = X_all_te['Policy Start Date'].dt.day\n# Add duration to policy start date\n# Convert 'Insurance Duration' to timedelta\nX_all_te['Duration Timedelta'] = pd.to_timedelta(X_all_te['Insurance Duration'], unit='D')\nX_all_te['Policy End Date'] = X_all_te['Policy Start Date']+X_all_te['Duration Timedelta']\nX_all_te['Year_End'] = X_all_te['Policy Start Date'].dt.year\nX_all_te['Month_End'] = X_all_te['Policy Start Date'].dt.month\nX_all_te['Day_End'] = X_all_te['Policy Start Date'].dt.day\n# drop policy start date and end date\nX_all_te.drop(['Policy Start Date','Policy End Date','Duration Timedelta'], axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:21.190427Z","iopub.execute_input":"2024-12-28T12:20:21.190727Z","iopub.status.idle":"2024-12-28T12:20:21.659687Z","shell.execute_reply.started":"2024-12-28T12:20:21.190697Z","shell.execute_reply":"2024-12-28T12:20:21.658989Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Clustering with K-Means","metadata":{}},{"cell_type":"code","source":"scaler = StandardScaler()\nscaled_tr = scaler.fit_transform(X_all_tr)\nscaled_te = scaler.fit_transform(X_all_te)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:21.660476Z","iopub.execute_input":"2024-12-28T12:20:21.660767Z","iopub.status.idle":"2024-12-28T12:20:22.527936Z","shell.execute_reply.started":"2024-12-28T12:20:21.660737Z","shell.execute_reply":"2024-12-28T12:20:22.527249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cluster using different k values\n#inertias = []\n#K = range(1, 11)  # Example range, adjust as necessary\n#for k in K:\n#    kmeans = KMeans(n_clusters=k, random_state=0)\n#    kmeans.fit(scaled_tr)\n#    inertias.append(kmeans.inertia_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:22.528591Z","iopub.execute_input":"2024-12-28T12:20:22.5288Z","iopub.status.idle":"2024-12-28T12:20:22.53207Z","shell.execute_reply.started":"2024-12-28T12:20:22.528782Z","shell.execute_reply":"2024-12-28T12:20:22.531278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot elbow method\n#plt.plot(K, inertias, 'bx-')\n#plt.xlabel('Number of clusters (k)')\n#plt.ylabel('Inertia')\n#plt.title('Elbow Method For Optimal k')\n#plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:22.533088Z","iopub.execute_input":"2024-12-28T12:20:22.533453Z","iopub.status.idle":"2024-12-28T12:20:22.553644Z","shell.execute_reply.started":"2024-12-28T12:20:22.533421Z","shell.execute_reply":"2024-12-28T12:20:22.552698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cluster using different k values\n#inertias = []\n#K = range(1, 11)  # Example range, adjust as necessary\n#for k in K:\n#    kmeans = KMeans(n_clusters=k, random_state=0)\n#    kmeans.fit(scaled_te)\n#    inertias.append(kmeans.inertia_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:22.554482Z","iopub.execute_input":"2024-12-28T12:20:22.554711Z","iopub.status.idle":"2024-12-28T12:20:22.568423Z","shell.execute_reply.started":"2024-12-28T12:20:22.554691Z","shell.execute_reply":"2024-12-28T12:20:22.567522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot elbow method\n#plt.plot(K, inertias, 'bx-')\n#plt.xlabel('Number of clusters (k)')\n#plt.ylabel('Inertia')\n#plt.title('Elbow Method For Optimal k')\n#plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:22.569154Z","iopub.execute_input":"2024-12-28T12:20:22.569341Z","iopub.status.idle":"2024-12-28T12:20:22.583356Z","shell.execute_reply.started":"2024-12-28T12:20:22.569324Z","shell.execute_reply":"2024-12-28T12:20:22.582722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"clusters_KM = (KMeans(n_clusters=5,\n                           random_state=0)\n            .fit(X_all_tr)\n            .labels_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:22.584048Z","iopub.execute_input":"2024-12-28T12:20:22.584271Z","iopub.status.idle":"2024-12-28T12:20:34.003767Z","shell.execute_reply.started":"2024-12-28T12:20:22.584253Z","shell.execute_reply":"2024-12-28T12:20:34.003074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Add the cluster labels to the dataframe\nX_all_tr['Cluster_KM'] = clusters_KM\n\n# Display the first few rows with cluster labels\nX_all_tr.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:34.004534Z","iopub.execute_input":"2024-12-28T12:20:34.00476Z","iopub.status.idle":"2024-12-28T12:20:34.026587Z","shell.execute_reply.started":"2024-12-28T12:20:34.00474Z","shell.execute_reply":"2024-12-28T12:20:34.025796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"clusters_KM = (KMeans(n_clusters=5,\n                           random_state=0)\n            .fit(X_all_te)\n            .labels_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:34.027337Z","iopub.execute_input":"2024-12-28T12:20:34.027577Z","iopub.status.idle":"2024-12-28T12:20:40.750444Z","shell.execute_reply.started":"2024-12-28T12:20:34.027555Z","shell.execute_reply":"2024-12-28T12:20:40.749723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Add the cluster labels to the dataframe\nX_all_te['Cluster_KM'] = clusters_KM\n\n# Display the first few rows with cluster labels\nX_all_te.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:40.751215Z","iopub.execute_input":"2024-12-28T12:20:40.751547Z","iopub.status.idle":"2024-12-28T12:20:40.771008Z","shell.execute_reply.started":"2024-12-28T12:20:40.751516Z","shell.execute_reply":"2024-12-28T12:20:40.770318Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Zero Variance Predictors","metadata":{}},{"cell_type":"code","source":"# Find features with variance equal zero or lower than 0.05\nto_drop = [col for col in X_all_tr.columns if np.var(X_all_tr[col]) ==0]\nto_drop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:40.771833Z","iopub.execute_input":"2024-12-28T12:20:40.772148Z","iopub.status.idle":"2024-12-28T12:20:40.926854Z","shell.execute_reply.started":"2024-12-28T12:20:40.772115Z","shell.execute_reply":"2024-12-28T12:20:40.925988Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Drop features with zero variance\nX_all_tr_v = X_all_tr.drop(X_all_tr[to_drop], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:40.927624Z","iopub.execute_input":"2024-12-28T12:20:40.927832Z","iopub.status.idle":"2024-12-28T12:20:40.99241Z","shell.execute_reply.started":"2024-12-28T12:20:40.927814Z","shell.execute_reply":"2024-12-28T12:20:40.991794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Find features with variance equal zero or lower than 0.05\nto_drop = [col for col in X_all_tr.columns if np.var(X_all_tr[col]) ==0]\nto_drop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:40.993084Z","iopub.execute_input":"2024-12-28T12:20:40.993317Z","iopub.status.idle":"2024-12-28T12:20:41.145121Z","shell.execute_reply.started":"2024-12-28T12:20:40.993295Z","shell.execute_reply":"2024-12-28T12:20:41.144247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Drop features with zero variance\nX_all_te_v = X_all_te.drop(X_all_te[to_drop], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:41.145994Z","iopub.execute_input":"2024-12-28T12:20:41.146258Z","iopub.status.idle":"2024-12-28T12:20:41.185807Z","shell.execute_reply.started":"2024-12-28T12:20:41.146225Z","shell.execute_reply":"2024-12-28T12:20:41.185021Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Normality Test","metadata":{}},{"cell_type":"code","source":"# Perform Normality test\nstat, p = shapiro(X_all_tr_v)\nprint('Statistics=%.3f, p=%.3f' % (stat, p))\n# Interpret the normality test result\nalpha = 0.05\nif p > alpha:\n    print('Sample looks Gaussian (fail to reject H0)')\nelse:\n    print('Sample does not look Gaussian (reject H0)')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:41.186511Z","iopub.execute_input":"2024-12-28T12:20:41.186711Z","iopub.status.idle":"2024-12-28T12:20:42.671332Z","shell.execute_reply.started":"2024-12-28T12:20:41.186694Z","shell.execute_reply":"2024-12-28T12:20:42.670682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Perform Normality test\nstat, p = shapiro(X_all_te_v)\nprint('Statistics=%.3f, p=%.3f' % (stat, p))\n# Interpret the normality test result\nalpha = 0.05\nif p > alpha:\n    print('Sample looks Gaussian (fail to reject H0)')\nelse:\n    print('Sample does not look Gaussian (reject H0)')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:42.672139Z","iopub.execute_input":"2024-12-28T12:20:42.672475Z","iopub.status.idle":"2024-12-28T12:20:43.800527Z","shell.execute_reply.started":"2024-12-28T12:20:42.672444Z","shell.execute_reply":"2024-12-28T12:20:43.799817Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Correlated Predictors","metadata":{}},{"cell_type":"code","source":"# Plot Correlation heatmap\ncorr_matrix = X_all_tr_v.corr(method='spearman')\nsns.set(rc = {'figure.figsize': (30, 30)})\nplt.figure()\nsns.heatmap(corr_matrix, square = True, annot=True, fmt='.2f')\nplt.title('Correlation Heatmap on train set',size=25)\nplt.yticks(fontsize=\"15\")\nplt.xticks(fontsize=\"15\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:43.801304Z","iopub.execute_input":"2024-12-28T12:20:43.801637Z","iopub.status.idle":"2024-12-28T12:20:54.712229Z","shell.execute_reply.started":"2024-12-28T12:20:43.801604Z","shell.execute_reply":"2024-12-28T12:20:54.711359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Select correlated features and removed it\n# Select upper triangle of correlation matrix\nupper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))\n# Find index of feature columns with correlation greater than 0.75\nto_drop = [column for column in upper.columns if any(upper[column].abs() > 0.75)]\nto_drop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:54.713224Z","iopub.execute_input":"2024-12-28T12:20:54.713508Z","iopub.status.idle":"2024-12-28T12:20:54.724655Z","shell.execute_reply.started":"2024-12-28T12:20:54.713485Z","shell.execute_reply":"2024-12-28T12:20:54.723933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop highly correlated features\nX_all_tr_f = X_all_tr_v.drop(X_all_tr_v[to_drop], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:54.725363Z","iopub.execute_input":"2024-12-28T12:20:54.725709Z","iopub.status.idle":"2024-12-28T12:20:54.817456Z","shell.execute_reply.started":"2024-12-28T12:20:54.725676Z","shell.execute_reply":"2024-12-28T12:20:54.81676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot Correlation heatmap\ncorr_matrix = X_all_te_v.corr(method='spearman')\nsns.set(rc = {'figure.figsize': (30, 30)})\nplt.figure()\nsns.heatmap(corr_matrix, square = True, annot=True, fmt='.2f')\nplt.title('Correlation Heatmap on test set',size=25)\nplt.yticks(fontsize=\"15\")\nplt.xticks(fontsize=\"15\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:20:54.818167Z","iopub.execute_input":"2024-12-28T12:20:54.818397Z","iopub.status.idle":"2024-12-28T12:21:02.540836Z","shell.execute_reply.started":"2024-12-28T12:20:54.818362Z","shell.execute_reply":"2024-12-28T12:21:02.539934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select correlated features and removed it\n# Select upper triangle of correlation matrix\nupper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))\n# Find index of feature columns with correlation greater than 0.75\nto_drop = [column for column in upper.columns if any(upper[column].abs() > 0.75)]\nto_drop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:21:02.541639Z","iopub.execute_input":"2024-12-28T12:21:02.541874Z","iopub.status.idle":"2024-12-28T12:21:02.552694Z","shell.execute_reply.started":"2024-12-28T12:21:02.541844Z","shell.execute_reply":"2024-12-28T12:21:02.551949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop highly correlated features\nX_all_te_f = X_all_te_v.drop(X_all_te_v[to_drop], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:21:02.553541Z","iopub.execute_input":"2024-12-28T12:21:02.553823Z","iopub.status.idle":"2024-12-28T12:21:02.622626Z","shell.execute_reply.started":"2024-12-28T12:21:02.553788Z","shell.execute_reply":"2024-12-28T12:21:02.621703Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LightGBM Modelling","metadata":{}},{"cell_type":"markdown","source":"### Fine-tuning the model","metadata":{}},{"cell_type":"code","source":"def RMSLE(y_true: list, y_pred: list) :\n    \"\"\"\n    The Root Mean Squared Log Error (RMSLE) metric using only NumPy\n    \n    :param y_true: The ground truth labels given in the dataset\n    :param y_pred: Our predictions\n    :return: The RMSLE score\n    \"\"\"\n    n = len(y_true)\n    RMSLE = np.sqrt(np.mean(np.square(np.log1p(y_pred) - np.log1p(y_true))))\n    return RMSLE","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:21:14.316621Z","iopub.execute_input":"2024-12-28T12:21:14.316906Z","iopub.status.idle":"2024-12-28T12:21:14.321306Z","shell.execute_reply.started":"2024-12-28T12:21:14.316882Z","shell.execute_reply":"2024-12-28T12:21:14.320489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fine-tuning\n\n#def objective(trial):\n#    params = {\n#        'objective':'Gamma',\n#        'random_state': 0,\n#        'n_estimators': trial.suggest_int('n_estimators', 50, 1000),\n#        'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.1),\n#        'max_depth': trial.suggest_int('max_depth', 3, 8),\n#        'num_leaves': trial.suggest_int('num_leaves', 20, 150),\n#        'min_child_samples': trial.suggest_int('min_child_samples', 5, 100),\n#        'n_jobs':-1,\n#        'verbosity': -1\n#    }\n\n#    lgbm = LGBMRegressor(**params)\n\n#    cv=KFold(n_splits=5, shuffle=True)\n\n#    rmsle_scores = []\n\n\n#    for train_index, val_index in cv.split(X_all_tr_f):\n#        X_tr, X_val = X_all_tr_f.iloc[train_index], X_all_tr_f.iloc[val_index]\n#        y_tr, y_val = y.iloc[train_index], y.iloc[val_index]\n\n#        lgbm.fit(X_tr, np.log1p(y_tr),  sample_weight=X_tr['Previous Claims'].values)\n#        pred_val = lgbm.predict(X_val)\n\n#        rmsle_score = RMSLE(np.log1p(y_val), pred_val)\n#        rmsle_scores.append(rmsle_score)\n\n#    return np.mean(rmsle_scores)\n\n#study = optuna.create_study(direction='minimize')\n#study.optimize(objective, n_trials=10)\n\n#best_params = study.best_params\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:21:16.85301Z","iopub.execute_input":"2024-12-28T12:21:16.853315Z","iopub.status.idle":"2024-12-28T12:40:46.758294Z","shell.execute_reply.started":"2024-12-28T12:21:16.853288Z","shell.execute_reply":"2024-12-28T12:40:46.757653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#best_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:40:53.670781Z","iopub.execute_input":"2024-12-28T12:40:53.671088Z","iopub.status.idle":"2024-12-28T12:40:53.676276Z","shell.execute_reply.started":"2024-12-28T12:40:53.671061Z","shell.execute_reply":"2024-12-28T12:40:53.675454Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Fit & Prediction","metadata":{}},{"cell_type":"code","source":"# Initialize and configure the LGBMRegressor with optimized parameters\nlgbm_tuned = LGBMRegressor(\n        objective='Gamma',\n        n_estimators= 973,\n        learning_rate= 0.010992925333895221,\n        max_depth= 8,\n        min_child_samples=55,\n        num_leaves= 23,\n        n_jobs=-1,\n        verbosity=-1,\n        random_state=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:42:23.27053Z","iopub.execute_input":"2024-12-28T12:42:23.270842Z","iopub.status.idle":"2024-12-28T12:42:23.275082Z","shell.execute_reply.started":"2024-12-28T12:42:23.270817Z","shell.execute_reply":"2024-12-28T12:42:23.274221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Perform the train-validation split\nX_train, X_val, y_train, y_val = train_test_split(X_all_tr_f, y, test_size=0.2, random_state=0)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:42:27.387984Z","iopub.execute_input":"2024-12-28T12:42:27.388301Z","iopub.status.idle":"2024-12-28T12:42:27.671033Z","shell.execute_reply.started":"2024-12-28T12:42:27.388275Z","shell.execute_reply":"2024-12-28T12:42:27.670079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cv_5=KFold(n_splits=5, shuffle=True)\nparam_grid = {}\nlgbm_model = GridSearchCV(lgbm_tuned,param_grid,cv=cv_5)\nlgbm_fitted=lgbm_model.fit(X_train, np.log1p(y_train),\n               sample_weight=X_train['Previous Claims'].values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:53:28.58507Z","iopub.execute_input":"2024-12-28T12:53:28.585355Z","iopub.status.idle":"2024-12-28T12:56:55.719741Z","shell.execute_reply.started":"2024-12-28T12:53:28.585331Z","shell.execute_reply":"2024-12-28T12:56:55.718986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score_lgbm = []\npredictions_tr_lgbm = lgbm_fitted.predict(X_train)\npredictions_val_lgbm = lgbm_fitted.predict(X_val)\n\nrmsle_train = RMSLE(np.log1p(y_train), predictions_tr_lgbm)\nrmsle_val = RMSLE(np.log1p(y_val), predictions_val_lgbm)\n\nscore_dict = {\n        'rmsle_train': rmsle_train,\n        'rmsle_val': rmsle_val,\n    }\n\nscore_lgbm.append(score_dict)\nscore_lgbm = pd.DataFrame(score_lgbm, columns = ['rmsle_train','rmsle_val'])\nscore_lgbm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:56:55.72123Z","iopub.execute_input":"2024-12-28T12:56:55.721556Z","iopub.status.idle":"2024-12-28T12:57:19.922344Z","shell.execute_reply.started":"2024-12-28T12:56:55.721525Z","shell.execute_reply":"2024-12-28T12:57:19.921732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions_te_lgbm = lgbm_fitted.predict(X_all_te_f)\npred_te_lgbm = np.expm1(predictions_te_lgbm)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:57:19.923606Z","iopub.execute_input":"2024-12-28T12:57:19.924082Z","iopub.status.idle":"2024-12-28T12:57:37.130726Z","shell.execute_reply.started":"2024-12-28T12:57:19.924057Z","shell.execute_reply":"2024-12-28T12:57:37.129944Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission['Premium Amount'] = pred_te_lgbm\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T12:57:37.131775Z","iopub.execute_input":"2024-12-28T12:57:37.132063Z","iopub.status.idle":"2024-12-28T12:57:38.445501Z","shell.execute_reply.started":"2024-12-28T12:57:37.13204Z","shell.execute_reply":"2024-12-28T12:57:38.444819Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Importance","metadata":{}},{"cell_type":"code","source":"# Perform the train-validation split\n# Step 1: Sample 30% from original X_train and y_train\nX_sample, _, y_sample, _ = train_test_split(X_train, y_train, test_size=0.7, random_state=0)\n\nX_train_sample, X_val_sample, y_train_sample, y_val_sample = train_test_split(\n    X_sample, y_sample, test_size=0.2, random_state=0\n)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T13:00:00.304831Z","iopub.execute_input":"2024-12-28T13:00:00.305208Z","iopub.status.idle":"2024-12-28T13:00:00.587383Z","shell.execute_reply.started":"2024-12-28T13:00:00.305177Z","shell.execute_reply":"2024-12-28T13:00:00.58644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train a LightGBM model and calculate SHAP values for global feature importance\n# Create and fit a LightGBM regressor with specified parameters\nLGBM_ = LGBMRegressor(\n        objective='Gamma',\n        n_estimators= 973,\n        learning_rate= 0.010992925333895221,\n        max_depth= 8,\n        min_child_samples=55,\n        num_leaves= 23,\n        n_jobs=-1,\n        verbosity=-1,\n        random_state=0).fit(X_train_sample, y_train_sample)\n# Create a SHAP explainer for the LightGBM model\nLGBM_explainer = shap.TreeExplainer(LGBM_)\n# Calculate SHAP values for the test dataset\nLGBM_shap_values = LGBM_explainer.shap_values(X_val_sample)\n# Set figure size for the plot\nplt.rcParams['figure.figsize'] = (5,5)\n# Set title for the SHAP feature importance plot\nplt.title(\"LGBM SHAP FEATURES IMPORTANCE\")\n# Plot SHAP summary plot for the LightGBM model\nshap.summary_plot(LGBM_shap_values, features=X_val_sample, feature_names=X_val_sample.columns,plot_type='bar')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T13:03:25.961022Z","iopub.execute_input":"2024-12-28T13:03:25.96131Z","iopub.status.idle":"2024-12-28T13:05:20.577873Z","shell.execute_reply.started":"2024-12-28T13:03:25.961288Z","shell.execute_reply":"2024-12-28T13:05:20.57697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}