{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"1f67f76f-2370-4184-9b17-b36cab4524f1","cell_type":"markdown","source":"Note that there is no dependance between the different variables. The data semms not reflect the reality.","metadata":{}},{"id":"06b97d12-f038-4301-8bdd-6570624c61ac","cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.preprocessing import LabelEncoder\nfrom skopt import BayesSearchCV\nfrom xgboost import XGBRegressor\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.model_selection import KFold\nfrom sklearn.linear_model import LassoCV\nfrom sklearn.tree import DecisionTreeRegressor","metadata":{},"outputs":[],"execution_count":null},{"id":"17127305-42da-4f39-95ba-5b41c904b3d9","cell_type":"markdown","source":"# Data importation","metadata":{}},{"id":"44a940a2-fa99-4b2f-bc87-97d0d5bb773f","cell_type":"code","source":"data_train=pd.read_csv('train.csv')\ndata_test=pd.read_csv('test.csv')","metadata":{},"outputs":[],"execution_count":null},{"id":"84bf87d7-ad26-405b-80c5-33d6633a259a","cell_type":"code","source":"data_train['Policy Start Date'] = pd.to_datetime(data_train['Policy Start Date'])\ndata_test['Policy Start Date'] = pd.to_datetime(data_test['Policy Start Date'])","metadata":{},"outputs":[],"execution_count":null},{"id":"392e5907-7b7a-4554-b702-fe7cf6dcb4bb","cell_type":"code","source":"data_train.info()","metadata":{},"outputs":[],"execution_count":null},{"id":"05710333-2567-42e9-8aed-543af99f4066","cell_type":"code","source":"data2=data_train.copy(deep=True)","metadata":{},"outputs":[],"execution_count":null},{"id":"d635f725-27a5-4107-b1b2-6021559194e3","cell_type":"markdown","source":"# Descriptive statistics","metadata":{}},{"id":"871b7e6e-b624-4f4c-b585-505f905687c5","cell_type":"markdown","source":"## Response variable","metadata":{}},{"id":"38573a60-8a1a-4ea6-89fb-8ca225b83b2b","cell_type":"code","source":"data2.columns","metadata":{},"outputs":[],"execution_count":null},{"id":"7a2bccdd-656d-4ed0-b8bd-6d63b1b55a16","cell_type":"code","source":"sns.displot(data=data2, x='Premium Amount')\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"id":"c3f7b32e-18bf-4c28-8a42-2738332c1429","cell_type":"markdown","source":"As the response variable is highky skewed, I will use the logarithm of this variable.","metadata":{}},{"id":"4f6a41d4-c3af-4294-b102-85c6e4591850","cell_type":"code","source":"data2['Pr_amount_log']=np.log(data2['Premium Amount'])\nplt.hist(data2['Pr_amount_log'])","metadata":{},"outputs":[],"execution_count":null},{"id":"dc52e8ab-750e-4179-a88d-179dd525e568","cell_type":"markdown","source":"# Explanatory variables","metadata":{}},{"id":"51ae4f2c-bd40-450b-8196-720e8630ecbf","cell_type":"code","source":"plt.figure(figsize=(10, 4))\nplt.subplot(1, 2, 1)\nsns.histplot(data_train['Age'], kde=True, bins=20, color='blue')\nplt.subplot(1, 2, 2)\nsns.scatterplot(data=data2, x='Age', y='Pr_amount_log')\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"id":"f81437c3-457a-4a62-aee1-a23a2b2ea859","cell_type":"code","source":"plt.figure(figsize=(10, 4))\nplt.subplot(2, 2, 1)\nsns.histplot(data_train['Annual Income'], kde=True, bins=20, color='blue')\nplt.subplot(2, 2, 2)\nsns.scatterplot(data=data2, x='Annual Income', y='Pr_amount_log')\nplt.subplot(2, 2, 3)\nsns.histplot(np.sqrt(data_train['Annual Income']), kde=True, bins=20, color='blue')\nplt.subplot(2, 2, 4)\nsns.histplot(np.log(data_train['Annual Income']), kde=True, bins=20, color='blue')\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"id":"215a33ae-31d6-4279-92c6-238898764e08","cell_type":"markdown","source":"As `Annual Income` is skewed, it could be usefull to change it using the squared root or the logarithm. After checking using different prediction models, it is not usefull to use a transformation.","metadata":{}},{"id":"6653aace-1d74-4e0d-867f-41f31df8afce","cell_type":"code","source":"data2['Annual_income_sqrt']=np.sqrt(data2['Annual Income'])\ndata2['Annual_income_log']=np.sqrt(data2['Annual Income'])","metadata":{},"outputs":[],"execution_count":null},{"id":"0b1e2609-1c1e-4e93-b239-e28dadff00d5","cell_type":"code","source":"plt.figure(figsize=(10, 4))\nplt.subplot(1, 2, 1)\nsns.histplot(data_train['Number of Dependents'], kde=True, bins=20, color='blue')\nplt.subplot(1, 2, 2)\nsns.scatterplot(data=data2, x='Number of Dependents', y='Pr_amount_log')\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"id":"290ac4be-b7bf-49a0-a529-c1fc902c72cb","cell_type":"code","source":"plt.figure(figsize=(10, 4))\nplt.subplot(1, 2, 1)\nsns.histplot(data_train['Health Score'], kde=True, bins=20, color='blue')\nplt.subplot(1, 2, 2)\nsns.scatterplot(data=data2, x='Health Score', y='Pr_amount_log')\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"id":"46ba8245-73df-4a03-8c62-95e87573f457","cell_type":"code","source":"plt.figure(figsize=(10, 4))\nplt.subplot(1, 2, 1)\nsns.histplot(data_train['Previous Claims'], kde=True, bins=20, color='blue')\nplt.subplot(1, 2, 2)\nsns.scatterplot(data=data2, x='Previous Claims', y='Pr_amount_log')\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"id":"e7b0f110-8b6d-48cf-8d7d-b7d70f4c9958","cell_type":"code","source":"plt.figure(figsize=(10, 4))\nplt.subplot(1, 2, 1)\nsns.histplot(data_train['Vehicle Age'], kde=True, bins=20, color='blue')\nplt.subplot(1, 2, 2)\nsns.scatterplot(data=data2, x='Vehicle Age', y='Pr_amount_log')\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"id":"729f72f6-8e92-4949-9fa7-365802d08f41","cell_type":"code","source":"plt.figure(figsize=(10, 4))\nplt.subplot(1, 2, 1)\nsns.histplot(data_train['Credit Score'], kde=True, bins=20, color='blue')\nplt.subplot(1, 2, 2)\nsns.scatterplot(data=data2, x='Credit Score', y='Pr_amount_log')\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"id":"3af15b51-c634-49e8-9273-595fa900cff2","cell_type":"code","source":"plt.figure(figsize=(10, 4))\nplt.subplot(1, 2, 1)\nsns.histplot(data_train['Insurance Duration'], kde=True, bins=20, color='blue')\nplt.subplot(1, 2, 2)\nsns.scatterplot(data=data2, x='Insurance Duration', y='Pr_amount_log')\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"id":"5fc37cff-691a-4133-9ca4-f049cc5ad7ab","cell_type":"code","source":"sns.scatterplot(data=data2, x='Policy Start Date', y='Pr_amount_log')","metadata":{},"outputs":[],"execution_count":null},{"id":"d84896b1-8284-4715-aa98-142a9aececd9","cell_type":"markdown","source":"Transformation of `Policy Start Date` as a difference with the first date.","metadata":{}},{"id":"c1ef1617-46a0-47b5-888a-cd626d9fa934","cell_type":"code","source":"data2['date']=(data2['Policy Start Date'] - min(data2['Policy Start Date'])).dt.total_seconds()//1000000","metadata":{},"outputs":[],"execution_count":null},{"id":"06a16ad9-96d2-491a-9c84-4147dc9c6214","cell_type":"code","source":"data2['date'].describe()","metadata":{},"outputs":[],"execution_count":null},{"id":"e5cced45-fb18-4bb2-8958-205ff8c31926","cell_type":"code","source":"plt.figure(figsize=(10, 4))\nplt.subplot(1, 2, 1)\nsns.histplot(data2['date'], kde=True, bins=20, color='blue')\nplt.subplot(1, 2, 2)\nsns.scatterplot(data=data2, x='date', y='Pr_amount_log')\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"id":"87669a64-f33d-41a3-ada9-eac20d378fe8","cell_type":"code","source":"fig, axs = plt.subplots(4,3, figsize=(10, 10))\nplt.figure(figsize=(5, 2))\nsns.barplot(ax=axs[0, 0],data=data_train, x=\"Gender\", y=\"Premium Amount\")\nsns.barplot(ax=axs[0, 1],data=data_train, x=\"Marital Status\", y=\"Premium Amount\")\nsns.barplot(ax=axs[0, 2],data=data_train, x=\"Education Level\", y=\"Premium Amount\")\nsns.barplot(ax=axs[1 ,0],data=data_train, x=\"Occupation\", y=\"Premium Amount\")\nsns.barplot(ax=axs[1, 1],data=data_train, x=\"Location\", y=\"Premium Amount\")\nsns.barplot(ax=axs[1, 2],data=data_train, x=\"Policy Type\", y=\"Premium Amount\")\nsns.barplot(ax=axs[2, 0],data=data_train, x=\"Smoking Status\", y=\"Premium Amount\")\nsns.barplot(ax=axs[2, 1],data=data_train, x=\"Exercise Frequency\", y=\"Premium Amount\")\nsns.barplot(ax=axs[2, 2],data=data_train, x=\"Property Type\", y=\"Premium Amount\")\nsns.barplot(ax=axs[3, 0],data=data_train, x=\"Customer Feedback\", y=\"Premium Amount\")\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"id":"d33caca6-aaa0-4f50-9a64-8623dd80eba9","cell_type":"code","source":"data2.columns","metadata":{},"outputs":[],"execution_count":null},{"id":"44221496-cc1f-4b91-b24c-53152e938cf3","cell_type":"markdown","source":"# Dealing with missing values","metadata":{}},{"id":"2ce341f1-3a22-43ae-b43a-59f8c268e7c4","cell_type":"markdown","source":"First see if some continuous variables are influenced by categorical ones. There is no impact so imputation can be simply done using the mode for the categorical variables and the mean or median for the continuous.","metadata":{}},{"id":"74673664-97b7-428d-aca1-1f9eda29dac9","cell_type":"code","source":"data_train.groupby('Gender')[['Annual Income', 'Age', 'Number of Dependents', 'Health Score', 'Previous Claims', \\\n                              'Vehicle Age', 'Credit Score', 'Insurance Duration']].mean()","metadata":{},"outputs":[],"execution_count":null},{"id":"bcb018e0-d07d-4b43-908c-d22ab7009b29","cell_type":"code","source":"fig, axs = plt.subplots(2,4, figsize=(10, 10))\nplt.figure(figsize=(5, 2))\nsns.boxplot(ax=axs[0, 0],x='Gender', y='Annual Income', data=data_train)\nsns.boxplot(ax=axs[0, 1],x='Gender', y='Age', data=data_train)\nsns.boxplot(ax=axs[0, 2],x='Gender', y='Number of Dependents', data=data_train)\nsns.boxplot(ax=axs[0, 3],x='Gender', y='Health Score', data=data_train)\nsns.boxplot(ax=axs[1, 0],x='Gender', y='Previous Claims', data=data_train)\nsns.boxplot(ax=axs[1, 1],x='Gender', y='Vehicle Age', data=data_train)\nsns.boxplot(ax=axs[1, 2],x='Gender', y='Credit Score', data=data_train)\nsns.boxplot(ax=axs[1, 3],x='Gender', y='Insurance Duration', data=data_train)","metadata":{},"outputs":[],"execution_count":null},{"id":"9bcd71dc-cedf-461b-9f75-4f7296feca04","cell_type":"code","source":"fig, axs = plt.subplots(2,4, figsize=(10, 10))\nplt.figure(figsize=(5, 2))\nsns.boxplot(ax=axs[0, 0],x='Marital Status', y='Annual Income', data=data_train)\nsns.boxplot(ax=axs[0, 1],x='Marital Status', y='Age', data=data_train)\nsns.boxplot(ax=axs[0, 2],x='Marital Status', y='Number of Dependents', data=data_train)\nsns.boxplot(ax=axs[0, 3],x='Marital Status', y='Health Score', data=data_train)\nsns.boxplot(ax=axs[1, 0],x='Marital Status', y='Previous Claims', data=data_train)\nsns.boxplot(ax=axs[1, 1],x='Marital Status', y='Vehicle Age', data=data_train)\nsns.boxplot(ax=axs[1, 2],x='Marital Status', y='Credit Score', data=data_train)\nsns.boxplot(ax=axs[1, 3],x='Marital Status', y='Insurance Duration', data=data_train)","metadata":{"scrolled":true},"outputs":[],"execution_count":null},{"id":"be5bb479-e39a-453b-95ed-2b0583140a31","cell_type":"code","source":"fig, axs = plt.subplots(2,4, figsize=(10, 10))\nplt.figure(figsize=(5, 2))\nsns.boxplot(ax=axs[0, 0],x='Education Level', y='Annual Income', data=data_train)\nsns.boxplot(ax=axs[0, 1],x='Education Level', y='Age', data=data_train)\nsns.boxplot(ax=axs[0, 2],x='Education Level', y='Number of Dependents', data=data_train)\nsns.boxplot(ax=axs[0, 3],x='Education Level', y='Health Score', data=data_train)\nsns.boxplot(ax=axs[1, 0],x='Education Level', y='Previous Claims', data=data_train)\nsns.boxplot(ax=axs[1, 1],x='Education Level', y='Vehicle Age', data=data_train)\nsns.boxplot(ax=axs[1, 2],x='Education Level', y='Credit Score', data=data_train)\nsns.boxplot(ax=axs[1, 3],x='Education Level', y='Insurance Duration', data=data_train)","metadata":{"scrolled":true},"outputs":[],"execution_count":null},{"id":"80869905-236d-48dd-acbe-6b126affbaad","cell_type":"code","source":"fig, axs = plt.subplots(2,4, figsize=(10, 10))\nplt.figure(figsize=(5, 2))\nsns.boxplot(ax=axs[0, 0],x='Occupation', y='Annual Income', data=data_train)\nsns.boxplot(ax=axs[0, 1],x='Occupation', y='Age', data=data_train)\nsns.boxplot(ax=axs[0, 2],x='Occupation', y='Number of Dependents', data=data_train)\nsns.boxplot(ax=axs[0, 3],x='Occupation', y='Health Score', data=data_train)\nsns.boxplot(ax=axs[1, 0],x='Occupation', y='Previous Claims', data=data_train)\nsns.boxplot(ax=axs[1, 1],x='Occupation', y='Vehicle Age', data=data_train)\nsns.boxplot(ax=axs[1, 2],x='Occupation', y='Credit Score', data=data_train)\nsns.boxplot(ax=axs[1, 3],x='Occupation', y='Insurance Duration', data=data_train)","metadata":{},"outputs":[],"execution_count":null},{"id":"8b2b4a6b-e4d3-4ef9-a97e-237f8d9e1d48","cell_type":"code","source":"fig, axs = plt.subplots(2,4, figsize=(10, 10))\nplt.figure(figsize=(5, 2))\nsns.boxplot(ax=axs[0, 0],x='Location', y='Annual Income', data=data_train)\nsns.boxplot(ax=axs[0, 1],x='Location', y='Age', data=data_train)\nsns.boxplot(ax=axs[0, 2],x='Location', y='Number of Dependents', data=data_train)\nsns.boxplot(ax=axs[0, 3],x='Location', y='Health Score', data=data_train)\nsns.boxplot(ax=axs[1, 0],x='Location', y='Previous Claims', data=data_train)\nsns.boxplot(ax=axs[1, 1],x='Location', y='Vehicle Age', data=data_train)\nsns.boxplot(ax=axs[1, 2],x='Location', y='Credit Score', data=data_train)\nsns.boxplot(ax=axs[1, 3],x='Location', y='Insurance Duration', data=data_train)","metadata":{"scrolled":true},"outputs":[],"execution_count":null},{"id":"aa3e5137-f507-48d5-8d37-b92a439b611e","cell_type":"code","source":"fig, axs = plt.subplots(2,4, figsize=(10, 10))\nplt.figure(figsize=(5, 2))\nsns.boxplot(ax=axs[0, 0],x='Policy Type', y='Annual Income', data=data_train)\nsns.boxplot(ax=axs[0, 1],x='Policy Type', y='Age', data=data_train)\nsns.boxplot(ax=axs[0, 2],x='Policy Type', y='Number of Dependents', data=data_train)\nsns.boxplot(ax=axs[0, 3],x='Policy Type', y='Health Score', data=data_train)\nsns.boxplot(ax=axs[1, 0],x='Policy Type', y='Previous Claims', data=data_train)\nsns.boxplot(ax=axs[1, 1],x='Policy Type', y='Vehicle Age', data=data_train)\nsns.boxplot(ax=axs[1, 2],x='Policy Type', y='Credit Score', data=data_train)\nsns.boxplot(ax=axs[1, 3],x='Policy Type', y='Insurance Duration', data=data_train)","metadata":{},"outputs":[],"execution_count":null},{"id":"149fcc28-4a06-4b9e-a725-9def5cd091bd","cell_type":"code","source":"fig, axs = plt.subplots(2,4, figsize=(10, 10))\nplt.figure(figsize=(5, 2))\nsns.boxplot(ax=axs[0, 0],x='Smoking Status', y='Annual Income', data=data_train)\nsns.boxplot(ax=axs[0, 1],x='Smoking Status', y='Age', data=data_train)\nsns.boxplot(ax=axs[0, 2],x='Smoking Status', y='Number of Dependents', data=data_train)\nsns.boxplot(ax=axs[0, 3],x='Smoking Status', y='Health Score', data=data_train)\nsns.boxplot(ax=axs[1, 0],x='Smoking Status', y='Previous Claims', data=data_train)\nsns.boxplot(ax=axs[1, 1],x='Smoking Status', y='Vehicle Age', data=data_train)\nsns.boxplot(ax=axs[1, 2],x='Smoking Status', y='Credit Score', data=data_train)\nsns.boxplot(ax=axs[1, 3],x='Smoking Status', y='Insurance Duration', data=data_train)","metadata":{},"outputs":[],"execution_count":null},{"id":"0b8cecb8-6e64-4374-9e88-678e0f507f7b","cell_type":"code","source":"# For categorical variable, imputation using the mode\ndata2['Marital Status']=data2['Marital Status'].fillna(data_train['Marital Status'].mode()[0])\ndata2['Occupation']=data2['Occupation'].fillna(data_train['Occupation'].mode()[0])\ndata2['Customer Feedback']=data2['Customer Feedback'].fillna(data_train['Customer Feedback'].mode()[0])","metadata":{},"outputs":[],"execution_count":null},{"id":"e0e70e1c-b972-4959-9e9c-ea098c04ffda","cell_type":"code","source":"# For numerical variables, imputation using mean or median. Using median when the distribution is symetric and mean otherwise\ndata2['Age']=data2['Age'].fillna(data_train['Age'].median())\ndata2['Annual Income']=data2['Annual Income'].fillna((data_train['Annual Income']).mean())\ndata2['Number of Dependents']=data2['Number of Dependents'].fillna(data_train['Number of Dependents'].median())\ndata2['Health Score']=data2['Health Score'].fillna(data_train['Health Score'].mean())\ndata2['Previous Claims']=data2['Previous Claims'].fillna(data_train['Previous Claims'].mean())\ndata2['Vehicle Age']=data2['Vehicle Age'].fillna(data_train['Vehicle Age'].median())\ndata2['Credit Score']=data2['Credit Score'].fillna(data_train['Credit Score'].mean())\ndata2['Insurance Duration']=data2['Insurance Duration'].fillna(data_train['Insurance Duration'].median())","metadata":{},"outputs":[],"execution_count":null},{"id":"7dccd22c-e573-40b6-b47c-b7e842d1eef0","cell_type":"code","source":"data2.info()","metadata":{"scrolled":true},"outputs":[],"execution_count":null},{"id":"19e3a455-f1ae-41e5-b271-951ad29c7e4d","cell_type":"markdown","source":"# Cleaning data","metadata":{}},{"id":"9eb5815a-119e-4f40-87be-47cbad4d3abc","cell_type":"code","source":"data2['Gender']=np.where(data2['Gender']=='Male',1,0)","metadata":{},"outputs":[],"execution_count":null},{"id":"bd36357d-49fe-435e-b298-19328af95d96","cell_type":"code","source":"d_marital=pd.get_dummies(data2['Marital Status'], drop_first=True)\nd_education=pd.get_dummies(data2['Education Level'], drop_first=True)\nd_occupation=pd.get_dummies(data2['Occupation'], drop_first=True)\nd_location=pd.get_dummies(data2['Location'], drop_first=True)\nd_policy=pd.get_dummies(data2['Policy Type'], drop_first=True)\nd_smoking=pd.get_dummies(data2['Smoking Status'], drop_first=True)\nd_exercise=pd.get_dummies(data2['Exercise Frequency'], drop_first=True)\nd_property=pd.get_dummies(data2['Property Type'], drop_first=True)\nd_customer=pd.get_dummies(data2['Customer Feedback'], drop_first=True)\n\nX=pd.concat([d_marital,d_education,d_occupation,d_location,d_policy,d_smoking,d_exercise,d_property,d_customer, \\\n    data2[['Gender', 'Age', 'Annual Income', 'Number of Dependents', 'Health Score', 'Previous Claims', 'Vehicle Age', 'Credit Score', \\\n           'Insurance Duration', 'date']]], axis=1)\ny=data2['Pr_amount_log']","metadata":{},"outputs":[],"execution_count":null},{"id":"7436e892-6f35-4495-80d7-bf7738c6dfba","cell_type":"markdown","source":"# Scale data","metadata":{}},{"id":"afabc745-1214-4d0a-9e6e-53b535e74a34","cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)","metadata":{},"outputs":[],"execution_count":null},{"id":"5e8bb81c-ccf3-4df1-8cc4-ccbc09b9d857","cell_type":"markdown","source":"# Preparing test data","metadata":{}},{"id":"84828371-3236-4d09-abc2-07c01f8611fd","cell_type":"code","source":"X_pred=data_test.copy(deep=True)\nX_pred['Annual_income_sqrt']=np.sqrt(X_pred['Annual Income'])\nX_pred['Annual_income_log']=np.log(X_pred['Annual Income'])\nX_pred['date']=(X_pred['Policy Start Date'] - min(data2['Policy Start Date'])).dt.total_seconds()//1000000","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"ed373423-9fcf-4283-a16a-7e24a49745ff","cell_type":"code","source":"# For categorical variable, imputation using th mode\nX_pred['Marital Status']=X_pred['Marital Status'].fillna(data_test['Marital Status'].mode()[0])\nX_pred['Occupation']=X_pred['Occupation'].fillna(data_test['Occupation'].mode()[0])\nX_pred['Customer Feedback']=X_pred['Customer Feedback'].fillna(data_test['Customer Feedback'].mode()[0])\n\n# For numerical variables, imputation using mean or median. Using median when the distribution is symetric and mean otherwise\nX_pred['Age']=X_pred['Age'].fillna(data_test['Age'].median())\nX_pred['Annual Income']=X_pred['Annual Income'].fillna((data_test['Annual Income']).mean())\nX_pred['Number of Dependents']=X_pred['Number of Dependents'].fillna(data_test['Number of Dependents'].median())\nX_pred['Health Score']=X_pred['Health Score'].fillna(data_test['Health Score'].mean())\nX_pred['Previous Claims']=X_pred['Previous Claims'].fillna(data_test['Previous Claims'].mean())\nX_pred['Vehicle Age']=X_pred['Vehicle Age'].fillna(data_test['Vehicle Age'].median())\nX_pred['Credit Score']=X_pred['Credit Score'].fillna(data_test['Credit Score'].mean())\nX_pred['Insurance Duration']=X_pred['Insurance Duration'].fillna(data_test['Insurance Duration'].median())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"1784e283-e808-469f-a655-1c7dbddb580b","cell_type":"code","source":"X_pred['Gender']=np.where(X_pred['Gender']=='Male',1,0)\n\nd_marital=pd.get_dummies(X_pred['Marital Status'], drop_first=True)\nd_education=pd.get_dummies(X_pred['Education Level'], drop_first=True)\nd_occupation=pd.get_dummies(X_pred['Occupation'], drop_first=True)\nd_location=pd.get_dummies(X_pred['Location'], drop_first=True)\nd_policy=pd.get_dummies(X_pred['Policy Type'], drop_first=True)\nd_smoking=pd.get_dummies(X_pred['Smoking Status'], drop_first=True)\nd_exercise=pd.get_dummies(X_pred['Exercise Frequency'], drop_first=True)\nd_property=pd.get_dummies(X_pred['Property Type'], drop_first=True)\nd_customer=pd.get_dummies(X_pred['Customer Feedback'], drop_first=True)\n\nX_pred2=pd.concat([d_marital,d_education,d_occupation,d_location,d_policy,d_smoking,d_exercise,d_property,d_customer, \\\n    X_pred[['Gender', 'Age', 'Annual Income', 'Number of Dependents', 'Health Score', 'Previous Claims', 'Vehicle Age', 'Credit Score', \\\n           'Insurance Duration', 'date']]], axis=1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"d6a4fc2d-074b-4dd3-b638-bd6ab0e408a0","cell_type":"code","source":"X_pred_scaled = scaler.transform(X_pred2)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"dce4276e-dc62-47b0-90e3-909514fdae49","cell_type":"markdown","source":"# Linear model","metadata":{}},{"id":"9c1b2983-fa48-43a7-aa06-c20b43e43f6e","cell_type":"code","source":"reg = LinearRegression()\nreg.fit(X_scaled, y)","metadata":{},"outputs":[],"execution_count":null},{"id":"a2d2eb63-0596-4881-b941-edc5c26d5d5f","cell_type":"code","source":"y_pred=reg.predict(X_pred_scaled)\nfinal=pd.DataFrame({'id': data_test['id'], 'Premium Amount': np.exp(y_pred)})\nprint(final.head())\nfinal.to_csv('CB_submisson.csv', index=False)","metadata":{},"outputs":[],"execution_count":null},{"id":"e5a70e19-d356-4d1e-a526-ba61e9f9234f","cell_type":"markdown","source":"# XGBoost regressor","metadata":{}},{"id":"9109b220-821e-498e-8846-cf7353011bca","cell_type":"code","source":"# Initialize the XGBoost regressor model\nmodel = XGBRegressor(random_state=42)\n\n# Hyperparameters for BayesSearchCV tuning\nsearch_spaces = {\n    'n_estimators': (10, 200),      \n    'max_depth': (3, 10),            \n    'reg_alpha': (0.001, 0.2, 'log-uniform'),  \n    'reg_lambda': (0.1, 100, 'log-uniform')     \n}\n\n# Set up BayesSearchCV for hyperparameter tuning\nsearch = BayesSearchCV(\n    estimator=model,\n    search_spaces=search_spaces,\n    n_iter=50,          # Number of iterations for optimization\n    cv=5,               # k-fold cross-validation\n    verbose=1,          # Display detailed logs\n    scoring=\"neg_root_mean_squared_error\",\n    random_state=42     \n)\n\n# Perform the Bayesian optimization with cross-validation\nsearch.fit(X_scaled, y)\n\nprint(\"Best params: \", search.best_params_)\nprint(\"Best RMSLE: \", -search.best_score_)","metadata":{"scrolled":true},"outputs":[],"execution_count":null},{"id":"edda7d00-f077-466d-8b3a-2527b88af012","cell_type":"code","source":"y_pred = search.predict(X_pred_scaled) \nfinal=pd.DataFrame({'id': data_test['id'], 'Premium Amount':np.exp(y_pred)})\nprint(final.head())\nfinal.to_csv('CB_submisson.csv', index=False)","metadata":{},"outputs":[],"execution_count":null},{"id":"7901b95f-84e1-462a-887a-28b93690935b","cell_type":"markdown","source":"# Ridge regression","metadata":{}},{"id":"5ebb7cbe-f325-4e42-9c42-607fa6081235","cell_type":"code","source":"alphas= np.array([5.00000000e+09, 3.78231664e+09, 2.86118383e+09, 2.16438064e+09,\n       1.63727458e+09, 1.23853818e+09, 9.36908711e+08, 7.08737081e+08,\n       5.36133611e+08, 4.05565415e+08, 3.06795364e+08, 2.32079442e+08,\n       1.75559587e+08, 1.32804389e+08, 1.00461650e+08, 7.59955541e+07,\n       5.74878498e+07, 4.34874501e+07, 3.28966612e+07, 2.48851178e+07,\n       1.88246790e+07, 1.42401793e+07, 1.07721735e+07, 8.14875417e+06,\n       6.16423370e+06, 4.66301673e+06, 3.52740116e+06, 2.66834962e+06,\n       2.01850863e+06, 1.52692775e+06, 1.15506485e+06, 8.73764200e+05,\n       6.60970574e+05, 5.00000000e+05, 3.78231664e+05, 2.86118383e+05,\n       2.16438064e+05, 1.63727458e+05, 1.23853818e+05, 9.36908711e+04,\n       7.08737081e+04, 5.36133611e+04, 4.05565415e+04, 3.06795364e+04,\n       2.32079442e+04, 1.75559587e+04, 1.32804389e+04, 1.00461650e+04,\n       7.59955541e+03, 5.74878498e+03, 4.34874501e+03, 3.28966612e+03,\n       2.48851178e+03, 1.88246790e+03, 1.42401793e+03, 1.07721735e+03,\n       8.14875417e+02, 6.16423370e+02, 4.66301673e+02, 3.52740116e+02,\n       2.66834962e+02, 2.01850863e+02, 1.52692775e+02, 1.15506485e+02,\n       8.73764200e+01, 6.60970574e+01, 5.00000000e+01, 3.78231664e+01,\n       2.86118383e+01, 2.16438064e+01, 1.63727458e+01, 1.23853818e+01,\n       9.36908711e+00, 7.08737081e+00, 5.36133611e+00, 4.05565415e+00,\n       3.06795364e+00, 2.32079442e+00, 1.75559587e+00, 1.32804389e+00,\n       1.00461650e+00, 7.59955541e-01, 5.74878498e-01, 4.34874501e-01,\n       3.28966612e-01, 2.48851178e-01, 1.88246790e-01, 1.42401793e-01,\n       1.07721735e-01, 8.14875417e-02, 6.16423370e-02, 4.66301673e-02,\n       3.52740116e-02, 2.66834962e-02, 2.01850863e-02, 1.52692775e-02,\n       1.15506485e-02, 8.73764200e-03, 6.60970574e-03, 5.00000000e-03])\n\nridgecv = RidgeCV(alphas = alphas, scoring = 'neg_root_mean_squared_error', cv=KFold(10))\nridgecv.fit(X_scaled, y)","metadata":{},"outputs":[],"execution_count":null},{"id":"e0ca8cee-1275-491a-9908-d2213a224c03","cell_type":"code","source":"y_pred = ridgecv.predict(X_pred_scaled)\nfinal=pd.DataFrame({'id': data_test['id'], 'Premium Amount':np.exp(y_pred)})\nprint(final.head())\nfinal.to_csv('CB_submisson.csv', index=False)","metadata":{},"outputs":[],"execution_count":null},{"id":"1f275e55-f9a1-48b7-9362-5910f26105d6","cell_type":"markdown","source":"# Lasso Regression","metadata":{}},{"id":"3e56efa2-5976-4f5d-a93d-c1a6ce84da15","cell_type":"code","source":"lassocv = LassoCV(cv=KFold(10), random_state=42)\nlassocv.fit(X_scaled, y)","metadata":{},"outputs":[],"execution_count":null},{"id":"9a4d8b81-528b-480e-b486-ec1461fc6a4e","cell_type":"code","source":"y_pred = lassocv.predict(X_pred_scaled)\nfinal=pd.DataFrame({'id': data_test['id'], 'Premium Amount':np.exp(y_pred)})\nprint(final.head())\nfinal.to_csv('CB_submisson.csv', index=False)","metadata":{},"outputs":[],"execution_count":null},{"id":"823a23a8-e970-439e-957e-e8aaec0b84cd","cell_type":"markdown","source":"# Decision Tree Regressor","metadata":{}},{"id":"682530a6-6728-461d-835a-9f1c21c02ef4","cell_type":"code","source":"dt= DecisionTreeRegressor( max_depth=2, random_state=12, criterion ='squared_error')\ndt.fit(X_scaled, y)","metadata":{},"outputs":[],"execution_count":null},{"id":"d6c32790-31af-4bdd-b447-a321506b37ed","cell_type":"code","source":"y_pred = dt.predict(X_pred_scaled)\nfinal=pd.DataFrame({'id': data_test['id'], 'Premium Amount':np.exp(y_pred)})\nprint(final.head())\nfinal.to_csv('CB_submisson.csv', index=False)","metadata":{},"outputs":[],"execution_count":null},{"id":"a1ad7cc3-d307-48e3-9a9b-e51ef0ac6c71","cell_type":"code","source":"model = DecisionTreeRegressor(random_state=12,criterion ='squared_error')\n\n# Hyperparameters for BayesSearchCV tuning\nsearch_spaces = {    \n    'max_depth': (2,4,6,8,10,12,14),\n    # Minimum number of samples required to split a node\n    'min_samples_split' : (10, 20, 50, 100),\n    # Minimum number of samples required at each leaf node\n    'min_samples_leaf' : (2, 5, 10, 20)\n}\n\n# Set up BayesSearchCV for hyperparameter tuning\nsearch = BayesSearchCV(\n    estimator=model,\n    search_spaces=search_spaces,\n    n_iter=50,          # Number of iterations for optimization\n    cv=5,               # k-fold cross-validation\n    verbose=1,          # Display detailed logs\n    scoring=\"neg_root_mean_squared_error\",\n    random_state=42     \n)\n\n# Perform the Bayesian optimization with cross-validation\nsearch.fit(X_scaled, y)\n\nprint(\"Best params: \", search.best_params_)\nprint(\"Best RMSLE: \", -search.best_score_)","metadata":{"scrolled":true},"outputs":[],"execution_count":null},{"id":"aef6430f-6cf2-4d03-9f95-5456e5890fea","cell_type":"code","source":"y_pred = search.predict(X_pred_scaled)\nfinal=pd.DataFrame({'id': data_test['id'], 'Premium Amount':np.exp(y_pred)})\nprint(final.head())\nfinal.to_csv('CB_submisson.csv', index=False)","metadata":{},"outputs":[],"execution_count":null}]}