{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"****\n# Introduction #\n****","metadata":{}},{"cell_type":"markdown","source":"**Private Score: 1.06689**","metadata":{}},{"cell_type":"markdown","source":"**This is a competition notebook, this notebook demonstrates my way of solving this problem**\n\n**The goal of this notebook is to correctly predict the Premium Amount**","metadata":{}},{"cell_type":"markdown","source":"****\n# Dataset Description #\n****\n**The dataset for this competition (both train and test) was generated from a deep learning model trained on the Insurance Premium Prediction dataset. Feature distributions are close to, but not exactly the same, as the original. Feel free to use the original dataset as part of this competition, both to explore differences as well as to see whether incorporating the original in training improves model performance.**","metadata":{}},{"cell_type":"markdown","source":"****\n# Reading and Displaying the Training dataset #\n****","metadata":{}},{"cell_type":"code","source":"# importing\nimport pandas as pd\nimport numpy as np\nimport warnings","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:12:47.272282Z","iopub.execute_input":"2024-12-28T02:12:47.272616Z","iopub.status.idle":"2024-12-28T02:12:47.656448Z","shell.execute_reply.started":"2024-12-28T02:12:47.27259Z","shell.execute_reply":"2024-12-28T02:12:47.655289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reading and displaying the dataset\npd.set_option('display.max_columns', None)\nwarnings.simplefilter(action='ignore', category=FutureWarning)\ndf=pd.read_csv(r'/kaggle/input/playground-series-s4e12/train.csv')\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:12:47.657786Z","iopub.execute_input":"2024-12-28T02:12:47.658288Z","iopub.status.idle":"2024-12-28T02:12:54.673927Z","shell.execute_reply.started":"2024-12-28T02:12:47.65825Z","shell.execute_reply":"2024-12-28T02:12:54.6725Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"****\n# Training Data Distribution #\n****","metadata":{}},{"cell_type":"code","source":"# importing\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:12:54.67605Z","iopub.execute_input":"2024-12-28T02:12:54.676428Z","iopub.status.idle":"2024-12-28T02:12:55.565417Z","shell.execute_reply.started":"2024-12-28T02:12:54.676397Z","shell.execute_reply":"2024-12-28T02:12:55.564155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:12:55.566902Z","iopub.execute_input":"2024-12-28T02:12:55.567394Z","iopub.status.idle":"2024-12-28T02:12:56.21475Z","shell.execute_reply.started":"2024-12-28T02:12:55.567367Z","shell.execute_reply":"2024-12-28T02:12:56.21337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Let us look at how our continuous features and target variable are distributed\n\ncols=['Age','Annual Income','Health Score','Credit Score','Premium Amount']\nfig,ax=plt.subplots(3,2,figsize=(20,10))\nax=ax.flatten()\ni=0\nfor col in cols:\n    sns.histplot(data=df,x=col,ax=ax[i],kde=True)\n    i+=1\n\nax[5].axis('off')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:12:56.216175Z","iopub.execute_input":"2024-12-28T02:12:56.216568Z","iopub.status.idle":"2024-12-28T02:13:22.346838Z","shell.execute_reply.started":"2024-12-28T02:12:56.216531Z","shell.execute_reply":"2024-12-28T02:13:22.345714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:13:22.348136Z","iopub.execute_input":"2024-12-28T02:13:22.348418Z","iopub.status.idle":"2024-12-28T02:13:22.355827Z","shell.execute_reply.started":"2024-12-28T02:13:22.348395Z","shell.execute_reply":"2024-12-28T02:13:22.354428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Now let us take a look at some of our categorical features distribution\n\ncols1=['Gender','Marital Status',\n       'Number of Dependents', 'Education Level', 'Occupation','Location', 'Policy Type', 'Previous Claims',\n      'Customer Feedback', 'Smoking Status', 'Exercise Frequency',\n       'Property Type','Insurance Duration','Vehicle Age']\nfig, ax = plt.subplots(4, 6, figsize=(40, 35))\nax = ax.flatten()\ni = 0\nfor col in cols1:\n    tdf = df[col].value_counts().reset_index()\n    tdf.columns = ['label', 'count']\n    if i < len(ax):\n        ax[i].pie(x=tdf['count'], labels=tdf['label'], autopct='%.2f%%')\n        i += 1\n        if i < len(ax):\n            sns.countplot(data=df, x=col, ax=ax[i])\n            i += 1\nax[14].axis('off')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:13:22.357047Z","iopub.execute_input":"2024-12-28T02:13:22.357327Z","iopub.status.idle":"2024-12-28T02:13:33.119935Z","shell.execute_reply.started":"2024-12-28T02:13:22.357305Z","shell.execute_reply":"2024-12-28T02:13:33.118516Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"****\n# Training Data Preprocessing #\n****","metadata":{}},{"cell_type":"code","source":"# Checking for any null values\nplt.figure(figsize=(20,5))\nsns.heatmap(df.isnull())\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:13:33.123292Z","iopub.execute_input":"2024-12-28T02:13:33.123604Z","iopub.status.idle":"2024-12-28T02:13:57.49045Z","shell.execute_reply.started":"2024-12-28T02:13:33.12358Z","shell.execute_reply":"2024-12-28T02:13:57.489243Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Finding the percentage of null values\ntdf = pd.DataFrame({\n    'Cols': df.columns,  \n    'percentage': ((df.isnull().sum()) / df.shape[0]*100)\n})\ntdf.sort_values(by='percentage',ascending=False,inplace=True)\ntdf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:13:57.492433Z","iopub.execute_input":"2024-12-28T02:13:57.492755Z","iopub.status.idle":"2024-12-28T02:13:58.11986Z","shell.execute_reply.started":"2024-12-28T02:13:57.492727Z","shell.execute_reply":"2024-12-28T02:13:58.118422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Taking care of the null values by fillind it with median and mode\ndf['Previous Claims']=df['Previous Claims'].fillna(df['Previous Claims'].mode()[0])\ndf['Occupation']=df['Occupation'].fillna(df['Occupation'].mode()[0])\ndf['Credit Score']=df['Credit Score'].fillna(df['Credit Score'].median())\ndf['Customer Feedback']=df['Customer Feedback'].fillna(df['Customer Feedback'].mode()[0])\ndf['Number of Dependents']=df['Number of Dependents'].fillna(df['Number of Dependents'].mode()[0])\ndf['Health Score']=df['Health Score'].fillna(df['Health Score'].median())\ndf['Annual Income']=df['Annual Income'].fillna(df['Annual Income'].median())\ndf['Age']=df['Age'].fillna(df['Age'].mode()[0])\ndf['Marital Status']=df['Marital Status'].fillna(df['Marital Status'].mode()[0])\ndf['Vehicle Age']=df['Vehicle Age'].fillna(df['Vehicle Age'].median())\ndf['Insurance Duration']=df['Insurance Duration'].fillna(df['Insurance Duration'].mode()[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:13:58.121405Z","iopub.execute_input":"2024-12-28T02:13:58.12179Z","iopub.status.idle":"2024-12-28T02:13:58.869922Z","shell.execute_reply.started":"2024-12-28T02:13:58.121746Z","shell.execute_reply":"2024-12-28T02:13:58.868942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tdf = pd.DataFrame({\n    'Cols': df.columns,  \n    'percentage': ((df.isnull().sum()) / df.shape[0]) *100 \n})\ntdf.sort_values(by='percentage',ascending=False,inplace=True)\ntdf['percentage']=tdf['percentage'].astype(int)\ntdf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:13:58.871133Z","iopub.execute_input":"2024-12-28T02:13:58.871496Z","iopub.status.idle":"2024-12-28T02:13:59.543966Z","shell.execute_reply.started":"2024-12-28T02:13:58.871469Z","shell.execute_reply":"2024-12-28T02:13:59.542998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Splitting our Policy Start Date feature into Year, Month and Date\ndf[['Year', 'Month', 'Date']]=df['Policy Start Date'].str.split('-',expand=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:13:59.545092Z","iopub.execute_input":"2024-12-28T02:13:59.54545Z","iopub.status.idle":"2024-12-28T02:14:02.980811Z","shell.execute_reply.started":"2024-12-28T02:13:59.545422Z","shell.execute_reply":"2024-12-28T02:14:02.979318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:02.982221Z","iopub.execute_input":"2024-12-28T02:14:02.982738Z","iopub.status.idle":"2024-12-28T02:14:03.014938Z","shell.execute_reply.started":"2024-12-28T02:14:02.982703Z","shell.execute_reply":"2024-12-28T02:14:03.013902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Splitting our Date feature into date and time\ndf.drop(columns='Policy Start Date',inplace=True)\ndf[['date','time']]=df['Date'].str.split(' ',expand=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:03.015975Z","iopub.execute_input":"2024-12-28T02:14:03.016258Z","iopub.status.idle":"2024-12-28T02:14:06.627313Z","shell.execute_reply.started":"2024-12-28T02:14:03.016235Z","shell.execute_reply":"2024-12-28T02:14:06.626167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Splitting the time feature into Hour, Min and Seconds\ndf.drop(columns='Date',inplace=True)\ndf[['Hour','Min','Seconds']]=df['time'].str.split(':',expand=True)\ndf.drop(columns='time',inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:06.628085Z","iopub.execute_input":"2024-12-28T02:14:06.628342Z","iopub.status.idle":"2024-12-28T02:14:11.503433Z","shell.execute_reply.started":"2024-12-28T02:14:06.62832Z","shell.execute_reply":"2024-12-28T02:14:11.502406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:11.504328Z","iopub.execute_input":"2024-12-28T02:14:11.504598Z","iopub.status.idle":"2024-12-28T02:14:12.589854Z","shell.execute_reply.started":"2024-12-28T02:14:11.504577Z","shell.execute_reply":"2024-12-28T02:14:12.588735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Converting to integer datatype\ndf[['Year','Month','date','Hour','Min']]=df[['Year','Month','date','Hour','Min']].astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:12.590799Z","iopub.execute_input":"2024-12-28T02:14:12.591138Z","iopub.status.idle":"2024-12-28T02:14:13.680423Z","shell.execute_reply.started":"2024-12-28T02:14:12.591111Z","shell.execute_reply":"2024-12-28T02:14:13.679294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Seconds']=df['Seconds'].astype(float)\ndf.drop(columns=['Hour','Min'],inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:13.681935Z","iopub.execute_input":"2024-12-28T02:14:13.682942Z","iopub.status.idle":"2024-12-28T02:14:14.06414Z","shell.execute_reply.started":"2024-12-28T02:14:13.682894Z","shell.execute_reply":"2024-12-28T02:14:14.063008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:14.065241Z","iopub.execute_input":"2024-12-28T02:14:14.065503Z","iopub.status.idle":"2024-12-28T02:14:14.654032Z","shell.execute_reply.started":"2024-12-28T02:14:14.065482Z","shell.execute_reply":"2024-12-28T02:14:14.652976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cols = df.select_dtypes(include=['object']).columns\nfor col in cols:\n    print(df[col].value_counts(),'\\n')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:14.655179Z","iopub.execute_input":"2024-12-28T02:14:14.655476Z","iopub.status.idle":"2024-12-28T02:14:15.876724Z","shell.execute_reply.started":"2024-12-28T02:14:14.655438Z","shell.execute_reply":"2024-12-28T02:14:15.875577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Encoding all the categorical features\n# We can also use One Hot Encoding to convert the categorical columns into Binary, but the complexity increases\nfrom sklearn.preprocessing import LabelEncoder\nle=LabelEncoder()\nfor col in cols:\n    df[col]=le.fit_transform(df[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:15.877753Z","iopub.execute_input":"2024-12-28T02:14:15.878122Z","iopub.status.idle":"2024-12-28T02:14:18.196145Z","shell.execute_reply.started":"2024-12-28T02:14:15.878089Z","shell.execute_reply":"2024-12-28T02:14:18.194786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:18.197307Z","iopub.execute_input":"2024-12-28T02:14:18.197817Z","iopub.status.idle":"2024-12-28T02:14:18.248714Z","shell.execute_reply.started":"2024-12-28T02:14:18.197776Z","shell.execute_reply":"2024-12-28T02:14:18.24751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().any()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:18.255151Z","iopub.execute_input":"2024-12-28T02:14:18.255584Z","iopub.status.idle":"2024-12-28T02:14:18.286053Z","shell.execute_reply.started":"2024-12-28T02:14:18.255549Z","shell.execute_reply":"2024-12-28T02:14:18.284837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:18.288316Z","iopub.execute_input":"2024-12-28T02:14:18.288737Z","iopub.status.idle":"2024-12-28T02:14:18.32769Z","shell.execute_reply.started":"2024-12-28T02:14:18.288702Z","shell.execute_reply":"2024-12-28T02:14:18.326504Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"****\n# Reading and Displaying the Testing Dataset #\n****","metadata":{}},{"cell_type":"code","source":"# Reading and displaying Testing data\nte=pd.read_csv(r\"/kaggle/input/playground-series-s4e12/test.csv\")\nte","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:18.328778Z","iopub.execute_input":"2024-12-28T02:14:18.329209Z","iopub.status.idle":"2024-12-28T02:14:22.551705Z","shell.execute_reply.started":"2024-12-28T02:14:18.329172Z","shell.execute_reply":"2024-12-28T02:14:22.550789Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"****\n# Testing Data Distribution #\n****","metadata":{}},{"cell_type":"code","source":"# Let us look at how our continuous features and target variable are distributed\n\ncols=['Age','Annual Income','Health Score','Credit Score']\nfig,ax=plt.subplots(2,2,figsize=(20,10))\nax=ax.flatten()\ni=0\nfor col in cols:\n    sns.histplot(data=te,x=col,ax=ax[i],kde=True)\n    i+=1\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:22.552637Z","iopub.execute_input":"2024-12-28T02:14:22.553065Z","iopub.status.idle":"2024-12-28T02:14:36.298178Z","shell.execute_reply.started":"2024-12-28T02:14:22.55302Z","shell.execute_reply":"2024-12-28T02:14:36.296938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Now let us take a look at some of our categorical features distribution\n\ncols1=['Gender','Marital Status',\n       'Number of Dependents', 'Education Level', 'Occupation','Location', 'Policy Type', 'Previous Claims',\n      'Customer Feedback', 'Smoking Status', 'Exercise Frequency',\n       'Property Type','Insurance Duration','Vehicle Age']\nfig, ax = plt.subplots(4, 6, figsize=(40, 35))\nax = ax.flatten()\ni = 0\nfor col in cols1:\n    tdf = te[col].value_counts().reset_index()\n    tdf.columns = ['label', 'count']\n    if i < len(ax):\n        ax[i].pie(x=tdf['count'], labels=tdf['label'], autopct='%.2f%%')\n        i += 1\n        if i < len(ax):\n            sns.countplot(data=df, x=col, ax=ax[i])\n            i += 1\nax[14].axis('off')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:36.29944Z","iopub.execute_input":"2024-12-28T02:14:36.299851Z","iopub.status.idle":"2024-12-28T02:14:41.148385Z","shell.execute_reply.started":"2024-12-28T02:14:36.299813Z","shell.execute_reply":"2024-12-28T02:14:41.147095Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"****\n# Testing Data Preprocessing #\n****","metadata":{}},{"cell_type":"code","source":"te.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:41.14948Z","iopub.execute_input":"2024-12-28T02:14:41.14994Z","iopub.status.idle":"2024-12-28T02:14:41.577648Z","shell.execute_reply.started":"2024-12-28T02:14:41.149905Z","shell.execute_reply":"2024-12-28T02:14:41.576545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"te['Previous Claims']=te['Previous Claims'].fillna(te['Previous Claims'].mode()[0])\nte['Occupation']=te['Occupation'].fillna(te['Occupation'].mode()[0])\nte['Credit Score']=te['Credit Score'].fillna(te['Credit Score'].median())\nte['Customer Feedback']=te['Customer Feedback'].fillna(te['Customer Feedback'].mode()[0])\nte['Number of Dependents']=te['Number of Dependents'].fillna(te['Number of Dependents'].mode()[0])\nte['Health Score']=te['Health Score'].fillna(te['Health Score'].median())\nte['Annual Income']=te['Annual Income'].fillna(te['Annual Income'].median())\nte['Age']=te['Age'].fillna(te['Age'].mode()[0])\nte['Marital Status']=te['Marital Status'].fillna(te['Marital Status'].mode()[0])\nte['Vehicle Age']=te['Vehicle Age'].fillna(te['Vehicle Age'].median())\nte['Insurance Duration']=te['Insurance Duration'].fillna(te['Insurance Duration'].mode()[0])\nte[['Year', 'Month', 'Date']]=te['Policy Start Date'].str.split('-',expand=True)\nte.drop(columns='Policy Start Date',inplace=True)\nte[['date','time']]=te['Date'].str.split(' ',expand=True)\nte.drop(columns='Date',inplace=True)\nte[['Hour','Min','Seconds']]=te['time'].str.split(':',expand=True)\nte.drop(columns='time',inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:41.578811Z","iopub.execute_input":"2024-12-28T02:14:41.579144Z","iopub.status.idle":"2024-12-28T02:14:50.418508Z","shell.execute_reply.started":"2024-12-28T02:14:41.579117Z","shell.execute_reply":"2024-12-28T02:14:50.417488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"te[['Year','Month','date','Hour','Min']]=te[['Year','Month','date','Hour','Min']].astype(int)\nte['Seconds']=df['Seconds'].astype(float)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:50.4195Z","iopub.execute_input":"2024-12-28T02:14:50.419815Z","iopub.status.idle":"2024-12-28T02:14:51.216112Z","shell.execute_reply.started":"2024-12-28T02:14:50.419787Z","shell.execute_reply":"2024-12-28T02:14:51.215056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cols = te.select_dtypes(include=['object']).columns\ncols\nle=LabelEncoder()\nfor col in cols:\n    te[col]=le.fit_transform(te[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:51.217223Z","iopub.execute_input":"2024-12-28T02:14:51.217535Z","iopub.status.idle":"2024-12-28T02:14:52.969014Z","shell.execute_reply.started":"2024-12-28T02:14:51.217497Z","shell.execute_reply":"2024-12-28T02:14:52.967598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"te.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:52.970069Z","iopub.execute_input":"2024-12-28T02:14:52.970375Z","iopub.status.idle":"2024-12-28T02:14:53.009844Z","shell.execute_reply.started":"2024-12-28T02:14:52.970348Z","shell.execute_reply":"2024-12-28T02:14:53.008757Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"****\n# Feature Engineering #\n****","metadata":{}},{"cell_type":"code","source":"# importing\nfrom sklearn.feature_selection import mutual_info_regression","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:14:53.010772Z","iopub.execute_input":"2024-12-28T02:14:53.011102Z","iopub.status.idle":"2024-12-28T02:14:53.398961Z","shell.execute_reply.started":"2024-12-28T02:14:53.011076Z","shell.execute_reply":"2024-12-28T02:14:53.397647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Now Let us calculate Mutual information and then proceed to visualize correlation along with mutual information\n# Calculating Mutual Information\ntdf=df.copy()  \nx=tdf.drop(columns='Premium Amount')\ny=tdf['Premium Amount']\nmi=mutual_info_regression(x,y)\nmi_df=pd.DataFrame({'Feature':x.columns,'Mutual Information':mi})\nmi_df=mi_df.sort_values(by='Mutual Information', ascending=False).reset_index(drop=True)\n\n# Visualizing Mutual Information and correlation \nfig,ax=plt.subplots(2,1,figsize=(15,30))\nsns.heatmap(df.corr(),annot=True,cmap='magma',ax=ax[0])\nax[0].set_title('Correlation')\nsns.barplot(x='Mutual Information',y='Feature', data=mi_df,ax=ax[1])\nax[1].set_title('Mutual Information')\nplt.tight_layout()\nplt.suptitle('Before Feature Engineering')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-28T02:14:53.400166Z","iopub.execute_input":"2024-12-28T02:14:53.400478Z","iopub.status.idle":"2024-12-28T02:24:27.030528Z","shell.execute_reply.started":"2024-12-28T02:14:53.40045Z","shell.execute_reply":"2024-12-28T02:24:27.029285Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature Engineering\ndf['Avg_Claims'] = df.groupby('id')['Previous Claims'].transform('mean')\ndf['Claims per Dependent'] = np.where(\n    df['Number of Dependents'] == 0, \n    df['Previous Claims'],  \n    df['Previous Claims'] / df['Number of Dependents']\n)\ndf['Dependents']=df['Number of Dependents']*df['Number of Dependents']\ndf['Claims_Square']=df['Previous Claims']*df['Previous Claims']\ndf['HC_For_Age']=df.groupby('Age')['Health Score'].transform('mean')\ndf['Claims_Health_Ratio']=df['Previous Claims']/df['Health Score']\ndf['Income_to_Dependents'] = np.where(\n    df['Number of Dependents'] == 0,\n    df['Annual Income'], \n    df['Annual Income'] / df['Number of Dependents']\n)\ndf['Claim_Norm']=df['Previous Claims']/(df['Previous Claims'].max())\ndf['Residual_Income']=df['Annual Income']-df['Previous Claims']\ndf['Log_Income']=np.log(df['Annual Income'])\ndf['Life']=2024-df['Year']\ndf[['s_Year','s_Month','s_date']]=df[['Year','Month','date']].apply(np.sin)\ndf[['c_Year','c_Month','c_date']]=df[['Year','Month','date']].apply(np.cos)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:24:27.031567Z","iopub.execute_input":"2024-12-28T02:24:27.031892Z","iopub.status.idle":"2024-12-28T02:24:27.487334Z","shell.execute_reply.started":"2024-12-28T02:24:27.031842Z","shell.execute_reply":"2024-12-28T02:24:27.486302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature Engineering for Testing data\nte['Avg_Claims'] = te.groupby('id')['Previous Claims'].transform('mean')\nte['Claims per Dependent'] = np.where(\n    te['Number of Dependents'] == 0, \n    te['Previous Claims'],  \n    te['Previous Claims'] / te['Number of Dependents']\n)\nte['Dependents'] = te['Number of Dependents'] * te['Number of Dependents']\nte['Claims_Square'] = te['Previous Claims'] * te['Previous Claims']\nte['HC_For_Age'] = te.groupby('Age')['Health Score'].transform('mean')\nte['Claims_Health_Ratio'] = te['Previous Claims'] / te['Health Score']\nte['Income_to_Dependents'] = np.where(\n    te['Number of Dependents'] == 0,\n    te['Annual Income'], \n    te['Annual Income'] / te['Number of Dependents']\n)\nte['Claim_Norm'] = te['Previous Claims'] / (te['Previous Claims'].max())\nte['Residual_Income'] = te['Annual Income'] - te['Previous Claims']\nte['Log_Income'] = np.log(te['Annual Income'])\nte.drop(columns=['Hour','Min'],inplace=True)\nte['Life']=2024-te['Year']\nte[['s_Year','s_Month','s_date']]=te[['Year','Month','date']].apply(np.sin)\nte[['c_Year','c_Month','c_date']]=te[['Year','Month','date']].apply(np.cos)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:24:27.488428Z","iopub.execute_input":"2024-12-28T02:24:27.488846Z","iopub.status.idle":"2024-12-28T02:24:27.980751Z","shell.execute_reply.started":"2024-12-28T02:24:27.488761Z","shell.execute_reply":"2024-12-28T02:24:27.979771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Mutual Information and Correlation after adding some features\ntdf=df.copy()  \nx=tdf.drop(columns='Premium Amount')\ny=tdf['Premium Amount']\nmi=mutual_info_regression(x,y)\nmi_df=pd.DataFrame({'Feature':x.columns,'Mutual Information':mi})\nmi_df=mi_df.sort_values(by='Mutual Information', ascending=False).reset_index(drop=True)\n\n# Visualizing Mutual Information and correlation \nfig,ax=plt.subplots(2,1,figsize=(15,30))\nsns.heatmap(df.corr(),annot=True,cmap='magma',ax=ax[0])\nax[0].set_title('Correlation')\nsns.barplot(x='Mutual Information',y='Feature', data=mi_df,ax=ax[1])\nax[1].set_title('Mutual Information')\nplt.tight_layout()\nplt.suptitle('After Feature Engineering')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-28T02:24:27.981673Z","iopub.execute_input":"2024-12-28T02:24:27.981948Z","iopub.status.idle":"2024-12-28T02:40:37.941324Z","shell.execute_reply.started":"2024-12-28T02:24:27.981926Z","shell.execute_reply":"2024-12-28T02:40:37.939848Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"****\n# Data Preparation #\n****","metadata":{}},{"cell_type":"code","source":"# importing \nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:40:37.942848Z","iopub.execute_input":"2024-12-28T02:40:37.943388Z","iopub.status.idle":"2024-12-28T02:40:37.94891Z","shell.execute_reply.started":"2024-12-28T02:40:37.943347Z","shell.execute_reply":"2024-12-28T02:40:37.947319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Splitting into features and target variable\nx_t=df.drop(columns=['Premium Amount','id'])\ny_t=df['Premium Amount']\n\ny_t=np.log1p(df['Premium Amount'])\nX_test=te.drop('id', axis=1)\n\n# Scaling the Features\nss=StandardScaler()\nx_t=ss.fit_transform(x_t)\nX_test= ss.transform(X_test)\n\n# Splitting the Training set into Training and Testing set\nx_tr,x_te, y_tr,y_te=train_test_split(x_t, y_t, test_size=0.15, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:40:37.950044Z","iopub.execute_input":"2024-12-28T02:40:37.950342Z","iopub.status.idle":"2024-12-28T02:40:40.161793Z","shell.execute_reply.started":"2024-12-28T02:40:37.950315Z","shell.execute_reply":"2024-12-28T02:40:40.160964Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"****\n# Model - LightBGM #\n****","metadata":{}},{"cell_type":"code","source":"# importing\nfrom lightgbm import LGBMRegressor, early_stopping\nfrom sklearn.model_selection import RandomizedSearchCV","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:40:40.162545Z","iopub.execute_input":"2024-12-28T02:40:40.1629Z","iopub.status.idle":"2024-12-28T02:40:42.399659Z","shell.execute_reply.started":"2024-12-28T02:40:40.162855Z","shell.execute_reply":"2024-12-28T02:40:42.39848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = LGBMRegressor(n_estimators=1000, learning_rate=0.01, max_depth=7, \n                      random_state=42,num_leaves=225,verbose=-1)\nmodel.fit(x_tr,y_tr, eval_set=[(x_te,y_te)], callbacks=[early_stopping(stopping_rounds=500)])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:40:42.400959Z","iopub.execute_input":"2024-12-28T02:40:42.401795Z","iopub.status.idle":"2024-12-28T02:42:30.829206Z","shell.execute_reply.started":"2024-12-28T02:40:42.401752Z","shell.execute_reply":"2024-12-28T02:42:30.828207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predicting\n# Predict on Test Data\npred=model.predict(X_test)\npred= np.expm1(pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:42:30.830156Z","iopub.execute_input":"2024-12-28T02:42:30.830441Z","iopub.status.idle":"2024-12-28T02:43:15.102025Z","shell.execute_reply.started":"2024-12-28T02:42:30.830418Z","shell.execute_reply":"2024-12-28T02:43:15.100892Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"****\n# Submission #\n****","metadata":{}},{"cell_type":"code","source":"# Submission\nsubmission = pd.DataFrame({'id': te['id'], 'Premium Amount':pred})\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T02:43:15.102986Z","iopub.execute_input":"2024-12-28T02:43:15.103354Z","iopub.status.idle":"2024-12-28T02:43:16.682454Z","shell.execute_reply.started":"2024-12-28T02:43:15.103317Z","shell.execute_reply":"2024-12-28T02:43:16.68117Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"****\n# Feedback and Suggestions #\n****","metadata":{}},{"cell_type":"markdown","source":"**If you liked this notebook or if you found it Helpful Kindly Upvote :)**\n\n**Please provide feedback and Suggestions to improve this notebook**","metadata":{}}]}