{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:31:55.629565Z","iopub.execute_input":"2024-12-11T05:31:55.629948Z","iopub.status.idle":"2024-12-11T05:31:56.098261Z","shell.execute_reply.started":"2024-12-11T05:31:55.629912Z","shell.execute_reply":"2024-12-11T05:31:56.097182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:32:35.769851Z","iopub.execute_input":"2024-12-11T05:32:35.770231Z","iopub.status.idle":"2024-12-11T05:32:36.610228Z","shell.execute_reply.started":"2024-12-11T05:32:35.770197Z","shell.execute_reply":"2024-12-11T05:32:36.609299Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Import Datasets","metadata":{}},{"cell_type":"code","source":"test_path = '/kaggle/input/playground-series-s4e12/test.csv'\ntrain_path = '/kaggle/input/playground-series-s4e12/train.csv'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:35:31.719681Z","iopub.execute_input":"2024-12-11T05:35:31.720303Z","iopub.status.idle":"2024-12-11T05:35:31.725505Z","shell.execute_reply.started":"2024-12-11T05:35:31.720265Z","shell.execute_reply":"2024-12-11T05:35:31.724266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_set = pd.read_csv(test_path)\ntrain_set = pd.read_csv(train_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:36:12.77011Z","iopub.execute_input":"2024-12-11T05:36:12.770504Z","iopub.status.idle":"2024-12-11T05:36:22.817761Z","shell.execute_reply.started":"2024-12-11T05:36:12.770472Z","shell.execute_reply":"2024-12-11T05:36:22.816783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = pd.DataFrame(test_set)\ntrain_df = pd.DataFrame(train_set)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:36:36.929776Z","iopub.execute_input":"2024-12-11T05:36:36.930198Z","iopub.status.idle":"2024-12-11T05:36:36.935461Z","shell.execute_reply.started":"2024-12-11T05:36:36.930164Z","shell.execute_reply":"2024-12-11T05:36:36.934118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:36:51.069271Z","iopub.execute_input":"2024-12-11T05:36:51.06971Z","iopub.status.idle":"2024-12-11T05:36:51.108427Z","shell.execute_reply.started":"2024-12-11T05:36:51.069675Z","shell.execute_reply":"2024-12-11T05:36:51.107455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Shape of train dataset -->',train_df.shape)\nprint('Shape of test dataset -->',test_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:37:17.569876Z","iopub.execute_input":"2024-12-11T05:37:17.570286Z","iopub.status.idle":"2024-12-11T05:37:17.576322Z","shell.execute_reply.started":"2024-12-11T05:37:17.570253Z","shell.execute_reply":"2024-12-11T05:37:17.575195Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Checking Null Values and data types","metadata":{}},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:37:53.290309Z","iopub.execute_input":"2024-12-11T05:37:53.290783Z","iopub.status.idle":"2024-12-11T05:37:53.952206Z","shell.execute_reply.started":"2024-12-11T05:37:53.290748Z","shell.execute_reply":"2024-12-11T05:37:53.951133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:38:09.371395Z","iopub.execute_input":"2024-12-11T05:38:09.372201Z","iopub.status.idle":"2024-12-11T05:38:10.005694Z","shell.execute_reply.started":"2024-12-11T05:38:09.372162Z","shell.execute_reply":"2024-12-11T05:38:10.004658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:39:27.580388Z","iopub.execute_input":"2024-12-11T05:39:27.580869Z","iopub.status.idle":"2024-12-11T05:39:28.006759Z","shell.execute_reply.started":"2024-12-11T05:39:27.580833Z","shell.execute_reply":"2024-12-11T05:39:28.005718Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA","metadata":{}},{"cell_type":"code","source":"sns.set(style=\"whitegrid\")\nnumeric_columns = train_df.select_dtypes(include=['number']).columns\nnumeric_columns_without_id = [col for col in numeric_columns if col != 'id']\n\nfor col in numeric_columns_without_id:\n    plt.figure(figsize=(12, 4))\n    plt.subplot(1, 2, 1)\n    sns.histplot(train_df[col], kde=True)\n    plt.subplot(1, 2, 2)\n    sns.boxplot(x=train_df[col])\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:40:19.850302Z","iopub.execute_input":"2024-12-11T05:40:19.850867Z","iopub.status.idle":"2024-12-11T05:41:11.979628Z","shell.execute_reply.started":"2024-12-11T05:40:19.850824Z","shell.execute_reply":"2024-12-11T05:41:11.978549Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Distribution of Annual Income with Gender","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 8))\nsns.boxplot(x = 'Gender' , y='Annual Income',data = train_df, hue = 'Gender')\nplt.ylabel(\"Gender\")\nplt.title('Distribution of Annual Income with Gender')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:42:32.131253Z","iopub.execute_input":"2024-12-11T05:42:32.132135Z","iopub.status.idle":"2024-12-11T05:42:33.627838Z","shell.execute_reply.started":"2024-12-11T05:42:32.132091Z","shell.execute_reply":"2024-12-11T05:42:33.626777Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Distribution of Annual Income with Education Level","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 8))\nsns.boxplot(x = 'Education Level' ,y='Annual Income',data = train_df,hue ='Education Level')\nplt.title('Distribution of Annual Income with Education Level')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:43:12.770515Z","iopub.execute_input":"2024-12-11T05:43:12.771549Z","iopub.status.idle":"2024-12-11T05:43:14.754242Z","shell.execute_reply.started":"2024-12-11T05:43:12.771505Z","shell.execute_reply":"2024-12-11T05:43:14.753173Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Distribution of Annual Income with Location","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 8))\nsns.boxplot(x = 'Location' ,y='Annual Income',data = train_df,hue ='Location')\nplt.title('Distribution of Annual Income with Location')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:43:52.090114Z","iopub.execute_input":"2024-12-11T05:43:52.09109Z","iopub.status.idle":"2024-12-11T05:43:53.963838Z","shell.execute_reply.started":"2024-12-11T05:43:52.091042Z","shell.execute_reply":"2024-12-11T05:43:53.962805Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Distribution of Health Score with Gender","metadata":{"execution":{"iopub.status.busy":"2024-12-11T05:44:11.125187Z","iopub.execute_input":"2024-12-11T05:44:11.125638Z","iopub.status.idle":"2024-12-11T05:44:11.130309Z","shell.execute_reply.started":"2024-12-11T05:44:11.1256Z","shell.execute_reply":"2024-12-11T05:44:11.129204Z"}}},{"cell_type":"code","source":"plt.figure(figsize=(15, 8))\nsns.boxplot(x = 'Gender' ,y='Health Score',data = train_df,hue ='Gender')\nplt.title('Distribution of Health Score with Gender')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:44:28.335112Z","iopub.execute_input":"2024-12-11T05:44:28.3356Z","iopub.status.idle":"2024-12-11T05:44:29.704508Z","shell.execute_reply.started":"2024-12-11T05:44:28.335561Z","shell.execute_reply":"2024-12-11T05:44:29.703268Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Distribution of Health Score with Occupation","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 8))\nsns.boxplot(x = 'Occupation' ,y='Health Score',data = train_df,hue ='Occupation')\nplt.title('Distribution of Health Score with Occupation')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:45:09.102094Z","iopub.execute_input":"2024-12-11T05:45:09.102841Z","iopub.status.idle":"2024-12-11T05:45:10.562059Z","shell.execute_reply.started":"2024-12-11T05:45:09.10278Z","shell.execute_reply":"2024-12-11T05:45:10.560935Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Distribution of Health Score with Location","metadata":{"execution":{"iopub.status.busy":"2024-12-11T05:45:30.140054Z","iopub.execute_input":"2024-12-11T05:45:30.140535Z","iopub.status.idle":"2024-12-11T05:45:30.145676Z","shell.execute_reply.started":"2024-12-11T05:45:30.140499Z","shell.execute_reply":"2024-12-11T05:45:30.144501Z"}}},{"cell_type":"code","source":"plt.figure(figsize=(15, 8))\nsns.boxplot(x = 'Location' ,y='Health Score',data = train_df,hue ='Location')\nplt.title('Distribution of Health Score with Location')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:45:46.960044Z","iopub.execute_input":"2024-12-11T05:45:46.960476Z","iopub.status.idle":"2024-12-11T05:45:48.383363Z","shell.execute_reply.started":"2024-12-11T05:45:46.960442Z","shell.execute_reply":"2024-12-11T05:45:48.382291Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### EDA For categorical columns","metadata":{}},{"cell_type":"code","source":"categorical_columns = ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location',\n       'Policy Type', 'Customer Feedback',\n       'Smoking Status', 'Exercise Frequency', 'Property Type']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:46:39.770457Z","iopub.execute_input":"2024-12-11T05:46:39.770843Z","iopub.status.idle":"2024-12-11T05:46:39.776153Z","shell.execute_reply.started":"2024-12-11T05:46:39.770806Z","shell.execute_reply":"2024-12-11T05:46:39.774813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(18,40))\nfor i,column in enumerate(categorical_columns, 1):\n    plt.subplot(5,2,i)\n    sns.countplot(x=column, data=train_df, hue = column)\n    plt.title(f'Distribution of {column}')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:46:49.250197Z","iopub.execute_input":"2024-12-11T05:46:49.250596Z","iopub.status.idle":"2024-12-11T05:47:04.702774Z","shell.execute_reply.started":"2024-12-11T05:46:49.250559Z","shell.execute_reply":"2024-12-11T05:47:04.701576Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Fill null values of Numerical Columns","metadata":{"execution":{"iopub.status.busy":"2024-12-11T05:50:47.449992Z","iopub.execute_input":"2024-12-11T05:50:47.450445Z","iopub.status.idle":"2024-12-11T05:50:47.455263Z","shell.execute_reply.started":"2024-12-11T05:50:47.450394Z","shell.execute_reply":"2024-12-11T05:50:47.454257Z"}}},{"cell_type":"markdown","source":"### Fill Null values of Age column, and Number of Dependents.","metadata":{}},{"cell_type":"markdown","source":"#### According to Histograms and boxplots, Age , Number of Dependents columns has uniform distribution. Hance , I dicided to fill null values of those variables randomly which values in it's range.","metadata":{}},{"cell_type":"code","source":"def random_fill_null(df,var):\n    non_null = df[var].dropna()\n    df[var] = df[var].apply(lambda x: np.random.choice(non_null) if pd.isna(x) else x)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:53:11.655176Z","iopub.execute_input":"2024-12-11T05:53:11.656033Z","iopub.status.idle":"2024-12-11T05:53:11.661488Z","shell.execute_reply.started":"2024-12-11T05:53:11.655995Z","shell.execute_reply":"2024-12-11T05:53:11.660142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_fill_null(test_df,'Age')\nrandom_fill_null(train_df,'Age')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:53:22.059503Z","iopub.execute_input":"2024-12-11T05:53:22.059922Z","iopub.status.idle":"2024-12-11T05:53:24.310892Z","shell.execute_reply.started":"2024-12-11T05:53:22.05989Z","shell.execute_reply":"2024-12-11T05:53:24.310018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_fill_null(test_df,'Number of Dependents')\nrandom_fill_null(train_df,'Number of Dependents')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:53:31.455106Z","iopub.execute_input":"2024-12-11T05:53:31.455861Z","iopub.status.idle":"2024-12-11T05:53:37.4616Z","shell.execute_reply.started":"2024-12-11T05:53:31.455821Z","shell.execute_reply":"2024-12-11T05:53:37.460502Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Fill Null values of Annual Income, Previous Claims, Credit Score, and Insurance Duration column","metadata":{}},{"cell_type":"markdown","source":"#### According to Histograms and boxplots, mentioned columns has skewed distribution. Hence, null values were filled using median.","metadata":{}},{"cell_type":"code","source":"def median_fill_null(df,var):\n    median_value = df[var].median()\n    df[var] = df[var].fillna(median_value)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:55:37.69544Z","iopub.execute_input":"2024-12-11T05:55:37.69596Z","iopub.status.idle":"2024-12-11T05:55:37.701464Z","shell.execute_reply.started":"2024-12-11T05:55:37.695925Z","shell.execute_reply":"2024-12-11T05:55:37.700217Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"median_fill_null(train_df,'Annual Income')\nmedian_fill_null(test_df,'Annual Income')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:55:46.879766Z","iopub.execute_input":"2024-12-11T05:55:46.880161Z","iopub.status.idle":"2024-12-11T05:55:46.941594Z","shell.execute_reply.started":"2024-12-11T05:55:46.880128Z","shell.execute_reply":"2024-12-11T05:55:46.940686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"median_fill_null(train_df,'Previous Claims')\nmedian_fill_null(test_df,'Previous Claims')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:55:54.935273Z","iopub.execute_input":"2024-12-11T05:55:54.935742Z","iopub.status.idle":"2024-12-11T05:55:54.997541Z","shell.execute_reply.started":"2024-12-11T05:55:54.935707Z","shell.execute_reply":"2024-12-11T05:55:54.996612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"median_fill_null(train_df,'Credit Score')\nmedian_fill_null(test_df,'Credit Score')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:56:02.515244Z","iopub.execute_input":"2024-12-11T05:56:02.515649Z","iopub.status.idle":"2024-12-11T05:56:02.572338Z","shell.execute_reply.started":"2024-12-11T05:56:02.515614Z","shell.execute_reply":"2024-12-11T05:56:02.571186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"median_fill_null(train_df,'Insurance Duration')\nmedian_fill_null(test_df,'Insurance Duration')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:56:12.694904Z","iopub.execute_input":"2024-12-11T05:56:12.695291Z","iopub.status.idle":"2024-12-11T05:56:12.754384Z","shell.execute_reply.started":"2024-12-11T05:56:12.695254Z","shell.execute_reply":"2024-12-11T05:56:12.753559Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Fill Null values of Health Score, and Vehicle Age column","metadata":{}},{"cell_type":"markdown","source":"#### Above mentioned variables are normally distributed. Hence, those null values were filled using the Mean.","metadata":{}},{"cell_type":"code","source":"def mean_fill_null(df,var):\n    mean_value = df[var].mean()\n    df[var] = df[var].fillna(mean_value)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:57:04.629545Z","iopub.execute_input":"2024-12-11T05:57:04.630336Z","iopub.status.idle":"2024-12-11T05:57:04.635151Z","shell.execute_reply.started":"2024-12-11T05:57:04.630299Z","shell.execute_reply":"2024-12-11T05:57:04.633902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mean_fill_null(train_df,'Health Score')\nmean_fill_null(test_df,'Health Score')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:57:12.290114Z","iopub.execute_input":"2024-12-11T05:57:12.290775Z","iopub.status.idle":"2024-12-11T05:57:12.318196Z","shell.execute_reply.started":"2024-12-11T05:57:12.290737Z","shell.execute_reply":"2024-12-11T05:57:12.316937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mean_fill_null(train_df,'Vehicle Age')\nmean_fill_null(test_df,'Vehicle Age')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:57:19.299275Z","iopub.execute_input":"2024-12-11T05:57:19.299657Z","iopub.status.idle":"2024-12-11T05:57:19.324523Z","shell.execute_reply.started":"2024-12-11T05:57:19.299626Z","shell.execute_reply":"2024-12-11T05:57:19.323591Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Fill Null values of Categorical Columns","metadata":{}},{"cell_type":"markdown","source":"#### Because of having lot of missing values, Unknown data is allocated to new category as unknown.","metadata":{}},{"cell_type":"code","source":"categorical_columns = ['Occupation','Marital Status','Customer Feedback']\n\ndef fill_catogory(df):\n    for column in categorical_columns:\n        df[column] = df[column].astype('category')\n        df[column] = df[column].cat.add_categories(['Unknown'])\n        df[column] = df[column].fillna('Unknown')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:58:14.190092Z","iopub.execute_input":"2024-12-11T05:58:14.191099Z","iopub.status.idle":"2024-12-11T05:58:14.19651Z","shell.execute_reply.started":"2024-12-11T05:58:14.191055Z","shell.execute_reply":"2024-12-11T05:58:14.195267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fill_catogory(train_df)\nfill_catogory(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:58:22.369884Z","iopub.execute_input":"2024-12-11T05:58:22.370228Z","iopub.status.idle":"2024-12-11T05:58:22.900919Z","shell.execute_reply.started":"2024-12-11T05:58:22.370197Z","shell.execute_reply":"2024-12-11T05:58:22.899756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:58:32.204313Z","iopub.execute_input":"2024-12-11T05:58:32.204761Z","iopub.status.idle":"2024-12-11T05:58:32.682645Z","shell.execute_reply.started":"2024-12-11T05:58:32.204726Z","shell.execute_reply":"2024-12-11T05:58:32.68158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T05:58:44.329504Z","iopub.execute_input":"2024-12-11T05:58:44.329863Z","iopub.status.idle":"2024-12-11T05:58:44.653315Z","shell.execute_reply.started":"2024-12-11T05:58:44.329833Z","shell.execute_reply":"2024-12-11T05:58:44.652268Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### All the null values have filled.","metadata":{}},{"cell_type":"markdown","source":"### Check number of unique values in each column.","metadata":{}},{"cell_type":"code","source":"for col in train_df.columns:\n    print(col,'---->',train_df[col].nunique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:00:54.330075Z","iopub.execute_input":"2024-12-11T06:00:54.330918Z","iopub.status.idle":"2024-12-11T06:00:55.321101Z","shell.execute_reply.started":"2024-12-11T06:00:54.330877Z","shell.execute_reply":"2024-12-11T06:00:55.319952Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Checking for Linear Relationship with Premium Amount","metadata":{}},{"cell_type":"code","source":"numeric_df = train_df.select_dtypes(include = ['number'])\ncorr_matrix = numeric_df.corr()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:01:34.809854Z","iopub.execute_input":"2024-12-11T06:01:34.810252Z","iopub.status.idle":"2024-12-11T06:01:35.306932Z","shell.execute_reply.started":"2024-12-11T06:01:34.810216Z","shell.execute_reply":"2024-12-11T06:01:35.305808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(corr_matrix['Premium Amount'].sort_values(ascending = False).to_string())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:01:45.370744Z","iopub.execute_input":"2024-12-11T06:01:45.371834Z","iopub.status.idle":"2024-12-11T06:01:45.3788Z","shell.execute_reply.started":"2024-12-11T06:01:45.371794Z","shell.execute_reply":"2024-12-11T06:01:45.377696Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Checking for Multicolinearity","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(20,20))\nsns.heatmap(corr_matrix,annot=True,cmap = 'coolwarm', fmt = \".2f\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:02:25.97024Z","iopub.execute_input":"2024-12-11T06:02:25.971126Z","iopub.status.idle":"2024-12-11T06:02:26.660519Z","shell.execute_reply.started":"2024-12-11T06:02:25.971084Z","shell.execute_reply":"2024-12-11T06:02:26.659424Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Here, All the correlation values are very low. So we can conclude that there is no strong linear relationship among variables. Hence, Multicollinearity cannot be identified.","metadata":{}},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"### One hot Encording for Categorical Variables","metadata":{}},{"cell_type":"markdown","source":"#### One-hot encoding is used to convert categorical data into a numerical format, making it easier for machine learning models to process.Categorical variables often have no inherent order. For instance, the categories \"red\", \"green\", and \"blue\" don’t have a meaningful numeric order, so simply assigning numbers (e.g., 1, 2, 3) could mislead the model into thinking there’s some kind of ordinal relationship. One-hot encoding avoids this issue by representing each category with a binary vector.","metadata":{}},{"cell_type":"code","source":"train_df_encoded = pd.get_dummies(train_df, \n                                  columns=['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location',\n                                           'Policy Type', 'Customer Feedback','Smoking Status',\n                                           'Exercise Frequency', 'Property Type'],\n                                  drop_first=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:03:50.002189Z","iopub.execute_input":"2024-12-11T06:03:50.002655Z","iopub.status.idle":"2024-12-11T06:03:50.921244Z","shell.execute_reply.started":"2024-12-11T06:03:50.002619Z","shell.execute_reply":"2024-12-11T06:03:50.920442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df_encoded = pd.get_dummies(test_df, \n                                  columns=['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location',\n                                           'Policy Type', 'Customer Feedback','Smoking Status',\n                                           'Exercise Frequency', 'Property Type'],\n                                  drop_first=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:04:00.224921Z","iopub.execute_input":"2024-12-11T06:04:00.225328Z","iopub.status.idle":"2024-12-11T06:04:00.770639Z","shell.execute_reply.started":"2024-12-11T06:04:00.225291Z","shell.execute_reply":"2024-12-11T06:04:00.769497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df_encoded.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:04:10.815266Z","iopub.execute_input":"2024-12-11T06:04:10.815695Z","iopub.status.idle":"2024-12-11T06:04:10.823251Z","shell.execute_reply.started":"2024-12-11T06:04:10.815657Z","shell.execute_reply":"2024-12-11T06:04:10.822009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df_encoded.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:04:17.795004Z","iopub.execute_input":"2024-12-11T06:04:17.795956Z","iopub.status.idle":"2024-12-11T06:04:17.802178Z","shell.execute_reply.started":"2024-12-11T06:04:17.795912Z","shell.execute_reply":"2024-12-11T06:04:17.801339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df_encoded.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:04:25.415134Z","iopub.execute_input":"2024-12-11T06:04:25.415546Z","iopub.status.idle":"2024-12-11T06:04:25.42266Z","shell.execute_reply.started":"2024-12-11T06:04:25.415511Z","shell.execute_reply":"2024-12-11T06:04:25.421604Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Create Features on Policy Start Date","metadata":{}},{"cell_type":"code","source":"import datetime","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:04:57.874656Z","iopub.execute_input":"2024-12-11T06:04:57.875079Z","iopub.status.idle":"2024-12-11T06:04:57.880067Z","shell.execute_reply.started":"2024-12-11T06:04:57.875042Z","shell.execute_reply":"2024-12-11T06:04:57.878948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def date_encode(df_encoded):\n    df_encoded['Policy Start Date'] = pd.to_datetime(df_encoded['Policy Start Date'], errors='coerce')\n    df_encoded['Year'] = df_encoded['Policy Start Date'].dt.year\n    df_encoded['Month'] = df_encoded['Policy Start Date'].dt.month\n    df_encoded['WEEK_OF_YEAR'] = df_encoded['Policy Start Date'].dt.isocalendar().week","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:05:09.999469Z","iopub.execute_input":"2024-12-11T06:05:10.000014Z","iopub.status.idle":"2024-12-11T06:05:10.007272Z","shell.execute_reply.started":"2024-12-11T06:05:09.999961Z","shell.execute_reply":"2024-12-11T06:05:10.005835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"date_encode(train_df_encoded)\ndate_encode(test_df_encoded)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:05:18.735523Z","iopub.execute_input":"2024-12-11T06:05:18.735919Z","iopub.status.idle":"2024-12-11T06:05:19.793384Z","shell.execute_reply.started":"2024-12-11T06:05:18.735885Z","shell.execute_reply":"2024-12-11T06:05:19.792226Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Transform Date features into sin and cosine","metadata":{}},{"cell_type":"markdown","source":"#### Typically, Date features such as month, week of year are cyclical. To transform this feature to numeric value. Data scientists use sine, and cosine transformations. Here, I have transformed month and week of year features into sine and cosine.","metadata":{}},{"cell_type":"code","source":"def periodic_transform(dff,variable):\n    dff[f\"{variable}_SIN\"] = np.sin(dff[variable] / dff[variable].max()*2*np.pi)\n    dff[f\"{variable}_COS\"] = np.cos(dff[variable] / dff[variable].max()*2*np.pi)\n    return dff","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:06:11.275263Z","iopub.execute_input":"2024-12-11T06:06:11.275718Z","iopub.status.idle":"2024-12-11T06:06:11.28183Z","shell.execute_reply.started":"2024-12-11T06:06:11.275683Z","shell.execute_reply":"2024-12-11T06:06:11.280631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cyclic_col = ['Month','WEEK_OF_YEAR']\n\nfor col in cyclic_col:\n    df_N = periodic_transform(train_df_encoded, col)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:06:27.014493Z","iopub.execute_input":"2024-12-11T06:06:27.014872Z","iopub.status.idle":"2024-12-11T06:06:27.177374Z","shell.execute_reply.started":"2024-12-11T06:06:27.014839Z","shell.execute_reply":"2024-12-11T06:06:27.176475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df_encoded = train_df_encoded.drop(['Policy Start Date','Month','WEEK_OF_YEAR'], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:06:37.87451Z","iopub.execute_input":"2024-12-11T06:06:37.874882Z","iopub.status.idle":"2024-12-11T06:06:37.971009Z","shell.execute_reply.started":"2024-12-11T06:06:37.874851Z","shell.execute_reply":"2024-12-11T06:06:37.969898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in cyclic_col:\n    df_N = periodic_transform(test_df_encoded, col)\n\ntest_df_encoded = test_df_encoded.drop(['Policy Start Date','Month','WEEK_OF_YEAR'], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:06:48.895294Z","iopub.execute_input":"2024-12-11T06:06:48.895944Z","iopub.status.idle":"2024-12-11T06:06:49.078482Z","shell.execute_reply.started":"2024-12-11T06:06:48.895878Z","shell.execute_reply":"2024-12-11T06:06:49.077329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df_encoded.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:07:24.635443Z","iopub.execute_input":"2024-12-11T06:07:24.635799Z","iopub.status.idle":"2024-12-11T06:07:24.642447Z","shell.execute_reply.started":"2024-12-11T06:07:24.635769Z","shell.execute_reply":"2024-12-11T06:07:24.641301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df_encoded.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:07:33.634863Z","iopub.execute_input":"2024-12-11T06:07:33.635242Z","iopub.status.idle":"2024-12-11T06:07:33.642863Z","shell.execute_reply.started":"2024-12-11T06:07:33.635208Z","shell.execute_reply":"2024-12-11T06:07:33.641757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df_encoded.head","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:07:58.399229Z","iopub.execute_input":"2024-12-11T06:07:58.399689Z","iopub.status.idle":"2024-12-11T06:07:58.604386Z","shell.execute_reply.started":"2024-12-11T06:07:58.399637Z","shell.execute_reply":"2024-12-11T06:07:58.603402Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Splitting Data","metadata":{}},{"cell_type":"code","source":"x = train_df_encoded.drop(['id','Premium Amount'],axis =1)\ny = train_df_encoded['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:08:33.164823Z","iopub.execute_input":"2024-12-11T06:08:33.165192Z","iopub.status.idle":"2024-12-11T06:08:33.246708Z","shell.execute_reply.started":"2024-12-11T06:08:33.165162Z","shell.execute_reply":"2024-12-11T06:08:33.245799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:08:44.754897Z","iopub.execute_input":"2024-12-11T06:08:44.755281Z","iopub.status.idle":"2024-12-11T06:08:44.941635Z","shell.execute_reply.started":"2024-12-11T06:08:44.755246Z","shell.execute_reply":"2024-12-11T06:08:44.940586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_train,x_test,y_train,y_test = train_test_split(x,y,test_size = 0.3,random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:08:55.894645Z","iopub.execute_input":"2024-12-11T06:08:55.895Z","iopub.status.idle":"2024-12-11T06:08:56.34599Z","shell.execute_reply.started":"2024-12-11T06:08:55.894971Z","shell.execute_reply":"2024-12-11T06:08:56.344899Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Scaling","metadata":{}},{"cell_type":"markdown","source":"#### Most of the features has no normal distribution. So, Applying min max scaler is wise.","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:10:00.174715Z","iopub.execute_input":"2024-12-11T06:10:00.175109Z","iopub.status.idle":"2024-12-11T06:10:00.180627Z","shell.execute_reply.started":"2024-12-11T06:10:00.175079Z","shell.execute_reply":"2024-12-11T06:10:00.179492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mm = MinMaxScaler()\nx_train_scaled = mm.fit_transform(x_train)\nx_test_scaled = mm.transform(x_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:10:08.03505Z","iopub.execute_input":"2024-12-11T06:10:08.035454Z","iopub.status.idle":"2024-12-11T06:10:10.889901Z","shell.execute_reply.started":"2024-12-11T06:10:08.035415Z","shell.execute_reply":"2024-12-11T06:10:10.88885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df_encoded = test_df_encoded.drop(['id'],axis =1)\ntest_scaled = mm.transform(test_df_encoded.values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:16:20.578271Z","iopub.execute_input":"2024-12-11T06:16:20.578748Z","iopub.status.idle":"2024-12-11T06:16:23.626514Z","shell.execute_reply.started":"2024-12-11T06:16:20.578711Z","shell.execute_reply":"2024-12-11T06:16:23.625529Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Deploy model","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error,mean_absolute_error\ndef model_acc(model,xtrain,ytrain,xtest):\n    model.fit(xtrain,ytrain)\n    y_pred = model.predict(xtest)\n    mse = mean_squared_error(y_test, y_pred)\n    mae = mean_absolute_error(y_test, y_pred)\n    rmse = np.sqrt(mse)\n\n    print(f\"MAE: {mae}\")\n    print(f\"MSE: {mse}\")\n    print(f\"RMSE: {rmse}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:17:39.93759Z","iopub.execute_input":"2024-12-11T06:17:39.93801Z","iopub.status.idle":"2024-12-11T06:17:39.944535Z","shell.execute_reply.started":"2024-12-11T06:17:39.937974Z","shell.execute_reply":"2024-12-11T06:17:39.94326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nlr = LinearRegression()\n\nfrom sklearn.linear_model import Lasso\nlasso = Lasso()\n\nfrom sklearn.tree import DecisionTreeRegressor\ndt = DecisionTreeRegressor()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:17:53.024194Z","iopub.execute_input":"2024-12-11T06:17:53.024754Z","iopub.status.idle":"2024-12-11T06:17:53.216488Z","shell.execute_reply.started":"2024-12-11T06:17:53.024715Z","shell.execute_reply":"2024-12-11T06:17:53.215332Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Linear Regression","metadata":{}},{"cell_type":"code","source":"model_acc(lr,x_train_scaled,y_train,x_test_scaled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:18:01.337868Z","iopub.execute_input":"2024-12-11T06:18:01.338227Z","iopub.status.idle":"2024-12-11T06:18:04.340618Z","shell.execute_reply.started":"2024-12-11T06:18:01.338197Z","shell.execute_reply":"2024-12-11T06:18:04.338554Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Lasso","metadata":{}},{"cell_type":"code","source":"model_acc(lasso,x_train_scaled,y_train,x_test_scaled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:18:14.137485Z","iopub.execute_input":"2024-12-11T06:18:14.138582Z","iopub.status.idle":"2024-12-11T06:18:14.701155Z","shell.execute_reply.started":"2024-12-11T06:18:14.138541Z","shell.execute_reply":"2024-12-11T06:18:14.696827Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Decision Tree Regression","metadata":{}},{"cell_type":"code","source":"model_acc(dt,x_train_scaled,y_train,x_test_scaled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:18:26.937533Z","iopub.execute_input":"2024-12-11T06:18:26.938197Z","iopub.status.idle":"2024-12-11T06:18:58.019555Z","shell.execute_reply.started":"2024-12-11T06:18:26.938158Z","shell.execute_reply":"2024-12-11T06:18:58.018236Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### LiightGBM","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:21:15.868004Z","iopub.execute_input":"2024-12-11T06:21:15.868456Z","iopub.status.idle":"2024-12-11T06:21:16.878546Z","shell.execute_reply.started":"2024-12-11T06:21:15.868414Z","shell.execute_reply":"2024-12-11T06:21:16.877432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_lgb_data = lgb.Dataset(x_train_scaled, label=y_train)\ntest_lgb_data = lgb.Dataset(x_test_scaled, label=y_test, reference=train_lgb_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:21:33.467953Z","iopub.execute_input":"2024-12-11T06:21:33.468605Z","iopub.status.idle":"2024-12-11T06:21:33.474017Z","shell.execute_reply.started":"2024-12-11T06:21:33.468564Z","shell.execute_reply":"2024-12-11T06:21:33.472916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params = {\n    'objective': 'regression',\n    'metric': 'l2',  # For regression, L2 loss (mean squared error) is common\n    'num_leaves': 31,\n    'learning_rate': 0.05,\n    'feature_fraction': 0.9,\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:21:46.878248Z","iopub.execute_input":"2024-12-11T06:21:46.878762Z","iopub.status.idle":"2024-12-11T06:21:46.884517Z","shell.execute_reply.started":"2024-12-11T06:21:46.878726Z","shell.execute_reply":"2024-12-11T06:21:46.883105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gbm = lgb.train(params, train_lgb_data, valid_sets=[test_lgb_data], num_boost_round=100)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:21:59.528612Z","iopub.execute_input":"2024-12-11T06:21:59.52901Z","iopub.status.idle":"2024-12-11T06:22:06.537666Z","shell.execute_reply.started":"2024-12-11T06:21:59.528973Z","shell.execute_reply":"2024-12-11T06:22:06.536546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = gbm.predict(x_test, num_iteration=gbm.best_iteration)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:23:23.177911Z","iopub.execute_input":"2024-12-11T06:23:23.179081Z","iopub.status.idle":"2024-12-11T06:23:23.670515Z","shell.execute_reply.started":"2024-12-11T06:23:23.179034Z","shell.execute_reply":"2024-12-11T06:23:23.669073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mse = mean_squared_error(y_test, y_pred)\nmae = mean_absolute_error(y_test, y_pred)\nrmse = np.sqrt(mse)\n\nprint(f\"MAE: {mae}\")\nprint(f\"MSE: {mse}\")\nprint(f\"RMSE: {rmse}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:23:33.977311Z","iopub.execute_input":"2024-12-11T06:23:33.97776Z","iopub.status.idle":"2024-12-11T06:23:33.989317Z","shell.execute_reply.started":"2024-12-11T06:23:33.977725Z","shell.execute_reply":"2024-12-11T06:23:33.98822Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### xgboost","metadata":{}},{"cell_type":"code","source":"import xgboost as xgb","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:24:37.617217Z","iopub.execute_input":"2024-12-11T06:24:37.617622Z","iopub.status.idle":"2024-12-11T06:24:37.80312Z","shell.execute_reply.started":"2024-12-11T06:24:37.617587Z","shell.execute_reply":"2024-12-11T06:24:37.801972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = xgb.DMatrix(x_train_scaled, label=y_train)\ntest_data = xgb.DMatrix(x_test_scaled, label=y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:24:41.137648Z","iopub.execute_input":"2024-12-11T06:24:41.138064Z","iopub.status.idle":"2024-12-11T06:24:42.821094Z","shell.execute_reply.started":"2024-12-11T06:24:41.13803Z","shell.execute_reply":"2024-12-11T06:24:42.820125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params = {\n    'objective': 'reg:squarederror',  # For regression tasks\n    'learning_rate': 0.1,  # Step size shrinkage\n    'max_depth': 5,  # Maximum depth of a tree\n    'alpha': 10,  # L1 regularization term on weights\n    'n_estimators': 100  # Number of boosting rounds (trees)\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:24:53.787648Z","iopub.execute_input":"2024-12-11T06:24:53.788026Z","iopub.status.idle":"2024-12-11T06:24:53.793561Z","shell.execute_reply.started":"2024-12-11T06:24:53.787996Z","shell.execute_reply":"2024-12-11T06:24:53.792397Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_xgb = xgb.train(params, train_data, num_boost_round=100)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:25:03.898079Z","iopub.execute_input":"2024-12-11T06:25:03.898932Z","iopub.status.idle":"2024-12-11T06:25:15.554573Z","shell.execute_reply.started":"2024-12-11T06:25:03.898892Z","shell.execute_reply":"2024-12-11T06:25:15.553549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model_xgb.predict(test_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:25:24.137855Z","iopub.execute_input":"2024-12-11T06:25:24.138218Z","iopub.status.idle":"2024-12-11T06:25:24.811668Z","shell.execute_reply.started":"2024-12-11T06:25:24.138186Z","shell.execute_reply":"2024-12-11T06:25:24.810456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mse = mean_squared_error(y_test, y_pred)\nmae = mean_absolute_error(y_test, y_pred)\nrmse = np.sqrt(mse)\n\nprint(f\"MAE: {mae}\")\nprint(f\"MSE: {mse}\")\nprint(f\"RMSE: {rmse}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:25:40.567284Z","iopub.execute_input":"2024-12-11T06:25:40.567737Z","iopub.status.idle":"2024-12-11T06:25:40.579864Z","shell.execute_reply.started":"2024-12-11T06:25:40.567701Z","shell.execute_reply":"2024-12-11T06:25:40.57866Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## PCA","metadata":{}},{"cell_type":"markdown","source":"#### After Encording data, we have 48 columns as features. Some times high dimensionality can cause to low accuracy. to reduce the number of dimensions we can perform PCA.   ","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA\npca = PCA()\npca.fit(x_train_scaled)\n\n# Explained variance ratio\nexplained_variance_ratio = pca.explained_variance_ratio_\ncumulative_explained_variance = explained_variance_ratio.cumsum()\n\n# Plot explained variance\nplt.figure(figsize=(10, 6))\nplt.plot(range(1, len(explained_variance_ratio) + 1), cumulative_explained_variance, marker='o', linestyle='--')\nplt.xlabel('Number of Principal Components')\nplt.ylabel('Cumulative Explained Variance')\nplt.title('Explained Variance vs. Number of Components')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:30:37.177629Z","iopub.execute_input":"2024-12-11T06:30:37.178065Z","iopub.status.idle":"2024-12-11T06:30:40.03484Z","shell.execute_reply.started":"2024-12-11T06:30:37.178029Z","shell.execute_reply":"2024-12-11T06:30:40.03376Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### When the number of components equals 32, Cumulative explained variance becomes constant. Hence Suitable least number of components equals 32.","metadata":{}},{"cell_type":"code","source":"pca = PCA(n_components=32)\nx_train_pca = pca.fit_transform(x_train_scaled)\nx_test_pca = pca.transform(x_test_scaled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:34:38.597437Z","iopub.execute_input":"2024-12-11T06:34:38.597852Z","iopub.status.idle":"2024-12-11T06:34:48.680794Z","shell.execute_reply.started":"2024-12-11T06:34:38.597816Z","shell.execute_reply":"2024-12-11T06:34:48.67923Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Linear regression for PCA data","metadata":{}},{"cell_type":"code","source":"model_acc(lr,x_train_pca,y_train,x_test_pca)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:37:49.818803Z","iopub.execute_input":"2024-12-11T06:37:49.819244Z","iopub.status.idle":"2024-12-11T06:37:51.643994Z","shell.execute_reply.started":"2024-12-11T06:37:49.819211Z","shell.execute_reply":"2024-12-11T06:37:51.641989Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Lasso for PCA data","metadata":{}},{"cell_type":"code","source":"model_acc(lasso,x_train_pca,y_train,x_test_pca)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:38:00.547685Z","iopub.execute_input":"2024-12-11T06:38:00.548087Z","iopub.status.idle":"2024-12-11T06:38:01.120592Z","shell.execute_reply.started":"2024-12-11T06:38:00.548048Z","shell.execute_reply":"2024-12-11T06:38:01.116706Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Decision tree regression for PCA data","metadata":{}},{"cell_type":"code","source":"model_acc(dt,x_train_pca,y_train,x_test_pca)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:38:09.302396Z","iopub.execute_input":"2024-12-11T06:38:09.302809Z","iopub.status.idle":"2024-12-11T06:40:19.604742Z","shell.execute_reply.started":"2024-12-11T06:38:09.302774Z","shell.execute_reply":"2024-12-11T06:40:19.603658Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### xgboost for PCA Data","metadata":{}},{"cell_type":"code","source":"train_pca_data = xgb.DMatrix(x_train_pca, label=y_train)\ntest_pca_data = xgb.DMatrix(x_test_pca, label=y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:40:31.81228Z","iopub.execute_input":"2024-12-11T06:40:31.812709Z","iopub.status.idle":"2024-12-11T06:40:32.652655Z","shell.execute_reply.started":"2024-12-11T06:40:31.812674Z","shell.execute_reply":"2024-12-11T06:40:32.651572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params = {\n    'objective': 'reg:squarederror',  # For regression tasks\n    'learning_rate': 0.1,  # Step size shrinkage\n    'max_depth': 5,  # Maximum depth of a tree\n    'alpha': 10,  # L1 regularization term on weights\n    'n_estimators': 100  # Number of boosting rounds (trees)\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:40:43.872664Z","iopub.execute_input":"2024-12-11T06:40:43.87309Z","iopub.status.idle":"2024-12-11T06:40:43.878192Z","shell.execute_reply.started":"2024-12-11T06:40:43.873056Z","shell.execute_reply":"2024-12-11T06:40:43.877049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_pca_xgb = xgb.train(params, train_pca_data, num_boost_round=100)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:40:53.072457Z","iopub.execute_input":"2024-12-11T06:40:53.072876Z","iopub.status.idle":"2024-12-11T06:41:06.946199Z","shell.execute_reply.started":"2024-12-11T06:40:53.07284Z","shell.execute_reply":"2024-12-11T06:41:06.94517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model_pca_xgb.predict(test_pca_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:41:10.977886Z","iopub.execute_input":"2024-12-11T06:41:10.978276Z","iopub.status.idle":"2024-12-11T06:41:11.634505Z","shell.execute_reply.started":"2024-12-11T06:41:10.978241Z","shell.execute_reply":"2024-12-11T06:41:11.633296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mse = mean_squared_error(y_test, y_pred)\nmae = mean_absolute_error(y_test, y_pred)\nrmse = np.sqrt(mse)\n\nprint(f\"MAE: {mae}\")\nprint(f\"MSE: {mse}\")\nprint(f\"RMSE: {rmse}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:41:23.682513Z","iopub.execute_input":"2024-12-11T06:41:23.683512Z","iopub.status.idle":"2024-12-11T06:41:23.698532Z","shell.execute_reply.started":"2024-12-11T06:41:23.683473Z","shell.execute_reply":"2024-12-11T06:41:23.697314Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### According to previous results XGBoost is the most suitable model without PCA. So finally I applied xgboost without PCA for the test set. ","metadata":{}},{"cell_type":"code","source":"test_scaled.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:45:18.900402Z","iopub.execute_input":"2024-12-11T06:45:18.901373Z","iopub.status.idle":"2024-12-11T06:45:18.909581Z","shell.execute_reply.started":"2024-12-11T06:45:18.901296Z","shell.execute_reply":"2024-12-11T06:45:18.908188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dtest = xgb.DMatrix(test_scaled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:45:45.94279Z","iopub.execute_input":"2024-12-11T06:45:45.943158Z","iopub.status.idle":"2024-12-11T06:45:47.05974Z","shell.execute_reply.started":"2024-12-11T06:45:45.943127Z","shell.execute_reply":"2024-12-11T06:45:47.058882Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test_pred = model_xgb.predict(dtest)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:45:57.642826Z","iopub.execute_input":"2024-12-11T06:45:57.643853Z","iopub.status.idle":"2024-12-11T06:45:59.143928Z","shell.execute_reply.started":"2024-12-11T06:45:57.643809Z","shell.execute_reply":"2024-12-11T06:45:59.142939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test_pred.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:46:26.90303Z","iopub.execute_input":"2024-12-11T06:46:26.904067Z","iopub.status.idle":"2024-12-11T06:46:26.912558Z","shell.execute_reply.started":"2024-12-11T06:46:26.904016Z","shell.execute_reply":"2024-12-11T06:46:26.911528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df = pd.DataFrame({\n    'id': test_df['id'],\n    'Premium Amount': y_test_pred\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:48:12.698781Z","iopub.execute_input":"2024-12-11T06:48:12.699212Z","iopub.status.idle":"2024-12-11T06:48:12.708636Z","shell.execute_reply.started":"2024-12-11T06:48:12.699176Z","shell.execute_reply":"2024-12-11T06:48:12.707631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T06:48:26.897888Z","iopub.execute_input":"2024-12-11T06:48:26.898284Z","iopub.status.idle":"2024-12-11T06:48:26.914096Z","shell.execute_reply.started":"2024-12-11T06:48:26.89825Z","shell.execute_reply":"2024-12-11T06:48:26.913024Z"}},"outputs":[],"execution_count":null}]}