{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":9178166,"sourceType":"datasetVersion","datasetId":5547076}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader,TensorDataset\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:17.192337Z","iopub.execute_input":"2024-12-06T12:58:17.192609Z","iopub.status.idle":"2024-12-06T12:58:22.144553Z","shell.execute_reply.started":"2024-12-06T12:58:17.192575Z","shell.execute_reply":"2024-12-06T12:58:22.143748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"competition_df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\noriginal_dataset_df = pd.read_csv('/kaggle/input/insurance-premium-prediction/Insurance Premium Prediction Dataset.csv')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:22.147655Z","iopub.execute_input":"2024-12-06T12:58:22.148001Z","iopub.status.idle":"2024-12-06T12:58:29.030068Z","shell.execute_reply.started":"2024-12-06T12:58:22.147977Z","shell.execute_reply":"2024-12-06T12:58:29.029338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"competition_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:29.031108Z","iopub.execute_input":"2024-12-06T12:58:29.031385Z","iopub.status.idle":"2024-12-06T12:58:29.064017Z","shell.execute_reply.started":"2024-12-06T12:58:29.031359Z","shell.execute_reply":"2024-12-06T12:58:29.06325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"original_dataset_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:29.065007Z","iopub.execute_input":"2024-12-06T12:58:29.065276Z","iopub.status.idle":"2024-12-06T12:58:29.081645Z","shell.execute_reply.started":"2024-12-06T12:58:29.065233Z","shell.execute_reply":"2024-12-06T12:58:29.080854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"competition_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:29.084181Z","iopub.execute_input":"2024-12-06T12:58:29.084537Z","iopub.status.idle":"2024-12-06T12:58:29.635888Z","shell.execute_reply.started":"2024-12-06T12:58:29.0845Z","shell.execute_reply":"2024-12-06T12:58:29.635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"original_dataset_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:29.636724Z","iopub.execute_input":"2024-12-06T12:58:29.636952Z","iopub.status.idle":"2024-12-06T12:58:29.767307Z","shell.execute_reply.started":"2024-12-06T12:58:29.636929Z","shell.execute_reply":"2024-12-06T12:58:29.766567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"competition_df.drop(columns='id',inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:29.768476Z","iopub.execute_input":"2024-12-06T12:58:29.768754Z","iopub.status.idle":"2024-12-06T12:58:29.920107Z","shell.execute_reply.started":"2024-12-06T12:58:29.768728Z","shell.execute_reply":"2024-12-06T12:58:29.919402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"original_dataset_df = original_dataset_df[sorted(original_dataset_df.columns)]\ncompetition_df = competition_df[sorted(competition_df.columns)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:29.920963Z","iopub.execute_input":"2024-12-06T12:58:29.921188Z","iopub.status.idle":"2024-12-06T12:58:30.136646Z","shell.execute_reply.started":"2024-12-06T12:58:29.921167Z","shell.execute_reply":"2024-12-06T12:58:30.135715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_df = pd.concat([original_dataset_df,competition_df],axis=0)\nnew_df = new_df.reset_index(drop=True)\nnew_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:30.137705Z","iopub.execute_input":"2024-12-06T12:58:30.137973Z","iopub.status.idle":"2024-12-06T12:58:31.212782Z","shell.execute_reply.started":"2024-12-06T12:58:30.137949Z","shell.execute_reply":"2024-12-06T12:58:31.21185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:31.214037Z","iopub.execute_input":"2024-12-06T12:58:31.214915Z","iopub.status.idle":"2024-12-06T12:58:31.87106Z","shell.execute_reply.started":"2024-12-06T12:58:31.214871Z","shell.execute_reply":"2024-12-06T12:58:31.870162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(((new_df.isna().sum()) / len(new_df))) * 100","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:31.872055Z","iopub.execute_input":"2024-12-06T12:58:31.872375Z","iopub.status.idle":"2024-12-06T12:58:32.521012Z","shell.execute_reply.started":"2024-12-06T12:58:31.872346Z","shell.execute_reply":"2024-12-06T12:58:32.520224Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_df = new_df.drop(columns='Customer Feedback')\n\nnew_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:32.522087Z","iopub.execute_input":"2024-12-06T12:58:32.52245Z","iopub.status.idle":"2024-12-06T12:58:33.290734Z","shell.execute_reply.started":"2024-12-06T12:58:32.522417Z","shell.execute_reply":"2024-12-06T12:58:33.289799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(((new_df.isna().sum()) / len(new_df))) * 100","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:33.291699Z","iopub.execute_input":"2024-12-06T12:58:33.29197Z","iopub.status.idle":"2024-12-06T12:58:33.886746Z","shell.execute_reply.started":"2024-12-06T12:58:33.291944Z","shell.execute_reply":"2024-12-06T12:58:33.885757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in new_df.columns:\n    print(new_df[col].value_counts(), '\\n')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:33.8902Z","iopub.execute_input":"2024-12-06T12:58:33.89054Z","iopub.status.idle":"2024-12-06T12:58:35.387238Z","shell.execute_reply.started":"2024-12-06T12:58:33.890513Z","shell.execute_reply":"2024-12-06T12:58:35.386393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_df.duplicated(keep=False).sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:35.388175Z","iopub.execute_input":"2024-12-06T12:58:35.388437Z","iopub.status.idle":"2024-12-06T12:58:37.041248Z","shell.execute_reply.started":"2024-12-06T12:58:35.388412Z","shell.execute_reply":"2024-12-06T12:58:37.040458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_features = ['Smoking Status','Property Type','Policy Type','Occupation','Marital Status','Location','Gender','Education Level','Exercise Frequency','Previous Claims','Number of Dependents','Insurance Duration']\n\nnew_df[cat_features].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:37.042425Z","iopub.execute_input":"2024-12-06T12:58:37.042702Z","iopub.status.idle":"2024-12-06T12:58:37.450249Z","shell.execute_reply.started":"2024-12-06T12:58:37.042675Z","shell.execute_reply":"2024-12-06T12:58:37.449473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_features = new_df.columns[[col not in cat_features + ['Policy Start Date'] for col in new_df.columns]]\nnum_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:37.451703Z","iopub.execute_input":"2024-12-06T12:58:37.452054Z","iopub.status.idle":"2024-12-06T12:58:37.458137Z","shell.execute_reply.started":"2024-12-06T12:58:37.452016Z","shell.execute_reply":"2024-12-06T12:58:37.457368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_df[num_features].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:37.459017Z","iopub.execute_input":"2024-12-06T12:58:37.459249Z","iopub.status.idle":"2024-12-06T12:58:37.931598Z","shell.execute_reply.started":"2024-12-06T12:58:37.459226Z","shell.execute_reply":"2024-12-06T12:58:37.93069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_df['Age'] = new_df['Age'].replace([np.inf, -np.inf], np.nan)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:37.932498Z","iopub.execute_input":"2024-12-06T12:58:37.932727Z","iopub.status.idle":"2024-12-06T12:58:37.947076Z","shell.execute_reply.started":"2024-12-06T12:58:37.932704Z","shell.execute_reply":"2024-12-06T12:58:37.94627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"eda = new_df.sample(5000)\npd.option_context('mode.use_inf_as_na', True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:37.948018Z","iopub.execute_input":"2024-12-06T12:58:37.948248Z","iopub.status.idle":"2024-12-06T12:58:37.989639Z","shell.execute_reply.started":"2024-12-06T12:58:37.948225Z","shell.execute_reply":"2024-12-06T12:58:37.988942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len((num_features).tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:37.990476Z","iopub.execute_input":"2024-12-06T12:58:37.990717Z","iopub.status.idle":"2024-12-06T12:58:37.996077Z","shell.execute_reply.started":"2024-12-06T12:58:37.990693Z","shell.execute_reply":"2024-12-06T12:58:37.995144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cols = 3\nrows = 4\nfig = plt.figure(figsize=(cols*3, rows*3))\nfor i, col in enumerate(num_features):\n    ax=fig.add_subplot(rows,cols,i+1)\n    sns.kdeplot(data=eda,x=col,ax=ax,fill=True)\nfig.tight_layout()  \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:37.997044Z","iopub.execute_input":"2024-12-06T12:58:37.997302Z","iopub.status.idle":"2024-12-06T12:58:39.063644Z","shell.execute_reply.started":"2024-12-06T12:58:37.997246Z","shell.execute_reply":"2024-12-06T12:58:39.062805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for feature in cat_features:\n    plt.figure(figsize=(10, 6))  # Optional: Set figure size for better clarity\n    sns.countplot(data=eda, x=feature, order=eda[feature].value_counts().index)\n    plt.title(f'Distribution of {feature}')\n    plt.xticks(rotation=45)  # Rotate labels if they overlap\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:39.064873Z","iopub.execute_input":"2024-12-06T12:58:39.065201Z","iopub.status.idle":"2024-12-06T12:58:41.250718Z","shell.execute_reply.started":"2024-12-06T12:58:39.065163Z","shell.execute_reply":"2024-12-06T12:58:41.249779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transformed_premium = np.log(eda['Premium Amount'])\neda['transformed_premium'] = transformed_premium\nsns.kdeplot(x=transformed_premium,fill=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:41.251866Z","iopub.execute_input":"2024-12-06T12:58:41.252197Z","iopub.status.idle":"2024-12-06T12:58:41.542191Z","shell.execute_reply.started":"2024-12-06T12:58:41.252164Z","shell.execute_reply":"2024-12-06T12:58:41.5414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transformed_income = np.log(eda['Annual Income'])\neda['transformed_Annual Income'] = transformed_income\nsns.kdeplot(x=transformed_income,fill=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:41.543216Z","iopub.execute_input":"2024-12-06T12:58:41.543508Z","iopub.status.idle":"2024-12-06T12:58:41.85082Z","shell.execute_reply.started":"2024-12-06T12:58:41.543483Z","shell.execute_reply":"2024-12-06T12:58:41.84995Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"eda[\"previous_claims_cat\"] = pd.cut(eda[\"Previous Claims\"], bins=[0., 1, 2, np.inf],\n                                                            labels=[0, 1, 2],right=False)\n\nsns.countplot(data=eda, x='previous_claims_cat', order=eda['previous_claims_cat'].value_counts().index)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:41.851881Z","iopub.execute_input":"2024-12-06T12:58:41.852135Z","iopub.status.idle":"2024-12-06T12:58:41.973095Z","shell.execute_reply.started":"2024-12-06T12:58:41.852111Z","shell.execute_reply":"2024-12-06T12:58:41.972103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"eda.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:41.974633Z","iopub.execute_input":"2024-12-06T12:58:41.975069Z","iopub.status.idle":"2024-12-06T12:58:41.993756Z","shell.execute_reply.started":"2024-12-06T12:58:41.975019Z","shell.execute_reply":"2024-12-06T12:58:41.992798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"eda.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:41.994837Z","iopub.execute_input":"2024-12-06T12:58:41.995237Z","iopub.status.idle":"2024-12-06T12:58:42.009693Z","shell.execute_reply.started":"2024-12-06T12:58:41.99519Z","shell.execute_reply":"2024-12-06T12:58:42.007945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = new_df.copy()\n\ndf[\"Previous Claims\"] = pd.cut(df[\"Previous Claims\"], \n                                bins=[0., 1, 2, np.inf],\n                                labels=['No claims','1 claim','2 or more'],right=False)\n\ndf['Year'] = ((df['Policy Start Date'].str.split().str[0].str.split('-')).str[0]).astype(int)\ndf['Month'] = ((df['Policy Start Date'].str.split().str[0].str.split('-')).str[1]).astype(int)\n\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:42.010884Z","iopub.execute_input":"2024-12-06T12:58:42.011291Z","iopub.status.idle":"2024-12-06T12:58:54.221903Z","shell.execute_reply.started":"2024-12-06T12:58:42.011228Z","shell.execute_reply":"2024-12-06T12:58:54.221016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.base import BaseEstimator,TransformerMixin\nfrom sklearn.utils.validation import check_array\n\nclass CyclicMonthEncoder(BaseEstimator,TransformerMixin):\n\n    def __init__(self,col_name='Month'):\n        \n        self.col_name = col_name\n\n    def fit(self,X,y=None):\n        \n        return self\n\n    def transform(self,X):\n\n        X_copy = X.copy()\n\n        if isinstance(X, pd.DataFrame):\n            X_copy = X.copy()\n        elif isinstance(X, np.ndarray):\n            X_copy = pd.DataFrame(X, columns=[self.col_name])\n        else:\n            raise TypeError(f\"Expected DataFrame or ndarray, got {type(X)}\")\n\n        if self.col_name not in X_copy.columns:\n            raise ValueError(f\"{self.col_name} not in the data\")\n\n        # Calculate sin and cos\n        sin_col = np.sin(2 * np.pi * (X_copy[self.col_name] - 1) / 11)\n        cos_col = np.cos(2 * np.pi * (X_copy[self.col_name] - 1) / 11)\n\n        # Return as NumPy array for compatibility with ColumnTransformer\n        return np.column_stack((sin_col, cos_col))\n\n    def get_feature_names_out(self,names=None):\n        \n        return [f\"{self.col_name}_sin\", f\"{self.col_name}_cos\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:54.222836Z","iopub.execute_input":"2024-12-06T12:58:54.223065Z","iopub.status.idle":"2024-12-06T12:58:54.314208Z","shell.execute_reply.started":"2024-12-06T12:58:54.223043Z","shell.execute_reply":"2024-12-06T12:58:54.313628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler,FunctionTransformer,OneHotEncoder,OrdinalEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline,make_pipeline\n\nnum_cols = ['Age','Credit Score','Health Score',\n            'Insurance Duration','Vehicle Age']\n\nord_cols = ['Exercise Frequency','Policy Type','Previous Claims']\n\ncat_cols = ['Education Level','Gender','Location',\n            'Marital Status','Property Type','Smoking Status']\n\nlog_cols = ['Annual Income']\n\nExcercise_order = ['Rarely','Daily','Weekly','Monthly']\n\nPolicy_order = ['Basic','Comprehensive','Premium']\n\nprev_claims_order = ['No claims','1 claim','2 or more']\n\nlog_transformer = FunctionTransformer(func=lambda x: np.log(x + 1),\n                                      inverse_func=lambda x: np.exp(x) - 1,\n                                      feature_names_out=\"one-to-one\")\n\nlog_pipeline = Pipeline([\n                ('imputer',SimpleImputer(strategy=\"median\")),\n                ('log_tr',log_transformer),\n                ('scaler',StandardScaler())\n])\n\nnum_pipeline = make_pipeline(SimpleImputer(strategy=\"median\"),\n                             StandardScaler())\n\ncat_pipeline = make_pipeline(SimpleImputer(strategy='most_frequent'),\n                             OneHotEncoder(handle_unknown='ignore',\n                                           sparse_output=False,\n                                           drop='if_binary'))\n\nmonth_pipeline = make_pipeline(SimpleImputer(strategy='most_frequent'),\n                               CyclicMonthEncoder())\n\nyear_pipeline = make_pipeline(SimpleImputer(strategy='most_frequent'),\n                              StandardScaler())\n\nord_pipeline = make_pipeline(SimpleImputer(strategy='most_frequent'),\n                             OrdinalEncoder(categories=[Excercise_order,Policy_order,prev_claims_order]))\n\npreprocessing = ColumnTransformer(\n    transformers=[\n        ('num',num_pipeline,num_cols),\n        ('cat',cat_pipeline,cat_cols),\n        ('ord',ord_pipeline,ord_cols),\n        ('cy',month_pipeline,['Month']),\n        ('',year_pipeline,['Year']),\n        ('log',log_pipeline,log_cols),\n    ],\nremainder='drop'\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:54.315186Z","iopub.execute_input":"2024-12-06T12:58:54.315455Z","iopub.status.idle":"2024-12-06T12:58:54.548936Z","shell.execute_reply.started":"2024-12-06T12:58:54.31543Z","shell.execute_reply":"2024-12-06T12:58:54.548291Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Maybe add income cat and use it for stratify**","metadata":{}},{"cell_type":"code","source":"sampled_data = df.sample(n=100000, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:54.550113Z","iopub.execute_input":"2024-12-06T12:58:54.550407Z","iopub.status.idle":"2024-12-06T12:58:54.677494Z","shell.execute_reply.started":"2024-12-06T12:58:54.550381Z","shell.execute_reply":"2024-12-06T12:58:54.676782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nunclean_X = sampled_data.drop(columns='Premium Amount')\nunclean_y = sampled_data[['Premium Amount']]\nunclean_X_train,unclean_X_test,unclean_y_train,unclean_y_test = train_test_split(unclean_X,unclean_y,test_size=0.2,shuffle=True)\n\ntarget_imputer = SimpleImputer(strategy=\"median\").fit(unclean_y_train)\nX_train = preprocessing.fit_transform(unclean_X_train)\ny_train = target_imputer.transform(unclean_y_train).reshape(-1)\n\n#X_test = preprocessing.transform(unclean_X_test)\n#y_test = target_imputer.transform(unclean_y_test).reshape(-1)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:54.678588Z","iopub.execute_input":"2024-12-06T12:58:54.678935Z","iopub.status.idle":"2024-12-06T12:58:55.046768Z","shell.execute_reply.started":"2024-12-06T12:58:54.678898Z","shell.execute_reply":"2024-12-06T12:58:55.045806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(X_train.shape,y_train.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:55.048047Z","iopub.execute_input":"2024-12-06T12:58:55.048438Z","iopub.status.idle":"2024-12-06T12:58:55.053377Z","shell.execute_reply.started":"2024-12-06T12:58:55.048399Z","shell.execute_reply":"2024-12-06T12:58:55.052466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.base import BaseEstimator, RegressorMixin\n\nclass RMSLELoss(nn.Module):\n    def __init__(self):\n        super(RMSLELoss, self).__init__()\n\n    def forward(self, y_pred, y_true):\n        # Apply log transformation to both predicted and true values\n        y_pred = torch.log(torch.clamp(y_pred, min=1e-8))  # Avoid log(0)\n        y_true = torch.log(torch.clamp(y_true, min=1e-8))  # Avoid log(0)\n        \n        # Compute Mean Squared Logarithmic Error\n        return torch.sqrt(torch.mean((y_pred - y_true) ** 2))\n        \nclass MLP(nn.Module):\n    def __init__(self, input_dim, hidden_dim,hidden2_dim, output_dim):\n        super(MLP, self).__init__()\n        self.fc1 = nn.Linear(input_dim, hidden_dim)\n        self.fc2 = nn.Linear(hidden_dim, hidden2_dim)\n        self.fc3 = nn.Linear(hidden2_dim, output_dim)\n        self.relu = nn.ReLU()\n\n    def forward(self, x):\n        x = self.fc1(x)\n        x = self.relu(x)\n        x = self.fc2(x)\n        x = self.relu(x)\n        x = self.fc3(x)\n        return x\n\nclass PyTorchRegressor(BaseEstimator, RegressorMixin):\n    def __init__(self,device, input_dim, hidden_dim=150,hidden2_dim=100, output_dim=1, lr=0.001, epochs=10, batch_size=32):\n        self.input_dim = input_dim\n        self.hidden_dim = hidden_dim\n        self.hidden2_dim = hidden2_dim\n        self.output_dim = output_dim\n        self.lr = lr\n        self.epochs = epochs\n        self.batch_size = batch_size\n        self.device = device\n        self.model = MLP(input_dim, hidden_dim,hidden2_dim, output_dim).to(device)\n        self.criterion = RMSLELoss()\n        self.optimizer = optim.Adam(self.model.parameters(), lr=self.lr)\n\n    def fit(self, X, y):\n        X_tensor = torch.tensor(X, dtype=torch.float32).to(self.device)\n        y_tensor = torch.tensor(y, dtype=torch.float32).to(self.device)\n\n        train_dataset = TensorDataset(X_tensor, y_tensor)\n        train_loader = DataLoader(train_dataset, batch_size=self.batch_size, shuffle=True)\n\n        for epoch in range(self.epochs):\n            self.model.train()  \n            running_loss = 0.0\n            for inputs, targets in train_loader:\n                self.optimizer.zero_grad()  \n                outputs = self.model(inputs)  \n                loss = self.criterion(outputs, targets) \n                loss.backward()  \n                self.optimizer.step()  \n                running_loss += loss.item()\n            avg_loss = running_loss / len(train_loader)\n            print(f'Epoch {epoch+1}/{self.epochs}, Loss: {avg_loss:.4f}')\n        return self\n\n    def predict(self, X):\n        X_tensor = torch.tensor(X, dtype=torch.float32).to(self.device)\n        self.model.eval() \n        with torch.no_grad():\n            predictions = self.model(X_tensor)\n        return predictions.cpu().numpy().flatten()  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:55.05449Z","iopub.execute_input":"2024-12-06T12:58:55.054743Z","iopub.status.idle":"2024-12-06T12:58:55.066835Z","shell.execute_reply.started":"2024-12-06T12:58:55.054701Z","shell.execute_reply":"2024-12-06T12:58:55.066108Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **BaseLine Model**","metadata":{}},{"cell_type":"code","source":"\"\"\"from sklearn.pipeline import make_pipeline\nfrom sklearn.linear_model import LinearRegression\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import mean_squared_log_error,make_scorer\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import cross_val_score\nfrom catboost import CatBoostRegressor\nfrom lightgbm import LGBMRegressor\n\ndef rmsle(y_true, y_pred):\n    y_true = np.maximum(0, y_true) \n    y_pred = np.maximum(0, y_pred)\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))\n\nscorer = make_scorer(rmsle,greater_is_better=False)\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(device)\nmodels = [\n    ('DL',PyTorchRegressor(device=device,input_dim=27, epochs=20)),\n    (\"LinearRegression\",LinearRegression()),\n    (\"XGBoost\",XGBRegressor(random_state=42)),\n    ('CatBoost',CatBoostRegressor(verbose=0,random_state=42)),\n    ('LBGM',LGBMRegressor(verbose=0,random_state=42)),\n]\nfor name,model in models:\n    pipeline = make_pipeline(preprocessing,model)\n    scores = - cross_val_score(pipeline,unclean_X_train,y_train,cv=5,scoring=scorer)\n    print(f\"{name}: Mean accuracy = {scores.mean():.4f}, Standard deviation = {scores.std():.4f}\")\"\"\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:58:55.067724Z","iopub.execute_input":"2024-12-06T12:58:55.067945Z","iopub.status.idle":"2024-12-06T13:05:45.168114Z","shell.execute_reply.started":"2024-12-06T12:58:55.067923Z","shell.execute_reply":"2024-12-06T13:05:45.167021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"unclean_X = df.drop(columns='Premium Amount')\nunclean_y = df[['Premium Amount']]\nunclean_X_train,unclean_X_test,unclean_y_train,unclean_y_test = train_test_split(unclean_X,unclean_y,test_size=0.2,shuffle=True)\n\ntarget_imputer = SimpleImputer(strategy=\"median\").fit(unclean_y_train)\nX_train = preprocessing.fit_transform(unclean_X_train)\ny_train = target_imputer.transform(unclean_y_train).reshape(-1)\n\nX_test = preprocessing.transform(unclean_X_test)\ny_test = target_imputer.transform(unclean_y_test).reshape(-1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T13:05:45.169442Z","iopub.execute_input":"2024-12-06T13:05:45.169772Z","iopub.status.idle":"2024-12-06T13:05:52.692755Z","shell.execute_reply.started":"2024-12-06T13:05:45.169743Z","shell.execute_reply":"2024-12-06T13:05:52.691801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = PyTorchRegressor(device=device,input_dim=27, epochs=10)\nmodel.fit(X_train,y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T13:05:52.693851Z","iopub.execute_input":"2024-12-06T13:05:52.694132Z","iopub.status.idle":"2024-12-06T13:17:00.582029Z","shell.execute_reply.started":"2024-12-06T13:05:52.694106Z","shell.execute_reply":"2024-12-06T13:17:00.581147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = model.predict(X_test)\nvalid_rmsle = rmsle(y_test,predictions)\nvalid_rmsle","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T13:17:00.583374Z","iopub.execute_input":"2024-12-06T13:17:00.583915Z","iopub.status.idle":"2024-12-06T13:17:00.630897Z","shell.execute_reply.started":"2024-12-06T13:17:00.583875Z","shell.execute_reply":"2024-12-06T13:17:00.630194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#pred = model.predict(X_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T13:17:00.632013Z","iopub.execute_input":"2024-12-06T13:17:00.632383Z","iopub.status.idle":"2024-12-06T13:17:00.636334Z","shell.execute_reply.started":"2024-12-06T13:17:00.632346Z","shell.execute_reply":"2024-12-06T13:17:00.635472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T13:17:00.641183Z","iopub.execute_input":"2024-12-06T13:17:00.641677Z","iopub.status.idle":"2024-12-06T13:17:00.645445Z","shell.execute_reply.started":"2024-12-06T13:17:00.641641Z","shell.execute_reply":"2024-12-06T13:17:00.644618Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_df = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\npred_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T13:17:33.551348Z","iopub.execute_input":"2024-12-06T13:17:33.551694Z","iopub.status.idle":"2024-12-06T13:17:36.743835Z","shell.execute_reply.started":"2024-12-06T13:17:33.551664Z","shell.execute_reply":"2024-12-06T13:17:36.742889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_df[\"Previous Claims\"] = pd.cut(pred_df[\"Previous Claims\"], \n                                bins=[0., 1, 2, np.inf],\n                                labels=['No claims','1 claim','2 or more'],right=False)\n\npred_df['Year'] = ((pred_df['Policy Start Date'].str.split().str[0].str.split('-')).str[0]).astype(int)\npred_df['Month'] = ((pred_df['Policy Start Date'].str.split().str[0].str.split('-')).str[1]).astype(int)\n\npred_df.head() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T13:17:36.745124Z","iopub.execute_input":"2024-12-06T13:17:36.745403Z","iopub.status.idle":"2024-12-06T13:17:43.960221Z","shell.execute_reply.started":"2024-12-06T13:17:36.745378Z","shell.execute_reply":"2024-12-06T13:17:43.95939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_pred = preprocessing.transform(pred_df)\nsub_df = pd.DataFrame()\nsub_df['id'] = pred_df['id']\nsub_df['Premium Amount'] = model.predict(X_pred)\n\nsub_df.to_csv('DL.csv',index=None)\nprint(\"saved model\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T13:17:43.961526Z","iopub.execute_input":"2024-12-06T13:17:43.961875Z","iopub.status.idle":"2024-12-06T13:17:47.260519Z","shell.execute_reply.started":"2024-12-06T13:17:43.961838Z","shell.execute_reply":"2024-12-06T13:17:47.259623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}