{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:39:45.320094Z","iopub.execute_input":"2024-12-01T08:39:45.320522Z","iopub.status.idle":"2024-12-01T08:39:45.724762Z","shell.execute_reply.started":"2024-12-01T08:39:45.320475Z","shell.execute_reply":"2024-12-01T08:39:45.72359Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Import libraries","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom catboost import CatBoostRegressor\nfrom sklearn.metrics import mean_squared_log_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:06:51.795057Z","iopub.execute_input":"2024-12-01T09:06:51.795357Z","iopub.status.idle":"2024-12-01T09:06:51.800402Z","shell.execute_reply.started":"2024-12-01T09:06:51.795326Z","shell.execute_reply":"2024-12-01T09:06:51.799381Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load data","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest_df = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:39:46.43114Z","iopub.execute_input":"2024-12-01T08:39:46.431566Z","iopub.status.idle":"2024-12-01T08:39:53.610748Z","shell.execute_reply.started":"2024-12-01T08:39:46.431533Z","shell.execute_reply":"2024-12-01T08:39:53.609761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = train_df.drop(columns = ['Policy Start Date', 'id'], axis = 1)\ntest_df = test_df.drop(columns = ['Policy Start Date'], axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:39:53.611907Z","iopub.execute_input":"2024-12-01T08:39:53.612199Z","iopub.status.idle":"2024-12-01T08:39:53.964054Z","shell.execute_reply.started":"2024-12-01T08:39:53.612169Z","shell.execute_reply":"2024-12-01T08:39:53.962848Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train Data Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Handling numerical columns","metadata":{}},{"cell_type":"code","source":"numeric_cols = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', \n                'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration']\n\nnumeric_imputer = SimpleImputer(strategy='median')\ntrain_df[numeric_cols] = numeric_imputer.fit_transform(train_df[numeric_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:39:53.966341Z","iopub.execute_input":"2024-12-01T08:39:53.966645Z","iopub.status.idle":"2024-12-01T08:39:55.705047Z","shell.execute_reply.started":"2024-12-01T08:39:53.966615Z","shell.execute_reply":"2024-12-01T08:39:55.703935Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Handling categorical columns","metadata":{}},{"cell_type":"code","source":"categorical_cols = ['Gender', 'Marital Status', 'Education Level', 'Occupation', \n                    'Location', 'Policy Type', 'Customer Feedback', \n                    'Smoking Status', 'Exercise Frequency', 'Property Type']\n\ncategorical_imputer = SimpleImputer(strategy='most_frequent')\ntrain_df[categorical_cols] = categorical_imputer.fit_transform(train_df[categorical_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:39:55.706224Z","iopub.execute_input":"2024-12-01T08:39:55.706548Z","iopub.status.idle":"2024-12-01T08:39:57.836477Z","shell.execute_reply.started":"2024-12-01T08:39:55.706515Z","shell.execute_reply":"2024-12-01T08:39:57.835426Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Encoding Categorical Features for CatBoost","metadata":{}},{"cell_type":"code","source":"categorical_features = ['Gender', 'Marital Status', 'Education Level', 'Occupation', \n                        'Location', 'Policy Type', 'Customer Feedback', \n                        'Smoking Status', 'Exercise Frequency', 'Property Type']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:39:57.837893Z","iopub.execute_input":"2024-12-01T08:39:57.838331Z","iopub.status.idle":"2024-12-01T08:39:57.843547Z","shell.execute_reply.started":"2024-12-01T08:39:57.838276Z","shell.execute_reply":"2024-12-01T08:39:57.842551Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Split into features (X) and target (y)","metadata":{}},{"cell_type":"code","source":"X = train_df.drop(columns=['Premium Amount'])\ny = train_df['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:39:57.844968Z","iopub.execute_input":"2024-12-01T08:39:57.845376Z","iopub.status.idle":"2024-12-01T08:39:58.053836Z","shell.execute_reply.started":"2024-12-01T08:39:57.845329Z","shell.execute_reply":"2024-12-01T08:39:58.052948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:39:58.055075Z","iopub.execute_input":"2024-12-01T08:39:58.055391Z","iopub.status.idle":"2024-12-01T08:39:58.953376Z","shell.execute_reply.started":"2024-12-01T08:39:58.055361Z","shell.execute_reply":"2024-12-01T08:39:58.952273Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train the CatBoost Model","metadata":{}},{"cell_type":"code","source":"model = CatBoostRegressor(iterations=1000, \n                          depth=10,       \n                          learning_rate=0.05, \n                          loss_function='RMSE',\n                          cat_features=categorical_features,\n                          random_seed=42,\n                          verbose=100)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:39:58.954608Z","iopub.execute_input":"2024-12-01T08:39:58.954966Z","iopub.status.idle":"2024-12-01T08:39:58.960542Z","shell.execute_reply.started":"2024-12-01T08:39:58.954931Z","shell.execute_reply":"2024-12-01T08:39:58.959502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.fit(X_train, y_train, cat_features=categorical_features, eval_set=(X_valid, y_valid), plot=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:39:58.961934Z","iopub.execute_input":"2024-12-01T08:39:58.962336Z","iopub.status.idle":"2024-12-01T09:06:50.903371Z","shell.execute_reply.started":"2024-12-01T08:39:58.96229Z","shell.execute_reply":"2024-12-01T09:06:50.902205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(X_valid)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:06:50.904964Z","iopub.execute_input":"2024-12-01T09:06:50.905419Z","iopub.status.idle":"2024-12-01T09:06:51.786841Z","shell.execute_reply.started":"2024-12-01T09:06:50.905364Z","shell.execute_reply":"2024-12-01T09:06:51.785947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rmsle = np.sqrt(mean_squared_log_error(y_valid, y_pred))\nprint(f'RMSLE: {rmsle}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:06:51.80167Z","iopub.execute_input":"2024-12-01T09:06:51.80216Z","iopub.status.idle":"2024-12-01T09:06:51.82708Z","shell.execute_reply.started":"2024-12-01T09:06:51.802099Z","shell.execute_reply":"2024-12-01T09:06:51.825853Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Test Data Preprocessing","metadata":{}},{"cell_type":"code","source":"test_data = test_df.drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:06:51.828654Z","iopub.execute_input":"2024-12-01T09:06:51.829007Z","iopub.status.idle":"2024-12-01T09:06:51.952293Z","shell.execute_reply.started":"2024-12-01T09:06:51.828972Z","shell.execute_reply":"2024-12-01T09:06:51.951103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_cols = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', \n                'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration']\n\nnumeric_imputer = SimpleImputer(strategy='median')\n\ntest_data[numeric_cols] = numeric_imputer.fit_transform(test_data[numeric_cols])\n\ncategorical_cols = ['Gender', 'Marital Status', 'Education Level', 'Occupation', \n                    'Location', 'Policy Type', 'Customer Feedback', \n                    'Smoking Status', 'Exercise Frequency', 'Property Type']\n\ncategorical_imputer = SimpleImputer(strategy='most_frequent')\ntest_data[categorical_cols] = categorical_imputer.fit_transform(test_data[categorical_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:06:51.953646Z","iopub.execute_input":"2024-12-01T09:06:51.954016Z","iopub.status.idle":"2024-12-01T09:06:54.472831Z","shell.execute_reply.started":"2024-12-01T09:06:51.953982Z","shell.execute_reply":"2024-12-01T09:06:54.471836Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Test Predictions","metadata":{}},{"cell_type":"code","source":"test_pred = model.predict(test_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:06:54.473948Z","iopub.execute_input":"2024-12-01T09:06:54.474235Z","iopub.status.idle":"2024-12-01T09:06:57.609651Z","shell.execute_reply.started":"2024-12-01T09:06:54.474205Z","shell.execute_reply":"2024-12-01T09:06:57.608624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:06:57.611015Z","iopub.execute_input":"2024-12-01T09:06:57.611427Z","iopub.status.idle":"2024-12-01T09:06:57.61958Z","shell.execute_reply.started":"2024-12-01T09:06:57.611372Z","shell.execute_reply":"2024-12-01T09:06:57.618479Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame({\n    'id': test_df['id'],\n    'Premium Amount': test_pred\n})\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)\n\nprint(\"Prediction file has been created\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:06:57.948296Z","iopub.status.idle":"2024-12-01T09:06:57.948711Z","shell.execute_reply.started":"2024-12-01T09:06:57.948526Z","shell.execute_reply":"2024-12-01T09:06:57.948547Z"}},"outputs":[],"execution_count":null}]}