{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-07T14:00:16.566898Z","iopub.execute_input":"2024-12-07T14:00:16.56745Z","iopub.status.idle":"2024-12-07T14:00:17.766583Z","shell.execute_reply.started":"2024-12-07T14:00:16.567398Z","shell.execute_reply":"2024-12-07T14:00:17.765411Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Importing Libraries","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import mean_squared_log_error\nfrom sklearn.model_selection import RandomizedSearchCV\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:25:40.398824Z","iopub.execute_input":"2024-12-06T16:25:40.399224Z","iopub.status.idle":"2024-12-06T16:25:40.405245Z","shell.execute_reply.started":"2024-12-06T16:25:40.399191Z","shell.execute_reply":"2024-12-06T16:25:40.404072Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loading the Data","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:25:13.757922Z","iopub.execute_input":"2024-12-06T16:25:13.758294Z","iopub.status.idle":"2024-12-06T16:25:18.591503Z","shell.execute_reply.started":"2024-12-06T16:25:13.758261Z","shell.execute_reply":"2024-12-06T16:25:18.590417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:25:18.59345Z","iopub.execute_input":"2024-12-06T16:25:18.593807Z","iopub.status.idle":"2024-12-06T16:25:19.241041Z","shell.execute_reply.started":"2024-12-06T16:25:18.593772Z","shell.execute_reply":"2024-12-06T16:25:19.239914Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Features (X) and target (y)","metadata":{}},{"cell_type":"code","source":"X = train_df.drop(columns=['id', 'Premium Amount', 'Policy Start Date'])\ny = train_df['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:25:19.242298Z","iopub.execute_input":"2024-12-06T16:25:19.242601Z","iopub.status.idle":"2024-12-06T16:25:19.42443Z","shell.execute_reply.started":"2024-12-06T16:25:19.242568Z","shell.execute_reply":"2024-12-06T16:25:19.423288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"continuous_features = [\n    'Age', 'Annual Income', 'Number of Dependents', 'Health Score', 'Previous Claims',\n    'Vehicle Age', 'Credit Score', 'Insurance Duration'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:25:19.426744Z","iopub.execute_input":"2024-12-06T16:25:19.427099Z","iopub.status.idle":"2024-12-06T16:25:19.4322Z","shell.execute_reply.started":"2024-12-06T16:25:19.427067Z","shell.execute_reply":"2024-12-06T16:25:19.430979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_features = [\n    'Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location',\n    'Policy Type', 'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:25:19.433817Z","iopub.execute_input":"2024-12-06T16:25:19.434285Z","iopub.status.idle":"2024-12-06T16:25:19.445258Z","shell.execute_reply.started":"2024-12-06T16:25:19.434239Z","shell.execute_reply":"2024-12-06T16:25:19.443932Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Imputers for handling missing values","metadata":{}},{"cell_type":"code","source":"imputer_continuous = SimpleImputer(strategy='median')\nimputer_categorical = SimpleImputer(strategy='most_frequent')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:25:19.446611Z","iopub.execute_input":"2024-12-06T16:25:19.447013Z","iopub.status.idle":"2024-12-06T16:25:19.461Z","shell.execute_reply.started":"2024-12-06T16:25:19.446978Z","shell.execute_reply":"2024-12-06T16:25:19.4599Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## One-hot encoder for categorical features","metadata":{}},{"cell_type":"code","source":"one_hot_encoder = OneHotEncoder(handle_unknown='ignore', drop='first')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:25:19.462563Z","iopub.execute_input":"2024-12-06T16:25:19.463388Z","iopub.status.idle":"2024-12-06T16:25:19.475906Z","shell.execute_reply.started":"2024-12-06T16:25:19.463339Z","shell.execute_reply":"2024-12-06T16:25:19.474481Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Preprocessing pipeline ","metadata":{}},{"cell_type":"code","source":"preprocessor = ColumnTransformer(\n    transformers=[\n        ('num', imputer_continuous, continuous_features),\n        ('cat', Pipeline(steps=[('imputer', imputer_categorical), ('onehot', one_hot_encoder)]), categorical_features)\n    ]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:25:19.47737Z","iopub.execute_input":"2024-12-06T16:25:19.47785Z","iopub.status.idle":"2024-12-06T16:25:19.489534Z","shell.execute_reply.started":"2024-12-06T16:25:19.477768Z","shell.execute_reply":"2024-12-06T16:25:19.488411Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train-Test Split","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:25:19.491097Z","iopub.execute_input":"2024-12-06T16:25:19.491478Z","iopub.status.idle":"2024-12-06T16:25:20.352207Z","shell.execute_reply.started":"2024-12-06T16:25:19.491442Z","shell.execute_reply":"2024-12-06T16:25:20.351135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'Training set size: {X_train.shape[0]} rows')\nprint(f'Test set size: {X_test.shape[0]} rows')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:25:20.355135Z","iopub.execute_input":"2024-12-06T16:25:20.355469Z","iopub.status.idle":"2024-12-06T16:25:20.361321Z","shell.execute_reply.started":"2024-12-06T16:25:20.355436Z","shell.execute_reply":"2024-12-06T16:25:20.360199Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Building the XGBRegressor Model","metadata":{}},{"cell_type":"code","source":"model = XGBRegressor(objective='reg:squarederror', n_estimators=1000, learning_rate=0.01)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:25:20.362873Z","iopub.execute_input":"2024-12-06T16:25:20.363215Z","iopub.status.idle":"2024-12-06T16:25:20.373079Z","shell.execute_reply.started":"2024-12-06T16:25:20.363183Z","shell.execute_reply":"2024-12-06T16:25:20.371998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pipeline = Pipeline(steps=[\n    ('preprocessor', preprocessor),\n    ('regressor', model)\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:25:20.374363Z","iopub.execute_input":"2024-12-06T16:25:20.3748Z","iopub.status.idle":"2024-12-06T16:25:20.384576Z","shell.execute_reply.started":"2024-12-06T16:25:20.374754Z","shell.execute_reply":"2024-12-06T16:25:20.383456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_dist = {\n    'regressor__n_estimators': [100, 500, 1000],  # Number of trees\n    'regressor__learning_rate': [0.01, 0.05, 0.1, 0.2],  # Learning rate\n    'regressor__max_depth': [3, 5, 7, 9],  # Maximum depth of trees\n    'regressor__subsample': [0.6, 0.8, 1.0],  # Subsampling ratio\n    'regressor__colsample_bytree': [0.6, 0.8, 1.0],  # Feature sampling ratio\n    'regressor__gamma': [0, 0.1, 0.2, 0.5]  # Minimum loss reduction for further partition\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:25:20.386008Z","iopub.execute_input":"2024-12-06T16:25:20.386328Z","iopub.status.idle":"2024-12-06T16:25:20.398315Z","shell.execute_reply.started":"2024-12-06T16:25:20.386296Z","shell.execute_reply":"2024-12-06T16:25:20.397131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_search = RandomizedSearchCV(\n    estimator=pipeline, \n    param_distributions=param_dist,\n    n_iter=10,\n    scoring='neg_root_mean_squared_error',\n    cv=3,\n    verbose=2,\n    random_state=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:25:44.911772Z","iopub.execute_input":"2024-12-06T16:25:44.912163Z","iopub.status.idle":"2024-12-06T16:25:44.917489Z","shell.execute_reply.started":"2024-12-06T16:25:44.91213Z","shell.execute_reply":"2024-12-06T16:25:44.916212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_search.fit(X_train, np.log1p(y_train))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:25:49.095842Z","iopub.execute_input":"2024-12-06T16:25:49.096271Z","iopub.status.idle":"2024-12-06T16:33:18.274508Z","shell.execute_reply.started":"2024-12-06T16:25:49.096235Z","shell.execute_reply":"2024-12-06T16:33:18.272917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Best parameters found:\", random_search.best_params_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:45:48.550865Z","iopub.execute_input":"2024-12-06T16:45:48.551992Z","iopub.status.idle":"2024-12-06T16:45:48.558298Z","shell.execute_reply.started":"2024-12-06T16:45:48.551949Z","shell.execute_reply":"2024-12-06T16:45:48.556815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_model = random_search.best_estimator_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:45:52.331005Z","iopub.execute_input":"2024-12-06T16:45:52.331376Z","iopub.status.idle":"2024-12-06T16:45:52.336346Z","shell.execute_reply.started":"2024-12-06T16:45:52.331338Z","shell.execute_reply":"2024-12-06T16:45:52.335108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = np.expm1(best_model.predict(X_test))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:45:55.89634Z","iopub.execute_input":"2024-12-06T16:45:55.896742Z","iopub.status.idle":"2024-12-06T16:45:57.130008Z","shell.execute_reply.started":"2024-12-06T16:45:55.896709Z","shell.execute_reply":"2024-12-06T16:45:57.128771Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluating the Model using RMSLE","metadata":{}},{"cell_type":"code","source":"rmsle = np.sqrt(mean_squared_log_error(y_test, y_pred))\nprint(f'Root Mean Squared Logarithmic Error (RMSLE): {rmsle}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:45:59.742703Z","iopub.execute_input":"2024-12-06T16:45:59.743091Z","iopub.status.idle":"2024-12-06T16:45:59.760908Z","shell.execute_reply.started":"2024-12-06T16:45:59.743057Z","shell.execute_reply":"2024-12-06T16:45:59.759846Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Visualizing Predictions","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.scatterplot(x=y_test, y=y_pred)\nplt.xlabel('True Premium Amount')\nplt.ylabel('Predicted Premium Amount')\nplt.title('True vs Predicted Premium Amounts')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:46:19.65316Z","iopub.execute_input":"2024-12-06T16:46:19.653567Z","iopub.status.idle":"2024-12-06T16:46:20.670468Z","shell.execute_reply.started":"2024-12-06T16:46:19.653524Z","shell.execute_reply":"2024-12-06T16:46:20.669255Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Test Data","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:46:25.041398Z","iopub.execute_input":"2024-12-06T16:46:25.041764Z","iopub.status.idle":"2024-12-06T16:46:27.961177Z","shell.execute_reply.started":"2024-12-06T16:46:25.041727Z","shell.execute_reply":"2024-12-06T16:46:27.960143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_features = test_df.drop(columns=['id', 'Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:46:38.974057Z","iopub.execute_input":"2024-12-06T16:46:38.974428Z","iopub.status.idle":"2024-12-06T16:46:39.091555Z","shell.execute_reply.started":"2024-12-06T16:46:38.974388Z","shell.execute_reply":"2024-12-06T16:46:39.090529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_predictions = np.expm1(best_model.predict(test_features))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:46:39.555518Z","iopub.execute_input":"2024-12-06T16:46:39.555927Z","iopub.status.idle":"2024-12-06T16:46:43.776637Z","shell.execute_reply.started":"2024-12-06T16:46:39.555864Z","shell.execute_reply":"2024-12-06T16:46:43.775789Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame({\n    'id': test_df['id'],\n    'Premium Amount': test_predictions\n})\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)\nprint(\"Prediction file has been created\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:46:43.777864Z","iopub.execute_input":"2024-12-06T16:46:43.778201Z","iopub.status.idle":"2024-12-06T16:46:45.046541Z","shell.execute_reply.started":"2024-12-06T16:46:43.778167Z","shell.execute_reply":"2024-12-06T16:46:45.045242Z"}},"outputs":[],"execution_count":null}]}