{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:29.520855Z","iopub.execute_input":"2024-12-18T06:44:29.52185Z","iopub.status.idle":"2024-12-18T06:44:30.666478Z","shell.execute_reply.started":"2024-12-18T06:44:29.521809Z","shell.execute_reply":"2024-12-18T06:44:30.665236Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Importing libraries","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, MinMaxScaler, LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.impute import SimpleImputer\nimport lightgbm as lgb\nfrom sklearn.metrics import mean_squared_log_error\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:50:16.298216Z","iopub.execute_input":"2024-12-18T06:50:16.298687Z","iopub.status.idle":"2024-12-18T06:50:17.22495Z","shell.execute_reply.started":"2024-12-18T06:50:16.29865Z","shell.execute_reply":"2024-12-18T06:50:17.224084Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loading the data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:32.286374Z","iopub.execute_input":"2024-12-18T06:44:32.286833Z","iopub.status.idle":"2024-12-18T06:44:38.484741Z","shell.execute_reply.started":"2024-12-18T06:44:32.286799Z","shell.execute_reply":"2024-12-18T06:44:38.48355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:38.487138Z","iopub.execute_input":"2024-12-18T06:44:38.487473Z","iopub.status.idle":"2024-12-18T06:44:39.167368Z","shell.execute_reply.started":"2024-12-18T06:44:38.487441Z","shell.execute_reply":"2024-12-18T06:44:39.166388Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Handling Missing Values","metadata":{}},{"cell_type":"code","source":"numerical_cols = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', \n                  'Previous Claims', 'Credit Score', 'Vehicle Age', 'Insurance Duration']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:39.168837Z","iopub.execute_input":"2024-12-18T06:44:39.169203Z","iopub.status.idle":"2024-12-18T06:44:39.175214Z","shell.execute_reply.started":"2024-12-18T06:44:39.169169Z","shell.execute_reply":"2024-12-18T06:44:39.174092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_imputer = SimpleImputer(strategy='median')\ndf[numerical_cols] = numerical_imputer.fit_transform(df[numerical_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:39.176532Z","iopub.execute_input":"2024-12-18T06:44:39.177367Z","iopub.status.idle":"2024-12-18T06:44:40.889951Z","shell.execute_reply.started":"2024-12-18T06:44:39.177321Z","shell.execute_reply":"2024-12-18T06:44:40.889039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_cols = ['Gender', 'Marital Status', 'Occupation', 'Location', 'Policy Type', \n                    'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:40.891286Z","iopub.execute_input":"2024-12-18T06:44:40.891724Z","iopub.status.idle":"2024-12-18T06:44:40.897493Z","shell.execute_reply.started":"2024-12-18T06:44:40.891671Z","shell.execute_reply":"2024-12-18T06:44:40.896331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_imputer = SimpleImputer(strategy='most_frequent')\ndf[categorical_cols] = categorical_imputer.fit_transform(df[categorical_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:40.89853Z","iopub.execute_input":"2024-12-18T06:44:40.898846Z","iopub.status.idle":"2024-12-18T06:44:42.745002Z","shell.execute_reply.started":"2024-12-18T06:44:40.898816Z","shell.execute_reply":"2024-12-18T06:44:42.743947Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"code","source":"df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\ndf['Policy Start Year'] = df['Policy Start Date'].dt.year\ndf['Policy Start Month'] = df['Policy Start Date'].dt.month\ndf['Policy Start DayOfWeek'] = df['Policy Start Date'].dt.dayofweek\ndf['Policy Start Quarter'] = df['Policy Start Date'].dt.quarter","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:42.746421Z","iopub.execute_input":"2024-12-18T06:44:42.746756Z","iopub.status.idle":"2024-12-18T06:44:43.401608Z","shell.execute_reply.started":"2024-12-18T06:44:42.746723Z","shell.execute_reply":"2024-12-18T06:44:43.400392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:43.404811Z","iopub.execute_input":"2024-12-18T06:44:43.40519Z","iopub.status.idle":"2024-12-18T06:44:45.018152Z","shell.execute_reply.started":"2024-12-18T06:44:43.405156Z","shell.execute_reply":"2024-12-18T06:44:45.017063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Age Category'] = pd.cut(df['Age'], bins=[0, 18, 30, 45, 60, np.inf], labels=['0-18', '19-30', '31-45', '46-60', '61+'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:45.019398Z","iopub.execute_input":"2024-12-18T06:44:45.01968Z","iopub.status.idle":"2024-12-18T06:44:45.064685Z","shell.execute_reply.started":"2024-12-18T06:44:45.019652Z","shell.execute_reply":"2024-12-18T06:44:45.063667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:45.065899Z","iopub.execute_input":"2024-12-18T06:44:45.066207Z","iopub.status.idle":"2024-12-18T06:44:46.709509Z","shell.execute_reply.started":"2024-12-18T06:44:45.066178Z","shell.execute_reply":"2024-12-18T06:44:46.708337Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Encoding Categorical Features","metadata":{}},{"cell_type":"code","source":"label_columns = ['Education Level', 'Smoking Status', 'Exercise Frequency', 'Occupation']\nlabel_encoder = LabelEncoder()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:46.710986Z","iopub.execute_input":"2024-12-18T06:44:46.71132Z","iopub.status.idle":"2024-12-18T06:44:46.71618Z","shell.execute_reply.started":"2024-12-18T06:44:46.71129Z","shell.execute_reply":"2024-12-18T06:44:46.71499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in label_columns:\n    df[col] = label_encoder.fit_transform(df[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:46.717475Z","iopub.execute_input":"2024-12-18T06:44:46.717891Z","iopub.status.idle":"2024-12-18T06:44:47.718692Z","shell.execute_reply.started":"2024-12-18T06:44:46.717859Z","shell.execute_reply":"2024-12-18T06:44:47.717535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.get_dummies(df, columns=['Gender', 'Marital Status', 'Location', 'Policy Type', \n                                 'Customer Feedback', 'Property Type', 'Age Category'], drop_first=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:47.719955Z","iopub.execute_input":"2024-12-18T06:44:47.72038Z","iopub.status.idle":"2024-12-18T06:44:48.853674Z","shell.execute_reply.started":"2024-12-18T06:44:47.720346Z","shell.execute_reply":"2024-12-18T06:44:48.852513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:48.855475Z","iopub.execute_input":"2024-12-18T06:44:48.855819Z","iopub.status.idle":"2024-12-18T06:44:48.914968Z","shell.execute_reply.started":"2024-12-18T06:44:48.855786Z","shell.execute_reply":"2024-12-18T06:44:48.913914Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Scaling","metadata":{}},{"cell_type":"code","source":"numerical_cols = ['Annual Income', 'Credit Score', 'Health Score', 'Vehicle Age', 'Insurance Duration']\nscaler = StandardScaler()\n\ndf[numerical_cols] = scaler.fit_transform(df[numerical_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:48.916399Z","iopub.execute_input":"2024-12-18T06:44:48.916756Z","iopub.status.idle":"2024-12-18T06:44:49.01304Z","shell.execute_reply.started":"2024-12-18T06:44:48.916686Z","shell.execute_reply":"2024-12-18T06:44:49.011856Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Outlier Handling","metadata":{}},{"cell_type":"code","source":"for col in numerical_cols:\n    lower_percentile = df[col].quantile(0.05)\n    upper_percentile = df[col].quantile(0.95)\n    df[col] = np.clip(df[col], lower_percentile, upper_percentile)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:49.014419Z","iopub.execute_input":"2024-12-18T06:44:49.01476Z","iopub.status.idle":"2024-12-18T06:44:49.28213Z","shell.execute_reply.started":"2024-12-18T06:44:49.014729Z","shell.execute_reply":"2024-12-18T06:44:49.280872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.drop(columns=['id', 'Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:49.283513Z","iopub.execute_input":"2024-12-18T06:44:49.283945Z","iopub.status.idle":"2024-12-18T06:44:49.358372Z","shell.execute_reply.started":"2024-12-18T06:44:49.283898Z","shell.execute_reply":"2024-12-18T06:44:49.357225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corr_matrix = df.corr()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:49.359872Z","iopub.execute_input":"2024-12-18T06:44:49.360322Z","iopub.status.idle":"2024-12-18T06:44:53.342593Z","shell.execute_reply.started":"2024-12-18T06:44:49.360268Z","shell.execute_reply":"2024-12-18T06:44:53.341361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_col = 'Premium Amount'\nhigh_corr_features = [col for col in corr_matrix.columns if abs(corr_matrix[target_col][col]) > 0.9 and col != target_col]\ndf = df.drop(columns=high_corr_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:53.344155Z","iopub.execute_input":"2024-12-18T06:44:53.344558Z","iopub.status.idle":"2024-12-18T06:44:53.44078Z","shell.execute_reply.started":"2024-12-18T06:44:53.344516Z","shell.execute_reply":"2024-12-18T06:44:53.439481Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"high_corr_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:44:59.817077Z","iopub.execute_input":"2024-12-18T06:44:59.817442Z","iopub.status.idle":"2024-12-18T06:44:59.823886Z","shell.execute_reply.started":"2024-12-18T06:44:59.817412Z","shell.execute_reply":"2024-12-18T06:44:59.822755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:45:03.107078Z","iopub.execute_input":"2024-12-18T06:45:03.10786Z","iopub.status.idle":"2024-12-18T06:45:03.380919Z","shell.execute_reply.started":"2024-12-18T06:45:03.107825Z","shell.execute_reply":"2024-12-18T06:45:03.379808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:45:53.152609Z","iopub.execute_input":"2024-12-18T06:45:53.153055Z","iopub.status.idle":"2024-12-18T06:45:53.215774Z","shell.execute_reply.started":"2024-12-18T06:45:53.152989Z","shell.execute_reply":"2024-12-18T06:45:53.21446Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Splitting the dataset","metadata":{}},{"cell_type":"code","source":"X = df.drop(columns=[\"Premium Amount\"])\ny = df[\"Premium Amount\"] ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:51:00.285112Z","iopub.execute_input":"2024-12-18T06:51:00.285538Z","iopub.status.idle":"2024-12-18T06:51:00.354672Z","shell.execute_reply.started":"2024-12-18T06:51:00.285493Z","shell.execute_reply":"2024-12-18T06:51:00.35334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:51:15.409118Z","iopub.execute_input":"2024-12-18T06:51:15.409625Z","iopub.status.idle":"2024-12-18T06:51:15.818902Z","shell.execute_reply.started":"2024-12-18T06:51:15.409581Z","shell.execute_reply":"2024-12-18T06:51:15.817327Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LightGBM Dataset","metadata":{}},{"cell_type":"code","source":"train_data_lgb = lgb.Dataset(X_train, label=y_train)\ntest_data_lgb = lgb.Dataset(X_test, label=y_test, reference=train_data_lgb)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:52:20.14994Z","iopub.execute_input":"2024-12-18T06:52:20.150356Z","iopub.status.idle":"2024-12-18T06:52:20.155398Z","shell.execute_reply.started":"2024-12-18T06:52:20.150323Z","shell.execute_reply":"2024-12-18T06:52:20.154297Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Hyperparameters","metadata":{}},{"cell_type":"code","source":"params = {\n    'objective': 'regression',  # Regression task\n    'metric': 'rmse',           # RMSE will be used internally during training\n    'boosting_type': 'gbdt',    # Gradient Boosting Decision Trees\n    'num_leaves': 31,           # Number of leaves in the tree\n    'learning_rate': 0.05,      # Learning rate\n    'feature_fraction': 0.9,    # Fraction of features to consider for each iteration\n    'bagging_fraction': 0.8,    # Fraction of samples to consider for each iteration\n    'bagging_freq': 5,          # Frequency for bagging\n    'verbose': -1               # Suppress LightGBM output\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:53:03.868878Z","iopub.execute_input":"2024-12-18T06:53:03.869338Z","iopub.status.idle":"2024-12-18T06:53:03.875146Z","shell.execute_reply.started":"2024-12-18T06:53:03.869301Z","shell.execute_reply":"2024-12-18T06:53:03.8739Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training the Model","metadata":{}},{"cell_type":"code","source":"num_round = 1000\n\nbst = lgb.train(\n    params,\n    train_data_lgb,\n    num_round,\n    valid_sets=[test_data_lgb]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:57:54.751975Z","iopub.execute_input":"2024-12-18T06:57:54.752955Z","iopub.status.idle":"2024-12-18T06:58:49.501412Z","shell.execute_reply.started":"2024-12-18T06:57:54.752913Z","shell.execute_reply":"2024-12-18T06:58:49.500472Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Make predictions","metadata":{}},{"cell_type":"code","source":"y_pred = bst.predict(X_test, num_iteration=bst.best_iteration)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:59:06.200037Z","iopub.execute_input":"2024-12-18T06:59:06.200854Z","iopub.status.idle":"2024-12-18T06:59:11.165586Z","shell.execute_reply.started":"2024-12-18T06:59:06.200816Z","shell.execute_reply":"2024-12-18T06:59:11.164477Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Calculate RMSLE (Root Mean Squared Logarithmic Error)","metadata":{}},{"cell_type":"code","source":"y_test_log = np.log1p(y_test)\ny_pred_log = np.log1p(y_pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:59:11.32045Z","iopub.execute_input":"2024-12-18T06:59:11.32083Z","iopub.status.idle":"2024-12-18T06:59:11.334513Z","shell.execute_reply.started":"2024-12-18T06:59:11.320798Z","shell.execute_reply":"2024-12-18T06:59:11.333346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"msle = mean_squared_log_error(y_test_log, y_pred_log)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:59:11.756163Z","iopub.execute_input":"2024-12-18T06:59:11.756586Z","iopub.status.idle":"2024-12-18T06:59:11.77508Z","shell.execute_reply.started":"2024-12-18T06:59:11.756548Z","shell.execute_reply":"2024-12-18T06:59:11.774089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rmsle = np.sqrt(msle)\nprint(f\"Root Mean Squared Logarithmic Error (RMSLE): {rmsle:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:59:12.335732Z","iopub.execute_input":"2024-12-18T06:59:12.336288Z","iopub.status.idle":"2024-12-18T06:59:12.34197Z","shell.execute_reply.started":"2024-12-18T06:59:12.336238Z","shell.execute_reply":"2024-12-18T06:59:12.340785Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Test data","metadata":{}},{"cell_type":"code","source":"tdf = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T07:15:24.472965Z","iopub.execute_input":"2024-12-18T07:15:24.473461Z","iopub.status.idle":"2024-12-18T07:15:27.283209Z","shell.execute_reply.started":"2024-12-18T07:15:24.473423Z","shell.execute_reply":"2024-12-18T07:15:27.282002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T07:15:27.285245Z","iopub.execute_input":"2024-12-18T07:15:27.285579Z","iopub.status.idle":"2024-12-18T07:15:30.126569Z","shell.execute_reply.started":"2024-12-18T07:15:27.285548Z","shell.execute_reply":"2024-12-18T07:15:30.125702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_cols = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', \n                  'Previous Claims', 'Credit Score', 'Vehicle Age', 'Insurance Duration']\n\nnumerical_imputer = SimpleImputer(strategy='median')\ntest_df[numerical_cols] = numerical_imputer.fit_transform(test_df[numerical_cols])\n\ncategorical_cols = ['Gender', 'Marital Status', 'Occupation', 'Location', 'Policy Type', \n                    'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type']\n\ncategorical_imputer = SimpleImputer(strategy='most_frequent')\ntest_df[categorical_cols] = categorical_imputer.fit_transform(test_df[categorical_cols])\n\ntest_df['Policy Start Date'] = pd.to_datetime(test_df['Policy Start Date'])\ntest_df['Policy Start Year'] = test_df['Policy Start Date'].dt.year\ntest_df['Policy Start Month'] = test_df['Policy Start Date'].dt.month\ntest_df['Policy Start DayOfWeek'] = test_df['Policy Start Date'].dt.dayofweek\ntest_df['Policy Start Quarter'] = test_df['Policy Start Date'].dt.quarter\n\ntest_df['Age Category'] = pd.cut(test_df['Age'], bins=[0, 18, 30, 45, 60, np.inf], labels=['0-18', '19-30', '31-45', '46-60', '61+'])\n\nlabel_columns = ['Education Level', 'Smoking Status', 'Exercise Frequency', 'Occupation']\nlabel_encoder = LabelEncoder()\n\nfor col in label_columns:\n    test_df[col] = label_encoder.fit_transform(test_df[col])\n\ntest_df = pd.get_dummies(test_df, columns=['Gender', 'Marital Status', 'Location', 'Policy Type', \n                                 'Customer Feedback', 'Property Type', 'Age Category'], drop_first=True)\n\nnumerical_cols = ['Annual Income', 'Credit Score', 'Health Score', 'Vehicle Age', 'Insurance Duration']\nscaler = StandardScaler()\n\ntest_df[numerical_cols] = scaler.fit_transform(test_df[numerical_cols])\n\n\nfor col in numerical_cols:\n    lower_percentile = test_df[col].quantile(0.05)\n    upper_percentile = test_df[col].quantile(0.95)\n    test_df[col] = np.clip(test_df[col], lower_percentile, upper_percentile)\n\ntest_df = test_df.drop(columns=['id', 'Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T07:15:30.127863Z","iopub.execute_input":"2024-12-18T07:15:30.128202Z","iopub.status.idle":"2024-12-18T07:15:34.352127Z","shell.execute_reply.started":"2024-12-18T07:15:30.12817Z","shell.execute_reply":"2024-12-18T07:15:34.351227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T07:15:53.607625Z","iopub.execute_input":"2024-12-18T07:15:53.608565Z","iopub.status.idle":"2024-12-18T07:15:53.66412Z","shell.execute_reply.started":"2024-12-18T07:15:53.608511Z","shell.execute_reply":"2024-12-18T07:15:53.662875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = bst.predict(test_df, num_iteration=bst.best_iteration)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T07:16:17.80882Z","iopub.execute_input":"2024-12-18T07:16:17.809566Z","iopub.status.idle":"2024-12-18T07:16:33.939494Z","shell.execute_reply.started":"2024-12-18T07:16:17.809527Z","shell.execute_reply":"2024-12-18T07:16:33.938276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T07:16:37.421192Z","iopub.execute_input":"2024-12-18T07:16:37.421615Z","iopub.status.idle":"2024-12-18T07:16:37.429325Z","shell.execute_reply.started":"2024-12-18T07:16:37.421578Z","shell.execute_reply":"2024-12-18T07:16:37.42829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.log1p(y_pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T07:17:17.493123Z","iopub.execute_input":"2024-12-18T07:17:17.494222Z","iopub.status.idle":"2024-12-18T07:17:17.514284Z","shell.execute_reply.started":"2024-12-18T07:17:17.494179Z","shell.execute_reply":"2024-12-18T07:17:17.513145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame({\n    'id': tdf['id'],\n    'Premium Amount': np.log1p(y_pred)\n})\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)\nprint(\"Prediction file has been created\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T07:20:04.135854Z","iopub.execute_input":"2024-12-18T07:20:04.136283Z","iopub.status.idle":"2024-12-18T07:20:05.775765Z","shell.execute_reply.started":"2024-12-18T07:20:04.136245Z","shell.execute_reply":"2024-12-18T07:20:05.77457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}