{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Notebook Version\n\n- Version 1 (11/30/2024)\n   * EDA\n   * Baseline modeling 1.0\n \n\n- Version 2 (12/02/2024)\n   * Baseline modeling 1.0 updated\n \n\n- Version 3 (12/04/2024)\n   * Baseline modeling 1.0 updated\n \n\n- Version 4 (12/04/2024)\n   * Fixing bug\n\n\n- Version 5 (12/04/2024)\n   * Fixing bug\n \n\n- Version 6 (12/04/2024)\n   * Fixing bug\n \n\n- Version 7 (12/06/2024)\n   * Baseline modeling 2.0 added\n\n# Loading Libraries","metadata":{}},{"cell_type":"code","source":"%%time\nimport pandas as pd; pd.set_option('display.max_columns', 100)\nimport numpy as np\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport gc\n\nimport matplotlib.pyplot as plt; plt.style.use('ggplot')\nimport seaborn as sns\n\nfrom sklearn.preprocessing import MinMaxScaler, StandardScaler, LabelEncoder\nfrom sklearn.pipeline import make_pipeline, Pipeline\nfrom sklearn.linear_model import Ridge, RidgeCV, Lasso, LassoCV\nfrom sklearn.model_selection import KFold, StratifiedKFold, train_test_split, GridSearchCV, RepeatedKFold, RepeatedStratifiedKFold\nfrom sklearn.inspection import PartialDependenceDisplay\nfrom sklearn.ensemble import RandomForestRegressor, HistGradientBoostingRegressor, GradientBoostingRegressor, ExtraTreesRegressor\nfrom sklearn.svm import SVR\n\nfrom ydf import RandomForestLearner, GradientBoostedTreesLearner\nimport ydf\n\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor, DMatrix\nimport xgboost as xgb\nfrom catboost import CatBoostRegressor, Pool","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T17:30:25.346715Z","iopub.execute_input":"2024-12-04T17:30:25.347069Z","iopub.status.idle":"2024-12-04T17:30:30.179729Z","shell.execute_reply.started":"2024-12-04T17:30:25.347039Z","shell.execute_reply":"2024-12-04T17:30:30.178913Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Reading Data Files","metadata":{}},{"cell_type":"code","source":"%%time\ntrain = pd.read_csv('../input/playground-series-s4e12/train.csv', index_col=0)\ntest = pd.read_csv('../input/playground-series-s4e12/test.csv', index_col=0)\n\nprint('The dimension of the train dataset is:', train.shape)\nprint('The dimension of the test dataset is:', test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:16:43.160696Z","iopub.execute_input":"2024-12-04T16:16:43.161733Z","iopub.status.idle":"2024-12-04T16:16:53.095973Z","shell.execute_reply.started":"2024-12-04T16:16:43.161692Z","shell.execute_reply":"2024-12-04T16:16:53.094936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T03:28:53.632242Z","iopub.execute_input":"2024-12-01T03:28:53.632493Z","iopub.status.idle":"2024-12-01T03:28:53.656907Z","shell.execute_reply.started":"2024-12-01T03:28:53.632469Z","shell.execute_reply":"2024-12-01T03:28:53.65611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T03:28:53.659133Z","iopub.execute_input":"2024-12-01T03:28:53.659701Z","iopub.status.idle":"2024-12-01T03:28:53.677486Z","shell.execute_reply.started":"2024-12-01T03:28:53.659672Z","shell.execute_reply":"2024-12-01T03:28:53.676699Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"First, let's check for missing values in the `train` and `test` data frames.","metadata":{}},{"cell_type":"code","source":"print('--- Train ---\\n')\nprint(100*train.isnull().sum() / train.shape[0])\nprint('\\n')\nprint('--- Test ---\\n')\nprint(100*test.isnull().sum() / test.shape[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T03:28:53.678534Z","iopub.execute_input":"2024-12-01T03:28:53.678878Z","iopub.status.idle":"2024-12-01T03:28:54.574396Z","shell.execute_reply.started":"2024-12-01T03:28:53.678842Z","shell.execute_reply":"2024-12-01T03:28:54.573607Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the above, we see that `Occupation` and `Previous Claims` are the features with the highest percentage of missing values. Next, we check for potential duplicates.","metadata":{}},{"cell_type":"code","source":"print(f\"There are {sum(train.duplicated())} duplicated rows in the train data frame.\")\n\nprint(\"\\n\")\nprint(f\"After dropping the Premium Amount column, there are {sum(train.drop(columns=['Premium Amount']).duplicated())} duplicated rows in the train data frame.\")\n\nprint(\"\\n\")\nprint(f\"There are {sum(test.duplicated())} duplicated rows in the test data frame.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T03:28:54.575351Z","iopub.execute_input":"2024-12-01T03:28:54.575618Z","iopub.status.idle":"2024-12-01T03:28:58.610064Z","shell.execute_reply.started":"2024-12-01T03:28:54.575592Z","shell.execute_reply":"2024-12-01T03:28:58.609145Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Finally, we check if there are any observations that appear in both the `train` and `test` data frames.","metadata":{}},{"cell_type":"code","source":"temp_train = train.drop(columns=['Premium Amount'], axis=1)\ntemp_test = test\n\ninner_join = pd.merge(temp_train, temp_test)\nprint(f\"There are {inner_join.shape[0]} observations that appear in both the train and test data frames\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T03:28:58.611103Z","iopub.execute_input":"2024-12-01T03:28:58.611376Z","iopub.status.idle":"2024-12-01T03:29:01.837389Z","shell.execute_reply.started":"2024-12-01T03:28:58.611349Z","shell.execute_reply":"2024-12-01T03:29:01.83637Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Exploration\n\nFirst, we start by exploring the distribution of `Premium Amount`.","metadata":{}},{"cell_type":"code","source":"sns.kdeplot(data=train, x='Premium Amount', color='steelblue', fill=True);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T03:29:01.83857Z","iopub.execute_input":"2024-12-01T03:29:01.838941Z","iopub.status.idle":"2024-12-01T03:29:06.312381Z","shell.execute_reply.started":"2024-12-01T03:29:01.838903Z","shell.execute_reply":"2024-12-01T03:29:06.311336Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the above, we see that the distribution is tri-modal and right-skewed. Next, we explore potential relationships between the input features and `Premium Amount`.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(15, 7))\n\nplt_1 = sns.scatterplot(data=train, x='Age', y='Premium Amount', ax=ax[0])\nplt_2 = sns.boxplot(data=train, x='Gender', y='Premium Amount', ax=ax[1]);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T03:29:06.313738Z","iopub.execute_input":"2024-12-01T03:29:06.314505Z","iopub.status.idle":"2024-12-01T03:29:11.271513Z","shell.execute_reply.started":"2024-12-01T03:29:06.314458Z","shell.execute_reply":"2024-12-01T03:29:11.270584Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the above charts, there is no interesting relationship that can be exploited for modeling purposes. ","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(15, 7))\n\nplt_1 = sns.boxplot(data=train, x='Number of Dependents', y='Premium Amount', ax=ax[0])\nplt_2 = sns.boxplot(data=train, x='Marital Status', y='Premium Amount', ax=ax[1]);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T03:29:11.274012Z","iopub.execute_input":"2024-12-01T03:29:11.274281Z","iopub.status.idle":"2024-12-01T03:29:12.517339Z","shell.execute_reply.started":"2024-12-01T03:29:11.274254Z","shell.execute_reply":"2024-12-01T03:29:12.51648Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the above charts, there is no interesting relationship that can be exploited for modeling purposes. ","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(15, 7))\n\nplt_1 = sns.boxplot(data=train, x='Education Level', y='Premium Amount', ax=ax[0])\nplt_2 = sns.boxplot(data=train, x='Occupation', y='Premium Amount', ax=ax[1]);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T03:29:12.518482Z","iopub.execute_input":"2024-12-01T03:29:12.518727Z","iopub.status.idle":"2024-12-01T03:29:14.057834Z","shell.execute_reply.started":"2024-12-01T03:29:12.518702Z","shell.execute_reply":"2024-12-01T03:29:14.057109Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the above charts, there is no interesting relationship that can be exploited for modeling purposes. ","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(15, 7))\n\nplt_1 = sns.scatterplot(data=train, x='Health Score', y='Premium Amount', ax=ax[0])\nplt_2 = sns.boxplot(data=train, x='Location', y='Premium Amount', ax=ax[1]);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T03:29:14.058888Z","iopub.execute_input":"2024-12-01T03:29:14.05924Z","iopub.status.idle":"2024-12-01T03:29:18.717001Z","shell.execute_reply.started":"2024-12-01T03:29:14.0592Z","shell.execute_reply":"2024-12-01T03:29:18.716059Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the above charts, there is no interesting relationship that can be exploited for modeling purposes. ","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(15, 7))\n\nplt_1 = sns.boxplot(data=train, x='Previous Claims', y='Premium Amount', ax=ax[0])\nplt_2 = sns.boxplot(data=train, x='Policy Type', y='Premium Amount', ax=ax[1]);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T03:29:18.718364Z","iopub.execute_input":"2024-12-01T03:29:18.718723Z","iopub.status.idle":"2024-12-01T03:29:19.891801Z","shell.execute_reply.started":"2024-12-01T03:29:18.718685Z","shell.execute_reply":"2024-12-01T03:29:19.891007Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the left panel, it seems that there is a slightly upward trend as `Previous Claims` increases. Also note that there is one observation with `Previous Claims=9`.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(18, 7))\n\nplt_1 = sns.boxplot(data=train, x='Vehicle Age', y='Premium Amount', ax=ax[0])\nplt_2 = sns.scatterplot(data=train, x='Credit Score', y='Premium Amount', ax=ax[1]);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T03:29:19.89301Z","iopub.execute_input":"2024-12-01T03:29:19.893419Z","iopub.status.idle":"2024-12-01T03:29:24.320696Z","shell.execute_reply.started":"2024-12-01T03:29:19.893376Z","shell.execute_reply":"2024-12-01T03:29:24.31978Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the above charts, there is no interesting relationship that can be exploited for modeling purposes. ","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(18, 7))\n\nplt_1 = sns.boxplot(data=train, x='Customer Feedback', y='Premium Amount', ax=ax[0])\nplt_2 = sns.boxplot(data=train, x='Insurance Duration', y='Premium Amount', ax=ax[1]);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T03:29:24.321885Z","iopub.execute_input":"2024-12-01T03:29:24.322149Z","iopub.status.idle":"2024-12-01T03:29:25.627631Z","shell.execute_reply.started":"2024-12-01T03:29:24.322124Z","shell.execute_reply":"2024-12-01T03:29:25.626617Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the above charts, there is no interesting relationship that can be exploited for modeling purposes. ","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(18, 7))\n\nplt_1 = sns.boxplot(data=train, x='Smoking Status', y='Premium Amount', ax=ax[0])\nplt_2 = sns.boxplot(data=train, x='Exercise Frequency', y='Premium Amount', ax=ax[1]);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T03:29:25.628652Z","iopub.execute_input":"2024-12-01T03:29:25.628936Z","iopub.status.idle":"2024-12-01T03:29:27.161906Z","shell.execute_reply.started":"2024-12-01T03:29:25.628908Z","shell.execute_reply":"2024-12-01T03:29:27.160998Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the above charts, there is no interesting relationship that can be exploited for modeling purposes. ","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(18, 7))\n\nplt_1 = sns.scatterplot(data=train, x='Annual Income', y='Premium Amount', ax=ax[0])\nplt_2 = sns.boxplot(data=train, x='Property Type', y='Premium Amount', ax=ax[1]);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T03:29:27.163147Z","iopub.execute_input":"2024-12-01T03:29:27.163446Z","iopub.status.idle":"2024-12-01T03:29:31.945468Z","shell.execute_reply.started":"2024-12-01T03:29:27.163419Z","shell.execute_reply":"2024-12-01T03:29:31.944638Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the above charts, there is no interesting relationship that can be exploited for modeling purposes. Based on the different considered chart, the data seems pretty random. ","metadata":{}},{"cell_type":"markdown","source":"# Baseline Modeling 1.0\n\nWe first preprocess the data as follows.","metadata":{}},{"cell_type":"code","source":"%%time \ndef feature_processing(df):\n\n    df['Gender'] = df['Gender'].map({'Female': 0, 'Male': 1})\n    df['Smoking Status'] = df['Smoking Status'].map({'No': 0, 'Yes': 1})\n    df['Previous Claims'] = df['Previous Claims'].clip(None, 8)    \n\n    return df\n\ntrain = feature_processing(train)\ntest = feature_processing(test)\n\ncat_cols = ['Marital Status',\n 'Education Level',\n 'Occupation',\n 'Location',\n 'Policy Type',\n 'Customer Feedback',\n 'Exercise Frequency',\n 'Property Type']\n\nfor col in cat_cols:\n    train[col] = train[col].astype('category')\n    test[col] = test[col].astype('category')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:17:03.756898Z","iopub.execute_input":"2024-12-04T16:17:03.757781Z","iopub.status.idle":"2024-12-04T16:17:05.122374Z","shell.execute_reply.started":"2024-12-04T16:17:03.757739Z","shell.execute_reply":"2024-12-04T16:17:05.121306Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Then, we define the input and target features for modeling purposes.","metadata":{}},{"cell_type":"code","source":"%%time \nX = train.drop(columns=['Policy Start Date', 'Premium Amount'], axis=1)\ny = np.log(train['Premium Amount'])\n\ntest_cv = test.drop(columns=['Policy Start Date'], axis=1)\nskf = RepeatedKFold(n_splits=10, n_repeats=1, random_state=42)\n\ndef root_mean_squared_log_error(y_true, y_pred):\n    y_pred = np.maximum(0, y_pred)\n    return np.sqrt(np.mean((np.log1p(y_true) - np.log1p(y_pred)) ** 2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:17:09.466482Z","iopub.execute_input":"2024-12-04T16:17:09.466829Z","iopub.status.idle":"2024-12-04T16:17:09.56196Z","shell.execute_reply.started":"2024-12-04T16:17:09.4668Z","shell.execute_reply":"2024-12-04T16:17:09.560907Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"First, we train the `LGBMRegressor` model over a 10-fold cross-validation strategy.","metadata":{}},{"cell_type":"code","source":"%%time\nlgb_params = {'learning_rate': 0.05002120462413359,\n 'n_estimators': 1000,\n 'max_depth': 11,\n 'reg_alpha': 0.6571665413916834,\n 'reg_lambda': 9.708571261009048,\n 'num_leaves': 30,\n 'colsample_bytree': 0.7984167676439333,\n 'verbose': -1,\n 'n_jobs': -1,\n 'device': 'gpu'}\n\nscores, lgb_test_preds = [], []\nfor i, (train_index, test_index) in enumerate(skf.split(X, y)):\n    \n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train, X_test = X.iloc[train_index], X.iloc[test_index]\n    y_train, y_test = y[train_index], y[test_index]\n            \n    lgb_md = LGBMRegressor(**lgb_params).fit(X_train, y_train)\n    preds = np.exp(lgb_md.predict(X_test))  \n\n    score = root_mean_squared_log_error(np.exp(y_test), preds)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n    \n    lgb_test_preds.append(np.exp(lgb_md.predict(test_cv)))\n\nlgb_oof_score = np.mean(scores)  \nlgb_std = np.std(scores)\nprint(f\"The 10-fold average oof RMSLE score of the LGBM model is {lgb_oof_score}\")\nprint(f\"The 10-fold std oof RMSLE score of the LGBM model is {lgb_std}\")  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T16:08:27.053507Z","iopub.execute_input":"2024-12-02T16:08:27.054332Z","iopub.status.idle":"2024-12-02T16:17:43.088519Z","shell.execute_reply.started":"2024-12-02T16:08:27.054299Z","shell.execute_reply":"2024-12-02T16:17:43.087857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nsubmission = pd.read_csv('../input/playground-series-s4e12/sample_submission.csv')\nsubmission['Premium Amount'] = np.mean(lgb_test_preds, axis=0)\nprint(submission.head())\n\nsubmission.to_csv('baseline_1_LGBM.csv', index=False)\n\ndel submission\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T16:26:51.452725Z","iopub.execute_input":"2024-12-02T16:26:51.453788Z","iopub.status.idle":"2024-12-02T16:26:53.103131Z","shell.execute_reply.started":"2024-12-02T16:26:51.453738Z","shell.execute_reply":"2024-12-02T16:26:53.102213Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Next, we train the `XGBClassifier` model over a 10-fold cross-validation strategy.","metadata":{}},{"cell_type":"code","source":"%%time\nxgb_params = {'device': 'cuda',\n 'objective': 'reg:squarederror',\n 'max_depth': 8,\n 'learning_rate': 0.06939190698873096,\n 'gamma': 4.375658606397604,\n 'min_child_weight': 48,\n 'colsample_bytree': 0.7772990920299161,\n 'n_jobs': -1}\n\ntest_cv_xgb = DMatrix(test_cv, enable_categorical=True)\n\nscores, xgb_oof_preds, xgb_test_preds = [], [], []\nfor i, (train_index, test_index) in enumerate(skf.split(X, y)):\n    \n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train, X_test = X.iloc[train_index], X.iloc[test_index]\n    y_train, y_test = y[train_index], y[test_index]\n    \n    d_train = DMatrix(X_train, y_train, enable_categorical=True)\n    d_test = DMatrix(X_test, y_test, enable_categorical=True)\n\n    xgb_md = xgb.train(params=xgb_params, dtrain=d_train, num_boost_round=1000)\n    preds = np.exp(xgb_md.predict(d_test))\n\n    score = root_mean_squared_log_error(np.exp(y_test), preds)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n    \n    xgb_test_preds.append(np.exp(xgb_md.predict(test_cv_xgb)))\n\nxgb_oof_score = np.mean(scores)  \nxgb_std = np.std(scores)\nprint(f\"The 10-fold average oof RMSLE score of the XGBoost model is {xgb_oof_score}\")\nprint(f\"The 10-fold std oof RMSLE score of the XGBoost model is {xgb_std}\")  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:17:25.9321Z","iopub.execute_input":"2024-12-04T16:17:25.932503Z","iopub.status.idle":"2024-12-04T16:18:47.601209Z","shell.execute_reply.started":"2024-12-04T16:17:25.932468Z","shell.execute_reply":"2024-12-04T16:18:47.600096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nsubmission = pd.read_csv('../input/playground-series-s4e12/sample_submission.csv')\nsubmission['Premium Amount'] = np.mean(lgb_test_preds, axis=0)\nprint(submission.head())\n\nsubmission.to_csv('baseline_1_XGB.csv', index=False)\n\ndel submission\ngc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Next, we train the `CatBoostRegressor` model over a 10-fold cross-validation strategy.","metadata":{}},{"cell_type":"code","source":"%%time\ncb_params = {'loss_function': 'RMSE',\n 'iterations': 944,\n 'depth': 9,\n 'l2_leaf_reg': 6.254098596713341,\n 'task_type': 'GPU'}\n\nX_cat = X.copy()\ntest_cat = test_cv.copy()\n\nfor col in cat_cols:\n    X_cat[col] = X_cat[col].astype('str')\n    test_cat[col] = test_cat[col].astype('str')\n\ntest_pool = Pool(data=test_cat, cat_features=cat_cols)\n\nscores, cat_test_preds = [], []\nfor i, (train_index, test_index) in enumerate(skf.split(X_cat, y)):\n    \n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train, X_test = X_cat.iloc[train_index], X_cat.iloc[test_index]\n    y_train, y_test = y[train_index], y[test_index]\n\n    model_pool = Pool(data=X_train, label=y_train, cat_features=cat_cols)\n    eval_pool = Pool(data=X_test, label=y_test, cat_features=cat_cols)\n            \n    cat_md = CatBoostRegressor(**cb_params).fit(model_pool, eval_set=eval_pool, verbose=0)\n    preds = np.exp(cat_md.predict(eval_pool))\n    \n    score = root_mean_squared_log_error(np.exp(y_test), preds)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n    \n    cat_test_preds.append(np.exp(cat_md.predict(test_pool)))\n\ncat_oof_score = np.mean(scores)  \ncat_std = np.std(scores)\nprint(f\"The 10-fold average oof RMSLE score of the CatBoost model is {cat_oof_score}\")\nprint(f\"The 10-fold std oof RMSLE score of the CatBoost model is {cat_std}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T16:18:07.859512Z","iopub.execute_input":"2024-12-02T16:18:07.859848Z","iopub.status.idle":"2024-12-02T16:25:21.683079Z","shell.execute_reply.started":"2024-12-02T16:18:07.859818Z","shell.execute_reply":"2024-12-02T16:25:21.682185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nsubmission = pd.read_csv('../input/playground-series-s4e12/sample_submission.csv')\nsubmission['Premium Amount'] = np.mean(cat_test_preds, axis=0)\nprint(submission.head())\n\nsubmission.to_csv('baseline_1_CatBoost.csv', index=False)\n\ndel submission, X, y, train, test, test_cv, X_cat, test_cat, test_pool, test_cv_xgb\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T16:26:43.37544Z","iopub.execute_input":"2024-12-02T16:26:43.375781Z","iopub.status.idle":"2024-12-02T16:26:45.028944Z","shell.execute_reply.started":"2024-12-02T16:26:43.375749Z","shell.execute_reply":"2024-12-02T16:26:45.028119Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Next, we train the `GradientBoostedTreesLearner` model over a 10-fold cross-validation strategy.","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/playground-series-s4e12/train.csv', index_col=0)\ntest = pd.read_csv('../input/playground-series-s4e12/test.csv', index_col=0)\n\ndef feature_processing(df):\n\n    df['Gender'] = df['Gender'].map({'Female': 0, 'Male': 1})\n    df['Smoking Status'] = df['Smoking Status'].map({'No': 0, 'Yes': 1})\n    df['Previous Claims'] = df['Previous Claims'].clip(None, 8)    \n\n    return df\n\ntrain = feature_processing(train)\ntest = feature_processing(test)\ntest_cv = test.drop(columns=['Policy Start Date'], axis=1)\n\nskf = RepeatedKFold(n_splits=10, n_repeats=1, random_state=42)\n\ndef root_mean_squared_log_error(y_true, y_pred):\n    y_pred = np.maximum(0, y_pred)\n    return np.sqrt(np.mean((np.log1p(y_true) - np.log1p(y_pred)) ** 2))\n\nX = train.drop(columns=['Policy Start Date'], axis=1)\nX['Premium Amount'] = np.log(X['Premium Amount'])","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-04T16:55:07.83713Z","iopub.execute_input":"2024-12-04T16:55:07.83774Z","iopub.status.idle":"2024-12-04T16:55:16.800425Z","shell.execute_reply.started":"2024-12-04T16:55:07.837706Z","shell.execute_reply":"2024-12-04T16:55:16.799733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nydf.verbose(-1)\nscores, ydf_test_preds = [], []\nfor i, (train_index, test_index) in enumerate(skf.split(X)):\n\n    print(f\"------------ Working on Fold {i} ------------\")\n            \n    X_train, X_test = X.iloc[train_index], X.iloc[test_index]\n    \n    ydf_md = GradientBoostedTreesLearner(label='Premium Amount', \n                                         task=ydf.Task.REGRESSION, \n                                         num_threads=10, \n                                         num_trees=1000, \n                                         max_depth=8).train(X_train)\n    ydf_pred = np.exp(ydf_md.predict(X_test))\n\n    score = root_mean_squared_log_error(np.exp(X_test['Premium Amount']), ydf_pred)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n\n    ydf_test_preds.append(np.exp(ydf_md.predict(test_cv)))\n\nydf_gb_oof_score = np.mean(scores)  \nydf_gb_std = np.std(scores)\nprint(f\"The 10-fold average oof RMSLE score of the GradientBoostedTreesLearner model is {ydf_gb_oof_score}\")\nprint(f\"The 10-fold std oof RMSLE score of the GradientBoostedTreesLearner model is {ydf_gb_std}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nsubmission = pd.read_csv('../input/playground-series-s4e12/sample_submission.csv')\nsubmission['Premium Amount'] = np.mean(ydf_test_preds, axis=0)\nprint(submission.head())\n\nsubmission.to_csv('baseline_1_gb.csv', index=False)\n\ndel submission, train, test, test_cv, X\ngc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Finally, we compare the out-of-fold performance of the considered models.","metadata":{}},{"cell_type":"code","source":"%%time\nmd = pd.DataFrame()\nmd['Model'] = ['LGBM', 'XGB', 'CatBoost', 'GradientBoostedTreesLearner']\nmd['10-fold RMSLE'] = [lgb_oof_score, xgb_oof_score, cat_oof_score, ydf_gb_oof_score]\nmd","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-04T16:20:05.630144Z","iopub.execute_input":"2024-12-04T16:20:05.630967Z","iopub.status.idle":"2024-12-04T16:20:05.658644Z","shell.execute_reply.started":"2024-12-04T16:20:05.630918Z","shell.execute_reply":"2024-12-04T16:20:05.657544Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Baseline Modeling 2.0\n\nIn this section, we engineer a couple of features from `Policy Start Date`.","metadata":{}},{"cell_type":"code","source":"%%time\ntrain = pd.read_csv('../input/playground-series-s4e12/train.csv', index_col=0)\ntest = pd.read_csv('../input/playground-series-s4e12/test.csv', index_col=0)\n\ndef feature_processing(df):\n\n    df['Gender'] = df['Gender'].map({'Female': 0, 'Male': 1})\n    df['Smoking Status'] = df['Smoking Status'].map({'No': 0, 'Yes': 1})\n    df['Previous Claims'] = df['Previous Claims'].clip(None, 8)\n        \n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    df['Year'] = df['Policy Start Date'].dt.year\n    df['Month'] = df['Policy Start Date'].dt.month\n\n    df = df.drop(columns=['Age', 'Policy Start Date'], axis=1)\n    return df\n\n\ntrain = feature_processing(train)\ntest = feature_processing(test)\n\nskf = RepeatedKFold(n_splits=10, n_repeats=1, random_state=42)\n\ndef root_mean_squared_log_error(y_true, y_pred):\n    y_pred = np.maximum(0, y_pred)\n    return np.sqrt(np.mean((np.log1p(y_true) - np.log1p(y_pred)) ** 2))\n\nX = train.copy()\nX['Premium Amount'] = np.log(X['Premium Amount'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T00:32:17.086186Z","iopub.execute_input":"2024-12-07T00:32:17.086735Z","iopub.status.idle":"2024-12-07T00:32:28.422279Z","shell.execute_reply.started":"2024-12-07T00:32:17.086703Z","shell.execute_reply":"2024-12-07T00:32:28.421354Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Next, we train the `RandomForestLearner` over a 10 fold-cross validation strategy.","metadata":{}},{"cell_type":"code","source":"%%time\nscores, ydf_test_preds = [], []\nfor i, (train_index, test_index) in enumerate(skf.split(X)):\n\n    print(f\"------------ Working on Fold {i} ------------\")\n            \n    X_train, X_test = X.iloc[train_index], X.iloc[test_index]\n    \n    ydf_md = RandomForestLearner(label='Premium Amount', \n                                 task=ydf.Task.REGRESSION, \n                                 num_threads=10, \n                                 num_trees=100).train(X_train)\n    ydf_pred = np.expm1(ydf_md.predict(X_test))\n\n    score = root_mean_squared_log_error(np.exp(X_test['Premium Amount']), ydf_pred)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n\n    ydf_test_preds.append(np.exp(ydf_md.predict(test)))\n\nydf_rf_oof_score = np.mean(scores)  \nydf_rf_std = np.std(scores)\nprint(f\"The 10-fold average oof RMSLE score of the RandomForestLearner model is {ydf_rf_oof_score}\")\nprint(f\"The 10-fold std oof RMSLE score of the RandomForestLearner model is {ydf_rf_std}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nsubmission = pd.read_csv('../input/playground-series-s4e12/sample_submission.csv')\nsubmission['Premium Amount'] = np.expm1(np.mean(np.log1p(ydf_test_preds), axis=0))\nprint(submission.head())\n\nsubmission.to_csv('baseline_2_rf.csv', index=False)\n\ndel train, test, X, submission\ngc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}