{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30805,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## IMPORTING THE MODULES","metadata":{},"attachments":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import OrdinalEncoder, MinMaxScaler, StandardScaler\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.metrics import mean_squared_log_error\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:29.981115Z","iopub.execute_input":"2024-12-12T07:10:29.981473Z","iopub.status.idle":"2024-12-12T07:10:31.126786Z","shell.execute_reply.started":"2024-12-12T07:10:29.981437Z","shell.execute_reply":"2024-12-12T07:10:31.126116Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## LOADING THE DATA","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:31.128582Z","iopub.execute_input":"2024-12-12T07:10:31.129252Z","iopub.status.idle":"2024-12-12T07:10:40.151942Z","shell.execute_reply.started":"2024-12-12T07:10:31.12921Z","shell.execute_reply":"2024-12-12T07:10:40.151213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:40.153055Z","iopub.execute_input":"2024-12-12T07:10:40.15335Z","iopub.status.idle":"2024-12-12T07:10:40.723833Z","shell.execute_reply.started":"2024-12-12T07:10:40.153323Z","shell.execute_reply":"2024-12-12T07:10:40.723086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:40.725089Z","iopub.execute_input":"2024-12-12T07:10:40.72571Z","iopub.status.idle":"2024-12-12T07:10:41.259344Z","shell.execute_reply.started":"2024-12-12T07:10:40.725668Z","shell.execute_reply":"2024-12-12T07:10:41.258476Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## DATA ENGINEERING","metadata":{}},{"cell_type":"code","source":"num_col = train.select_dtypes(include='float').columns\ncat_col = train.select_dtypes(include='object').columns\ntarget = \"Premium Amount\"\nnum_col = num_col.drop(target)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:41.261221Z","iopub.execute_input":"2024-12-12T07:10:41.261483Z","iopub.status.idle":"2024-12-12T07:10:41.440485Z","shell.execute_reply.started":"2024-12-12T07:10:41.261458Z","shell.execute_reply":"2024-12-12T07:10:41.439532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def fill_missing_values(df):\n    numeric_cols = df.select_dtypes(include=['number']).columns\n\n    for col in numeric_cols:\n        if df[col].isnull().sum() > 0:  \n            if df[col].skew() > 1 or df[col].skew() < -1:\n                df[col].fillna(df[col].median(), inplace=True)\n                print(f\"for column {col} we used median\")\n            else:\n                df[col].fillna(df[col].mean(), inplace=True)\n                print(f\"for column {col} we used mean\")\n    \n    categorical_cols = df.select_dtypes(include=['object']).columns\n    for col in categorical_cols:\n        if df[col].isnull().sum() > 0:\n            df[col].fillna(df[col].mode()[0], inplace=True)\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:41.441693Z","iopub.execute_input":"2024-12-12T07:10:41.442103Z","iopub.status.idle":"2024-12-12T07:10:41.448493Z","shell.execute_reply.started":"2024-12-12T07:10:41.442064Z","shell.execute_reply":"2024-12-12T07:10:41.447605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = fill_missing_values(train)\ntest = fill_missing_values(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:41.449484Z","iopub.execute_input":"2024-12-12T07:10:41.449753Z","iopub.status.idle":"2024-12-12T07:10:43.763995Z","shell.execute_reply.started":"2024-12-12T07:10:41.449727Z","shell.execute_reply":"2024-12-12T07:10:43.763032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Income to Age Ratio'] = train['Annual Income'] / train['Age']\ntest['Income to Age Ratio'] = test['Annual Income'] / test['Age']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:43.765133Z","iopub.execute_input":"2024-12-12T07:10:43.76541Z","iopub.status.idle":"2024-12-12T07:10:43.78153Z","shell.execute_reply.started":"2024-12-12T07:10:43.765385Z","shell.execute_reply":"2024-12-12T07:10:43.780892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Health_Smoking_Interaction'] = train['Health Score'] * (train['Smoking Status'] == 1.0).astype(int)\ntest['Health_Smoking_Interaction'] = test['Health Score'] * (test['Smoking Status'] == 1.0).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:43.782405Z","iopub.execute_input":"2024-12-12T07:10:43.782636Z","iopub.status.idle":"2024-12-12T07:10:43.944815Z","shell.execute_reply.started":"2024-12-12T07:10:43.782612Z","shell.execute_reply":"2024-12-12T07:10:43.943908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Policy Start Date'] = pd.to_datetime(train['Policy Start Date'])\ntrain['Policy Year'] = train['Policy Start Date'].dt.year\ntrain['Policy Month'] = train['Policy Start Date'].dt.month\ntrain['Policy Quarter'] = train['Policy Start Date'].dt.quarter","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:43.945937Z","iopub.execute_input":"2024-12-12T07:10:43.94621Z","iopub.status.idle":"2024-12-12T07:10:44.437223Z","shell.execute_reply.started":"2024-12-12T07:10:43.946184Z","shell.execute_reply":"2024-12-12T07:10:44.43641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test['Policy Start Date'] = pd.to_datetime(test['Policy Start Date'])\ntest['Policy Year'] = test['Policy Start Date'].dt.year\ntest['Policy Month'] = test['Policy Start Date'].dt.month\ntest['Policy Quarter'] = test['Policy Start Date'].dt.quarter","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:44.438288Z","iopub.execute_input":"2024-12-12T07:10:44.43864Z","iopub.status.idle":"2024-12-12T07:10:44.767119Z","shell.execute_reply.started":"2024-12-12T07:10:44.438603Z","shell.execute_reply":"2024-12-12T07:10:44.766193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.drop('Policy Start Date', axis=1)\ntest = test.drop('Policy Start Date', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:44.768358Z","iopub.execute_input":"2024-12-12T07:10:44.768737Z","iopub.status.idle":"2024-12-12T07:10:45.116753Z","shell.execute_reply.started":"2024-12-12T07:10:44.7687Z","shell.execute_reply":"2024-12-12T07:10:45.115761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_cols = train.select_dtypes(include=['number']) \nskewness_values = numeric_cols.skew()\n\nprint(skewness_values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:45.117881Z","iopub.execute_input":"2024-12-12T07:10:45.118179Z","iopub.status.idle":"2024-12-12T07:10:45.410654Z","shell.execute_reply.started":"2024-12-12T07:10:45.118152Z","shell.execute_reply":"2024-12-12T07:10:45.409717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_col = train.select_dtypes(include='float').columns\ncat_col = train.select_dtypes(include='object').columns\ntarget = \"Premium Amount\"\nnum_col = num_col.drop(target)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:45.414001Z","iopub.execute_input":"2024-12-12T07:10:45.41428Z","iopub.status.idle":"2024-12-12T07:10:45.999396Z","shell.execute_reply.started":"2024-12-12T07:10:45.414252Z","shell.execute_reply":"2024-12-12T07:10:45.998404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def apply_transformations(df):\n    scaler = StandardScaler()\n\n    log_transform_cols = ['Annual Income', 'Previous Claims', 'Income to Age Ratio']\n    standard_scale_cols = ['Age', 'Health Score', 'Vehicle Age', 'Credit Score', 'Insurance Duration', \n                           'Policy Year', 'Policy Month', 'Policy Quarter']\n\n    for col in log_transform_cols:\n        if (df[col] > 0).all():  \n            df[col] = np.log(df[col] + 1) \n\n    df[standard_scale_cols] = scaler.fit_transform(df[standard_scale_cols])\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:46.000597Z","iopub.execute_input":"2024-12-12T07:10:46.000984Z","iopub.status.idle":"2024-12-12T07:10:46.006687Z","shell.execute_reply.started":"2024-12-12T07:10:46.000944Z","shell.execute_reply":"2024-12-12T07:10:46.005869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = apply_transformations(train)\ntest = apply_transformations(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:46.007775Z","iopub.execute_input":"2024-12-12T07:10:46.008052Z","iopub.status.idle":"2024-12-12T07:10:46.399344Z","shell.execute_reply.started":"2024-12-12T07:10:46.008027Z","shell.execute_reply":"2024-12-12T07:10:46.398525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"encoder = OrdinalEncoder(handle_unknown='use_encoded_value', unknown_value=-1)\n\ntrain[cat_col] = encoder.fit_transform(train[cat_col].astype(str))\ntest[cat_col] = encoder.transform(test[cat_col].astype(str))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:46.400368Z","iopub.execute_input":"2024-12-12T07:10:46.400637Z","iopub.status.idle":"2024-12-12T07:10:50.725418Z","shell.execute_reply.started":"2024-12-12T07:10:46.400612Z","shell.execute_reply":"2024-12-12T07:10:50.724684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Premium Amount'] = np.log1p(train['Premium Amount'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:50.726361Z","iopub.execute_input":"2024-12-12T07:10:50.7266Z","iopub.status.idle":"2024-12-12T07:10:50.735079Z","shell.execute_reply.started":"2024-12-12T07:10:50.726577Z","shell.execute_reply":"2024-12-12T07:10:50.734389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(train.drop(target, axis=1), train[target], test_size=0.2, random_state=105)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:50.736057Z","iopub.execute_input":"2024-12-12T07:10:50.736276Z","iopub.status.idle":"2024-12-12T07:10:51.357218Z","shell.execute_reply.started":"2024-12-12T07:10:50.736254Z","shell.execute_reply":"2024-12-12T07:10:51.356173Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## LGB MODEL","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\nfrom skopt import BayesSearchCV\nfrom skopt.space import Real, Integer\nfrom collections import OrderedDict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:51.358318Z","iopub.execute_input":"2024-12-12T07:10:51.358682Z","iopub.status.idle":"2024-12-12T07:10:54.148769Z","shell.execute_reply.started":"2024-12-12T07:10:51.358645Z","shell.execute_reply":"2024-12-12T07:10:54.147838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"search_space = {\n    'learning_rate': Real(0.0001, 0.05, prior='uniform'),\n    'n_estimators': Integer(100, 5000),\n    'max_depth': Integer(5, 50),\n    'min_child_weight': Real(0.1, 10.0, prior='uniform'),\n    'min_child_samples': Integer(10, 500),\n    'subsample': Real(0.5, 1.0, prior='uniform'),\n    'subsample_freq': Integer(1, 10),\n    'colsample_bytree': Real(0.1, 1.0, prior='uniform'),\n    'num_leaves': Integer(20, 50),\n}\n\nlgbm = lgb.LGBMRegressor()\n\nclass PrintCVFolds:\n    def __init__(self):\n        self.fold = 0\n\n    def __call__(self, env):\n        if hasattr(env, 'iteration'):\n            if env.iteration == 0:\n                self.fold += 1\n                print(f\"Training fold {self.fold}\")\n\nopt = BayesSearchCV(\n    lgbm,\n    search_space,\n    n_iter=10, \n    scoring='neg_mean_squared_error',\n    cv=2, \n    verbose=0,\n    n_jobs=-1,\n    random_state=42,\n    refit=False,\n)\n\nprint_cv_folds = PrintCVFolds()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:54.149994Z","iopub.execute_input":"2024-12-12T07:10:54.150711Z","iopub.status.idle":"2024-12-12T07:10:54.165258Z","shell.execute_reply.started":"2024-12-12T07:10:54.150672Z","shell.execute_reply":"2024-12-12T07:10:54.164405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#opt.fit(train.drop(target, axis=1), train[target], callback=[print_cv_folds])\nbest_params_ = OrderedDict([('colsample_bytree', 0.8311563895216271), ('learning_rate', 0.5015284578259513), ('max_depth', 42), ('min_child_samples', 403), ('min_child_weight', 5.278218047738398), ('n_estimators', 568), ('num_leaves', 43), ('subsample', 0.9363151766549711), ('subsample_freq', 9), ('device', 'gpu'), ('objective','regression_l2')])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:54.166099Z","iopub.execute_input":"2024-12-12T07:10:54.166328Z","iopub.status.idle":"2024-12-12T07:10:54.17763Z","shell.execute_reply.started":"2024-12-12T07:10:54.166306Z","shell.execute_reply":"2024-12-12T07:10:54.176898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Best Parameters: {best_params_}\")\n\nlgb_model = lgb.LGBMRegressor(**dict(best_params_)) \nlgb_model.fit(train.drop(target, axis=1), train[target])\n\nlgb.plot_importance(lgb_model, importance_type=\"gain\", figsize=(8, 6), max_num_features=20, color=\"black\",\n                    title=\"LightGBM Feature Importance (Gain)\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:10:54.178633Z","iopub.execute_input":"2024-12-12T07:10:54.178911Z","iopub.status.idle":"2024-12-12T07:11:20.104479Z","shell.execute_reply.started":"2024-12-12T07:10:54.178879Z","shell.execute_reply":"2024-12-12T07:11:20.103746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds = lgb_model.predict(X_test)\n\npreds = preds.reshape(-1, 1)\n\nrmse = np.sqrt(mean_squared_log_error(np.expm1(y_test), np.expm1(preds)))\n\nplt.figure(figsize=(12, 6))\nplt.plot(y_test.values[:400], label='Actual', color='blue')\nplt.plot(preds[:400], label='Predicted', color='red', linestyle='--')\nplt.xlabel('Sample Index')\nplt.ylabel('Value')\nplt.title(f'Actual vs. Predicted (RMSLE: {rmse:.4f})')\nplt.legend()\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:11:20.1056Z","iopub.execute_input":"2024-12-12T07:11:20.105975Z","iopub.status.idle":"2024-12-12T07:11:23.532994Z","shell.execute_reply.started":"2024-12-12T07:11:20.105936Z","shell.execute_reply":"2024-12-12T07:11:23.532144Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## XGB ","metadata":{}},{"cell_type":"code","source":"import xgboost as xgb","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:11:23.534098Z","iopub.execute_input":"2024-12-12T07:11:23.534408Z","iopub.status.idle":"2024-12-12T07:11:23.687348Z","shell.execute_reply.started":"2024-12-12T07:11:23.534375Z","shell.execute_reply":"2024-12-12T07:11:23.686421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"regressor=xgb.XGBRegressor(eval_metric='rmsle', tree_method='gpu_hist')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:11:23.688444Z","iopub.execute_input":"2024-12-12T07:11:23.688722Z","iopub.status.idle":"2024-12-12T07:11:23.692762Z","shell.execute_reply.started":"2024-12-12T07:11:23.688696Z","shell.execute_reply":"2024-12-12T07:11:23.69189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_grid = {\n    'max_depth': [3, 5, 7],\n    'learning_rate': [0.001, 0.01, 0.1],\n    'n_estimators': [50, 100, 200],\n    'subsample': [0.8, 0.9],\n    'colsample_bytree': [0.8, 0.9],\n    'min_child_weight': [1, 3, 5]\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:11:23.693872Z","iopub.execute_input":"2024-12-12T07:11:23.694124Z","iopub.status.idle":"2024-12-12T07:11:23.704175Z","shell.execute_reply.started":"2024-12-12T07:11:23.694099Z","shell.execute_reply":"2024-12-12T07:11:23.70331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#search = GridSearchCV(regressor, param_grid, cv=2).fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:11:23.705178Z","iopub.execute_input":"2024-12-12T07:11:23.705453Z","iopub.status.idle":"2024-12-12T07:11:23.711773Z","shell.execute_reply.started":"2024-12-12T07:11:23.705427Z","shell.execute_reply":"2024-12-12T07:11:23.711114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_params_ = {\n    'colsample_bytree': 0.9,\n    'learning_rate': 0.1,\n    'max_depth': 9,\n    'n_estimators' : 20000,\n    'min_child_weight': 5,\n    'n_estimators': 100,\n    'subsample': 0.9,\n    'tree_method': 'gpu_hist',\n    'gpu_id': 0,                \n    'predictor': 'gpu_predictor',  \n}\n\nprint(f\"The best hyperparameters are {best_params_}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:11:23.712741Z","iopub.execute_input":"2024-12-12T07:11:23.713074Z","iopub.status.idle":"2024-12-12T07:11:23.720709Z","shell.execute_reply.started":"2024-12-12T07:11:23.713039Z","shell.execute_reply":"2024-12-12T07:11:23.720023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"regressor = xgb.XGBRegressor(**best_params_)\n\nregressor.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:11:23.721573Z","iopub.execute_input":"2024-12-12T07:11:23.721877Z","iopub.status.idle":"2024-12-12T07:11:26.733724Z","shell.execute_reply.started":"2024-12-12T07:11:23.721818Z","shell.execute_reply":"2024-12-12T07:11:26.733033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = regressor.predict(X_test)\n\nrmsle = np.sqrt(mean_squared_log_error(np.expm1(y_test), np.expm1(y_pred)))\nprint(f\"Root Mean Squared Log Error (RMSLE): {rmsle}\")\n\nxgb.plot_importance(regressor, importance_type=\"gain\", max_num_features=12, color=\"black\",\n                    title=\"XGB Feature Importance (Gain)\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:11:26.735237Z","iopub.execute_input":"2024-12-12T07:11:26.735778Z","iopub.status.idle":"2024-12-12T07:11:27.068084Z","shell.execute_reply.started":"2024-12-12T07:11:26.735737Z","shell.execute_reply":"2024-12-12T07:11:27.067257Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CATBOOST REGRESSOR","metadata":{}},{"cell_type":"code","source":"from catboost import CatBoostRegressor\nfrom sklearn.model_selection import RandomizedSearchCV","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:11:27.069306Z","iopub.execute_input":"2024-12-12T07:11:27.069672Z","iopub.status.idle":"2024-12-12T07:11:27.224273Z","shell.execute_reply.started":"2024-12-12T07:11:27.069622Z","shell.execute_reply":"2024-12-12T07:11:27.223572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_CBR = CatBoostRegressor()\n\nparam_dist = {\n    'iterations': [100, 300, 500, 1000],   \n    'depth': [4, 5, 6, 9, 12],               \n    'l2_leaf_reg': [1e-3, 1e-2, 1e-1], \n    'leaf_estimation_iterations': [15],\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:11:27.225186Z","iopub.execute_input":"2024-12-12T07:11:27.22542Z","iopub.status.idle":"2024-12-12T07:11:27.233052Z","shell.execute_reply.started":"2024-12-12T07:11:27.225396Z","shell.execute_reply":"2024-12-12T07:11:27.232161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_search = RandomizedSearchCV(model_CBR, param_distributions=param_dist, \n                                   n_iter=10, cv=3, verbose=1, random_state=42, n_jobs=-1)\n\n#random_search.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:11:27.234103Z","iopub.execute_input":"2024-12-12T07:11:27.234369Z","iopub.status.idle":"2024-12-12T07:11:27.242077Z","shell.execute_reply.started":"2024-12-12T07:11:27.234344Z","shell.execute_reply":"2024-12-12T07:11:27.241392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#random_search.best_params_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:11:27.243042Z","iopub.execute_input":"2024-12-12T07:11:27.243288Z","iopub.status.idle":"2024-12-12T07:11:27.254031Z","shell.execute_reply.started":"2024-12-12T07:11:27.243263Z","shell.execute_reply":"2024-12-12T07:11:27.253241Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_params_ = {'leaf_estimation_iterations': 15,\n 'l2_leaf_reg': 0.001,\n 'iterations': 500,\n 'depth': 9}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:11:27.255026Z","iopub.execute_input":"2024-12-12T07:11:27.255718Z","iopub.status.idle":"2024-12-12T07:11:27.262691Z","shell.execute_reply.started":"2024-12-12T07:11:27.255692Z","shell.execute_reply":"2024-12-12T07:11:27.261872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_model = CatBoostRegressor(**best_params_)\ncat_model.fit(train.drop(target, axis=1), train[target])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:11:27.263682Z","iopub.execute_input":"2024-12-12T07:11:27.263949Z","iopub.status.idle":"2024-12-12T07:12:44.077311Z","shell.execute_reply.started":"2024-12-12T07:11:27.263925Z","shell.execute_reply":"2024-12-12T07:12:44.076379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = cat_model.predict(X_test)\n\nrmsle = np.sqrt(mean_squared_log_error(np.expm1(y_test), np.expm1(y_pred)))\nprint(f\"Root Mean Squared Log Error (RMSLE): {rmsle}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:12:44.078494Z","iopub.execute_input":"2024-12-12T07:12:44.078845Z","iopub.status.idle":"2024-12-12T07:12:44.302003Z","shell.execute_reply.started":"2024-12-12T07:12:44.078817Z","shell.execute_reply":"2024-12-12T07:12:44.301115Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## TREES MODEL","metadata":{}},{"cell_type":"code","source":"import ydf\nimport numpy.typing as npty\nfrom typing import Tuple","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:12:44.303115Z","iopub.execute_input":"2024-12-12T07:12:44.303384Z","iopub.status.idle":"2024-12-12T07:12:44.379054Z","shell.execute_reply.started":"2024-12-12T07:12:44.303358Z","shell.execute_reply":"2024-12-12T07:12:44.378372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tuner = ydf.RandomSearchTuner(num_trials=50, automatic_search_space=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:12:44.380058Z","iopub.execute_input":"2024-12-12T07:12:44.380561Z","iopub.status.idle":"2024-12-12T07:12:44.384777Z","shell.execute_reply.started":"2024-12-12T07:12:44.380531Z","shell.execute_reply":"2024-12-12T07:12:44.383885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# If predictions are close to -1, numerical instabilities will distort the\n# results. The predictions are therefore capped slightly above -1.\nPREDICTION_MINIMUM = -1 + 1e-6\n\ndef loss_msle(\n    labels: npty.NDArray[np.float32],\n    predictions: npty.NDArray[np.float32],\n    weights: npty.NDArray[np.float32],\n) -> np.float32:\n  clipped_pred = np.maximum(PREDICTION_MINIMUM, predictions)\n  return np.sum((np.log1p(clipped_pred) - np.log1p(labels))**2) / len(labels)\n\ndef initial_predictions_msle(\n    labels: npty.NDArray[np.float32], _: npty.NDArray[np.float32]\n) -> npty.NDArray[np.float32]:\n  return np.exp(np.mean(np.log1p(labels))) - 1\n\ndef grad_msle(\n    labels: npty.NDArray[np.float32], predictions: npty.NDArray[np.float32]\n) -> npty.NDArray[np.float32]:\n  gradient = (2/ len(labels))*(np.log1p(predictions) - np.log1p(labels)) / (predictions + 1)\n  return gradient\n\ndef hessian_msle(\n    labels: npty.NDArray[np.float32], predictions: npty.NDArray[np.float32]\n) -> npty.NDArray[np.float32]:\n  hessian =  (2/ len(labels))*(1 - np.log1p(predictions) + np.log1p(labels)) / (predictions + 1)**2\n  return hessian\n\ndef gradient_and_hessian_msle(\n    labels: npty.NDArray[np.float32], predictions: npty.NDArray[np.float32]\n) -> Tuple[npty.NDArray[np.float32], npty.NDArray[np.float32]]:\n  clipped_pred = np.maximum(PREDICTION_MINIMUM, predictions)\n  return [grad_msle(labels, clipped_pred), hessian_msle(labels, clipped_pred)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:12:44.389501Z","iopub.execute_input":"2024-12-12T07:12:44.389762Z","iopub.status.idle":"2024-12-12T07:12:44.400332Z","shell.execute_reply.started":"2024-12-12T07:12:44.389735Z","shell.execute_reply":"2024-12-12T07:12:44.399647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Construct the loss object.\nmsle_custom_loss = ydf.RegressionLoss(\n    initial_predictions=initial_predictions_msle,\n    gradient_and_hessian=gradient_and_hessian_msle,\n    loss=loss_msle,\n    activation=ydf.Activation.IDENTITY,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:12:44.401366Z","iopub.execute_input":"2024-12-12T07:12:44.401623Z","iopub.status.idle":"2024-12-12T07:12:44.413059Z","shell.execute_reply.started":"2024-12-12T07:12:44.401583Z","shell.execute_reply":"2024-12-12T07:12:44.41236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"learner = ydf.RandomForestLearner(\n    label=target,\n    task=ydf.Task.REGRESSION,\n    include_all_columns=True,\n    num_trees=500,\n    #tuner=tuner,\n    #loss=msle_custom_loss\n)\nmodel = learner.train(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:19:02.963177Z","iopub.execute_input":"2024-12-12T07:19:02.963538Z","iopub.status.idle":"2024-12-12T07:35:08.521129Z","shell.execute_reply.started":"2024-12-12T07:19:02.963507Z","shell.execute_reply":"2024-12-12T07:35:08.520101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"learner = ydf.GradientBoostedTreesLearner(l2_regularization=2.0,\n    label=target,\n    task=ydf.Task.REGRESSION,\n    include_all_columns=True,\n    num_trees=500,\n    #tuner=tuner,\n    loss=msle_custom_loss\n)\nmodel1 = learner.train(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:42:51.150328Z","iopub.execute_input":"2024-12-12T07:42:51.151239Z","iopub.status.idle":"2024-12-12T07:53:31.875348Z","shell.execute_reply.started":"2024-12-12T07:42:51.15119Z","shell.execute_reply":"2024-12-12T07:53:31.87441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(X_test)\n\nrmsle = np.sqrt(mean_squared_log_error(np.expm1(y_test), np.expm1(y_pred)))\nprint(f\"Root Mean Squared Log Error (RMSLE) for the random forest model: {rmsle}\")\n\ny_pred = model1.predict(X_test)\n\nrmsle = np.sqrt(mean_squared_log_error(np.expm1(y_test), np.expm1(y_pred)))\nprint(f\"Root Mean Squared Log Error (RMSLE) for the graident boosted trees model: {rmsle}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:53:31.877132Z","iopub.execute_input":"2024-12-12T07:53:31.877779Z","iopub.status.idle":"2024-12-12T07:53:40.555785Z","shell.execute_reply.started":"2024-12-12T07:53:31.877725Z","shell.execute_reply":"2024-12-12T07:53:40.554872Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## SUBMISSION","metadata":{}},{"cell_type":"code","source":"subs = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\nlgb_preds = lgb_model.predict(test)\nxgb_preds = regressor.predict(test)\ncat_preds = cat_model.predict(test)\nydf_preds = model.predict(test)\ngbt_model = model1.predict(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:54:27.599257Z","iopub.execute_input":"2024-12-12T07:54:27.60003Z","iopub.status.idle":"2024-12-12T07:55:09.590557Z","shell.execute_reply.started":"2024-12-12T07:54:27.599983Z","shell.execute_reply":"2024-12-12T07:55:09.589763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# RANDOM FOREST MODEL does not perform so it is not included in finaly averaging","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T08:00:29.204507Z","iopub.execute_input":"2024-12-12T08:00:29.204865Z","iopub.status.idle":"2024-12-12T08:00:29.209102Z","shell.execute_reply.started":"2024-12-12T08:00:29.204819Z","shell.execute_reply":"2024-12-12T08:00:29.208238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds = (lgb_preds + cat_preds + xgb_preds + gbt_model) / 4","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:59:09.427436Z","iopub.execute_input":"2024-12-12T07:59:09.428321Z","iopub.status.idle":"2024-12-12T07:59:09.436316Z","shell.execute_reply.started":"2024-12-12T07:59:09.428285Z","shell.execute_reply":"2024-12-12T07:59:09.435288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"subs['Premium Amount'] = np.expm1(preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:59:14.135576Z","iopub.execute_input":"2024-12-12T07:59:14.136262Z","iopub.status.idle":"2024-12-12T07:59:14.143455Z","shell.execute_reply.started":"2024-12-12T07:59:14.136226Z","shell.execute_reply":"2024-12-12T07:59:14.142378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"subs[target]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:59:15.053184Z","iopub.execute_input":"2024-12-12T07:59:15.053761Z","iopub.status.idle":"2024-12-12T07:59:15.06084Z","shell.execute_reply.started":"2024-12-12T07:59:15.053726Z","shell.execute_reply":"2024-12-12T07:59:15.05999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"subs[['id', target]].to_csv('submission.csv', index = None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T07:59:16.871661Z","iopub.execute_input":"2024-12-12T07:59:16.87251Z","iopub.status.idle":"2024-12-12T07:59:18.270218Z","shell.execute_reply.started":"2024-12-12T07:59:16.872472Z","shell.execute_reply":"2024-12-12T07:59:18.269098Z"}},"outputs":[],"execution_count":null}]}