{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport optuna\nimport shap\nimport missingno as msno\n\n\nfrom datetime import datetime\nfrom catboost import Pool, CatBoostRegressor, EShapCalcType, EFeaturesSelectionAlgorithm\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import KFold, train_test_split\nfrom sklearn.metrics import mean_squared_error, mean_squared_log_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T01:24:04.194405Z","iopub.execute_input":"2025-04-11T01:24:04.194643Z","iopub.status.idle":"2025-04-11T01:24:14.806033Z","shell.execute_reply.started":"2025-04-11T01:24:04.194619Z","shell.execute_reply":"2025-04-11T01:24:14.805306Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\n\ndf_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\n\ndf_train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T01:24:14.807631Z","iopub.execute_input":"2025-04-11T01:24:14.807987Z","iopub.status.idle":"2025-04-11T01:24:22.966724Z","shell.execute_reply.started":"2025-04-11T01:24:14.807968Z","shell.execute_reply":"2025-04-11T01:24:22.9661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['Health Score']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T01:24:22.967382Z","iopub.execute_input":"2025-04-11T01:24:22.967622Z","iopub.status.idle":"2025-04-11T01:24:22.977818Z","shell.execute_reply.started":"2025-04-11T01:24:22.967603Z","shell.execute_reply":"2025-04-11T01:24:22.977044Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data exploration - what did one do to deserve such premiums?","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)\ndf_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T01:24:22.978592Z","iopub.execute_input":"2025-04-11T01:24:22.978847Z","iopub.status.idle":"2025-04-11T01:24:23.106306Z","shell.execute_reply.started":"2025-04-11T01:24:22.978822Z","shell.execute_reply":"2025-04-11T01:24:23.105639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['Health Score'].dropna().astype(int).nunique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T01:24:23.107025Z","iopub.execute_input":"2025-04-11T01:24:23.107718Z","iopub.status.idle":"2025-04-11T01:24:23.152107Z","shell.execute_reply.started":"2025-04-11T01:24:23.107697Z","shell.execute_reply":"2025-04-11T01:24:23.151529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Percentages of missing rows per column\ndf_train.isna().mean().mul(100)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T01:24:23.152798Z","iopub.execute_input":"2025-04-11T01:24:23.152995Z","iopub.status.idle":"2025-04-11T01:24:23.745499Z","shell.execute_reply.started":"2025-04-11T01:24:23.152979Z","shell.execute_reply":"2025-04-11T01:24:23.744803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Creating a matrix of missing data\nplt.figure(figsize=(10, 6))\nmsno.matrix(df_train)\nplt.title('Matrix of missing values')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T01:24:23.747979Z","iopub.execute_input":"2025-04-11T01:24:23.748246Z","iopub.status.idle":"2025-04-11T01:24:29.949492Z","shell.execute_reply.started":"2025-04-11T01:24:23.748227Z","shell.execute_reply":"2025-04-11T01:24:29.948573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['Health Score'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T01:24:29.950278Z","iopub.execute_input":"2025-04-11T01:24:29.950476Z","iopub.status.idle":"2025-04-11T01:24:30.055629Z","shell.execute_reply.started":"2025-04-11T01:24:29.950458Z","shell.execute_reply":"2025-04-11T01:24:30.054806Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**What's missing?**\n\nMissing data doesn't appear to be concentrated in single rows, but spread out. Even so, number of missing values per row might hold valuable information.","metadata":{}},{"cell_type":"markdown","source":"**How unique are our features?**\n\nToo many categories can impair training speed and may bring nothing to the table if they have a low count. Shall we take a look?","metadata":{}},{"cell_type":"code","source":"numerical_vars = ['Age', 'Annual Income', 'Health Score',\n                  'Previous Claims', 'Vehicle Age', 'Credit Score',\n                  'Insurance Duration', 'Premium Amount']\n\n\n\n# Melting the dataframe\nmelted_cats = df_train.drop(['id', 'Policy Start Date']+numerical_vars,\n                            axis =1).melt(var_name='Feat', value_name='Cat')\n\nfg = sns.FacetGrid(melted_cats, col='Feat', col_wrap=3, height=4, sharex=False, sharey=False)\n\n\n# Mapping bar plots to a grid\ndef plot_feature(data, **kwargs):\n    feature = data['Feat'].iloc[0]  # Get the current feature being plotted\n    sns.countplot(data=data, x='Cat', order=data['Cat'].value_counts().index, **kwargs, palette = 'pastel')\n\nfg.map_dataframe(plot_feature)\n\n# Plot adjustments\nfg.set_axis_labels(x_var='', y_var='count')\nfg.set_titles(\"{col_name}\")\nfg.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T01:24:30.056326Z","iopub.execute_input":"2025-04-11T01:24:30.056622Z","iopub.status.idle":"2025-04-11T01:24:50.295002Z","shell.execute_reply.started":"2025-04-11T01:24:30.056592Z","shell.execute_reply":"2025-04-11T01:24:50.294232Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The number of unique values across all features is consistent, with each feature having a similar or identical count of distinct categories.","metadata":{}},{"cell_type":"code","source":"melted_nums = df_train[numerical_vars].melt(var_name='Feat',\n                                            value_name='Value')\n\n# Create a FacetGrid for box plots\nfg = sns.FacetGrid(melted_nums, col='Feat', col_wrap=3, height=5, sharey=False)\n\n# Mapping box plots to the grid\nfg.map_dataframe(sns.boxplot, y='Value', color='aquamarine')\n\n# Plot adjustments\nfg.set_axis_labels(x_var='', y_var='value')\nfg.set_titles(\"{col_name}\")\nfg.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T01:24:50.295807Z","iopub.execute_input":"2025-04-11T01:24:50.296066Z","iopub.status.idle":"2025-04-11T01:25:00.083043Z","shell.execute_reply.started":"2025-04-11T01:24:50.296046Z","shell.execute_reply":"2025-04-11T01:25:00.08216Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*Actual Income*, *Previous Claims* and *Premium Amount* have visible outliers and also are not symmetrical. Apart from these features *Health Score* shows signs of its distribution being right skewed.","metadata":{}},{"cell_type":"markdown","source":"**Affecting Premium Amounts**\n\nIt's worth taking a look how insurance duration and previous claim interact with each other, as well as the Premium Amount.","metadata":{}},{"cell_type":"code","source":"work_ps = df_train.dropna().pivot_table(\n    index='Insurance Duration',  \n    columns= 'Previous Claims',  \n    values='Premium Amount',\n    aggfunc='mean', \n    fill_value=0\n)\n\n\nplt.figure(figsize=(10, 6))\nsns.heatmap(work_ps, annot=True, cmap=\"rocket_r\", cbar=True).invert_yaxis()\nplt.title('Premium Amount based on its duration and number of claims')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T01:25:00.084681Z","iopub.execute_input":"2025-04-11T01:25:00.084954Z","iopub.status.idle":"2025-04-11T01:25:01.199351Z","shell.execute_reply.started":"2025-04-11T01:25:00.084935Z","shell.execute_reply":"2025-04-11T01:25:01.198548Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*Previous Claims* have a stronger effect on the *Premium Amount* than the duration of insurance. Nevertheless, longer duration can also mean more time for a possible claim to happen. A ratio of claims vs duration could be useful for training a model.","metadata":{}},{"cell_type":"markdown","source":"**Missing values in *Occupation***\n\nSince *Occupation* has the second most misisng rows, it's worth checking if the data can be imputed by predicting with other existing features. First the relation between *Annual Income* and *Occupation*.","metadata":{}},{"cell_type":"code","source":"df_train.groupby(\"Occupation\")['Annual Income'].agg(['mean', 'median'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T01:25:01.200172Z","iopub.execute_input":"2025-04-11T01:25:01.200361Z","iopub.status.idle":"2025-04-11T01:25:01.317789Z","shell.execute_reply.started":"2025-04-11T01:25:01.200346Z","shell.execute_reply":"2025-04-11T01:25:01.317199Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"It seems that the *Occupation* variable is not that easy to impute from *Annual Income* - all occupations, more or less, have the same earnings.","metadata":{}},{"cell_type":"code","source":"df_train.groupby(\"Occupation\")['Premium Amount'].mean()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T01:25:01.31847Z","iopub.execute_input":"2025-04-11T01:25:01.318667Z","iopub.status.idle":"2025-04-11T01:25:01.413964Z","shell.execute_reply.started":"2025-04-11T01:25:01.318652Z","shell.execute_reply":"2025-04-11T01:25:01.413325Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"They also pay the same premiums.","metadata":{}},{"cell_type":"markdown","source":"# Preparing data for training","metadata":{}},{"cell_type":"markdown","source":"After checking [Backpacker's notebook](https://www.kaggle.com/code/backpaker/rid-catboost-nonlog) I wondered what the reason was for score discrepency. Was it hyperparameters, feature engineering or use of an entirely different model? Apparently, during the fitting of a CatBoostRegressor model it is imperative to assign *Health Score* to [cat_features](https://catboost.ai/docs/en/concepts/python-reference_catboostregressor).\n\nThanks to this little change, mean scores for cv went from around 1.04 to 1.03-ish.\n\nAnother thing came to attention after creating a variable by mistake. *HealthScore* made from *Health Score* and transformed into an integer then into string, achieved lower scores in the absence of *Health Score* variable. What was surprising, the cv scores got better when both variables (this time *HealtScore* is an integer and *Health Score* a string) were introduced to the model.","metadata":{}},{"cell_type":"code","source":"def feature_changer(df):\n\n    global numerical_vars\n    vars = numerical_vars.copy()\n\n    vars = [var for var in vars if var not in ['Health Score', 'Previous Claims']]\n    if 'Premium Amount' in vars:\n      vars.remove('Premium Amount')\n\n\n    # Counter for missing values per row\n    df['MissingValuesCount'] = df.isna().sum(axis=1)\n    df['MissingHealth'] = df['Health Score'].isna().astype(int)\n    # Number of days passed between last policy started and current policy\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n\n    df['Days Passed'] = (df['Policy Start Date'].max() - df['Policy Start Date']).dt.days\n    df['Year'] = df['Policy Start Date'].dt.year\n    df['Day'] = df['Policy Start Date'].dt.day\n    df['Month'] = df['Policy Start Date'].dt.month\n\n    # Creating Ratio variables, as well as categorizing continuous ones\n    df['Claims v Duration'] = df['Previous Claims'] / df['Insurance Duration']\n    df['Health vs Claims'] = df['Health Score'] / df['Previous Claims']\n    df['Cat Credit Score'] = df['Credit Score'].copy()\n    df['Int Credit Score'] = df['Credit Score'].apply(lambda x: int(x) if pd.notna(x) else x)\n    df['Int Annual Income'] = df['Annual Income'].apply(lambda x: int(x) if pd.notna(x) else x)\n    vars+= ['Days Passed']\n    cats = [col for col in df.columns if col not in vars]\n\n    # Filling missing data\n    df['HealthScore'] = df['Health Score'].apply(lambda x: int(x) if pd.notna(x) else x)\n    df[cats] = df[cats].fillna('None').astype('string')\n    df[vars] = df[vars].fillna(0).astype(float)\n\n    df = df.drop(['id', 'Policy Start Date'], axis = 1)\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T01:25:01.414635Z","iopub.execute_input":"2025-04-11T01:25:01.414863Z","iopub.status.idle":"2025-04-11T01:25:01.4229Z","shell.execute_reply.started":"2025-04-11T01:25:01.414847Z","shell.execute_reply":"2025-04-11T01:25:01.422221Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Adding multiple variables as categorical and continuous ones**\n\nDecided to experiment with creating copies of variables and converting them to strings so they could pass as cat_features. CV improved but the whole process also became more computationally expensive.","metadata":{}},{"cell_type":"code","source":"X, y = df_train.drop('Premium Amount', axis=1), df_train['Premium Amount']\n\nX = feature_changer(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T01:25:01.423695Z","iopub.execute_input":"2025-04-11T01:25:01.423952Z","iopub.status.idle":"2025-04-11T01:25:15.52943Z","shell.execute_reply.started":"2025-04-11T01:25:01.42393Z","shell.execute_reply":"2025-04-11T01:25:15.528678Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Taking first steps towards shaping and training data","metadata":{}},{"cell_type":"code","source":"class CatBoostAndOptunaParams:\n    def __init__(self,\n                 X,\n                 y,\n                 n_folds=10,\n                 hyperparam_bounds=None,\n                 random_state=42,\n                 task_type='GPU',\n                 verbose=0):\n        self.X = X\n        self.y = y\n        self.n_folds = n_folds\n        self.random_state = random_state\n        self.task_type = task_type\n        self.verbose = verbose\n        self.hyperparam_bounds = hyperparam_bounds or {\n            'iterations': (500,700),\n            'depth': (7, 9),\n            'learning_rate': (0.05, 0.2),\n            'l2_leaf_reg': (0.3, 1.0),\n            'loss_function': ['RMSE']\n        }\n\n        # Placeholders\n        self.hyperparams_history = []\n        self._oof_predictions = []\n        self.best_score = float('inf')\n\n    def _suggest_hyperparams(self, trial):\n        return {\n            'iterations': trial.suggest_int('iterations', *self.hyperparam_bounds['iterations']),\n            'depth': trial.suggest_int('depth', *self.hyperparam_bounds['depth']),\n            'learning_rate': trial.suggest_float('learning_rate', *self.hyperparam_bounds['learning_rate'], log=True),\n            'l2_leaf_reg': trial.suggest_float('l2_leaf_reg', *self.hyperparam_bounds['l2_leaf_reg'], log=True),\n            'loss_function': trial.suggest_categorical('loss_function', self.hyperparam_bounds['loss_function'])\n        }\n\n\n    def objective(self, trial):\n\n        params = self._suggest_hyperparams(trial)\n\n        # Assigning string type variables to cat_features for fitting\n        cat_features = [col for col in self.X.select_dtypes(include=['object', 'string']).columns]\n\n        # K-Fold cross-validation\n        folds = KFold(n_splits=self.n_folds, shuffle = True, random_state=self.random_state)\n        fold_rmsle = []\n        oof_preds = np.zeros(len(self.y))\n\n\n        for fold, (train_idx, val_idx) in enumerate(folds.split(self.X, self.y)):\n            X_train, y_train = self.X.iloc[train_idx], self.y.iloc[train_idx]\n            X_val, y_val = self.X.iloc[val_idx], self.y.iloc[val_idx]\n\n            # Model creation and training\n            model = CatBoostRegressor(\n                iterations=params['iterations'],\n                depth=params['depth'],\n                learning_rate=params['learning_rate'],\n                loss_function=params['loss_function'],\n                random_state=self.random_state,\n                l2_leaf_reg=params['l2_leaf_reg'],\n                task_type=self.task_type,\n                verbose=self.verbose\n            )\n            model.fit(X_train, y_train, cat_features=cat_features)\n\n\n            val_pred = model.predict(X_val)\n            val_pred = np.maximum(0, val_pred) # Clipping prediction values\n\n            # Getting oof predictions\n            oof_preds[val_idx] = val_pred\n\n\n            fold_rmsle.append(mean_squared_error(y_val, val_pred, squared = False))\n\n        # Saving oof predictions\n        self._oof_predictions.append(oof_preds.copy())\n\n\n        # Calculating mean rmsle across folds\n        mean_rmsle = np.mean(fold_rmsle)\n\n        self.hyperparams_history.append({**params, 'RMSE': mean_rmsle})\n        return mean_rmsle\n\n    def optimize(self, n_trials=5, direction=\"minimize\"):\n        study = optuna.create_study(direction=direction)\n        study.optimize(self.objective, n_trials=n_trials)\n        self.best_params = study.best_params\n        self.best_value = study.best_value\n        return study\n\n    @property\n    def oof(self):\n        if self._oof_predictions is None:\n            raise ValueError(\"No OOF predictions available\")\n        return self._oof_predictions\n\n    @property\n    def history(self):\n        if not self.hyperparams_history:\n            raise ValueError(\"No hyperparameter history available\")\n        self.hyperparams_history = sorted(self.hyperparams_history, key=lambda x: x['RMSE'])\n        return self.hyperparams_history ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T01:25:15.530307Z","iopub.execute_input":"2025-04-11T01:25:15.530573Z","iopub.status.idle":"2025-04-11T01:25:15.543698Z","shell.execute_reply.started":"2025-04-11T01:25:15.53055Z","shell.execute_reply":"2025-04-11T01:25:15.542996Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**OOF Predictions**\n\nOut-of-fold predictins come from each trial and then an average of results is taken.","metadata":{}},{"cell_type":"code","source":"y_log1p = np.log1p(y)\n\nhyperparam_search = CatBoostAndOptunaParams(X, y_log1p, n_folds = 5)\nfirst_study = hyperparam_search.optimize(n_trials=5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T01:25:15.544474Z","iopub.execute_input":"2025-04-11T01:25:15.544662Z","iopub.status.idle":"2025-04-11T02:04:49.386998Z","shell.execute_reply.started":"2025-04-11T01:25:15.544647Z","shell.execute_reply":"2025-04-11T02:04:49.386273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"oof_preds = hyperparam_search.oof\n\n#Sum of predictions from all trials\nmean_squared_log_error(y, sum(np.expm1(oof_preds))/len(oof_preds), squared = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:04:49.38794Z","iopub.execute_input":"2025-04-11T02:04:49.388214Z","iopub.status.idle":"2025-04-11T02:04:49.424674Z","shell.execute_reply.started":"2025-04-11T02:04:49.388192Z","shell.execute_reply":"2025-04-11T02:04:49.423953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.2, random_state=42)\n\ny_train_log1p = np.log1p(y_train)\ny_test_log1p = np.log1p(y_test)\n\ncat_features = [col for col in X_train.select_dtypes(include=['object', 'string']).columns]\n\n#Model on training data\nmodel = CatBoostRegressor(\n        **first_study.best_params,\n        task_type = 'GPU',\n        random_state=42,\n        verbose = 0,\n        cat_features=cat_features)\n\n\nmodel.fit(X_train, y_train, cat_features = cat_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:04:49.425453Z","iopub.execute_input":"2025-04-11T02:04:49.425788Z","iopub.status.idle":"2025-04-11T02:06:30.484622Z","shell.execute_reply.started":"2025-04-11T02:04:49.42577Z","shell.execute_reply":"2025-04-11T02:06:30.483875Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Testing the efficacy of created models and making final predictions","metadata":{}},{"cell_type":"code","source":"shap.initjs()\nexplainer = shap.TreeExplainer(model)\nshap_values = explainer.shap_values(Pool(X_train, y_train, cat_features=cat_features))\nshap.force_plot(explainer.expected_value, shap_values[169,:], X_train.iloc[169,:])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:06:30.485525Z","iopub.execute_input":"2025-04-11T02:06:30.485739Z","iopub.status.idle":"2025-04-11T02:09:48.567154Z","shell.execute_reply.started":"2025-04-11T02:06:30.485721Z","shell.execute_reply":"2025-04-11T02:09:48.566505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"shap.force_plot(explainer.expected_value, shap_values[196,:], X_train.iloc[196,:])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:09:48.567854Z","iopub.execute_input":"2025-04-11T02:09:48.568135Z","iopub.status.idle":"2025-04-11T02:09:48.573451Z","shell.execute_reply.started":"2025-04-11T02:09:48.568113Z","shell.execute_reply":"2025-04-11T02:09:48.572924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Feature importance\nshap.summary_plot(shap_values, X_train, plot_type=\"bar\", max_display=10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:09:48.576025Z","iopub.execute_input":"2025-04-11T02:09:48.576283Z","iopub.status.idle":"2025-04-11T02:09:50.518007Z","shell.execute_reply.started":"2025-04-11T02:09:48.576266Z","shell.execute_reply":"2025-04-11T02:09:50.517245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"exp = explainer(X_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:21:02.286851Z","iopub.execute_input":"2025-04-11T02:21:02.28745Z","iopub.status.idle":"2025-04-11T02:23:06.199281Z","shell.execute_reply.started":"2025-04-11T02:21:02.287428Z","shell.execute_reply":"2025-04-11T02:23:06.198507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"shap.plots.beeswarm(exp[:500])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:28:09.986457Z","iopub.execute_input":"2025-04-11T02:28:09.987159Z","iopub.status.idle":"2025-04-11T02:28:25.061028Z","shell.execute_reply.started":"2025-04-11T02:28:09.987135Z","shell.execute_reply":"2025-04-11T02:28:25.060156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"shap.plots.heatmap(exp[:500])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:26:28.968179Z","iopub.execute_input":"2025-04-11T02:26:28.968627Z","iopub.status.idle":"2025-04-11T02:26:44.207475Z","shell.execute_reply.started":"2025-04-11T02:26:28.968602Z","shell.execute_reply":"2025-04-11T02:26:44.206635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"shap.plots.waterfall(exp[22])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:19:16.601932Z","iopub.execute_input":"2025-04-11T02:19:16.602638Z","iopub.status.idle":"2025-04-11T02:19:31.891264Z","shell.execute_reply.started":"2025-04-11T02:19:16.602611Z","shell.execute_reply":"2025-04-11T02:19:31.890616Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*Health Score* as a continuous variable doesn't have the same importance for CatBoost as a categorical one. Variable *HealthScore* that was created by a mistake, also boasts a higher importance than other features, despite the existance of original *Health Score*. *HealthScore* (which is *Health Score* transformed into an integer) has fewer unique values than *Health Score* so it might provide some information about cutoff points.","metadata":{}},{"cell_type":"code","source":"train_pool = Pool(X_train, y_train_log1p, feature_names = X_train.columns.tolist(), cat_features=cat_features)\ntest_pool = Pool(X_test, y_test_log1p, feature_names = X_test.columns.tolist(), cat_features=cat_features)\n\n\nmodel = CatBoostRegressor(**first_study.best_params,\n        random_state=42,\n        task_type='GPU',\n        verbose = 0,\n        cat_features=cat_features)\n\nsummary = model.select_features(\n    train_pool,\n    eval_set=test_pool,\n    features_for_select='0-29',\n    num_features_to_select=14,\n    steps=16,\n    algorithm=EFeaturesSelectionAlgorithm.RecursiveByShapValues,\n    shap_calc_type=EShapCalcType.Regular,\n    train_final_model=True,\n    logging_level='Silent',\n    plot=False\n)\n\nsummary","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:09:50.55633Z","iopub.status.idle":"2025-04-11T02:09:50.55661Z","shell.execute_reply.started":"2025-04-11T02:09:50.556463Z","shell.execute_reply":"2025-04-11T02:09:50.556479Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Trimming down the number of features bears some fruit, as the rmsle improves a tiny bit.","metadata":{}},{"cell_type":"code","source":"#Function for creating oof predictions (on test data)\ndef cv_test_preds(X, y, X_test, hyperparams, n_folds=5, random_state=42, task_type='GPU', verbose=0):\n\n    folds = KFold(n_splits=n_folds, shuffle=True, random_state=random_state)\n    test_preds = []\n\n    for fold, (train_idx, val_idx) in enumerate(folds.split(X, y)):\n\n        X_train, y_train = X.iloc[train_idx], y.iloc[train_idx]\n\n        # Finding categorical features\n        cat_features = [col for col in X.select_dtypes(include=['object', 'string']).columns]\n\n\n        model = CatBoostRegressor(\n            iterations=hyperparams[0]['iterations'],\n            depth=hyperparams[0]['depth'],\n            learning_rate=hyperparams[0]['learning_rate'],\n            loss_function=hyperparams[0]['loss_function'],\n            random_state=random_state,\n            l2_leaf_reg=hyperparams[0]['l2_leaf_reg'],\n            task_type=task_type,\n            verbose=verbose\n        )\n        model.fit(X_train, y_train, cat_features=cat_features)\n\n        # Making predictions\n        test_preds.append(model.predict(X_test))\n\n    return test_preds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:09:50.557914Z","iopub.status.idle":"2025-04-11T02:09:50.558601Z","shell.execute_reply.started":"2025-04-11T02:09:50.558474Z","shell.execute_reply":"2025-04-11T02:09:50.558487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Getting hyperparameters for cut data\nX_selected = X.drop(summary['eliminated_features_names'], axis = 1)\nhyperparam_search = CatBoostAndOptunaParams(X_selected, y_log1p,  n_folds = 5, \n                                           hyperparam_bounds = {\n            'iterations': (500,700),\n            'depth': (7, 8),\n            'learning_rate': (0.05, 0.2),\n            'l2_leaf_reg': (0.4, 1.0),\n            'loss_function': ['RMSE']\n        })\n\n\ncut_study = hyperparam_search.optimize(n_trials=5)\n\nprint(\"Best Parameters:\", cut_study.best_params)\nprint(\"Best RMSLE:\", cut_study.best_value)","metadata":{"execution":{"iopub.status.busy":"2025-04-11T02:09:50.559305Z","iopub.status.idle":"2025-04-11T02:09:50.559572Z","shell.execute_reply.started":"2025-04-11T02:09:50.559428Z","shell.execute_reply":"2025-04-11T02:09:50.559443Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Sum of predictions from all trials\noof_cut_preds = hyperparam_search.oof\nmean_squared_log_error(y, sum(np.expm1(oof_cut_preds))/len(oof_cut_preds), squared = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:09:50.560203Z","iopub.status.idle":"2025-04-11T02:09:50.560434Z","shell.execute_reply.started":"2025-04-11T02:09:50.560318Z","shell.execute_reply":"2025-04-11T02:09:50.560331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = feature_changer(df_test)\n\nX_final_preds = cv_test_preds(X_selected,\n                              y_log1p,\n                              X_test.drop(summary['eliminated_features_names'], axis =1),\n                              hyperparam_search.history,\n                              n_folds = 5)\n\npreds = sum(np.expm1(X_final_preds))/len(X_final_preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:09:50.561203Z","iopub.status.idle":"2025-04-11T02:09:50.561399Z","shell.execute_reply.started":"2025-04-11T02:09:50.561306Z","shell.execute_reply":"2025-04-11T02:09:50.561315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\n\nsample_submission['Premium Amount'] = preds\nsample_submission.to_csv(\"submission.csv\", index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:09:50.562117Z","iopub.status.idle":"2025-04-11T02:09:50.562329Z","shell.execute_reply.started":"2025-04-11T02:09:50.562233Z","shell.execute_reply":"2025-04-11T02:09:50.562242Z"}},"outputs":[],"execution_count":null}]}