{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-30T10:23:28.197294Z","iopub.execute_input":"2024-12-30T10:23:28.197716Z","iopub.status.idle":"2024-12-30T10:23:29.330513Z","shell.execute_reply.started":"2024-12-30T10:23:28.197663Z","shell.execute_reply":"2024-12-30T10:23:29.329228Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q scikit-learn==1.5.2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T10:23:29.332644Z","iopub.execute_input":"2024-12-30T10:23:29.333262Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Importing required libraries","metadata":{"_kg_hide-output":true}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings('ignore')\nimport re\nfrom sklearn.model_selection import KFold, cross_val_score\nfrom sklearn.metrics import root_mean_squared_error\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor, plot_importance, early_stopping, log_evaluation\nfrom catboost import CatBoostRegressor, Pool\nimport optuna\nfrom sklearn.ensemble import StackingClassifier\nfrom sklearn.linear_model import Ridge","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    train_path = '/kaggle/input/playground-series-s4e12/train.csv'\n    test_path = '/kaggle/input/playground-series-s4e12/test.csv'\n    sub_path = '/kaggle/input/playground-series-s4e12/sample_submission.csv'\n\n    target = 'Premium Amount'\n    n_folds = 5\n    random_state = 42\n    n_trials = 250","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Loading the data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(Config.train_path, index_col='id')\ntest = pd.read_csv(Config.test_path, index_col='id')\n\nprint(f'Train shape: {train.shape}')\nprint(f'Test shape: {test.shape}')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.info()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe().T","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.describe().T","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(14,8))\nsns.heatmap(train.isnull(), cbar=False, yticklabels=False)\nplt.title('Missing Values')\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Exploratory data analysis","metadata":{}},{"cell_type":"markdown","source":"## 3.1 The target variable","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(6, 4))\nsns.histplot(data=train, x=Config.target, kde=True, bins=20)\nplt.title(f'Distribution of {Config.target}')\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3.2 Numerical features","metadata":{}},{"cell_type":"code","source":"numerical_features = train.select_dtypes(exclude=['object']).columns.to_list()\nnumerical_features.remove(Config.target)\n\nfig, axs = plt.subplots(nrows=len(numerical_features), ncols=2, figsize=(20, 6*len(numerical_features)))\n\n#axs = axs.flatten()\n\ndf_copy = train.copy()\n\nfor i, feat in enumerate(numerical_features):\n\n    num_unique = df_copy[feat].nunique()\n    discrete = True if num_unique <= 50 else False\n\n    if num_unique < 10:\n        binned_feat = feat\n    else:\n        df_copy[f'{feat}_binned'] = pd.cut(df_copy[feat], bins=10)\n        binned_feat = f'{feat}_binned'\n        \n    data = pd.concat([df_copy[Config.target], df_copy[binned_feat]], axis=1)\n\n    # Create countplots\n    sns.histplot(x=feat, data=df_copy, discrete=discrete, ax=axs[i, 0])\n    axs[i, 0].set_title(f'Countplot of {feat}')\n    axs[i, 0].tick_params(axis='x', rotation=45)\n    \n    sns.boxplot(x=binned_feat, y=Config.target, data=data, ax=axs[i, 1])\n    #axs[i].set_ylim(0, 5000)\n    axs[i, 1].set_title(f'Boxplot of binned {feat} vs {Config.target}')\n    axs[i, 1].tick_params(axis='x', rotation=45)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3.3 Categorical Features","metadata":{}},{"cell_type":"code","source":"categorical_features = train.select_dtypes(include=['object']).columns.to_list()\n# Plot number of unique values\nfig, ax = plt.subplots(1,1, figsize=(8,5))\nsns.barplot(x=train[categorical_features].nunique().sort_values(ascending=False).values,\n           y=train[categorical_features].nunique().sort_values(ascending=False).index,\n           ax=ax)\nax.set_title('Number of unique values in categorical columns')\n\nfor i, value in enumerate(train[categorical_features].nunique().sort_values(ascending=False).values):\n    ax.text(value, i, f'{value}', va='center')\n\nsns.despine()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_features_filtered = [col for col in train.select_dtypes(include='object').columns.to_list() if col != 'Policy Start Date']\n\nfig, axs = plt.subplots(nrows=len(categorical_features_filtered), ncols=2, figsize=(15, 6*len(categorical_features_filtered)))\n\nfor i, feat in enumerate(categorical_features_filtered):\n\n    sns.countplot(data=train, x=feat, ax=axs[i, 0])\n    axs[i, 0].set_title(f'Countplot of {feat}')\n    axs[i, 0].tick_params(axis='x', rotation=45)\n\n    data = pd.concat([train[Config.target], train[feat]], axis=1)\n\n    sns.boxplot(x=feat, y=Config.target, data=data, ax=axs[i, 1])\n    axs[i,1].set_title(f'Boxplot: {feat} vs {Config.target}')\n    axs[i, 1].tick_params(axis='x', rotation=45)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_features_df = train.select_dtypes(exclude='object')\n\ncorr = numerical_features_df.corr()\nmask = np.triu(np.ones_like(corr, dtype=bool), k=1)\n\nplt.figure(figsize=(8,6))\n\nsns.heatmap(corr, mask=mask, annot=True, fmt='.4f', vmin=-1, vmax=1)\n\nplt.title('Correlation Heatmap for Numerical Features')\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Data Processing","metadata":{}},{"cell_type":"markdown","source":"As part of the data processing, I will convert the `Policy Start Date` into a pandas datetime object and extract the year, month and day. For efficiency,  I'll also convert categorical columns into 'category' data type. ","metadata":{}},{"cell_type":"code","source":"def process_data():\n    train = pd.read_csv(Config.train_path, index_col='id')\n    test = pd.read_csv(Config.test_path, index_col='id')\n\n    train['Policy Start Date'] = pd.to_datetime(train['Policy Start Date'])\n    test['Policy Start Date'] = pd.to_datetime(test['Policy Start Date'])\n    train['year'] = train['Policy Start Date'].dt.year\n    test['year'] = test['Policy Start Date'].dt.year\n    train.drop('Policy Start Date', axis=1, inplace=True)\n    test.drop('Policy Start Date', axis=1, inplace=True)\n\n    categorical_columns = test.select_dtypes(include='object').columns.tolist()\n    train[categorical_columns] = train[categorical_columns].astype(str).astype('category')\n    test[categorical_columns] = test[categorical_columns].astype(str).astype('category')\n\n    X = train.drop(Config.target, axis=1)\n    y = np.log1p(train[Config.target])\n    X_test = test\n\n    return X, y, X_test","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Hyperparameter Tuning w/ Optuna","metadata":{}},{"cell_type":"code","source":"def objective(trial, model_type):\n\n    if model_type == 'ridge':\n        X = pd.DataFrame(oof_predictions)\n        X_tmp, y, _ = process_data()\n    else:\n        X, y, _ = process_data()\n\n    kf = KFold(n_splits=Config.n_folds, shuffle=True, random_state=Config.random_state)\n\n    rmse_scores = []\n\n    model = create_model(trial, model_type=model_type)\n\n    for train_idx, val_idx in kf.split(X, y):\n        X_train, X_val = X.iloc[train_idx, :], X.iloc[val_idx, :]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n        if model_type == 'lgbm':\n            model.fit(X_train, y_train,\n                eval_set=[(X_val, y_val)],\n                eval_metric='rmse',\n                callbacks=[\n                     log_evaluation(period=200),\n                     early_stopping(stopping_rounds=200)])\n\n            y_preds = model.predict(X_val)\n            score = root_mean_squared_error(y_val, np.maximum(y_preds, 0))\n            rmse_scores.append(score)\n\n        elif model_type == 'xgb':\n            model.fit(X_train, \n                    y_train, \n                    eval_set=[(X_val, y_val)], \n                    verbose=200\n                    )\n\n            y_preds = model.predict(X_val)\n            score = root_mean_squared_error(y_val, np.maximum(y_preds, 0))\n            rmse_scores.append(score)\n\n        elif model_type == 'cb':\n            cat_cols = X_train.select_dtypes(include='category').columns.tolist()\n            if len(cat_cols) > 0:\n                train_pool = Pool(X_train, y_train, cat_features=cat_cols)\n                val_pool = Pool(X_val, y_val, cat_features=cat_cols)\n\n            model.fit(\n                X=train_pool, \n                eval_set=val_pool, \n                verbose=200, \n                early_stopping_rounds=100,\n                use_best_model=True\n                )\n            \n            y_preds = model.predict(X_val)\n            score = root_mean_squared_error(y_val, np.maximum(y_preds, 0))\n            rmse_scores.append(score)\n\n        elif model_type == 'ridge':\n            model.fit(X_train, y_train)\n\n            y_preds = model.predict(X_val)\n            score = root_mean_squared_error(y_val, np.maximum(y_preds, 0))\n            rmse_scores.append(score)\n\n        avg_rmse_score = np.mean(rmse_scores)\n        #print(f'Mean RMSLE: {avg_rmse_score:.5f}')\n\n    return avg_rmse_score\n\ndef create_model(trial, model_type):\n\n    if model_type == 'lgbm':\n\n        params = {'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n                 'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.08, log=True),\n                 'min_child_samples': trial.suggest_int('min_child_samples', 10, 200),\n                 'min_child_weight': trial.suggest_float('min_child_weight', 0.5, 0.8),\n                 'num_leaves': trial.suggest_int('num_leaves', 10, 500),\n                 'reg_alpha': trial.suggest_float('reg_alpha', 0.1, 30.0, log=True),\n                 'reg_lambda': trial.suggest_float('reg_lambda', 0.1, 50.0, log=True),\n                 'subsample': trial.suggest_float('subsample', 0.1, 1.0),\n                 'max_depth': trial.suggest_int('max_depth', 3, 15),\n                 'num_iterations': trial.suggest_int('num_iterations', 500, 2000),\n                 'n_jobs': -1,\n                 'verbose': -1,\n                 'random_state': Config.random_state}\n\n        model = LGBMRegressor(**params)\n\n    elif model_type == 'xgb':\n        \n        params = {\"colsample_bylevel\": trial.suggest_float(\"colsample_bylevel\", 0.9, 1.0),\n                \"colsample_bynode\": trial.suggest_float(\"colsample_bynode\", 0.9, 1.0),\n                \"colsample_bytree\": trial.suggest_float(\"colsample_bynode\",0.9, 1.0),\n                \"early_stopping_rounds\": 100,\n                \"enable_categorical\": True,\n                \"eval_metric\": \"rmse\",\n                \"gamma\": trial.suggest_float(\"gamma\",4.0, 5.0, log=True),\n                \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.035, 0.04, log=True),\n                \"max_depth\": trial.suggest_int(\"max_depth\", 10, 20),\n                \"max_leaves\": trial.suggest_int(\"max_leaves\", 70, 90),\n                \"min_child_weight\": trial.suggest_int(\"min_child_weight\",20, 25),\n                \"n_estimators\": 2000,\n                \"n_jobs\": -1,\n                \"random_state\": Config.random_state,\n                \"reg_alpha\": trial.suggest_float(\"reg_alpha\", 10.0, 90.0, log=True),\n                \"reg_lambda\": trial.suggest_float(\"reg_lambda\", 1.0, 5.0, log=True),\n                \"subsample\": trial.suggest_float(\"subsample\", 0.9, 1.0),\n                \"verbosity\": 0\n            }\n        \n        model = XGBRegressor(**params)\n\n    elif model_type == 'cb':\n        \n        params = {\"border_count\": trial.suggest_int(\"border_count\",100, 250),\n                \"colsample_bylevel\": trial.suggest_float(\"colsample_bylevel\", 0.7, 0.8),\n                \"depth\": trial.suggest_int(\"depth\", 3, 10),\n                \"eval_metric\": \"RMSE\",\n                \"iterations\": 3000,\n                \"l2_leaf_reg\": trial.suggest_float(\"l2_leaf_reg\", 10.0, 15.0, log=True),\n                \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.01, 0.1, log=True),\n                \"min_child_samples\": trial.suggest_int(\"min_child_samples\", 40, 60),\n                \"random_state\": Config.random_state,\n                \"random_strength\": trial.suggest_float(\"random_strength\", 0.8, 0.95),\n                \"subsample\": trial.suggest_float(\"subsample\", 0.8, 1.0),\n                \"verbose\": False\n            }\n        \n        model = CatBoostRegressor(**params)\n\n    elif model_type == 'ridge':\n\n        params = {'random_state': Config.random_state,\n             'alpha': trial.suggest_float('alpha', 1e-4, 100.0, log=True),\n             'tol': trial.suggest_float('tol', 1e-6, 1e-2, log=True)}\n\n        model = Ridge(**params)\n\n    return model","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"run = 0\n\nif run == 1:\n    study_lgbm = optuna.create_study(sampler=optuna.samplers.TPESampler(n_startup_trials=50,\n                                                                        multivariate=True,\n                                                                       seed=Config.random_state),\n                                    direction='minimize')\n    study_lgbm.optimize(lambda trial: objective(trial, 'lgbm'), n_trials=Config.n_trials)\n    print('Best value:', study_lgbm.best_value)\n    print('Best trial:', study_lgbm.best_trial.params)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. Model Training","metadata":{}},{"cell_type":"code","source":"class MLModel:\n    def __init__(self, model, config=Config):\n        self.model = model\n        self.config = config\n\n    def fit_predict(self, X, y, X_test):\n\n        scores = []\n        oof_predictions = np.zeros(X.shape[0])\n        test_predictions = np.zeros(X_test.shape[0])\n        \n        kf = KFold(n_splits=self.config.n_folds, shuffle=True, random_state=self.config.random_state)\n\n        for fold, (train_idx, val_idx) in enumerate(kf.split(X, y)):\n            X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n            y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n            if isinstance(self.model, LGBMRegressor):\n                model = self.model\n                model.fit(X_train, y_train,\n                         eval_set=[(X_val, y_val)],\n                         eval_metric = 'rmse',\n                         callbacks = [log_evaluation(period=200),\n                                     early_stopping(stopping_rounds=100)])\n                \n                y_preds = model.predict(X_val)\n                test_pred = model.predict(X_test)\n\n            elif isinstance(self.model, XGBRegressor):\n                model = self.model\n                model.fit(X_train, \n                          y_train, \n                          eval_set=[(X_val, y_val)], \n                          verbose=200)\n\n                y_preds = model.predict(X_val)\n                test_pred = model.predict(X_test)\n\n            elif isinstance(self.model, CatBoostRegressor):\n                model = self.model\n\n                cat_feats = X_train.select_dtypes(include='category').columns.tolist()\n\n                train_pool = Pool(X_train, y_train, cat_features=cat_feats)\n                val_pool = Pool(X_val, y_val, cat_features=cat_feats)\n                test_pool = Pool(X_test, cat_features=cat_feats)\n\n                model.fit(X=train_pool, \n                          eval_set=val_pool, \n                          verbose=200, \n                          early_stopping_rounds=100,\n                          use_best_model=True\n                        )\n\n                y_preds = model.predict(val_pool)\n                test_pred = model.predict(test_pool)\n\n            else:\n                model = self.model\n                model.fit(X_train, y_train)\n                y_preds = model.predict(X_val)\n                test_pred = model.predict(X_test)\n\n            score = root_mean_squared_error(y_val, np.maximum(y_preds, 0))\n            scores.append(score)\n            oof_predictions[val_idx] = y_preds\n            test_predictions += test_pred / self.config.n_folds\n\n            print(f'\\n Fold {fold + 1} RMSLE: {score:.5f} \\n')\n\n        overall_score = root_mean_squared_error(y, np.maximum(oof_predictions, 0))\n        avg_score = np.mean(scores)\n\n        print(f'Overall RMSLE: {overall_score:.5f}')\n\n        return oof_predictions, test_predictions, scores","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_params = {'colsample_bytree': 0.9677970556671236,\n               'learning_rate': 0.011354521823228342,\n               'min_child_samples': 64,\n               'min_child_weight': 0.7015770814200998,\n               'num_leaves': 63,\n               'reg_alpha': 0.12213822248708686,\n               'reg_lambda': 35.5746069644439,\n               'subsample': 0.14721176204281672,\n               'max_depth': 14,\n               'num_iterations': 2000,\n               'n_jobs': -1,\n               'verbose': -1,\n               'random_state': Config.random_state}\n\nxgb_params = {'colsample_bylevel': 0.9503501013092244,\n              'colsample_bynode': 0.954152071720617,\n              'gamma': 4.983355914436968,\n              'learning_rate': 0.036883003914177,\n              'max_depth': 12,\n              'max_leaves': 89,\n              'min_child_weight': 21,\n              'reg_alpha': 67.43179327919249,\n              'reg_lambda': 4.923729929732029,\n              'subsample': 0.9061676815547584,\n              'n_estimators': 2000,\n              'early_stopping_rounds': 100,\n              'enable_categorical': True,\n              'eval_metric': 'rmse',\n              'n_jobs': -1,\n              'random_state': Config.random_state,\n              'verbosity': 0}\n\ncb_params = {'border_count': 247,\n             'colsample_bylevel': 0.7513363892382967,\n             'depth': 10,\n             'l2_leaf_reg': 13.07985492400343,\n             'learning_rate': 0.051755844016817154,\n             'min_child_samples': 50,\n             'random_strength': 0.8369711705860599,\n             'subsample': 0.965308188617597,\n             'iterations': 2000,\n             'eval_metric': 'RMSE', \n             'random_state': Config.random_state,\n             'verbose': False}","metadata":{"trusted":true,"_kg_hide-output":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rmsles = {}\noof_predictions = {}\ntest_predictions = {}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6.1 LightGBM","metadata":{}},{"cell_type":"code","source":"X, y, X_test = process_data()\nlgbm_model = LGBMRegressor(**lgbm_params)\nml_model = MLModel(lgbm_model)\noof_predictions['lgbm'], test_predictions['lgbm'], rmsles['lgbm'] = ml_model.fit_predict(X, y, X_test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6.2 XGBoost","metadata":{}},{"cell_type":"code","source":"X, y, X_test = process_data()\nxgb_model = XGBRegressor(**xgb_params)\nml_model = MLModel(xgb_model)\noof_predictions['xgb'], test_predictions['xgb'], rmsles['xgb'] = ml_model.fit_predict(X, y, X_test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6.3 CatBoost","metadata":{}},{"cell_type":"code","source":"X, y, X_test = process_data()\ncb_model = CatBoostRegressor(**cb_params)\nml_model = MLModel(cb_model)\noof_predictions['cb'], test_predictions['cb'], rmsles['cb'] = ml_model.fit_predict(X, y, X_test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6.4 Ensemble","metadata":{}},{"cell_type":"code","source":"X = pd.DataFrame(oof_predictions)\nX_test = pd.DataFrame(test_predictions)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"run = 0\nif run == 1:\n    study_ridge = optuna.create_study(sampler=optuna.samplers.TPESampler(n_startup_trials=50,\n                                                                        multivariate=True,\n                                                                        seed=Config.random_state),\n                                                                        direction='minimize')\n    study_ridge.optimize(lambda trial: objective(trial, 'ridge'), n_trials=Config.n_trials)\n    print('Best value:', study_ridge.best_value)\n    print('Best trial:', study_ridge.best_trial.params)","metadata":{"trusted":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ridge_params = {'random_state': Config.random_state,\n                'alpha': 78.82422632578647,\n                'tol': 0.004861208451680029}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ridge_model = Ridge(ridge_params)\nml_model = MLModel(xgb_model)\noof_predictions['ridge'], test_predictions['ridge'], rmsles['ridge'] = ml_model.fit_predict(X, y, X_test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_predictions","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6.5 Submission","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv(Config.sub_path)\nsub[Config.target] = np.expm1(test_predictions['ridge'])\nsub.to_csv('submission_ridge_ensemble.csv', index=False)\nsub.to_csv('submission.csv', index=False)\nsub.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}