{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom catboost import CatBoostRegressor\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_log_error\nfrom sklearn.preprocessing import LabelEncoder, OrdinalEncoder\nfrom sklearn.preprocessing import OneHotEncoder\n\n\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:13:35.61602Z","iopub.execute_input":"2024-12-03T14:13:35.616503Z","iopub.status.idle":"2024-12-03T14:13:37.269642Z","shell.execute_reply.started":"2024-12-03T14:13:35.616464Z","shell.execute_reply":"2024-12-03T14:13:37.26833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\n\nsample = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\n\ntrain.drop('id', axis=1, inplace=True)\ntest.drop('id', axis=1, inplace=True) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:13:39.89445Z","iopub.execute_input":"2024-12-03T14:13:39.89498Z","iopub.status.idle":"2024-12-03T14:13:51.058877Z","shell.execute_reply.started":"2024-12-03T14:13:39.894946Z","shell.execute_reply":"2024-12-03T14:13:51.057911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def date(Df):\n\n    Df['Policy Start Date'] = pd.to_datetime(Df['Policy Start Date'])\n    Df['Year'] = Df['Policy Start Date'].dt.year\n    Df['Day'] = Df['Policy Start Date'].dt.day\n    Df['Month'] = Df['Policy Start Date'].dt.month\n    Df['Month_name'] = Df['Policy Start Date'].dt.month_name()\n    Df['Day_of_week'] = Df['Policy Start Date'].dt.day_name()\n    Df['Week'] = Df['Policy Start Date'].dt.isocalendar().week\n    Df['Year_sin'] = np.sin(2 * np.pi * Df['Year'])\n    Df['Year_cos'] = np.cos(2 * np.pi * Df['Year'])\n    Df['Month_sin'] = np.sin(2 * np.pi * Df['Month'] / 12) \n    Df['Month_cos'] = np.cos(2 * np.pi * Df['Month'] / 12)\n    Df['Day_sin'] = np.sin(2 * np.pi * Df['Day'] / 31)  \n    Df['Day_cos'] = np.cos(2 * np.pi * Df['Day'] / 31)\n    Df['Group']=(Df['Year']-2020)*48+Df['Month']*4+Df['Day']//7\n    \n    Df.drop('Policy Start Date', axis=1, inplace=True)\n\n    return Df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:14:02.732355Z","iopub.execute_input":"2024-12-03T14:14:02.732761Z","iopub.status.idle":"2024-12-03T14:14:02.741754Z","shell.execute_reply.started":"2024-12-03T14:14:02.732729Z","shell.execute_reply":"2024-12-03T14:14:02.740502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = date(train)\ntest = date(test)\n\ncat_cols = [col for col in train.columns if train[col].dtype == 'object']\nfeature_cols = list(test.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:14:08.920953Z","iopub.execute_input":"2024-12-03T14:14:08.921372Z","iopub.status.idle":"2024-12-03T14:14:12.125797Z","shell.execute_reply.started":"2024-12-03T14:14:08.921337Z","shell.execute_reply":"2024-12-03T14:14:12.124683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CategoricalEncoder:\n    def __init__(self, train, test):\n        self.train = train\n        self.test = test\n\n    def frequency_encode(self, cat_cols, feature_cols, drop_org=False):\n        combined = pd.concat([self.train, self.test], axis=0, ignore_index=True)\n\n        new_cat_cols = [] \n        for col in cat_cols:\n            freq_encoding = combined[col].value_counts().to_dict()\n            \n            self.train[f\"{col}_freq\"] = self.train[col].map(freq_encoding).astype('float')\n            self.test[f\"{col}_freq\"] = self.test[col].map(freq_encoding).astype('float')\n\n            new_col_name = f\"{col}_freq\"\n            new_cat_cols.append(new_col_name)\n            feature_cols.append(new_col_name)\n            if drop_org:\n                feature_cols.remove(col)\n\n        return self.train, self.test, new_cat_cols, feature_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:14:12.127572Z","iopub.execute_input":"2024-12-03T14:14:12.128005Z","iopub.status.idle":"2024-12-03T14:14:12.136461Z","shell.execute_reply.started":"2024-12-03T14:14:12.127963Z","shell.execute_reply":"2024-12-03T14:14:12.135385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"encoder = CategoricalEncoder(train, test)\ntrain, test, cat_cols, feature_cols = encoder.frequency_encode(cat_cols, feature_cols, drop_org=True)\n\ntrain = train[feature_cols + ['Premium Amount']]\ntest = test[feature_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:14:19.260372Z","iopub.execute_input":"2024-12-03T14:14:19.26073Z","iopub.status.idle":"2024-12-03T14:14:23.85113Z","shell.execute_reply.started":"2024-12-03T14:14:19.260699Z","shell.execute_reply":"2024-12-03T14:14:23.85019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:14:36.690483Z","iopub.execute_input":"2024-12-03T14:14:36.690883Z","iopub.status.idle":"2024-12-03T14:14:36.723389Z","shell.execute_reply.started":"2024-12-03T14:14:36.690849Z","shell.execute_reply":"2024-12-03T14:14:36.722288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.isnull().sum())\nprint(test.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:14:42.587161Z","iopub.execute_input":"2024-12-03T14:14:42.587581Z","iopub.status.idle":"2024-12-03T14:14:42.689866Z","shell.execute_reply.started":"2024-12-03T14:14:42.587546Z","shell.execute_reply":"2024-12-03T14:14:42.688657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.fillna(-111)\ntest = test.fillna(-111)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:14:46.68231Z","iopub.execute_input":"2024-12-03T14:14:46.682679Z","iopub.status.idle":"2024-12-03T14:14:46.987549Z","shell.execute_reply.started":"2024-12-03T14:14:46.682649Z","shell.execute_reply":"2024-12-03T14:14:46.986636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.isnull().sum())\nprint(test.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:14:50.224772Z","iopub.execute_input":"2024-12-03T14:14:50.225128Z","iopub.status.idle":"2024-12-03T14:14:50.324697Z","shell.execute_reply.started":"2024-12-03T14:14:50.225099Z","shell.execute_reply":"2024-12-03T14:14:50.323581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train.drop('Premium Amount', axis=1)  \ny = train['Premium Amount']\n\ny_log = np.log1p(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:15:28.950045Z","iopub.execute_input":"2024-12-03T14:15:28.951058Z","iopub.status.idle":"2024-12-03T14:15:29.130182Z","shell.execute_reply.started":"2024-12-03T14:15:28.951016Z","shell.execute_reply":"2024-12-03T14:15:29.129064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rmsle(y_true, y_pred):\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:15:39.373572Z","iopub.execute_input":"2024-12-03T14:15:39.37397Z","iopub.status.idle":"2024-12-03T14:15:39.378987Z","shell.execute_reply.started":"2024-12-03T14:15:39.373934Z","shell.execute_reply":"2024-12-03T14:15:39.377858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_model():\n    kf = KFold(n_splits=5, shuffle=True, random_state=42)\n    oof = np.zeros(len(X))\n    models = []\n\n    for fold, (train_idx, valid_idx) in enumerate(kf.split(X)):\n        print(f\"Fold {fold + 1}\")\n        X_train, X_valid = X.iloc[train_idx], X.iloc[valid_idx]\n        y_train, y_valid = y_log.iloc[train_idx], y_log.iloc[valid_idx]\n\n        model = CatBoostRegressor(\n            iterations=3000,\n            learning_rate=0.05,\n            depth=6,\n            eval_metric=\"RMSE\",\n            random_seed=42,\n            verbose=200,\n            task_type='CPU',\n            l2_leaf_reg =  0.7,\n        )\n        \n        model.fit(X_train,\n                  y_train,\n                  eval_set=(X_valid, y_valid), \n                  early_stopping_rounds=300,\n                  # cat_features=cat_cols,\n                 )\n        models.append(model)\n        oof[valid_idx] = np.maximum(0, model.predict(X_valid))\n        fold_rmsle = rmsle(np.expm1(y_valid), np.expm1(oof[valid_idx]))\n        print(f\"Fold {fold + 1} RMSLE: {fold_rmsle}\")\n        \n    return models, oof","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:19:25.378034Z","iopub.execute_input":"2024-12-03T14:19:25.379144Z","iopub.status.idle":"2024-12-03T14:19:25.38688Z","shell.execute_reply.started":"2024-12-03T14:19:25.379101Z","shell.execute_reply":"2024-12-03T14:19:25.385715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"models,oof = train_model()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:19:34.559002Z","iopub.execute_input":"2024-12-03T14:19:34.559422Z","iopub.status.idle":"2024-12-03T14:34:07.274759Z","shell.execute_reply.started":"2024-12-03T14:19:34.559389Z","shell.execute_reply":"2024-12-03T14:34:07.273667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(rmsle(y, np.expm1(oof)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:34:46.554792Z","iopub.execute_input":"2024-12-03T14:34:46.555182Z","iopub.status.idle":"2024-12-03T14:34:46.637905Z","shell.execute_reply.started":"2024-12-03T14:34:46.555148Z","shell.execute_reply":"2024-12-03T14:34:46.636889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_predictions = np.zeros(len(test))\n\nfor model in models:\n    test_predictions += np.maximum(0, np.expm1(model.predict(test))) / len(models)\n\n\nsample['Premium Amount'] = test_predictions\nsample.to_csv('submission.csv', index = False)\nsample.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:35:04.182938Z","iopub.execute_input":"2024-12-03T14:35:04.18335Z","iopub.status.idle":"2024-12-03T14:35:09.618711Z","shell.execute_reply.started":"2024-12-03T14:35:04.183318Z","shell.execute_reply":"2024-12-03T14:35:09.617273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}