{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30805,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import math\nimport optuna\nimport random\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport xgboost as xgb\nimport seaborn as sns\nimport matplotlib.pyplot as plt \n\nfrom optuna import *\nfrom plotnine import *\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import KFold, RepeatedKFold\nfrom sklearn.compose import ColumnTransformer\n# from sklearn.metrics import root_mean_squared_error\nfrom sklearn.metrics import *\nfrom sklearn.preprocessing import OrdinalEncoder, OneHotEncoder\n\nwarnings.filterwarnings(\"ignore\")\npd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\npd.set_option('display.width', 1000)\n\ndef start_pipeline(df):\n    return df.copy()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:17:37.311708Z","iopub.execute_input":"2024-12-15T09:17:37.312046Z","iopub.status.idle":"2024-12-15T09:17:42.960383Z","shell.execute_reply.started":"2024-12-15T09:17:37.312006Z","shell.execute_reply":"2024-12-15T09:17:42.959686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\", index_col = 'id')\ntest = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\", index_col = 'id')\n\nprint(f'First 5 rows of the datasets:')\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:17:48.300739Z","iopub.execute_input":"2024-12-15T09:17:48.301077Z","iopub.status.idle":"2024-12-15T09:17:56.822727Z","shell.execute_reply.started":"2024-12-15T09:17:48.301046Z","shell.execute_reply":"2024-12-15T09:17:56.821633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.info())\nprint(50*\"-\")\ndisplay(train.describe().T)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:18:21.796749Z","iopub.execute_input":"2024-12-15T09:18:21.797092Z","iopub.status.idle":"2024-12-15T09:18:22.91672Z","shell.execute_reply.started":"2024-12-15T09:18:21.797061Z","shell.execute_reply":"2024-12-15T09:18:22.915546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Policy Start Date'] = pd.to_datetime(train['Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:18:16.62715Z","iopub.execute_input":"2024-12-15T09:18:16.627958Z","iopub.status.idle":"2024-12-15T09:18:16.969328Z","shell.execute_reply.started":"2024-12-15T09:18:16.627923Z","shell.execute_reply":"2024-12-15T09:18:16.968499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 5))\nsns.heatmap(train.isnull(), yticklabels = '')\nplt.title(\"Missing/Null Values\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:18:29.007577Z","iopub.execute_input":"2024-12-15T09:18:29.007886Z","iopub.status.idle":"2024-12-15T09:18:48.335189Z","shell.execute_reply.started":"2024-12-15T09:18:29.00786Z","shell.execute_reply":"2024-12-15T09:18:48.334037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_col = train.select_dtypes(include = ['object', 'category']).columns\nn_unique_per_col = (train[categorical_col].apply(lambda col: len(col.unique()))\n                         .reset_index()\n                         .sort_values(by = 0, ascending = False)\n                         .rename(columns = {0 : \"value\"})\n                         )\n\n\n(\n    ggplot(n_unique_per_col, aes(x = \"reorder(index, value)\", y = \"value\"))\n    + geom_bar(stat = 'identity', fill = 'steelblue', width = 0.7)\n    + coord_flip()\n    + theme_minimal()\n    + theme(\n        figure_size=(10, 8),\n        axis_text_y=element_text(size=10),\n        plot_title=element_text(size=14)\n    )\n    + labs(\n        x = \"Count\",\n        y = \"Columns\",\n        title = \"Number of Unique values per Categorical Column\"\n    )\n    + geom_text(\n        aes(label='value'),\n        nudge_y= 10,  # Adjust horizontal position of labels\n        size=8,\n        color='black'\n    )\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:18:56.598841Z","iopub.execute_input":"2024-12-15T09:18:56.599194Z","iopub.status.idle":"2024-12-15T09:18:58.151845Z","shell.execute_reply.started":"2024-12-15T09:18:56.599164Z","shell.execute_reply":"2024-12-15T09:18:58.150978Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# xploratory data analysis","metadata":{}},{"cell_type":"code","source":"correlation = train.select_dtypes(exclude = ['object', 'category']).corr()\n\nplt.figure(figsize=(12,5))\nmask = np.zeros_like(correlation)\nmask[np.triu_indices_from(mask)] = True\nsns.heatmap(correlation, annot=True, mask = mask, vmin = -1, vmax = 1,\n            cmap = 'viridis')\nplt.title(\"Correlation of Numerical Features in Train Data\")\nplt.xticks(rotation = 45, ha = 'right')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:19:04.298418Z","iopub.execute_input":"2024-12-15T09:19:04.298732Z","iopub.status.idle":"2024-12-15T09:19:05.151108Z","shell.execute_reply.started":"2024-12-15T09:19:04.298707Z","shell.execute_reply":"2024-12-15T09:19:05.150255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.set_style('whitegrid')\nplt.figure(figsize=(12, 5))\nsns.kdeplot(\n    train, \n    x = 'Credit Score', \n    fill = True, \n    alpha = 0.2,\n    hue = 'Property Type'\n)\nsns.despine()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:19:14.024162Z","iopub.execute_input":"2024-12-15T09:19:14.025069Z","iopub.status.idle":"2024-12-15T09:19:19.004083Z","shell.execute_reply.started":"2024-12-15T09:19:14.025018Z","shell.execute_reply":"2024-12-15T09:19:19.003293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 5))\nsns.histplot(\n    train,\n    x = 'Vehicle Age',\n    bins = 25,\n    hue = 'Marital Status'\n)\nsns.despine()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:19:31.207552Z","iopub.execute_input":"2024-12-15T09:19:31.207893Z","iopub.status.idle":"2024-12-15T09:19:33.208892Z","shell.execute_reply.started":"2024-12-15T09:19:31.207861Z","shell.execute_reply":"2024-12-15T09:19:33.208011Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"class Config:\n    train_path = \"/kaggle/input/playground-series-s4e12/train.csv\"\n    test_path = \"/kaggle/input/playground-series-s4e12/test.csv\"\n    submit_path = \"/kaggle/input/playground-series-s4e12/sample_submission.csv\"\n\n    target = \"Premium Amount\"\n    seed = 42\n    n_folds = 5\n    n = 80000 # number of sample\n\ndef get_data(model, \n             sampling: bool = True, \n             config : Config = Config) -> tuple:\n    \"\"\" Grabbing the features and target from the data \"\"\"\n    \n    assert (model in ['cb', 'xgb', 'lgb', 'lr']), \\\n    \"model can only be catboost, xgboost, lgbm, or logistic\"\n\n    if sampling:\n        train = pd.read_csv(Config.train_path, index_col = 'id').sample(config.n, \n                                                                        random_state = config.seed )\n    else:\n        train = pd.read_csv(Config.train_path, index_col = 'id')\n        \n    test = pd.read_csv(Config.test_path, index_col = 'id')\n    features_to_drop = ['Policy Start Date']\n\n    # create/modify features\n    train['Gender'] = train['Gender'].map({'Female' : 0, 'Male' : 1})\n    test['Gender'] = test['Gender'].map({'Female' : 0, 'Male' : 1})\n    train['Smoking Status'] = train['Smoking Status'].map({'No': 0, 'Yes': 1})\n    test['Smoking Status'] = test['Smoking Status'].map({'No': 0, 'Yes': 1})\n\n    # dropping features\n    train = train.drop(features_to_drop, axis = 1)\n    test = test.drop(features_to_drop, axis = 1)\n\n    if model == 'cb':\n        cat_col = test.columns.tolist()\n    else:\n        cat_col = test.select_dtypes(include = ['object']).columns.tolist()\n\n    train[cat_col] = train[cat_col].astype(str).astype('category')\n    test[cat_col] = test[cat_col].astype(str).astype('category')\n    \n    X = train.drop(Config.target, axis = 1)\n    y = train[Config.target]\n    X_test = test\n    return X, y, X_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:24:40.999325Z","iopub.execute_input":"2024-12-15T09:24:40.999705Z","iopub.status.idle":"2024-12-15T09:24:41.011311Z","shell.execute_reply.started":"2024-12-15T09:24:40.999674Z","shell.execute_reply":"2024-12-15T09:24:41.010413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Trainer:\n    def __init__(self, estimator, cfg: Config = Config,\n                 combine_original = False):\n        self.estimator = estimator\n        self.cfg = cfg\n        self.combine_original = combine_original\n\n    def fit_predict(self, X, y, \n                    X_test):\n        print(f\"Training {self.estimator.__class__.__name__}\\n\")\n        scores = []\n        oof_prediction = np.zeros(X.shape[0])\n        test_prediction = np.zeros(X_test.shape[0])\n\n        cv = RepeatedKFold(n_splits = self.cfg.n_folds, \n                           n_repeats = 1,\n                           # shuffle = True,\n                           random_state = self.cfg.seed)\n        cv_split = cv.split(X, y)\n        for i, (train_idx, val_idx) in enumerate(cv_split):\n            # Input and Target\n            x_train, y_train = X.iloc[train_idx], y.iloc[train_idx]\n            x_val, y_val = X.iloc[val_idx], y.iloc[val_idx]\n            \n            # x_train = transformer.fit_transform(x_train)\n            # x_val = transformer.transform(x_val)\n\n            # Estimator Fitting\n            y_train = np.log1p(y_train)\n            self.estimator.fit(x_train, y_train)\n\n            # Predictions\n            y_prediction = self.estimator.predict(x_val)\n            oof_prediction[val_idx] = y_prediction\n\n            ## Test Prediction\n            temp_test_pred = self.estimator.predict(X_test)\n            test_prediction += np.log1p(temp_test_pred) / self.cfg.n_folds\n\n            ## Val Prediction\n            y_val = np.log1p(y_val)\n            score = math.sqrt(mean_squared_error(y_val, y_prediction))\n            # score = root_mean_squared_log_error(y_val, y_prediction)\n            scores.append(score)\n            print(f'{5*\"-\"} Fold {i + 1} | RMSLE: {score:.4f}')\n        \n        print(f\"------- Average Score with {self.cfg.n_folds} Folds: {np.mean(scores):.4f} +/- {np.std(scores):.4f}\")\n        return scores, oof_prediction, test_prediction\n\n    def fit(self, X, y\n           ):\n        scores = []\n        cv = KFold(n_splits = self.cfg.n_folds, shuffle=True,\n                   random_state=self.cfg.seed)\n        cv_split = cv.split(X, y)\n        for i, (train_idx, val_idx) in enumerate(cv_split):\n            # Input and Target\n            x_train, y_train = X.iloc[train_idx], y.iloc[train_idx]\n            x_val, y_val = X.iloc[val_idx], y.iloc[val_idx]\n\n            y_train = np.log1p(y_train)\n            self.estimator.fit(x_train, y_train)\n\n            # Predictions\n            y_val = np.log1p(y_val)\n            y_predict = self.estimator.predict(x_val)\n            score = math.sqrt(mean_squared_error(y_val, y_predict))\n            # score = root_mean_squared_log_error(y_val, y_predict)\n            scores.append(score)\n        return np.mean(scores)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:22:33.003158Z","iopub.execute_input":"2024-12-15T09:22:33.003521Z","iopub.status.idle":"2024-12-15T09:22:33.014311Z","shell.execute_reply.started":"2024-12-15T09:22:33.003491Z","shell.execute_reply":"2024-12-15T09:22:33.013374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"config = Config()\nxgb_params = {\n    \"random_state\" : config.seed,\n    \"verbosity\" : 0,\n    \"n_jobs\" : -1,\n    \"objective\" : \"reg:squarederror\",\n    \"eval_metric\": \"rmse\"\n}\n\nlgb_params = {\n    \"random_state\" : config.seed,\n    \"verbose\" : -1,\n    \"n_jobs\": -1,\n    \"objective\" : \"regression\"\n}\nmodels = [\n    ('xgb', XGBRegressor(**xgb_params, enable_categorical = True)),\n    ('lgb', LGBMRegressor(**lgb_params))\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:22:37.196487Z","iopub.execute_input":"2024-12-15T09:22:37.196828Z","iopub.status.idle":"2024-12-15T09:22:37.202228Z","shell.execute_reply.started":"2024-12-15T09:22:37.196795Z","shell.execute_reply":"2024-12-15T09:22:37.201182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score = {}\noof_pred = {}\ntest_pred = {}\nfor name, est in models:\n    X_train, y_train, X_test = get_data(name)\n    # pipe = Pipeline([\n    #     ('transform', transform),\n    #     (name, est)\n    # ])\n    # X_train = transform.fit_transform(X_train)\n    est_trainer = Trainer(est)\n    score[name], oof_pred[name], test_pred[name] = est_trainer.fit_predict(X_train, \n                                                          y_train, X_test)\n    print(f\"{50*'-'}\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:24:51.899105Z","iopub.execute_input":"2024-12-15T09:24:51.899425Z","iopub.status.idle":"2024-12-15T09:25:25.11868Z","shell.execute_reply.started":"2024-12-15T09:24:51.899397Z","shell.execute_reply":"2024-12-15T09:25:25.117731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.DataFrame(score)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:26:30.708292Z","iopub.execute_input":"2024-12-15T09:26:30.708673Z","iopub.status.idle":"2024-12-15T09:26:30.718307Z","shell.execute_reply.started":"2024-12-15T09:26:30.708642Z","shell.execute_reply":"2024-12-15T09:26:30.717488Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Optimizing Base Model","metadata":{}},{"cell_type":"code","source":"def objective(trial : Trial,\n              x: pd.DataFrame,\n              y: np.ndarray | pd.Series,\n              model_name: str,\n              cfg: Config = Config) -> float:\n    assert (model_name in ['xgb', 'cb', 'lgb']), \\\n    \"model can only be xgb, cb, or lgb\"\n\n    if model_name == 'xgb':\n        param ={\n            # control model complexity\n            \"max_depth\" : trial.suggest_int(\"max_depth\", 1, 10),\n            \"min_child_weight\": trial.suggest_float(\"min_child_weight\",0, 10),\n            \"gamma\": trial.suggest_float(\"gamma\", 0, 10),\n\n            # regularization\n            \"alpha\" : trial.suggest_float(\"alpha\",0, 1),\n            \"lambda\": trial.suggest_float(\"lambda\",0, 1),\n\n            # add randomness\n            \"eta\" : trial.suggest_float(\"eta\", 0, 1),\n            \"subsample\" : trial.suggest_float(\"subsample\", 0, 1),\n            \"colsample_bytree\" : trial.suggest_float(\"colsample_bytree\", 0, 1),\n\n            \"n_estimators\" : trial.suggest_int(\"n_estimators\", 500, 1500)\n        }\n        model = XGBRegressor(**param, eval_metric = \"rmse\", \n                              objective = \"reg:squarederror\",\n                              verbosity = 0, random_state = cfg.seed, \n                              n_jobs = 4, enable_categorical=True)\n    \n    elif model_name == 'lgb':\n        param = {\n                'objective': 'regression',\n                'boosting_type': 'gbdt',\n                'random_state': cfg.seed,\n                'learning_rate': trial.suggest_loguniform('learning_rate', 0.005, 0.1),\n                'num_leaves': trial.suggest_int('num_leaves', 20, 150),\n                'max_depth': trial.suggest_int('max_depth', 6, 15),\n                'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 1, 100),\n                'feature_fraction': trial.suggest_uniform('feature_fraction', 0.4, 0.8),\n                'bagging_fraction': trial.suggest_uniform('bagging_fraction', 0.4, 0.8),\n                'bagging_freq': trial.suggest_int('bagging_freq', 1, 5),\n                'n_estimators': trial.suggest_int(\"n_estimators\", 500, 1500),\n                'verbose' : -1,\n                'device':'gpu'\n                }\n        model = LGBMRegressor(**param)\n    est_trainer = Trainer(model)\n    return est_trainer.fit(x, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T06:59:23.549636Z","iopub.execute_input":"2024-12-05T06:59:23.549967Z","iopub.status.idle":"2024-12-05T06:59:23.559251Z","shell.execute_reply.started":"2024-12-05T06:59:23.549941Z","shell.execute_reply":"2024-12-05T06:59:23.558309Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_name = 'lgb'\nX, y, X_test = get_data(model_name)\n# ft = create_transformer(X)\n\n# optuna.logging.set_verbosity(optuna.logging.WARNING)\nstudy = create_study(\n    study_name = 'optimization',\n    direction = 'minimize',\n    pruner = optuna.pruners.HyperbandPruner(),\n    sampler = optuna.samplers.TPESampler()\n)\nstudy.optimize(lambda trial: objective(trial, X, y, model_name, config),\n               n_trials = 50, show_progress_bar=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:22:45.144043Z","iopub.execute_input":"2024-12-05T07:22:45.144715Z","iopub.status.idle":"2024-12-05T07:59:20.403619Z","shell.execute_reply.started":"2024-12-05T07:22:45.14468Z","shell.execute_reply":"2024-12-05T07:59:20.402583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_score = {}\nfinal_oof = {}\nfinal_test = {}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:27:15.723665Z","iopub.execute_input":"2024-12-15T09:27:15.724294Z","iopub.status.idle":"2024-12-15T09:27:15.728101Z","shell.execute_reply.started":"2024-12-15T09:27:15.72426Z","shell.execute_reply":"2024-12-15T09:27:15.727096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nbest_param_lgb = {\n    'learning_rate': 0.0065411123979757175,\n     'num_leaves': 26,\n     'max_depth': 11,\n     'min_data_in_leaf': 69,\n     'feature_fraction': 0.7509045202739365,\n     'bagging_fraction': 0.7162728512004927,\n     'bagging_freq': 3,\n     'n_estimators': 1095}\nlgb = LGBMRegressor(**best_param_lgb, objective = \"regression\",\n                    verbose = -1)\nX, y, X_test = get_data('lgb', sampling = False)\nest_trainer = Trainer(lgb, config)\nfinal_score['lgb'], final_oof['lgb'], final_test['lgb'] = est_trainer.fit_predict(X, y, X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:27:18.426758Z","iopub.execute_input":"2024-12-15T09:27:18.427102Z","iopub.status.idle":"2024-12-15T09:36:34.483467Z","shell.execute_reply.started":"2024-12-15T09:27:18.427069Z","shell.execute_reply":"2024-12-15T09:36:34.482589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_param_xgb = {'max_depth': 9,\n                 'min_child_weight': 1.8234305487323932,\n                 'gamma': 5.596712843204808,\n                 'alpha': 0.7107093548012449,\n                 'lambda': 0.7444412824149143,\n                 'eta': 0.3996772555834951,\n                 'subsample': 0.9888215159841199,\n                 'colsample_bytree': 0.5043488959628755,\n                 'n_estimators': 1093}\nxgb = XGBRegressor(**best_param_xgb, \n                   eval_metric = \"rmse\", \n                   objective = \"reg:squarederror\",\n                   verbosity = 0, random_state = config.seed, \n                   n_jobs = 4, enable_categorical=True)\nX, y, X_test = get_data('xgb', sampling = False)\nest_trainer = Trainer(xgb, config)\nfinal_score['xgb'], final_oof['xgb'], final_test['xgb'] = est_trainer.fit_predict(X, y, X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:39:06.60005Z","iopub.execute_input":"2024-12-15T09:39:06.600902Z","iopub.status.idle":"2024-12-15T09:40:40.732778Z","shell.execute_reply.started":"2024-12-15T09:39:06.600862Z","shell.execute_reply":"2024-12-15T09:40:40.732078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.ensemble import RandomForestRegressor\nfinal_df = pd.DataFrame(final_oof)\nlog_y = np.log1p(y)\nlr = LinearRegression()\n\ncv_method = RepeatedKFold(n_repeats=1,n_splits= 10, random_state=config.seed)\ncv_score = cross_val_score(lr, final_df, log_y, \n                           scoring = \"neg_root_mean_squared_error\", cv = cv_method)\nprint(f\"CV Score: {abs(cv_score.mean())}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:43:48.31073Z","iopub.execute_input":"2024-12-15T09:43:48.311088Z","iopub.status.idle":"2024-12-15T09:43:50.587908Z","shell.execute_reply.started":"2024-12-15T09:43:48.311057Z","shell.execute_reply":"2024-12-15T09:43:50.586076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.log1p(train['Premium Amount']).plot(kind = 'hist')\n\nfinal_df.plot(kind = 'hist')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:53:06.139128Z","iopub.execute_input":"2024-12-15T09:53:06.139499Z","iopub.status.idle":"2024-12-15T09:53:06.945399Z","shell.execute_reply.started":"2024-12-15T09:53:06.139466Z","shell.execute_reply":"2024-12-15T09:53:06.944405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Final Prediction\n\nfinal_model = LinearRegression()\nfinal_model.fit(final_df, log_y)\n\n# Predict on Test Set\nfinal_test_df = pd.DataFrame(final_test)\nfinal_test_pred = final_model.predict(final_test_df)\nfinal_test_pred = np.expm1(final_test_pred) # exponentiate the value\nfinal_test_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:46:21.137008Z","iopub.execute_input":"2024-12-15T09:46:21.137376Z","iopub.status.idle":"2024-12-15T09:46:21.305814Z","shell.execute_reply.started":"2024-12-15T09:46:21.137328Z","shell.execute_reply":"2024-12-15T09:46:21.301639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Final Prediction\nmodel_name = 'lgb'\n# final_model = XGBRegressor(**best_param_xgb, \n#                    eval_metric = \"rmse\", \n#                    objective = \"reg:squarederror\",\n#                    verbosity = 0, random_state = config.seed, \n#                    n_jobs = 4, enable_categorical=True)\n\nfinal_model = LGBMRegressor(**best_param_lgb, objective = \"regression\",\n                    verbose = -1)\nX, y, X_test = get_data(model_name, sampling = False)\ny = np.log1p(y)\nfinal_model.fit(X, y)\nfinal_prediction = np.expm1(final_model.predict(X_test))\n\n\n# Submitting\nsubmit = pd.read_csv(config.submit_path, index_col = 'id')\nsubmit['Premium Amount'] = final_prediction\nsubmit.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T09:59:06.820608Z","iopub.execute_input":"2024-12-15T09:59:06.82097Z","iopub.status.idle":"2024-12-15T10:01:09.627683Z","shell.execute_reply.started":"2024-12-15T09:59:06.820941Z","shell.execute_reply":"2024-12-15T10:01:09.626807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit.to_csv(\"submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T10:02:04.166816Z","iopub.execute_input":"2024-12-15T10:02:04.167205Z","iopub.status.idle":"2024-12-15T10:02:05.507302Z","shell.execute_reply.started":"2024-12-15T10:02:04.167173Z","shell.execute_reply":"2024-12-15T10:02:05.506598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}