{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import accuracy_score, mean_squared_error\nimport warnings\n\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:25:28.018623Z","iopub.execute_input":"2024-12-09T15:25:28.019509Z","iopub.status.idle":"2024-12-09T15:25:32.006684Z","shell.execute_reply.started":"2024-12-09T15:25:28.019438Z","shell.execute_reply":"2024-12-09T15:25:32.005303Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv', index_col='id', engine='pyarrow')\ntest_df = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv', index_col='id', engine='pyarrow')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:25:32.009205Z","iopub.execute_input":"2024-12-09T15:25:32.009736Z","iopub.status.idle":"2024-12-09T15:25:35.707945Z","shell.execute_reply.started":"2024-12-09T15:25:32.009643Z","shell.execute_reply":"2024-12-09T15:25:35.706732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:25:35.709489Z","iopub.execute_input":"2024-12-09T15:25:35.709852Z","iopub.status.idle":"2024-12-09T15:25:35.746455Z","shell.execute_reply.started":"2024-12-09T15:25:35.709787Z","shell.execute_reply":"2024-12-09T15:25:35.744937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:25:35.747848Z","iopub.execute_input":"2024-12-09T15:25:35.74824Z","iopub.status.idle":"2024-12-09T15:25:35.773918Z","shell.execute_reply.started":"2024-12-09T15:25:35.748205Z","shell.execute_reply":"2024-12-09T15:25:35.77245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def date_separator(x):\n    return pd.Series([x.day, x.month, x.year])\n\ntrain_df[['day', 'month', 'year']] = train_df['Policy Start Date'].apply(date_separator)\ntest_df[['day', 'month', 'year']] = test_df['Policy Start Date'].apply(date_separator)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:25:35.778128Z","iopub.execute_input":"2024-12-09T15:25:35.778535Z","iopub.status.idle":"2024-12-09T15:29:26.366412Z","shell.execute_reply.started":"2024-12-09T15:25:35.778501Z","shell.execute_reply":"2024-12-09T15:29:26.364696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:29:26.367857Z","iopub.execute_input":"2024-12-09T15:29:26.368434Z","iopub.status.idle":"2024-12-09T15:29:27.012991Z","shell.execute_reply.started":"2024-12-09T15:29:26.368383Z","shell.execute_reply":"2024-12-09T15:29:27.01145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:29:27.015464Z","iopub.execute_input":"2024-12-09T15:29:27.015837Z","iopub.status.idle":"2024-12-09T15:29:27.641223Z","shell.execute_reply.started":"2024-12-09T15:29:27.015803Z","shell.execute_reply":"2024-12-09T15:29:27.639802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target = 'Premium Amount'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:29:27.642537Z","iopub.execute_input":"2024-12-09T15:29:27.64339Z","iopub.status.idle":"2024-12-09T15:29:27.649953Z","shell.execute_reply.started":"2024-12-09T15:29:27.643348Z","shell.execute_reply":"2024-12-09T15:29:27.647903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_features = train_df.drop(target, axis=1).select_dtypes(include=np.number).columns.values\nnumerical_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:29:27.651747Z","iopub.execute_input":"2024-12-09T15:29:27.652214Z","iopub.status.idle":"2024-12-09T15:29:27.955912Z","shell.execute_reply.started":"2024-12-09T15:29:27.652176Z","shell.execute_reply":"2024-12-09T15:29:27.954198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_features = train_df.drop(target, axis=1).select_dtypes(include='object').columns.values\ncategorical_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:29:27.959084Z","iopub.execute_input":"2024-12-09T15:29:27.959459Z","iopub.status.idle":"2024-12-09T15:29:28.799746Z","shell.execute_reply.started":"2024-12-09T15:29:27.959428Z","shell.execute_reply":"2024-12-09T15:29:28.798548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.experimental import enable_iterative_imputer  # noqa\nfrom sklearn.impute import IterativeImputer\n\n# Separate numeric and non-numeric columns\nnumeric_cols = train_df.select_dtypes(include=[\"number\"])\nnon_numeric_cols = train_df.select_dtypes(exclude=[\"number\"])\n\n# Initialize the Iterative Imputer\nimp_mean = IterativeImputer(random_state=0)\n\n# Apply the imputer to numeric columns only\nnumeric_cols_imputed = pd.DataFrame(\n    imp_mean.fit_transform(numeric_cols),\n    columns=numeric_cols.columns,\n    index=numeric_cols.index\n)\n\n# Combine the imputed numeric data with the original non-numeric data\ntrain_df_imputed = pd.concat([numeric_cols_imputed, non_numeric_cols], axis=1)\n\n# Print missing values before and after imputation\nprint(\"Missing values before imputation:\")\nprint(train_df.isnull().sum())\n\nprint(\"\\nMissing values after imputation:\")\nprint(train_df_imputed.isnull().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:35:22.965966Z","iopub.execute_input":"2024-12-09T15:35:22.966652Z","iopub.status.idle":"2024-12-09T15:36:14.698571Z","shell.execute_reply.started":"2024-12-09T15:35:22.966595Z","shell.execute_reply":"2024-12-09T15:36:14.696864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:04.316229Z","iopub.execute_input":"2024-12-09T15:38:04.316622Z","iopub.status.idle":"2024-12-09T15:38:05.779397Z","shell.execute_reply.started":"2024-12-09T15:38:04.316593Z","shell.execute_reply":"2024-12-09T15:38:05.778232Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[numerical_features].astype(np.float_).describe().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:05.781213Z","iopub.execute_input":"2024-12-09T15:38:05.781537Z","iopub.status.idle":"2024-12-09T15:38:06.727689Z","shell.execute_reply.started":"2024-12-09T15:38:05.781508Z","shell.execute_reply":"2024-12-09T15:38:06.726486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.describe(include='O').T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:06.728877Z","iopub.execute_input":"2024-12-09T15:38:06.729227Z","iopub.status.idle":"2024-12-09T15:38:08.389206Z","shell.execute_reply.started":"2024-12-09T15:38:06.729195Z","shell.execute_reply":"2024-12-09T15:38:08.388033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, FunctionTransformer, LabelEncoder, OneHotEncoder\nfrom sklearn.pipeline import make_pipeline, Pipeline\nfrom sklearn.compose import ColumnTransformer, make_column_selector, make_column_transformer\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import SimpleImputer, IterativeImputer\nimport category_encoders as ce\n\npreprocessing = ColumnTransformer([\n    ('num', make_pipeline(SimpleImputer(strategy='mean'), FunctionTransformer(), StandardScaler()), numerical_features),\n    ('cat', make_pipeline(SimpleImputer(strategy='most_frequent'), ce.cat_boost.CatBoostEncoder()), categorical_features)\n], remainder='drop')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:08.391709Z","iopub.execute_input":"2024-12-09T15:38:08.392195Z","iopub.status.idle":"2024-12-09T15:38:08.903632Z","shell.execute_reply.started":"2024-12-09T15:38:08.392147Z","shell.execute_reply":"2024-12-09T15:38:08.902632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train_df.copy()\ny = X.pop(target)\ny = np.log1p(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:08.904954Z","iopub.execute_input":"2024-12-09T15:38:08.905501Z","iopub.status.idle":"2024-12-09T15:38:09.528894Z","shell.execute_reply.started":"2024-12-09T15:38:08.905466Z","shell.execute_reply":"2024-12-09T15:38:09.527899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = preprocessing.fit_transform(X, y)\ntestProcessed = preprocessing.transform(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:11.277654Z","iopub.execute_input":"2024-12-09T15:38:11.278106Z","iopub.status.idle":"2024-12-09T15:38:22.841914Z","shell.execute_reply.started":"2024-12-09T15:38:11.27807Z","shell.execute_reply":"2024-12-09T15:38:22.840352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lightgbm import LGBMRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:22.844232Z","iopub.execute_input":"2024-12-09T15:38:22.844725Z","iopub.status.idle":"2024-12-09T15:38:23.998711Z","shell.execute_reply.started":"2024-12-09T15:38:22.844677Z","shell.execute_reply":"2024-12-09T15:38:23.997366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_params = {\n    'objective': 'regression',\n    'metric': 'rmse',\n    'boosting_type': 'gbdt',\n    'num_leaves': 31,\n    'max_depth': -1,\n    'learning_rate': 0.05,\n    'n_estimators': 1000,\n    'subsample': 0.8,  \n    'colsample_bytree': 0.8,\n    'reg_alpha': 0.1,\n    'reg_lambda': 0.1\n}\n\nlgb_model = LGBMRegressor(**lgb_params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:24.000259Z","iopub.execute_input":"2024-12-09T15:38:24.00092Z","iopub.status.idle":"2024-12-09T15:38:24.008636Z","shell.execute_reply.started":"2024-12-09T15:38:24.00088Z","shell.execute_reply":"2024-12-09T15:38:24.006346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:24.011997Z","iopub.execute_input":"2024-12-09T15:38:24.013357Z","iopub.status.idle":"2024-12-09T15:38:24.258345Z","shell.execute_reply.started":"2024-12-09T15:38:24.013311Z","shell.execute_reply":"2024-12-09T15:38:24.257091Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_model = XGBRegressor(\n    n_estimators=1000,\n    learning_rate=0.05,\n    max_depth=6,\n    random_state=42,\n    n_jobs=-1,\n    verbose=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:24.259743Z","iopub.execute_input":"2024-12-09T15:38:24.260117Z","iopub.status.idle":"2024-12-09T15:38:24.265433Z","shell.execute_reply.started":"2024-12-09T15:38:24.260083Z","shell.execute_reply":"2024-12-09T15:38:24.264194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.neural_network import MLPRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:24.266715Z","iopub.execute_input":"2024-12-09T15:38:24.267147Z","iopub.status.idle":"2024-12-09T15:38:24.289268Z","shell.execute_reply.started":"2024-12-09T15:38:24.267114Z","shell.execute_reply":"2024-12-09T15:38:24.287782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mlp = MLPRegressor(   \n    hidden_layer_sizes=(100, 100),\n    activation='relu',\n    solver='adam',\n    learning_rate_init=0.001,\n    max_iter= 1000,\n    batch_size='auto',\n    early_stopping=True,\n    tol=1e-4,\n    verbose=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:24.290847Z","iopub.execute_input":"2024-12-09T15:38:24.291411Z","iopub.status.idle":"2024-12-09T15:38:24.301559Z","shell.execute_reply.started":"2024-12-09T15:38:24.291348Z","shell.execute_reply":"2024-12-09T15:38:24.299963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:24.303128Z","iopub.execute_input":"2024-12-09T15:38:24.303492Z","iopub.status.idle":"2024-12-09T15:38:24.442305Z","shell.execute_reply.started":"2024-12-09T15:38:24.30346Z","shell.execute_reply":"2024-12-09T15:38:24.44081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rf_model = RandomForestRegressor(\n    n_estimators=100, \n    max_depth=6, \n    n_jobs=-1,\n    verbose=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:24.444153Z","iopub.execute_input":"2024-12-09T15:38:24.444675Z","iopub.status.idle":"2024-12-09T15:38:24.450441Z","shell.execute_reply.started":"2024-12-09T15:38:24.444622Z","shell.execute_reply":"2024-12-09T15:38:24.449258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import StackingRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:24.453306Z","iopub.execute_input":"2024-12-09T15:38:24.453636Z","iopub.status.idle":"2024-12-09T15:38:24.468297Z","shell.execute_reply.started":"2024-12-09T15:38:24.453605Z","shell.execute_reply":"2024-12-09T15:38:24.466998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta_model = LinearRegression()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:24.470089Z","iopub.execute_input":"2024-12-09T15:38:24.47062Z","iopub.status.idle":"2024-12-09T15:38:24.481603Z","shell.execute_reply.started":"2024-12-09T15:38:24.470552Z","shell.execute_reply":"2024-12-09T15:38:24.480373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from catboost import CatBoostRegressor\n\ncatboost_params = {\n    'iterations': 1000,\n    'learning_rate': 0.05,\n    'depth': 6,\n    'l2_leaf_reg': 3,\n    'loss_function': 'RMSE',\n    'border_count': 32,\n    'thread_count': -1,\n    'early_stopping_rounds': 50,\n    'verbose': True\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:24.483327Z","iopub.execute_input":"2024-12-09T15:38:24.483747Z","iopub.status.idle":"2024-12-09T15:38:24.920395Z","shell.execute_reply.started":"2024-12-09T15:38:24.483713Z","shell.execute_reply":"2024-12-09T15:38:24.919123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_models = [\n    ('xgb', xgb_model), \n    ('lgb', lgb_model),\n    # ('rf', rf_model),\n    ('mlp', mlp)\n    # ('catboost', CatBoostRegressor(**catboost_params))\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:33.188356Z","iopub.execute_input":"2024-12-09T15:38:33.18879Z","iopub.status.idle":"2024-12-09T15:38:33.194464Z","shell.execute_reply.started":"2024-12-09T15:38:33.188745Z","shell.execute_reply":"2024-12-09T15:38:33.193228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = StackingRegressor(estimators=base_models, final_estimator=meta_model, n_jobs=-1, verbose=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:36.45181Z","iopub.execute_input":"2024-12-09T15:38:36.452242Z","iopub.status.idle":"2024-12-09T15:38:36.458716Z","shell.execute_reply.started":"2024-12-09T15:38:36.452206Z","shell.execute_reply":"2024-12-09T15:38:36.457352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.fit(X,y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T15:38:39.568475Z","iopub.execute_input":"2024-12-09T15:38:39.568837Z","iopub.status.idle":"2024-12-09T17:23:04.661751Z","shell.execute_reply.started":"2024-12-09T15:38:39.568808Z","shell.execute_reply":"2024-12-09T17:23:04.659381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(testProcessed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T17:23:04.666705Z","iopub.execute_input":"2024-12-09T17:23:04.668345Z","iopub.status.idle":"2024-12-09T17:24:16.856296Z","shell.execute_reply.started":"2024-12-09T17:23:04.668277Z","shell.execute_reply":"2024-12-09T17:24:16.852659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")\nsub[target] = np.expm1(y_pred)\nsub.to_csv(\"submission_updates.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T17:24:16.858827Z","iopub.execute_input":"2024-12-09T17:24:16.859386Z","iopub.status.idle":"2024-12-09T17:24:18.963115Z","shell.execute_reply.started":"2024-12-09T17:24:16.859325Z","shell.execute_reply":"2024-12-09T17:24:18.961783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}