{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Importing python libraries","metadata":{}},{"cell_type":"code","source":"\nimport numpy as np \nimport pandas as pd\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:29:37.027226Z","iopub.execute_input":"2025-08-22T06:29:37.027864Z","iopub.status.idle":"2025-08-22T06:29:37.036772Z","shell.execute_reply.started":"2025-08-22T06:29:37.027836Z","shell.execute_reply":"2025-08-22T06:29:37.035826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest_data = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:29:38.330458Z","iopub.execute_input":"2025-08-22T06:29:38.331197Z","iopub.status.idle":"2025-08-22T06:29:46.99859Z","shell.execute_reply.started":"2025-08-22T06:29:38.331165Z","shell.execute_reply":"2025-08-22T06:29:46.997475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:29:46.999924Z","iopub.execute_input":"2025-08-22T06:29:47.000186Z","iopub.status.idle":"2025-08-22T06:29:47.709856Z","shell.execute_reply.started":"2025-08-22T06:29:47.000166Z","shell.execute_reply":"2025-08-22T06:29:47.708796Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Data Cleaning & Transformation","metadata":{}},{"cell_type":"code","source":"train_data.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:29:51.139728Z","iopub.execute_input":"2025-08-22T06:29:51.14002Z","iopub.status.idle":"2025-08-22T06:29:51.83697Z","shell.execute_reply.started":"2025-08-22T06:29:51.14Z","shell.execute_reply":"2025-08-22T06:29:51.835943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.nunique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:29:51.838258Z","iopub.execute_input":"2025-08-22T06:29:51.838611Z","iopub.status.idle":"2025-08-22T06:29:52.991475Z","shell.execute_reply.started":"2025-08-22T06:29:51.838581Z","shell.execute_reply":"2025-08-22T06:29:52.990559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.drop(columns=['id', 'Policy Start Date'], inplace=True)\ntest_data.drop(columns=['id', 'Policy Start Date'], inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:29:54.864119Z","iopub.execute_input":"2025-08-22T06:29:54.864467Z","iopub.status.idle":"2025-08-22T06:29:55.201594Z","shell.execute_reply.started":"2025-08-22T06:29:54.864441Z","shell.execute_reply":"2025-08-22T06:29:55.200612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# extract X and y from training data\nX = train_data.drop(['Premium Amount'], axis=1)\ny = train_data['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:29:57.199239Z","iopub.execute_input":"2025-08-22T06:29:57.199599Z","iopub.status.idle":"2025-08-22T06:29:57.367609Z","shell.execute_reply.started":"2025-08-22T06:29:57.199573Z","shell.execute_reply":"2025-08-22T06:29:57.366596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# extract the numerical and categorical columns\nnumeric_cols = X.select_dtypes(include=\"number\").columns\ncategorical_cols = X.select_dtypes(include=\"object\").columns\n\nprint(\"Numerical Columns are:\\n\", numeric_cols)\nprint()\nprint(\"Categorical Columns are: \\n\", categorical_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:29:57.462915Z","iopub.execute_input":"2025-08-22T06:29:57.463979Z","iopub.status.idle":"2025-08-22T06:29:57.555273Z","shell.execute_reply.started":"2025-08-22T06:29:57.463949Z","shell.execute_reply":"2025-08-22T06:29:57.554278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# fill the missing values\nfrom sklearn.impute import SimpleImputer\n\n# Numeric imputer (median)\nnum_imputer = SimpleImputer(strategy='median')\nX[numeric_cols] = num_imputer.fit_transform(X[numeric_cols])\n\n# Categorical imputer (mode)\ncat_imputer = SimpleImputer(strategy='most_frequent')\nX[categorical_cols] = cat_imputer.fit_transform(X[categorical_cols])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:29:59.906969Z","iopub.execute_input":"2025-08-22T06:29:59.907281Z","iopub.status.idle":"2025-08-22T06:30:03.733661Z","shell.execute_reply.started":"2025-08-22T06:29:59.907257Z","shell.execute_reply":"2025-08-22T06:30:03.732488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# now, checking null counts of every column\nX.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:30:03.735129Z","iopub.execute_input":"2025-08-22T06:30:03.736Z","iopub.status.idle":"2025-08-22T06:30:04.358016Z","shell.execute_reply.started":"2025-08-22T06:30:03.735969Z","shell.execute_reply":"2025-08-22T06:30:04.357073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# a informative summary\nfor col in categorical_cols:\n    print(col)\n    print(\"Uniques count: \", X[col].nunique())\n    print(\"The values are : \\n\", pd.DataFrame(X[col].value_counts().reset_index()))\n    print()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:30:06.537777Z","iopub.execute_input":"2025-08-22T06:30:06.538098Z","iopub.status.idle":"2025-08-22T06:30:08.087157Z","shell.execute_reply.started":"2025-08-22T06:30:06.538075Z","shell.execute_reply":"2025-08-22T06:30:08.08599Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Model Preprocessings:","metadata":{}},{"cell_type":"code","source":"# split data into train and test parts\nfrom sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.15, random_state=42)\n\nprint(\"X_train shape: \", X_train.shape)\nprint(\"y_train shape: \", y_train.shape)\nprint(\"X_test shape: \", X_test.shape)\nprint(\"y_test shape: \", y_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:30:11.436762Z","iopub.execute_input":"2025-08-22T06:30:11.437641Z","iopub.status.idle":"2025-08-22T06:30:12.193605Z","shell.execute_reply.started":"2025-08-22T06:30:11.437599Z","shell.execute_reply":"2025-08-22T06:30:12.192588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:30:13.709284Z","iopub.execute_input":"2025-08-22T06:30:13.70964Z","iopub.status.idle":"2025-08-22T06:30:13.734933Z","shell.execute_reply.started":"2025-08-22T06:30:13.709617Z","shell.execute_reply":"2025-08-22T06:30:13.733763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# filling the categorical values with corresponding numerical values\nfrom sklearn.preprocessing import OrdinalEncoder\noe = OrdinalEncoder()\n\nX_train[categorical_cols] = oe.fit_transform(X_train[categorical_cols])\nX_test[categorical_cols] = oe.transform(X_test[categorical_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:30:13.958315Z","iopub.execute_input":"2025-08-22T06:30:13.95874Z","iopub.status.idle":"2025-08-22T06:30:17.147819Z","shell.execute_reply.started":"2025-08-22T06:30:13.958717Z","shell.execute_reply":"2025-08-22T06:30:17.146717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# scaling the numerical cols types data\nfrom sklearn.preprocessing import StandardScaler\nsc = StandardScaler()\n\nX_train[numeric_cols] = sc.fit_transform(X_train[numeric_cols])\nX_test[numeric_cols] = sc.transform(X_test[numeric_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:30:17.149498Z","iopub.execute_input":"2025-08-22T06:30:17.149851Z","iopub.status.idle":"2025-08-22T06:30:17.396066Z","shell.execute_reply.started":"2025-08-22T06:30:17.149827Z","shell.execute_reply":"2025-08-22T06:30:17.39506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:30:17.397034Z","iopub.execute_input":"2025-08-22T06:30:17.397311Z","iopub.status.idle":"2025-08-22T06:30:17.425298Z","shell.execute_reply.started":"2025-08-22T06:30:17.397282Z","shell.execute_reply":"2025-08-22T06:30:17.424277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"as the data is very huge, so I will draw a sample out of it, and train models on that sample data.","metadata":{}},{"cell_type":"code","source":"Xy_sample = X_train.join(y_train).sample(20000, random_state=42)\nX_train_sample = Xy_sample.drop(\"Premium Amount\", axis=1)\ny_train_sample = Xy_sample[\"Premium Amount\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:30:18.309927Z","iopub.execute_input":"2025-08-22T06:30:18.310661Z","iopub.status.idle":"2025-08-22T06:30:18.662748Z","shell.execute_reply.started":"2025-08-22T06:30:18.310627Z","shell.execute_reply":"2025-08-22T06:30:18.661816Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Model Trainings & Evaluations:","metadata":{}},{"cell_type":"code","source":"# developing normal models\nfrom sklearn.linear_model import LinearRegression, Lasso, Ridge, ElasticNet\nfrom sklearn.metrics import mean_squared_error\nimport numpy as np\n\nmodels = {\n    \"LinearRegression\": LinearRegression(),\n    \"Lasso\": Lasso(alpha=0.001, max_iter=10000),\n    \"Ridge\": Ridge(alpha=1.0),\n    \"ElasticNet\": ElasticNet(alpha=0.001, l1_ratio=0.5, max_iter=10000)\n}\n\nresults = {}\n\nfor name, model in models.items():\n    model.fit(X_train_sample, y_train_sample)\n    y_pred = model.predict(X_test)\n    rmse = np.sqrt(mean_squared_error(y_test, y_pred))\n    results[name] = rmse\n\n# Print results\nfor model, rmse in results.items():\n    print(f\"{model}: RMSE = {rmse:.2f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:30:20.929846Z","iopub.execute_input":"2025-08-22T06:30:20.930195Z","iopub.status.idle":"2025-08-22T06:30:21.055846Z","shell.execute_reply.started":"2025-08-22T06:30:20.930172Z","shell.execute_reply":"2025-08-22T06:30:21.052407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeRegressor\nfrom sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor\n\nmodels = {\n    \"DecisionTree\": DecisionTreeRegressor(random_state=42, max_depth=10),\n    \"RandomForest\": RandomForestRegressor(random_state=42, n_estimators=100, max_depth=10),\n    \"GradientBoosting\": GradientBoostingRegressor(random_state=42, n_estimators=100, learning_rate=0.1, max_depth=5)\n}\n\nresults = {}\n\nfor name, model in models.items():\n    # model training on sample data\n    model.fit(X_train_sample, y_train_sample)\n\n    # prediction on the whole test data\n    y_pred = model.predict(X_test)\n    rmse = np.sqrt(mean_squared_error(y_test, y_pred))\n    results[name] = rmse\n\n# Print results\nfor model, rmse in results.items():\n    print(f\"{model}: RMSE = {rmse:.2f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:30:43.530909Z","iopub.execute_input":"2025-08-22T06:30:43.531226Z","iopub.status.idle":"2025-08-22T06:31:00.074798Z","shell.execute_reply.started":"2025-08-22T06:30:43.531206Z","shell.execute_reply":"2025-08-22T06:31:00.073797Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### RMSE results shows:\n\n1. **Random Forest** → 854.54 (lowest RMSE → best so far)\n\n2. **Gradient Boosting** → 858.54 (close second)\n\n3. **Decision Tree** → 924.83 (overfits, worse)\n\nLinear/Lasso/Ridge/ElasticNet → ~863.25 (all similar, not great)\n\n### Conclusion:\n\nRandom Forest currently performs best.\n\nGradient Boosting is also strong and may outperform RF with tuning (n_estimators, learning_rate, max_depth).\n\nLinear models don’t capture non-linearities in your data.\n\nSo the “perfect” one (for now) is Random Forest, but we should do hyperparameter tuning (GridSearchCV/Optuna) on Random Forest & Gradient Boosting for best results.","metadata":{}},{"cell_type":"code","source":"# finding the best parameters of both best algorithms using GridSearchCV\n\nfrom sklearn.model_selection import GridSearchCV\n\n# Models and param grids\nmodels = {\n    \"RandomForest\": (RandomForestRegressor(random_state=42, n_jobs=-1),\n                     {\n                         'n_estimators': [100, 200],\n                         'max_depth': [10, 20, None],\n                         'min_samples_split': [2, 5],\n                         'min_samples_leaf': [1, 2],\n                         'max_features': ['sqrt', 'log2']\n                     }),\n    \"GradientBoosting\": (GradientBoostingRegressor(random_state=42),\n                         {\n                             'n_estimators': [100, 200],\n                             'learning_rate': [0.05, 0.1],\n                             'max_depth': [3, 5],\n                             'min_samples_split': [2, 5],\n                             'min_samples_leaf': [1, 2]\n                         })\n}\n\nresults = {}\n\nfor name, (model, params) in models.items():\n    grid = GridSearchCV(model, params, cv=3,\n                        scoring='neg_root_mean_squared_error',\n                        n_jobs=-1, verbose=1)\n    grid.fit(X_train_sample, y_train_sample)\n    results[name] = {\"best_params\": grid.best_params_, \"best_rmse\": -grid.best_score_}\n\n# Print results\nfor name, res in results.items():\n    print(f\"{name}: Best RMSE = {res['best_rmse']:.2f}, Params = {res['best_params']}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:31:12.142204Z","iopub.execute_input":"2025-08-22T06:31:12.142974Z","iopub.status.idle":"2025-08-22T06:37:34.473065Z","shell.execute_reply.started":"2025-08-22T06:31:12.142946Z","shell.execute_reply":"2025-08-22T06:37:34.472188Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Since GridSearchCV uses cross-validation, it shows the average RMSE, which may not reflect the true performance**. To get the real result, we need to retrain the model with the best parameters on the full training data and then evaluate on the test set.","metadata":{}},{"cell_type":"code","source":"\n# re-building the both best models with best parameters, to get true results.\n\n# Best params from GridSearchCV\nbest_rf_params = {'max_depth': 10, 'max_features': 'sqrt', 'min_samples_leaf': 2, \n                  'min_samples_split': 5, 'n_estimators': 200}\n\nbest_gb_params = {'learning_rate': 0.05, 'max_depth': 5, 'min_samples_leaf': 2, \n                  'min_samples_split': 5, 'n_estimators': 100}\n\n# Refit models on full training set\nrf_best = RandomForestRegressor(**best_rf_params, random_state=42)\nrf_best.fit(X_train_sample, y_train_sample)\n\ngb_best = GradientBoostingRegressor(**best_gb_params, random_state=42)\ngb_best.fit(X_train_sample, y_train_sample)\n\n# Evaluate on test set\nrf_preds = rf_best.predict(X_test)\ngb_preds = gb_best.predict(X_test)\n\nrf_rmse = np.sqrt(mean_squared_error(y_test, rf_preds))\ngb_rmse = np.sqrt(mean_squared_error(y_test, gb_preds))\n\nprint(f\"Final RandomForest RMSE on Test set: {rf_rmse:.2f}\")\nprint(f\"Final GradientBoosting RMSE on Test set: {gb_rmse:.2f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-22T06:38:26.447951Z","iopub.execute_input":"2025-08-22T06:38:26.448272Z","iopub.status.idle":"2025-08-22T06:38:40.602553Z","shell.execute_reply.started":"2025-08-22T06:38:26.448247Z","shell.execute_reply":"2025-08-22T06:38:40.601571Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Results**: \n\nRandomForest RMSE = 858.42, GradientBoosting RMSE = 856.63.\n\nThis confirms that our analysis and observations led us to the best result.","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Conclusion:\n\n**GradientBoosting** is selected as the final model since it **achieved the lowest RMSE** (856.63 vs 858.42 for RandomForest) and generally provides more consistent performance by reducing variance compared to RandomForest.","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}