{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Understand the Competition and Data ","metadata":{}},{"cell_type":"markdown","source":"<div style=\"border: 2px solid #999; border-radius: 5px; padding: 15px; color: inherit;\">\n  <h2 style=\"margin-top:0;\">Dataset Overview</h2>\n  \n  <p>\n    The dataset for this competition was generated by a deep learning model originally trained on the \n    <strong>Insurance Premium Prediction</strong> dataset. While the feature distributions are similar to \n    the source data, they aren’t identical—adding complexity and opportunity for experimentation.\n  </p>\n  \n  <p>\n    You are free to leverage the original dataset for comparison or potentially incorporate it into your \n    training pipeline to see if it boosts performance.\n  </p>\n  \n  <h3>Files Provided</h3>\n  <ul>\n    <li><strong>train.csv</strong><br>\n      Contains feature columns plus the target, <code>Premium Amount</code>.\n    </li>\n    <li><strong>test.csv</strong><br>\n      Contains the same features without <code>Premium Amount</code>; your task is to predict these values.\n    </li>\n    <li><strong>sample_submission.csv</strong><br>\n      A reference template (two columns: <code>Id</code> and <code>Premium Amount</code>) for your final predictions.\n    </li>\n  </ul>\n  \n  <p>\n    Because the data originated from a learned model, there may be intricate relationships among features. \n    Exploring how they compare to the original dataset could help refine your approach \n    to feature engineering and model tuning.\n  </p>\n</div>\n","metadata":{}},{"cell_type":"markdown","source":"## Initial look at the data","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import LabelEncoder","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-27T21:09:44.001115Z","iopub.execute_input":"2024-12-27T21:09:44.001516Z","iopub.status.idle":"2024-12-27T21:09:44.275486Z","shell.execute_reply.started":"2024-12-27T21:09:44.001487Z","shell.execute_reply":"2024-12-27T21:09:44.274344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train=pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest=pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T21:09:46.346181Z","iopub.execute_input":"2024-12-27T21:09:46.346946Z","iopub.status.idle":"2024-12-27T21:09:57.665578Z","shell.execute_reply.started":"2024-12-27T21:09:46.346902Z","shell.execute_reply":"2024-12-27T21:09:57.664627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:13:45.07781Z","iopub.execute_input":"2024-12-27T18:13:45.078153Z","iopub.status.idle":"2024-12-27T18:13:45.1035Z","shell.execute_reply.started":"2024-12-27T18:13:45.078126Z","shell.execute_reply":"2024-12-27T18:13:45.102492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:13:45.105054Z","iopub.execute_input":"2024-12-27T18:13:45.105406Z","iopub.status.idle":"2024-12-27T18:13:45.140732Z","shell.execute_reply.started":"2024-12-27T18:13:45.105377Z","shell.execute_reply":"2024-12-27T18:13:45.139471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.shape)\nprint(test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:13:45.141764Z","iopub.execute_input":"2024-12-27T18:13:45.142113Z","iopub.status.idle":"2024-12-27T18:13:45.159152Z","shell.execute_reply.started":"2024-12-27T18:13:45.142073Z","shell.execute_reply":"2024-12-27T18:13:45.158069Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Basic data checks","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:13:45.160249Z","iopub.execute_input":"2024-12-27T18:13:45.16064Z","iopub.status.idle":"2024-12-27T18:13:45.807483Z","shell.execute_reply.started":"2024-12-27T18:13:45.160611Z","shell.execute_reply":"2024-12-27T18:13:45.806596Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Exploratory Data Analysis (EDA)","metadata":{}},{"cell_type":"markdown","source":"## Misssing Data","metadata":{}},{"cell_type":"code","source":" def show_missing_data(data):  \n    #Let's look missing data\n    missing_data= data.isnull().sum().sort_values(ascending=False)\n    missing_data = missing_data[missing_data > 0]\n    data_types = train.dtypes[missing_data.index]\n    \n    # Combine into a DataFrame\n    missing_data_df = pd.DataFrame({'Missing Data': missing_data, 'Data Type': data_types})\n    \n    # Display the result\n    print(missing_data_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T21:10:10.57243Z","iopub.execute_input":"2024-12-27T21:10:10.572822Z","iopub.status.idle":"2024-12-27T21:10:10.578323Z","shell.execute_reply.started":"2024-12-27T21:10:10.572792Z","shell.execute_reply":"2024-12-27T21:10:10.577189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show_missing_data(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:13:45.816789Z","iopub.execute_input":"2024-12-27T18:13:45.817154Z","iopub.status.idle":"2024-12-27T18:13:46.459546Z","shell.execute_reply.started":"2024-12-27T18:13:45.817127Z","shell.execute_reply":"2024-12-27T18:13:46.458414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show_missing_data(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:13:46.460799Z","iopub.execute_input":"2024-12-27T18:13:46.461205Z","iopub.status.idle":"2024-12-27T18:13:46.882109Z","shell.execute_reply.started":"2024-12-27T18:13:46.461165Z","shell.execute_reply":"2024-12-27T18:13:46.880813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:13:46.883084Z","iopub.execute_input":"2024-12-27T18:13:46.883341Z","iopub.status.idle":"2024-12-27T18:13:47.588805Z","shell.execute_reply.started":"2024-12-27T18:13:46.88332Z","shell.execute_reply":"2024-12-27T18:13:47.587921Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## High-Correlation Check","metadata":{}},{"cell_type":"code","source":"numeric_cols = train.select_dtypes(include=[np.number]).columns\ncorr_matrix = train[numeric_cols].corr()\nplt.figure(figsize=(12, 8))\nsns.heatmap(corr_matrix, annot=False, cmap='coolwarm')\nplt.title(\"Correlation Heatmap\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:13:47.589763Z","iopub.execute_input":"2024-12-27T18:13:47.590061Z","iopub.status.idle":"2024-12-27T18:13:48.41888Z","shell.execute_reply.started":"2024-12-27T18:13:47.590036Z","shell.execute_reply":"2024-12-27T18:13:48.417661Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Data Cleaning & Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Filling missings data","metadata":{}},{"cell_type":"code","source":"def impute_missing_data(data):\n    \n    categorical = data.select_dtypes(include='object').columns\n    for i in categorical:\n        \n        data[i] = data[i].fillna('Unknown')\n    \n   \n    numeric = data.select_dtypes(exclude='object').columns\n    for i in numeric:\n        \n        data[i] = data[i].fillna(data[i].mean())\n        \n    return data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T21:10:17.321084Z","iopub.execute_input":"2024-12-27T21:10:17.321492Z","iopub.status.idle":"2024-12-27T21:10:17.327078Z","shell.execute_reply.started":"2024-12-27T21:10:17.32146Z","shell.execute_reply":"2024-12-27T21:10:17.32576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train=impute_missing_data(train)\ntest=impute_missing_data(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T21:10:19.905173Z","iopub.execute_input":"2024-12-27T21:10:19.905566Z","iopub.status.idle":"2024-12-27T21:10:22.097991Z","shell.execute_reply.started":"2024-12-27T21:10:19.905538Z","shell.execute_reply":"2024-12-27T21:10:22.096785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show_missing_data(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T21:10:24.458887Z","iopub.execute_input":"2024-12-27T21:10:24.459314Z","iopub.status.idle":"2024-12-27T21:10:25.107943Z","shell.execute_reply.started":"2024-12-27T21:10:24.459279Z","shell.execute_reply":"2024-12-27T21:10:25.106807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show_missing_data(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:13:50.950951Z","iopub.execute_input":"2024-12-27T18:13:50.951243Z","iopub.status.idle":"2024-12-27T18:13:51.380071Z","shell.execute_reply.started":"2024-12-27T18:13:50.951218Z","shell.execute_reply":"2024-12-27T18:13:51.379112Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Feature Engineering ","metadata":{}},{"cell_type":"markdown","source":"## To separate the Policy Start Date feature into day, month and year.","metadata":{}},{"cell_type":"code","source":"def features_time(data):\n    data['Policy Start Date'] = pd.to_datetime(data['Policy Start Date'])\n\n    data['Policy_Year'] = data['Policy Start Date'].dt.year\n    data['Policy_Month'] = data['Policy Start Date'].dt.month\n    data['Policy_Day'] = data['Policy Start Date'].dt.day\n   \n    data = data.drop('Policy Start Date', axis=1)\n\n    return data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T21:10:28.170839Z","iopub.execute_input":"2024-12-27T21:10:28.171248Z","iopub.status.idle":"2024-12-27T21:10:28.178079Z","shell.execute_reply.started":"2024-12-27T21:10:28.171212Z","shell.execute_reply":"2024-12-27T21:10:28.176735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train=features_time(train)\ntest=features_time(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T21:10:30.162513Z","iopub.execute_input":"2024-12-27T21:10:30.162885Z","iopub.status.idle":"2024-12-27T21:10:31.636714Z","shell.execute_reply.started":"2024-12-27T21:10:30.162857Z","shell.execute_reply":"2024-12-27T21:10:31.635589Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Encoding categorical data","metadata":{}},{"cell_type":"code","source":"def encode_categorical(data):\n    cat=data.select_dtypes(include='object').columns\n    le=LabelEncoder()\n    for col in cat:\n        \n     data[col]=le.fit_transform(data[col])\n    return data  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T21:10:34.943575Z","iopub.execute_input":"2024-12-27T21:10:34.943974Z","iopub.status.idle":"2024-12-27T21:10:34.949521Z","shell.execute_reply.started":"2024-12-27T21:10:34.94394Z","shell.execute_reply":"2024-12-27T21:10:34.948218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train=encode_categorical(train)\ntest=encode_categorical(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T21:10:36.816701Z","iopub.execute_input":"2024-12-27T21:10:36.81709Z","iopub.status.idle":"2024-12-27T21:10:41.010398Z","shell.execute_reply.started":"2024-12-27T21:10:36.817057Z","shell.execute_reply":"2024-12-27T21:10:41.009311Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Split the data into training and validation","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T21:10:44.200779Z","iopub.execute_input":"2024-12-27T21:10:44.201161Z","iopub.status.idle":"2024-12-27T21:10:44.205943Z","shell.execute_reply.started":"2024-12-27T21:10:44.201132Z","shell.execute_reply":"2024-12-27T21:10:44.20463Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop the ID column\ntrain_data = train.drop(columns=['id'])\n\n# Separate Features and Target\nX = train_data.drop(columns=['Premium Amount'])\ny = train_data['Premium Amount']\n\n# Train-Test Split (for validation during training)\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\n\n# Prepare test data (Drop ID and ensure consistency with train features)\nX_test = test.drop(columns=['id'])\nid_test=test['id']\n\n\nX_train.columns = [col.replace(\" \", \"_\") for col in X_train.columns]\nX_val.columns = [col.replace(\" \", \"_\") for col in X_val.columns]\nX_test.columns = [col.replace(\" \", \"_\") for col in X_test.columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T21:12:20.657519Z","iopub.execute_input":"2024-12-27T21:12:20.65794Z","iopub.status.idle":"2024-12-27T21:12:21.49212Z","shell.execute_reply.started":"2024-12-27T21:12:20.657912Z","shell.execute_reply":"2024-12-27T21:12:21.491084Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Baseline Model","metadata":{}},{"cell_type":"markdown","source":"## Establish a simple baseline to measure progress","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score, roc_auc_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T21:12:25.956711Z","iopub.execute_input":"2024-12-27T21:12:25.957133Z","iopub.status.idle":"2024-12-27T21:12:25.962126Z","shell.execute_reply.started":"2024-12-27T21:12:25.957099Z","shell.execute_reply":"2024-12-27T21:12:25.960991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\n\n# Initialize a baseline regression model\nbaseline_model = RandomForestRegressor(random_state=42)\n\n# Train the model\nbaseline_model.fit(X_train, y_train)\n\n# Predict on the validation set\ny_pred = baseline_model.predict(X_val)\n\n# Evaluate the model using RMSE (Root Mean Squared Error)\nrmse = mean_squared_error(y_val, y_pred, squared=False)\n\nprint(f\"Baseline Model RMSE: {rmse:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:13:57.692299Z","iopub.execute_input":"2024-12-27T18:13:57.69265Z","iopub.status.idle":"2024-12-27T18:41:34.481442Z","shell.execute_reply.started":"2024-12-27T18:13:57.692615Z","shell.execute_reply":"2024-12-27T18:41:34.479736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predictions vs. Actual Values\nplt.figure(figsize=(10, 6))\nsns.scatterplot(x=y_val, y=y_pred, alpha=0.6)\nplt.plot([y_val.min(), y_val.max()], [y_val.min(), y_val.max()], color='red', linestyle='--')\nplt.xlabel('Actual Premium Amount')\nplt.ylabel('Predicted Premium Amount')\nplt.title('Predictions vs Actual Values')\nplt.show()\n\n# Residual Analysis\nresiduals = y_val - y_pred\n\nplt.figure(figsize=(10, 6))\nsns.histplot(residuals, kde=True, bins=30, color='blue', alpha=0.7)\nplt.axvline(x=0, color='red', linestyle='--')\nplt.xlabel('Residuals (Actual - Predicted)')\nplt.ylabel('Frequency')\nplt.title('Residual Distribution')\nplt.show()\n\n# Residuals vs. Predicted\nplt.figure(figsize=(10, 6))\nsns.scatterplot(x=y_pred, y=residuals, alpha=0.6)\nplt.axhline(y=0, color='red', linestyle='--')\nplt.xlabel('Predicted Premium Amount')\nplt.ylabel('Residuals')\nplt.title('Residuals vs Predicted')\nplt.show()\n\n# Calculate metrics for documentation\nrmse = mean_squared_error(y_val, y_pred, squared=False)\nmae = np.mean(np.abs(residuals))\n\nprint(f\"Baseline Model RMSE: {rmse:.4f}\")\nprint(f\"Baseline Model MAE (Mean Absolute Error): {mae:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:41:34.482964Z","iopub.execute_input":"2024-12-27T18:41:34.483373Z","iopub.status.idle":"2024-12-27T18:41:37.864278Z","shell.execute_reply.started":"2024-12-27T18:41:34.483342Z","shell.execute_reply":"2024-12-27T18:41:37.863175Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Analysis of Results\n\n## Predictions vs. Actual Values\nThe predicted values closely cluster around lower premium amounts but deviate significantly for higher premiums. This indicates that the model struggles with higher ranges, possibly due to insufficient representation in the training data or feature inadequacy.\n\n---\n\n## Residual Distribution\nThe residuals form a roughly bell-shaped curve but are slightly skewed. This could indicate unaddressed non-linear relationships in the data or an opportunity for feature transformation.\n\n---\n\n## Residuals vs. Predicted\nThere is a noticeable pattern in the residuals, with a downward slope indicating systematic errors. This suggests the model isn't fully capturing some aspect of the relationship between features and the target variable.\n\n---\n\n## Metrics\n- **RMSE:** 844.63 - Indicates the model’s predictions have a significant average error.\n- **MAE:** 644.88 - Highlights the model is generally off by this value, showing room for improvement, especially for outliers.\n","metadata":{}},{"cell_type":"markdown","source":"## Feature Importance Analysis","metadata":{}},{"cell_type":"code","source":"# Get feature importances from the trained RandomForestRegressor model\nimportances = baseline_model.feature_importances_\n\n# Sort the feature importances in descending order\nindices = np.argsort(importances)[::-1]\n\n# Plotting feature importance\nplt.figure(figsize=(10, 6))\nplt.title(\"Feature Importance for Baseline Model (Random Forest)\")\nplt.bar(range(len(importances)), importances[indices], align=\"center\")\nplt.xticks(range(len(importances)), X_train.columns[indices], rotation=90)\nplt.xlim([-1, len(importances)])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:41:37.865304Z","iopub.execute_input":"2024-12-27T18:41:37.865634Z","iopub.status.idle":"2024-12-27T18:41:39.588105Z","shell.execute_reply.started":"2024-12-27T18:41:37.865599Z","shell.execute_reply":"2024-12-27T18:41:39.587041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_importance_df = pd.DataFrame({\n    'Feature': X_train.columns,\n    'Importance': importances\n})\n\n# Sort the DataFrame by importance (descending order)\nfeature_importance_df = feature_importance_df.sort_values(by='Importance', ascending=False)\n\n# Display the feature importance table\nprint(feature_importance_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:41:39.589169Z","iopub.execute_input":"2024-12-27T18:41:39.589486Z","iopub.status.idle":"2024-12-27T18:41:39.597842Z","shell.execute_reply.started":"2024-12-27T18:41:39.589451Z","shell.execute_reply":"2024-12-27T18:41:39.59684Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Looking at the feature importance\n\nTop Features:\n\nHealth_Score (0.1318): This is the most important feature in the model, indicating that it plays a significant role in predicting the target variable. It seems to have the highest predictive power.\nAnnual_Income (0.1254): This feature is also highly important, suggesting that income levels are a strong predictor of the target variable.\nCredit_Score (0.0968): The credit score also contributes significantly to the model's predictions, which is typical for many financial or insurance-related models.\n___\nModerate Importance:\n\nAge (0.0788): Age has a moderate level of importance, which might be expected in insurance or financial models, as age can affect health or insurance premiums.\nPolicy_Day (0.0715) and Vehicle_Age (0.0627): These features also show moderate importance, which could be relevant in a policy-related model where the type of policy and age of the vehicle matter.\nInsurance_Duration (0.0444): This might indicate the length of time a customer has been insured, which could influence their claim behavior or risk profile.\n___\nLower Importance:\n\nPolicy_Year (0.0387): This feature has a relatively low importance, meaning it doesn’t contribute as much to the model's predictive power.\nNumber_of_Dependents (0.0340): Although it has some importance, it might not be as crucial as income or health score.\nPolicy_Month (0.0307): Policy-related temporal features like this may not be as important in the model compared to other factors like credit or health.\n___\nLeast Important:\n\nGender (0.0108) and Smoking_Status (0.0108): These features are the least important based on this model’s performance. They might still be somewhat useful but don’t have a significant impact on the predictions.\nOther features like Location (0.0190), Property_Type (0.0190), Policy_Type (0.0189), etc., also have relatively low importance compared to the top features.","metadata":{}},{"cell_type":"markdown","source":"## Dropping Features\nLet's try reducing the low importance inputs, will the accuracy of our model increase?","metadata":{}},{"cell_type":"code","source":"features_to_drop = ['Gender', 'Smoking_Status']\nX_train_reduced = X_train.drop(columns=features_to_drop)\nX_val_reduced = X_val.drop(columns=features_to_drop)\nX_test_reduced = X_test.drop(columns=features_to_drop)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Retrain model on reduced dataset\nbaseline_model.fit(X_train_reduced, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:43:24.332174Z","iopub.execute_input":"2024-12-27T18:43:24.33256Z","iopub.status.idle":"2024-12-27T19:09:22.479365Z","shell.execute_reply.started":"2024-12-27T18:43:24.332519Z","shell.execute_reply":"2024-12-27T19:09:22.477068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on the validation set\ny_pred = baseline_model.predict(X_val_reduced)\n\n# Evaluate the model using RMSE (Root Mean Squared Error)\nrmse = mean_squared_error(y_val, y_pred, squared=False)\n\nprint(f\"Baseline Model RMSE: {rmse:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T19:10:38.238461Z","iopub.execute_input":"2024-12-27T19:10:38.239032Z","iopub.status.idle":"2024-12-27T19:11:04.059781Z","shell.execute_reply.started":"2024-12-27T19:10:38.238982Z","shell.execute_reply":"2024-12-27T19:11:04.058582Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Result\nWe can see RMSE error is incresed. We will left the features.\n\nLet's try other algoritms.","metadata":{}},{"cell_type":"markdown","source":"# 6. Ensembling","metadata":{}},{"cell_type":"markdown","source":"## We will implements and evaluates four different regression models: GradientBoosting, XGBoost, LightGBM, and CatBoost. ","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingRegressor\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.metrics import mean_squared_error\n\n# Initialize models\nmodels = {\n    'GradientBoosting': GradientBoostingRegressor(random_state=42),\n    'XGBoost': XGBRegressor(random_state=42),\n    'LightGBM': LGBMRegressor(random_state=42),\n    'CatBoost': CatBoostRegressor(verbose=0, random_state=42)\n}\n\n\n# Train and evaluate models\nresults = {}\nfor name, model in models.items():\n    # Train the model\n    model.fit(X_train, y_train)\n    \n    # Predict on validation set\n    y_val_pred = model.predict(X_val)\n    \n    # Evaluate the model using Mean Squared Error\n    mse = mean_squared_error(y_val, y_val_pred)\n    results[name] = mse\n    print(f\"{name}: MSE = {mse}\")\n\n# Compare results\nprint(\"\\nModel Performance:\")\nfor name, mse in results.items():\n    print(f\"{name}: MSE = {mse}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T21:13:37.062166Z","iopub.execute_input":"2024-12-27T21:13:37.062587Z","iopub.status.idle":"2024-12-27T21:20:51.564561Z","shell.execute_reply.started":"2024-12-27T21:13:37.062559Z","shell.execute_reply":"2024-12-27T21:20:51.562982Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Conclusion\nWhile LightGBM emerged as the best-performing model with the lowest MSE, further optimization and exploration of ensemble techniques could enhance predictive accuracy.","metadata":{}},{"cell_type":"markdown","source":"# 7. Model Tuning and Validation","metadata":{}},{"cell_type":"markdown","source":"## Optimize model performance through parameter tuning and validation.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import RandomizedSearchCV\nimport lightgbm as lgb\n\n# Define the parameter grid\nparam_dist = {\n    'n_estimators': [100, 200, 300],\n    'learning_rate': [0.01, 0.05, 0.1, 0.2],\n    'max_depth': [-1, 3, 5, 7],\n    'num_leaves': [31, 50, 70],\n    'min_child_samples': [10, 20, 30],\n    'subsample': [0.6, 0.8, 1.0],\n    'colsample_bytree': [0.6, 0.8, 1.0]\n}\n\n# Initialize the model\nlgb_model = lgb.LGBMRegressor(random_state=42)\n\n# Perform RandomizedSearchCV\nrandom_search = RandomizedSearchCV(\n    estimator=lgb_model,\n    param_distributions=param_dist,\n    n_iter=50,  # Number of parameter combinations to try\n    scoring='neg_mean_squared_error',\n    cv=5,\n    random_state=42,\n    n_jobs=-1\n)\n\nrandom_search.fit(X_train, y_train)\n\n# Best parameters and score\nprint(\"Best Parameters:\", random_search.best_params_)\nprint(\"Best CV Score (MSE):\", -random_search.best_score_)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Implement LightGBM with the best parameters from hyperparameter tuning","metadata":{}},{"cell_type":"code","source":"best_params = {\n    'colsample_bytree': 0.6,\n    'learning_rate': 0.1366666666666667,\n    'max_depth': 5,\n    'min_child_samples': 20,\n    'n_estimators': 500,\n    'num_leaves': 20,\n    'reg_alpha': 1,\n    'reg_lambda': 0,\n    'subsample': 0.7,\n    'random_state': 42\n}\n\n# Initialize LightGBM with the best parameters\nbest_lgbm = LGBMRegressor(**best_params)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T21:59:29.044233Z","iopub.execute_input":"2024-12-27T21:59:29.044712Z","iopub.status.idle":"2024-12-27T21:59:53.633411Z","shell.execute_reply.started":"2024-12-27T21:59:29.044672Z","shell.execute_reply":"2024-12-27T21:59:53.632069Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 8. Final Training and Submission","metadata":{}},{"cell_type":"markdown","source":"## Final training.","metadata":{}},{"cell_type":"code","source":"# Train the model\nbest_lgbm.fit(X, y)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Make predict and save the submission","metadata":{}},{"cell_type":"code","source":"predicted = best_lgbm.predict(X_test)\npredicted = predicted.astype(float)\n\nsubmission = pd.DataFrame({'id': id_test, 'Premium Amount': predicted})\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T21:59:58.46705Z","iopub.execute_input":"2024-12-27T21:59:58.467466Z","iopub.status.idle":"2024-12-27T22:00:10.652532Z","shell.execute_reply.started":"2024-12-27T21:59:58.467427Z","shell.execute_reply":"2024-12-27T22:00:10.651377Z"}},"outputs":[],"execution_count":null}]}