{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:06:39.49941Z","iopub.execute_input":"2024-12-03T06:06:39.499876Z","iopub.status.idle":"2024-12-03T06:06:39.509487Z","shell.execute_reply.started":"2024-12-03T06:06:39.499838Z","shell.execute_reply":"2024-12-03T06:06:39.508168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.core.display import display, HTML\n# Add custom CSS\ndisplay(HTML(\"\"\"\n<style>\n    body {\n        background-color: #1e1e1e; /* Dark background */\n        color: #e0e0e0; /* Light text color for visibility */\n        font-family: 'Arial', sans-serif;\n    }\n    .jp-Notebook {\n        background-color: #2d2d2d; /* Slightly lighter dark for notebook */\n        color: #e0e0e0; /* Light text color */\n    }\n    table.dataframe {\n        border: 1px solid #444;\n        border-collapse: collapse;\n    }\n    table.dataframe th, table.dataframe td {\n        border: 1px solid #444;\n        padding: 5px;\n        text-align: left;\n        background-color: #333; /* Dark table background */\n        color: #e0e0e0; /* Light text */\n    }\n</style>\n\"\"\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:06:39.9786Z","iopub.execute_input":"2024-12-03T06:06:39.979564Z","iopub.status.idle":"2024-12-03T06:06:39.987963Z","shell.execute_reply.started":"2024-12-03T06:06:39.979515Z","shell.execute_reply":"2024-12-03T06:06:39.986731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Data manipulation\nimport numpy as np\nimport pandas as pd\n\n# Data visualization\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Machine learning\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import KNNImputer\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.base import BaseEstimator, TransformerMixin\n\nfrom bokeh.plotting import figure,show\nfrom bokeh.io import output_notebook\nfrom bokeh.palettes import Viridis256\n\nfrom scipy import stats\nfrom scipy.stats import norm\nfrom sklearn.preprocessing import StandardScaler\nimport gc\n# Suppress warnings\nimport warnings\nwarnings.filterwarnings('ignore')\n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:06:40.425412Z","iopub.execute_input":"2024-12-03T06:06:40.425846Z","iopub.status.idle":"2024-12-03T06:06:40.437071Z","shell.execute_reply.started":"2024-12-03T06:06:40.42579Z","shell.execute_reply":"2024-12-03T06:06:40.435824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install summarytools\nfrom summarytools import dfSummary","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:06:40.439286Z","iopub.execute_input":"2024-12-03T06:06:40.439663Z","iopub.status.idle":"2024-12-03T06:06:50.607672Z","shell.execute_reply.started":"2024-12-03T06:06:40.439627Z","shell.execute_reply":"2024-12-03T06:06:50.606114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Set sns color values\ncustom_palette=sns.color_palette(\"coolwarm\",as_cmap=True)\nsns.set_theme(style=\"whitegrid\",palette=\"coolwarm\")\n\n#set matplotlib style\nplt.style.use('ggplot')# Try styles like 'fivethirtyeight', 'bmh'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:06:50.610554Z","iopub.execute_input":"2024-12-03T06:06:50.611026Z","iopub.status.idle":"2024-12-03T06:06:50.623692Z","shell.execute_reply.started":"2024-12-03T06:06:50.610977Z","shell.execute_reply":"2024-12-03T06:06:50.622479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.set_theme(\n    context='notebook',\n    style='darkgrid',\n    palette='muted',\n    font='Arial',\n    font_scale=1.2,\n    rc={\n        \"axes.facecolor\": \"#eaeaf2\",\n        \"axes.grid\": True,\n        \"grid.color\": \"#ffffff\",\n        \"axes.edgecolor\": \"#4b4b4b\"\n    }\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:06:50.625239Z","iopub.execute_input":"2024-12-03T06:06:50.626034Z","iopub.status.idle":"2024-12-03T06:06:50.640245Z","shell.execute_reply.started":"2024-12-03T06:06:50.625995Z","shell.execute_reply":"2024-12-03T06:06:50.63887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv').drop(['id'], axis = 1)\ntest_df = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\nsubmission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:06:50.643139Z","iopub.execute_input":"2024-12-03T06:06:50.643883Z","iopub.status.idle":"2024-12-03T06:06:59.128492Z","shell.execute_reply.started":"2024-12-03T06:06:50.643845Z","shell.execute_reply":"2024-12-03T06:06:59.127193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df.shape)\nprint(test_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:06:59.129975Z","iopub.execute_input":"2024-12-03T06:06:59.130343Z","iopub.status.idle":"2024-12-03T06:06:59.136323Z","shell.execute_reply.started":"2024-12-03T06:06:59.130307Z","shell.execute_reply":"2024-12-03T06:06:59.135178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Check train data\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:06:59.138015Z","iopub.execute_input":"2024-12-03T06:06:59.13848Z","iopub.status.idle":"2024-12-03T06:06:59.170811Z","shell.execute_reply.started":"2024-12-03T06:06:59.138431Z","shell.execute_reply":"2024-12-03T06:06:59.169584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Check test data\ntest_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:06:59.172178Z","iopub.execute_input":"2024-12-03T06:06:59.172555Z","iopub.status.idle":"2024-12-03T06:06:59.198327Z","shell.execute_reply.started":"2024-12-03T06:06:59.17252Z","shell.execute_reply":"2024-12-03T06:06:59.197134Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#instead of describe use summarytools module as this is more comprehensive\ndfSummary(train_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:06:59.199662Z","iopub.execute_input":"2024-12-03T06:06:59.200052Z","iopub.status.idle":"2024-12-03T06:07:12.20828Z","shell.execute_reply.started":"2024-12-03T06:06:59.200019Z","shell.execute_reply":"2024-12-03T06:07:12.207117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dfSummary(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:07:12.209709Z","iopub.execute_input":"2024-12-03T06:07:12.2101Z","iopub.status.idle":"2024-12-03T06:07:21.454897Z","shell.execute_reply.started":"2024-12-03T06:07:12.210059Z","shell.execute_reply":"2024-12-03T06:07:21.453599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:07:21.456487Z","iopub.execute_input":"2024-12-03T06:07:21.456902Z","iopub.status.idle":"2024-12-03T06:07:21.464532Z","shell.execute_reply.started":"2024-12-03T06:07:21.456864Z","shell.execute_reply":"2024-12-03T06:07:21.46337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cont_features=train_df.select_dtypes(include=[np.number]).columns.drop(\"Premium Amount\")\nnum_features=len(cont_features)\ncols=2\nrows=(num_features//cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:07:21.466614Z","iopub.execute_input":"2024-12-03T06:07:21.467504Z","iopub.status.idle":"2024-12-03T06:07:21.515813Z","shell.execute_reply.started":"2024-12-03T06:07:21.467449Z","shell.execute_reply":"2024-12-03T06:07:21.514548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig,axes=plt.subplots(rows,cols,figsize=(20,rows*5))\nfig.suptitle(\"Continuous Features vs Premium Amount\",fontsize=16, fontweight=\"bold\")\naxes=axes.flatten()\n\n#plot each feature aggainst premium amount\nfor i,ax in enumerate(axes):\n    if i < num_features:\n        feature = cont_features[i]\n        sns.scatterplot(\n            x=train_df[feature],\n            y=train_df['Premium Amount'],\n            ax=ax,\n            color='teal',\n            edgecolor='black'\n         )\n        ax.set_title(f\"{feature} vs Premium Amount\" )\n        ax.set_xlabel(feature)\n        ax.set_ylabel(\"Premium Amount\")\n    else:\n        ax.axis('off')\n\nplt.tight_layout(rect=[0,0,1,0.95])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:07:21.517098Z","iopub.execute_input":"2024-12-03T06:07:21.517449Z","iopub.status.idle":"2024-12-03T06:07:42.413933Z","shell.execute_reply.started":"2024-12-03T06:07:21.517402Z","shell.execute_reply":"2024-12-03T06:07:42.412331Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This doesnt help much as most of numerical features are sort of categorical features only if we divide them into different bins.","metadata":{}},{"cell_type":"code","source":"#Check categorical features.\ncat_features=train_df.drop([\"Policy Start Date\"],axis=1).select_dtypes(include=['object']).columns\nnum_features=len(cat_features)\ncols=2\nrows=(num_features//cols)\nprint(rows)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:07:42.41741Z","iopub.execute_input":"2024-12-03T06:07:42.417835Z","iopub.status.idle":"2024-12-03T06:07:42.814219Z","shell.execute_reply.started":"2024-12-03T06:07:42.417794Z","shell.execute_reply":"2024-12-03T06:07:42.813011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(rows, cols, figsize=(16, rows * 5))\nfig.suptitle(\"Bar Chart of Categorical Features \", fontsize=16, fontweight=\"bold\")\n\naxes = axes.flatten()\n\n# Plot each categorical feature with a bar chart\nfor i, ax in enumerate(axes):\n    if i < len(cat_features):\n        feature = cat_features[i]\n        category_counts = train_df[feature].value_counts()\n        sns.barplot(\n            x=category_counts.index,\n            y=category_counts.values,\n            ax=ax,\n            palette=\"Set2\"\n        )\n        ax.set_title(f\"Bar Chart: Distribution of {feature}\")\n        ax.set_xlabel(feature)\n        ax.set_ylabel(\"Count\")\n    else:\n        # Turn off axes for empty subplots\n        ax.axis('off')\n\n# Adjust layout\nplt.tight_layout(rect=[0, 0, 1, 0.95])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:07:42.815508Z","iopub.execute_input":"2024-12-03T06:07:42.81592Z","iopub.status.idle":"2024-12-03T06:07:46.494335Z","shell.execute_reply.started":"2024-12-03T06:07:42.81588Z","shell.execute_reply":"2024-12-03T06:07:46.492691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#It seems like annual income is not distributed normally. Check normality of it.\nsns.distplot(train_df['Annual Income'],fit=norm)\nfig=plt.figure()\nres=stats.probplot(train_df['Annual Income'],plot=plt)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:07:46.496225Z","iopub.execute_input":"2024-12-03T06:07:46.496694Z","iopub.status.idle":"2024-12-03T06:07:58.202178Z","shell.execute_reply.started":"2024-12-03T06:07:46.496646Z","shell.execute_reply":"2024-12-03T06:07:58.200869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Applying log transformation\ntrain_df['Annual Income']=np.log(train_df['Annual Income'])\ntrain_df['Annual Income'].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:07:58.203749Z","iopub.execute_input":"2024-12-03T06:07:58.204247Z","iopub.status.idle":"2024-12-03T06:07:58.227233Z","shell.execute_reply.started":"2024-12-03T06:07:58.204197Z","shell.execute_reply":"2024-12-03T06:07:58.226151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Transformed histogram and normal probability plots\nsns.distplot(train_df['Annual Income'],fit=norm)\nfig=plt.figure()\nres=stats.probplot(train_df['Annual Income'],plot=plt)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:07:58.228539Z","iopub.execute_input":"2024-12-03T06:07:58.228924Z","iopub.status.idle":"2024-12-03T06:08:09.717537Z","shell.execute_reply.started":"2024-12-03T06:07:58.228888Z","shell.execute_reply":"2024-12-03T06:08:09.716369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check corelation matrix\ncorrmat=train_df[cont_features].corr()\nf, ax = plt.subplots(figsize=(12,9))\nsns.heatmap(corrmat,vmax=1,square=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:08:09.719159Z","iopub.execute_input":"2024-12-03T06:08:09.7196Z","iopub.status.idle":"2024-12-03T06:08:10.56009Z","shell.execute_reply.started":"2024-12-03T06:08:09.719562Z","shell.execute_reply":"2024-12-03T06:08:10.558695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Check linear regression\nfrom sklearn.linear_model import LinearRegression\nmodel=LinearRegression()\ntr_df=train_df.drop([\"Policy Start Date\", \"Premium Amount\"],axis=1)\n#model.fit(tr_df,train_df[\"Premium Amount\"])\n#print(\"Model Coefficients :\\n\")\n#for i in range(tr_df.shape[1]):\n#    print(tr_df.columns[i],\"=\",model.coef_[i].round(5))\ntr_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:22:21.907459Z","iopub.execute_input":"2024-12-03T06:22:21.907938Z","iopub.status.idle":"2024-12-03T06:22:22.074957Z","shell.execute_reply.started":"2024-12-03T06:22:21.907889Z","shell.execute_reply":"2024-12-03T06:22:22.073701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rmsle(y_true, y_pred):\n    \n    return np.sqrt(np.mean(np.square(np.log1p(y_pred) - np.log1p(y_true))))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:21:09.915604Z","iopub.execute_input":"2024-12-03T06:21:09.916074Z","iopub.status.idle":"2024-12-03T06:21:09.921728Z","shell.execute_reply.started":"2024-12-03T06:21:09.916038Z","shell.execute_reply":"2024-12-03T06:21:09.920368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#from sklearn.metrics import root_mean_squared_log_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T05:52:01.710073Z","iopub.execute_input":"2024-12-03T05:52:01.710527Z","iopub.status.idle":"2024-12-03T05:52:01.723015Z","shell.execute_reply.started":"2024-12-03T05:52:01.71048Z","shell.execute_reply":"2024-12-03T05:52:01.721843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#X_t = tr_df.copy()\nfrom sklearn.preprocessing import OneHotEncoder\nencoder = OneHotEncoder(handle_unknown='ignore')\ny = train_df[\"Premium Amount\"]\n#X= pd.get_dummies(tr_df, drop_first=True)\n#encoder.fit(tr_df)\n#X=encoder.transform(tr_df)\nX=tr_df.copy()\n#del tr_df\ngc.collect()\nX.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:22:27.664396Z","iopub.execute_input":"2024-12-03T06:22:27.664921Z","iopub.status.idle":"2024-12-03T06:22:28.052109Z","shell.execute_reply.started":"2024-12-03T06:22:27.66488Z","shell.execute_reply":"2024-12-03T06:22:28.051007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Importing required libraries\nfrom sklearn.model_selection import KFold\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import mean_squared_error\nnp.random.seed(42)\n\n# K-Fold splitting\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\n# Placeholder for results\nmodels = []\nval_scores = []\n'''\n# Train and validate using KFold\nfor train_index, val_index in kf.split(X):\n    X_train, X_val = X.iloc[train_index], X.iloc[val_index]\n    y_train, y_val = y.iloc[train_index], y.iloc[val_index]\n\n    # XGB model\n    xgb_model = XGBRegressor(\n        objective='reg:squarederror',\n        n_estimators=100,  \n        learning_rate=0.1,\n        max_depth=3,\n        tree_method='hist', \n        enable_categorical=False \n    )\n\n    xgb_model.fit(X_train, y_train)\n\n    y_pred_val = xgb_model.predict(X_val)\n    val_score = rmsle(y_val, y_pred_val)\n    val_scores.append(val_score)\n    models.append(xgb_model)\n\n# Feature importance from the last model\nfeature_importance = models[-1].feature_importances_\n\n# Display validation results and feature importances\naverage_val_score = np.mean(val_scores)\nfeature_importance_dict = {col: imp for col, imp in zip(X.columns, feature_importance)}\n\n# Create and display the DataFrame\nfeature_importance_df = pd.DataFrame({\n    \"Feature\": list(feature_importance_dict.keys()),\n    \"Importance\": list(feature_importance_dict.values()),\n    \"Average Validation RMSE\": [average_val_score] * len(feature_importance_dict)\n})\n\nprint(\"Feature Importance and Validation Scores:\")\nprint(feature_importance_df)\n'''\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:22:41.489182Z","iopub.execute_input":"2024-12-03T06:22:41.48959Z","iopub.status.idle":"2024-12-03T06:22:41.499731Z","shell.execute_reply.started":"2024-12-03T06:22:41.489556Z","shell.execute_reply":"2024-12-03T06:22:41.498551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define RMSLE as a custom evaluation metric\ndef rmsle_eval(y_pred, dtrain):\n    y_true = dtrain.get_label()\n    rmsle_value = np.sqrt(np.mean(np.square(np.log1p(y_pred) - np.log1p(y_true))))\n    return 'rmsle', rmsle_value\n\n# Placeholder for train and validation errors\ntrain_errors = []\nval_errors = []\nbest_model = None\nbest_val_error = float('inf')  \n\nfor train_index, val_index in kf.split(X):\n    X_train, X_val = X.iloc[train_index], X.iloc[val_index]\n    y_train, y_val = y.iloc[train_index], y.iloc[val_index]\n\n    encoder = OneHotEncoder(handle_unknown='ignore')\n\n    # Fit encoder on training data only\n    encoder.fit(X_train)\n\n    # Transform both training and testing data\n    X_train_encoded = encoder.transform(X_train)\n    X_val_encoded = encoder.transform(X_val)\n\n    # XGB model\n    xgb_model = XGBRegressor(\n        objective='reg:squarederror',\n        n_estimators=100,\n        learning_rate=0.1,\n        max_depth=3,\n        tree_method='hist',\n        enable_categorical=False\n    )\n\n  \n    eval_set = [(X_train_encoded, y_train), (X_val_encoded, y_val)]\n    xgb_model.fit(\n        X_train_encoded,\n        y_train,\n        eval_set=eval_set,\n        eval_metric=rmsle_eval,\n        verbose=True\n    )\n\n    # Calculate train and validation RMSLE\n    y_train_pred = xgb_model.predict(X_train_encoded)\n    y_val_pred = xgb_model.predict(X_val_encoded)\n    train_error = rmsle(y_train, y_train_pred)\n    val_error = rmsle(y_val, y_val_pred)\n\n    train_errors.append(train_error)\n    val_errors.append(val_error)\n\n    # Update the best model if this fold's validation error is the lowest\n    if val_error < best_val_error:\n        best_val_error = val_error\n        best_model = xgb_model\n\n# Feature importance from the best model\nfeature_importance = best_model.feature_importances_\nfeature_importance_dict = {col: imp for col, imp in zip(X.columns, feature_importance)}\n\n# Display feature importance\nfeature_importance_df = pd.DataFrame({\n    \"Feature\": list(feature_importance_dict.keys()),\n    \"Importance\": list(feature_importance_dict.values())\n}).sort_values(by=\"Importance\", ascending=False)\n\n# Plot train and validation errors for each split\nplt.figure(figsize=(10, 6))\nplt.plot(range(1, len(train_errors) + 1), train_errors, label='Train RMSLE', marker='o')\nplt.plot(range(1, len(val_errors) + 1), val_errors, label='Validation RMSLE', marker='o')\nplt.xlabel('Fold Number')\nplt.ylabel('RMSLE')\nplt.title('Train and Validation RMSLE for Each Fold')\nplt.legend()\nplt.grid()\nplt.show()\n\n# Display feature importance\nprint(\"Feature Importance from the Best Model:\")\nprint(feature_importance_df)\n\n# Display train and validation errors\nprint(\"Train RMSLE for each fold:\", train_errors)\nprint(\"Validation RMSLE for each fold:\", val_errors)\n\ndel X_train, X_val, y_train, y_val \ngc.collect() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:23:13.129403Z","iopub.execute_input":"2024-12-03T06:23:13.129946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select the top 10 features based on importance\ntop_features = feature_importance_df.sort_values(by=\"Importance\", ascending=False).head(10)[\"Feature\"].tolist()\n\n# Filter the dataset to include only the top 10 features\nX_top_features = X[top_features]\n\n# Retrain the XGB model using the top 10 features\nxgb_model_top_features = XGBRegressor(\n    objective='reg:squarederror',\n    n_estimators=100,  \n    learning_rate=0.1,\n    max_depth=3,\n    tree_method='hist',  \n    enable_categorical=False\n)\n\nxgb_model_top_features.fit(X_top_features, y)\ndel X_top_features\ngc.collect() \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T05:54:18.522917Z","iopub.execute_input":"2024-12-03T05:54:18.523271Z","iopub.status.idle":"2024-12-03T05:54:23.883608Z","shell.execute_reply.started":"2024-12-03T05:54:18.52324Z","shell.execute_reply":"2024-12-03T05:54:23.882222Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# WIP as cant get past this , notebook reloads because of memory full. \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T05:54:23.885289Z","iopub.execute_input":"2024-12-03T05:54:23.885676Z","iopub.status.idle":"2024-12-03T05:54:23.890372Z","shell.execute_reply.started":"2024-12-03T05:54:23.885639Z","shell.execute_reply":"2024-12-03T05:54:23.88918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#test_df= pd.get_dummies(test_df.drop([\"Policy Start Date\"],axis=1), drop_first=True)\ntest_df=encoder.transform(test_df)\ntest_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T05:55:42.353414Z","iopub.execute_input":"2024-12-03T05:55:42.357039Z","iopub.status.idle":"2024-12-03T05:55:43.51836Z","shell.execute_reply.started":"2024-12-03T05:55:42.356949Z","shell.execute_reply":"2024-12-03T05:55:43.516626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = test_df[top_features]\ndel test_df\ngc.collect()\nX_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T05:56:02.204918Z","iopub.execute_input":"2024-12-03T05:56:02.205364Z","iopub.status.idle":"2024-12-03T05:56:02.496216Z","shell.execute_reply.started":"2024-12-03T05:56:02.205323Z","shell.execute_reply":"2024-12-03T05:56:02.495075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on test data\n\ny_test_pred = xgb_model_top_features.predict(X_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T05:56:06.9936Z","iopub.execute_input":"2024-12-03T05:56:06.994053Z","iopub.status.idle":"2024-12-03T05:56:08.328185Z","shell.execute_reply.started":"2024-12-03T05:56:06.994004Z","shell.execute_reply":"2024-12-03T05:56:08.327177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T05:56:13.321297Z","iopub.execute_input":"2024-12-03T05:56:13.322381Z","iopub.status.idle":"2024-12-03T05:56:13.500815Z","shell.execute_reply.started":"2024-12-03T05:56:13.322337Z","shell.execute_reply":"2024-12-03T05:56:13.499866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission['Premium Amount'] = y_test_pred\nprint(submission)\nsubmission.to_csv('submission.csv', index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T05:56:39.168485Z","iopub.execute_input":"2024-12-03T05:56:39.168953Z","iopub.status.idle":"2024-12-03T05:56:40.3755Z","shell.execute_reply.started":"2024-12-03T05:56:39.168914Z","shell.execute_reply":"2024-12-03T05:56:40.374041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}