{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":9178166,"sourceType":"datasetVersion","datasetId":5547076}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Data Preparation","metadata":{}},{"cell_type":"markdown","source":"### Load Required libraries and packages","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:28:16.969105Z","iopub.execute_input":"2024-12-06T01:28:16.96991Z","iopub.status.idle":"2024-12-06T01:28:19.401224Z","shell.execute_reply.started":"2024-12-06T01:28:16.969862Z","shell.execute_reply":"2024-12-06T01:28:19.400338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Load Dataset\n\ntrain_df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv',index_col=0)\ntest_df = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv',index_col = 0)\noriginal_df = pd.read_csv('/kaggle/input/insurance-premium-prediction/Insurance Premium Prediction Dataset.csv',index_col=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:28:28.695803Z","iopub.execute_input":"2024-12-06T01:28:28.696301Z","iopub.status.idle":"2024-12-06T01:28:38.430944Z","shell.execute_reply.started":"2024-12-06T01:28:28.696266Z","shell.execute_reply":"2024-12-06T01:28:38.430231Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"First, we start our data cleaning and preprocessing with EDA. \nLet's check the missing values, duplicates, unique data first to get high overview.","metadata":{}},{"cell_type":"code","source":"print(\"The shape of data is \",train_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:28:38.432451Z","iopub.execute_input":"2024-12-06T01:28:38.432761Z","iopub.status.idle":"2024-12-06T01:28:38.43757Z","shell.execute_reply.started":"2024-12-06T01:28:38.432735Z","shell.execute_reply":"2024-12-06T01:28:38.436673Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_features = train_df.select_dtypes(['object','string']).columns.tolist()\ncategorical_features.remove('Policy Start Date')\nprint(\"Categorical Features \\n\",categorical_features)\nnumeric_features = train_df.select_dtypes(['int64','float64']).columns.tolist()\nnumeric_features.remove('Premium Amount')\nprint(\"Numerical  Features \\n\",numeric_features)\ntarget = 'Premium Amount'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:28:38.438456Z","iopub.execute_input":"2024-12-06T01:28:38.438723Z","iopub.status.idle":"2024-12-06T01:28:38.62983Z","shell.execute_reply.started":"2024-12-06T01:28:38.438699Z","shell.execute_reply":"2024-12-06T01:28:38.628898Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Lets observe the missing values and thei distribution.","metadata":{}},{"cell_type":"code","source":"missing_percent = (train_df.isna().sum()/len(train_df) )* 100\n# Sort by percentage in descending order\nmissing_percent_sorted = missing_percent[missing_percent > 0].sort_values(ascending=False)\n\n# Display the result\nprint(\"Missing Values Percentage:\")\nprint(missing_percent_sorted)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:28:38.631532Z","iopub.execute_input":"2024-12-06T01:28:38.631811Z","iopub.status.idle":"2024-12-06T01:28:39.169145Z","shell.execute_reply.started":"2024-12-06T01:28:38.631785Z","shell.execute_reply":"2024-12-06T01:28:39.168034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.heatmap(train_df.isna(), \n            cmap='viridis',  # Choose a color map\n            cbar=False,      # Disable color bar for simplicity\n            yticklabels=False)  # Remove y-axis labels for cleaner look","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:28:39.170365Z","iopub.execute_input":"2024-12-06T01:28:39.171064Z","iopub.status.idle":"2024-12-06T01:29:06.9319Z","shell.execute_reply.started":"2024-12-06T01:28:39.171022Z","shell.execute_reply":"2024-12-06T01:29:06.931082Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# describe numeric features\ntrain_df.describe().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:29:06.933211Z","iopub.execute_input":"2024-12-06T01:29:06.933876Z","iopub.status.idle":"2024-12-06T01:29:07.507403Z","shell.execute_reply.started":"2024-12-06T01:29:06.933833Z","shell.execute_reply":"2024-12-06T01:29:07.506547Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Study the relationship between variables**\n\nLets observe the numeric features first. The numeric features in the dataset are:\n\n- **Age**: The age of the policyholder in years.\n- **Annual Income**: The annual income of the policyholder (in local currency).\n- **Number of Dependents**: Total dependents financially supported by the policyholder.\n- **Health Score**: A numeric score representing the policyholder's health status.\n- **Previous Claims**: The count of claims made by the policyholder in the past.\n- **Vehicle Age**: The age of the insured vehicle in years.\n- **Credit Score**: A numerical representation of the policyholder's creditworthiness.\n- **Insurance Duration**: The duration of the insurance policy (in years).\n- **Premium Amount**: The amount paid for the insurance policy (in local currency).\n","metadata":{}},{"cell_type":"code","source":"def plot_histogram_and_boxplot(data, feature):\n    \"\"\"\n    Plots a histogram and a boxplot for the given feature.\n    \n    Args:\n    data (pd.DataFrame): The DataFrame containing the data.\n    feature (str): The column name of the feature to plot.\n    \"\"\"\n    plt.figure(figsize=(12, 5))\n    \n    # Histogram\n    plt.subplot(1, 2, 1)\n    sns.histplot(data[feature], kde=True, bins=20, color='blue')\n    plt.title(f'Histogram of {feature}')\n    plt.xlabel(feature)\n    plt.ylabel('Frequency')\n    \n    # Boxplot\n    plt.subplot(1, 2, 2)\n    sns.boxplot(x=data[feature], color='green')\n    plt.title(f'Boxplot of {feature}')\n    plt.xlabel(feature)\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:29:07.508636Z","iopub.execute_input":"2024-12-06T01:29:07.509004Z","iopub.status.idle":"2024-12-06T01:29:07.51516Z","shell.execute_reply.started":"2024-12-06T01:29:07.508963Z","shell.execute_reply":"2024-12-06T01:29:07.51423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Loop through features and call the function\nfor feature in numeric_features:\n    plot_histogram_and_boxplot(train_df, feature)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:29:07.516137Z","iopub.execute_input":"2024-12-06T01:29:07.516364Z","iopub.status.idle":"2024-12-06T01:29:45.397838Z","shell.execute_reply.started":"2024-12-06T01:29:07.516342Z","shell.execute_reply":"2024-12-06T01:29:45.396953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef plot_categorical_and_boxplot(data, cat_feature):\n    \"\"\"\n    Plots a countplot for the categorical feature and a boxplot \n    for the relationship between the categorical and numerical features.\n    \n    Args:\n    data (pd.DataFrame): The DataFrame containing the data.\n    cat_feature (str): The column name of the categorical feature.\n    num_feature (str): The column name of the numerical feature.\n    \"\"\"\n    plt.figure(figsize=(12, 5))\n    \n    # Countplot\n    plt.subplot(1, 2, 1)\n    sns.countplot(x=data[cat_feature], palette='pastel')\n    plt.title(f'Countplot of {cat_feature}')\n    plt.xlabel(cat_feature)\n    plt.ylabel('Count')\n    \n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:29:45.398852Z","iopub.execute_input":"2024-12-06T01:29:45.399082Z","iopub.status.idle":"2024-12-06T01:29:45.404837Z","shell.execute_reply.started":"2024-12-06T01:29:45.399059Z","shell.execute_reply":"2024-12-06T01:29:45.403874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Loop through categorical features and call the function\nfor cat_feature in categorical_features:\n    plot_categorical_and_boxplot(train_df, cat_feature)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:29:45.407555Z","iopub.execute_input":"2024-12-06T01:29:45.407891Z","iopub.status.idle":"2024-12-06T01:29:53.446777Z","shell.execute_reply.started":"2024-12-06T01:29:45.407865Z","shell.execute_reply":"2024-12-06T01:29:53.445889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\ndef plot_boxplot(data, cat_feature, num_feature):\n    \"\"\"\n    Plots a boxplot for the relationship between a categorical feature and a numerical feature.\n    \n    Args:\n    data (pd.DataFrame): The DataFrame containing the data.\n    cat_feature (str): The column name of the categorical feature.\n    num_feature (str): The column name of the numerical feature.\n    \"\"\"\n    plt.figure(figsize=(8, 5))\n    sns.boxplot(x=data[cat_feature], y=data[num_feature], palette='Set2')\n    plt.title(f'{num_feature} by {cat_feature}')\n    plt.xlabel(cat_feature)\n    plt.ylabel(num_feature)\n    plt.tight_layout()\n    plt.show()\n\n\n\n# List of categorical features and the numerical feature\n\nnumerical_feature = 'Premium Amount'\n\n# Loop through categorical features and call the function\nfor cat_feature in categorical_features:\n    plot_boxplot(train_df, cat_feature, numerical_feature)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:29:53.448422Z","iopub.execute_input":"2024-12-06T01:29:53.44872Z","iopub.status.idle":"2024-12-06T01:30:01.296467Z","shell.execute_reply.started":"2024-12-06T01:29:53.448693Z","shell.execute_reply":"2024-12-06T01:30:01.295577Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"All the features seem to have balanced data which i found bit weird. Moving further We will see the bivariate analysis to understtand the features better.","metadata":{}},{"cell_type":"code","source":"# # Loop through all combinations of feature pairs\n# for feature in numeric_features:\n#     plt.figure(figsize=(8, 5))\n#     sns.scatterplot(x=feature, y='Premium Amount', data=train_df)\n#     plt.title(f'Scatter Plot of {feature} vs Premium Price')\n#     plt.xlabel(feature)\n#     plt.ylabel('Premium Price')\n#     plt.tight_layout()\n#     plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:30:01.297914Z","iopub.execute_input":"2024-12-06T01:30:01.298306Z","iopub.status.idle":"2024-12-06T01:30:01.302706Z","shell.execute_reply.started":"2024-12-06T01:30:01.298267Z","shell.execute_reply":"2024-12-06T01:30:01.301772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# visualize correlations betwwen multiple numeric vairables at once. \n# stronger relations are closer to 1\n\ncorrelation_matrix = train_df[numeric_features].corr()\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm', fmt='.2f')\nplt.title('Correlation Heatmap')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:30:01.303627Z","iopub.execute_input":"2024-12-06T01:30:01.30389Z","iopub.status.idle":"2024-12-06T01:30:02.096338Z","shell.execute_reply.started":"2024-12-06T01:30:01.303855Z","shell.execute_reply":"2024-12-06T01:30:02.095102Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There is no strong corelation at all from observed Heatmap and Scatter plot as well. Lets deal with missing values \n### handling missing values","metadata":{}},{"cell_type":"code","source":"print(\"Missing value percent wise \\n\\n\",missing_percent_sorted)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:30:02.097444Z","iopub.execute_input":"2024-12-06T01:30:02.097839Z","iopub.status.idle":"2024-12-06T01:30:02.106004Z","shell.execute_reply.started":"2024-12-06T01:30:02.097794Z","shell.execute_reply":"2024-12-06T01:30:02.104743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Occupation and Previous Claims cover the large porportion of data, so rather imputation we can drop \n## Previous Claims and Occupation for now\n\n# Drop from train_df\ntrain_df.drop(['Occupation', 'Previous Claims'], axis=1, inplace=True)\n\n\n# Drop from test_df\ntest_df.drop(['Occupation', 'Previous Claims'], axis=1, inplace=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:30:02.107639Z","iopub.execute_input":"2024-12-06T01:30:02.108036Z","iopub.status.idle":"2024-12-06T01:30:02.423189Z","shell.execute_reply.started":"2024-12-06T01:30:02.107988Z","shell.execute_reply":"2024-12-06T01:30:02.422172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_percent = (train_df.isna().sum()/len(train_df) )* 100\n# Sort by percentage in descending order\nmissing_percent_sorted = missing_percent[missing_percent > 0].sort_values(ascending=False)\n\n# Display the result\nprint(\"Missing Values Percentage:\")\nprint(missing_percent_sorted)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:30:02.42456Z","iopub.execute_input":"2024-12-06T01:30:02.424958Z","iopub.status.idle":"2024-12-06T01:30:02.947344Z","shell.execute_reply.started":"2024-12-06T01:30:02.424919Z","shell.execute_reply":"2024-12-06T01:30:02.946362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Impute for train_df\n\n# Credit Score has central tendency with low outliers, so we use mean\ntrain_df['Credit Score'] = train_df['Credit Score'].fillna(train_df['Credit Score'].mean())\n\n# Number of Dependents has categorical values, so we use mode\ntrain_df['Number of Dependents'] = train_df['Number of Dependents'].fillna(train_df['Number of Dependents'].mode()[0])\n\n# Customer Feedback has categorical values, so we use mode\ntrain_df['Customer Feedback'] = train_df['Customer Feedback'].fillna(train_df['Customer Feedback'].mode()[0])\n\n# Age has no outliers, so we use median\ntrain_df['Age'] = train_df['Age'].fillna(train_df['Age'].median())\n\n# Health Scores has nno outliers, so we use median\ntrain_df['Health Score'] = train_df['Health Score'].fillna(train_df['Health Score'].median())\n\n\n# Annual Income has outliers, so we use mean\ntrain_df['Annual Income'] = train_df['Annual Income'].fillna(train_df['Annual Income'].mean())\n\n\n# Marital Status is categorical, so we use mode\ntrain_df['Marital Status'] = train_df['Marital Status'].fillna(train_df['Marital Status'].mode()[0])\n\n# Vehicle Age has low missing data, use mean\ntrain_df['Vehicle Age'] = train_df['Vehicle Age'].fillna(train_df['Vehicle Age'].mean())\n\n# Insurance Duration has no outliers, so we use median\ntrain_df['Insurance Duration'] = train_df['Insurance Duration'].fillna(train_df['Insurance Duration'].median())\n\n\n\n\n# Impute for test_df (same strategy as train_df)\n\n# Credit Score has central tendency with low outliers, so we use mean\ntest_df['Credit Score'] = test_df['Credit Score'].fillna(test_df['Credit Score'].mean())\n\n# Number of Dependents has categorical values, so we use mode\ntest_df['Number of Dependents'] = test_df['Number of Dependents'].fillna(test_df['Number of Dependents'].mode()[0])\n\n# Customer Feedback has categorical values, so we use mode\ntest_df['Customer Feedback'] = test_df['Customer Feedback'].fillna(test_df['Customer Feedback'].mode()[0])\n\n# Age has no outliers, so we use median\ntest_df['Age'] = test_df['Age'].fillna(test_df['Age'].median())\n\n# Health Scores has nno outliers, so we use median\ntest_df['Health Score'] = test_df['Health Score'].fillna(test_df['Health Score'].median())\n\n# Annual Income has outliers, so we use mean\ntest_df['Annual Income'] = test_df['Annual Income'].fillna(test_df['Annual Income'].mean())\n\n# Marital Status is categorical, so we use mode\ntest_df['Marital Status'] = test_df['Marital Status'].fillna(test_df['Marital Status'].mode()[0])\n\n# Vehicle Age has low missing data, use mean\ntest_df['Vehicle Age'] = test_df['Vehicle Age'].fillna(test_df['Vehicle Age'].mean())\n\n# Insurance Duration has no outliers, so we use median\ntest_df['Insurance Duration'] = test_df['Insurance Duration'].fillna(test_df['Insurance Duration'].median())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:30:02.948641Z","iopub.execute_input":"2024-12-06T01:30:02.949014Z","iopub.status.idle":"2024-12-06T01:30:03.693367Z","shell.execute_reply.started":"2024-12-06T01:30:02.948987Z","shell.execute_reply":"2024-12-06T01:30:03.692703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Test for Duplicates and outliers\ntrain_df = train_df.drop_duplicates()\ntest_df  = test_df.drop_duplicates()\n\n## There was not much outliers observerd earlier so we will move on with the model train preparation","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:30:03.694385Z","iopub.execute_input":"2024-12-06T01:30:03.694659Z","iopub.status.idle":"2024-12-06T01:30:06.694039Z","shell.execute_reply.started":"2024-12-06T01:30:03.694633Z","shell.execute_reply":"2024-12-06T01:30:06.693133Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Feature Engineering","metadata":{}},{"cell_type":"code","source":"def feature_eng(data):\n    # Make a copy of the data to avoid modifying the original dataframe\n    df = data.copy()\n\n    # Ensure 'Policy Start Date' exists and convert to datetime, handle missing values\n    if 'Policy Start Date' in df.columns:\n        df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'], errors='coerce')  # Handle errors during conversion (NaT for invalid dates)\n        df['Policy Start Year'] = df['Policy Start Date'].dt.year.astype(str)\n        df.drop(columns=['Policy Start Date'], inplace=True)  # Remove 'Policy Start Date'\n\n    # Handling skewness in numeric features\n    for column in numeric_features:\n        if column in df.columns:\n            skewness = df[column].skew()\n            if skewness > 1 or skewness < -1:  # Skewness threshold\n                df[column] = np.log1p(df[column])  # Apply log transformation\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:30:06.695124Z","iopub.execute_input":"2024-12-06T01:30:06.695369Z","iopub.status.idle":"2024-12-06T01:30:06.701699Z","shell.execute_reply.started":"2024-12-06T01:30:06.695345Z","shell.execute_reply":"2024-12-06T01:30:06.700693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cleaned_train_df = feature_eng(train_df)\ncleaned_test_df = feature_eng(test_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:30:06.703089Z","iopub.execute_input":"2024-12-06T01:30:06.703568Z","iopub.status.idle":"2024-12-06T01:30:08.506181Z","shell.execute_reply.started":"2024-12-06T01:30:06.70353Z","shell.execute_reply":"2024-12-06T01:30:08.505418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cleaned_train_df.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:30:08.507257Z","iopub.execute_input":"2024-12-06T01:30:08.507614Z","iopub.status.idle":"2024-12-06T01:30:08.513769Z","shell.execute_reply.started":"2024-12-06T01:30:08.507577Z","shell.execute_reply":"2024-12-06T01:30:08.512898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cleaned_train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:30:08.514895Z","iopub.execute_input":"2024-12-06T01:30:08.515224Z","iopub.status.idle":"2024-12-06T01:30:09.018469Z","shell.execute_reply.started":"2024-12-06T01:30:08.515187Z","shell.execute_reply":"2024-12-06T01:30:09.017643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cleaned_train_df['Policy Start Year'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:30:09.019561Z","iopub.execute_input":"2024-12-06T01:30:09.019835Z","iopub.status.idle":"2024-12-06T01:30:09.122928Z","shell.execute_reply.started":"2024-12-06T01:30:09.019808Z","shell.execute_reply":"2024-12-06T01:30:09.122014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_features = cleaned_train_df.select_dtypes(['float64','int64']).columns.tolist()\ncategorical_features = cleaned_train_df.select_dtypes(['object']).columns.tolist()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:30:09.123981Z","iopub.execute_input":"2024-12-06T01:30:09.124275Z","iopub.status.idle":"2024-12-06T01:30:09.727227Z","shell.execute_reply.started":"2024-12-06T01:30:09.124249Z","shell.execute_reply":"2024-12-06T01:30:09.726253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_features.remove('Premium Amount')\nnumeric_features\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:30:09.728376Z","iopub.execute_input":"2024-12-06T01:30:09.728666Z","iopub.status.idle":"2024-12-06T01:30:09.73455Z","shell.execute_reply.started":"2024-12-06T01:30:09.728639Z","shell.execute_reply":"2024-12-06T01:30:09.733688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:30:09.735685Z","iopub.execute_input":"2024-12-06T01:30:09.735928Z","iopub.status.idle":"2024-12-06T01:30:09.746543Z","shell.execute_reply.started":"2024-12-06T01:30:09.735904Z","shell.execute_reply":"2024-12-06T01:30:09.745684Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model Training","metadata":{}},{"cell_type":"code","source":"# importing required libraries\n\nfrom sklearn.model_selection import train_test_split, KFold, RepeatedStratifiedKFold\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error, r2_score, mean_squared_log_error\n\nfrom catboost import CatBoostRegressor\n\nimport optuna\nfrom optuna.samplers import TPESampler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:39:56.157302Z","iopub.execute_input":"2024-12-06T01:39:56.157663Z","iopub.status.idle":"2024-12-06T01:39:56.162231Z","shell.execute_reply.started":"2024-12-06T01:39:56.157633Z","shell.execute_reply":"2024-12-06T01:39:56.161313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df_encoded = pd.get_dummies(cleaned_train_df, drop_first=True)\ntest_df_encoded = pd.get_dummies(cleaned_test_df, drop_first=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:39:57.733951Z","iopub.execute_input":"2024-12-06T01:39:57.734566Z","iopub.status.idle":"2024-12-06T01:39:59.844637Z","shell.execute_reply.started":"2024-12-06T01:39:57.734517Z","shell.execute_reply":"2024-12-06T01:39:59.843902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = cleaned_train_df.drop('Premium Amount', axis=1)  \ny = cleaned_train_df['Premium Amount']\n\ny_log = np.log1p(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:39:59.846123Z","iopub.execute_input":"2024-12-06T01:39:59.846407Z","iopub.status.idle":"2024-12-06T01:40:00.022908Z","shell.execute_reply.started":"2024-12-06T01:39:59.846382Z","shell.execute_reply":"2024-12-06T01:40:00.022075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rmsle(y_true, y_pred):\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:40:00.023937Z","iopub.execute_input":"2024-12-06T01:40:00.024225Z","iopub.status.idle":"2024-12-06T01:40:00.028336Z","shell.execute_reply.started":"2024-12-06T01:40:00.024198Z","shell.execute_reply":"2024-12-06T01:40:00.027339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def objective(trial):\n    # Hyperparameter suggestions\n    param = {\n        'iterations': trial.suggest_int('iterations', 250, 1500),\n        \"depth\": trial.suggest_int(\"depth\", 1, 12),\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.05, 0.1),\n        \"min_data_in_leaf\": trial.suggest_int(\"min_data_in_leaf\", 1, 100),\n        \"l2_leaf_reg\": trial.suggest_float(\"l2_leaf_reg\", 0, 10),\n        \"bagging_temperature\": trial.suggest_float(\"bagging_temperature\", 0, 10),\n    }\n\n    # Cross-validation setup\n    cv = KFold(n_splits=5, shuffle=True, random_state=42)\n    rmsles = []\n    oof = np.zeros(len(y_log))  # Use transformed target length\n\n    # Perform CV\n    for train_idx, val_idx in cv.split(X, y_log):  # Use `y_log` instead of `y`\n        X_train_fold, X_val_fold = X.iloc[train_idx], X.iloc[val_idx]\n        y_train_fold, y_val_fold = y_log.iloc[train_idx], y_log.iloc[val_idx]\n\n        # Initialize model with suggested parameters\n        model = CatBoostRegressor(\n            cat_features=categorical_features,\n            eval_metric=\"RMSE\",\n            random_seed=42,\n            verbose=0,\n            task_type='GPU',\n            **param  # Include suggested parameters\n        )\n\n        # Fit the model\n        model.fit(\n            X_train_fold,\n            y_train_fold,\n            eval_set=(X_val_fold, y_val_fold),\n            early_stopping_rounds=300,\n        )\n\n        # Make predictions\n        oof[val_idx] = np.maximum(0, model.predict(X_val_fold))\n        fold_rmsle = rmsle(np.expm1(y_val_fold), np.expm1(oof[val_idx]))\n        rmsles.append(fold_rmsle)\n\n    print(f'Trial {trial.number}: Mean RMSLE = {np.mean(rmsles):.4f}')\n\n    # Return mean RMSLE as the optimization target\n    return np.mean(rmsles)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:52:09.673617Z","iopub.execute_input":"2024-12-06T01:52:09.673962Z","iopub.status.idle":"2024-12-06T01:52:09.682078Z","shell.execute_reply.started":"2024-12-06T01:52:09.673934Z","shell.execute_reply":"2024-12-06T01:52:09.681128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"catboost_study = optuna.create_study(direction='minimize', \n                                    sampler=TPESampler(n_startup_trials=100, seed=42, multivariate=True))\ncatboost_study.optimize(objective, n_trials=50)\ncatboost_combined_data_best_params = catboost_study.best_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:03:06.589133Z","iopub.execute_input":"2024-12-06T02:03:06.590061Z","iopub.status.idle":"2024-12-06T02:12:20.151703Z","shell.execute_reply.started":"2024-12-06T02:03:06.590013Z","shell.execute_reply":"2024-12-06T02:12:20.150707Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Number of finished trials: {}\".format(len(catboost_study.trials)))\n\nprint(\"Best trial:\")\ntrial = catboost_study.best_trial\n\nprint(\"  Value: {}\".format(trial.value))\n\nprint(\"  Params: \")\nfor key, value in trial.params.items():\n    print(\"    {}: {}\".format(key, value))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:12:20.153227Z","iopub.execute_input":"2024-12-06T02:12:20.153586Z","iopub.status.idle":"2024-12-06T02:12:20.159769Z","shell.execute_reply.started":"2024-12-06T02:12:20.153556Z","shell.execute_reply":"2024-12-06T02:12:20.158767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Training the final CatBoost Model\n\ndef train_model():\n    kf = KFold(n_splits=5, shuffle=True, random_state=42)\n    oof = np.zeros(len(X))\n    models = []\n\n    for fold, (train_idx, valid_idx) in enumerate(kf.split(X)):\n        print(f\"Fold {fold + 1}\")\n        X_train, X_valid = X.iloc[train_idx], X.iloc[valid_idx]\n        y_train, y_valid = y_log.iloc[train_idx], y_log.iloc[valid_idx]\n\n        model = CatBoostRegressor(\n            **catboost_combined_data_best_params,            \n           cat_features=categorical_features,\n            eval_metric=\"RMSE\",\n            random_seed=42,\n            verbose=0,\n            task_type='GPU',\n        )\n        \n        model.fit(X_train,\n                  y_train,\n                  eval_set=(X_valid, y_valid), \n                  early_stopping_rounds=300,\n                  # cat_features=cat_cols,\n                 )\n        models.append(model)\n        oof[valid_idx] = np.maximum(0, model.predict(X_valid))\n        fold_rmsle = rmsle(np.expm1(y_valid), np.expm1(oof[valid_idx]))\n        print(f\"Fold {fold + 1} RMSLE: {fold_rmsle}\")\n        \n    return models, oof\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:12:36.90831Z","iopub.execute_input":"2024-12-06T02:12:36.908932Z","iopub.status.idle":"2024-12-06T02:12:36.915529Z","shell.execute_reply.started":"2024-12-06T02:12:36.908902Z","shell.execute_reply":"2024-12-06T02:12:36.914428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"models,oof = train_model()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:12:48.725574Z","iopub.execute_input":"2024-12-06T02:12:48.72596Z","iopub.status.idle":"2024-12-06T02:18:05.516876Z","shell.execute_reply.started":"2024-12-06T02:12:48.725929Z","shell.execute_reply":"2024-12-06T02:18:05.515838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(rmsle(y, np.expm1(oof)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:18:28.015174Z","iopub.execute_input":"2024-12-06T02:18:28.01556Z","iopub.status.idle":"2024-12-06T02:18:28.039299Z","shell.execute_reply.started":"2024-12-06T02:18:28.015526Z","shell.execute_reply":"2024-12-06T02:18:28.03808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_predictions = np.zeros(len(cleaned_test_df))\nfor model in models:\n    test_predictions += np.maximum(0, np.expm1(model.predict(cleaned_test_df))) / len(models)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:18:33.70331Z","iopub.execute_input":"2024-12-06T02:18:33.703687Z","iopub.status.idle":"2024-12-06T02:18:40.307311Z","shell.execute_reply.started":"2024-12-06T02:18:33.703655Z","shell.execute_reply":"2024-12-06T02:18:40.306534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:18:44.928733Z","iopub.execute_input":"2024-12-06T02:18:44.929085Z","iopub.status.idle":"2024-12-06T02:18:44.93512Z","shell.execute_reply.started":"2024-12-06T02:18:44.929056Z","shell.execute_reply":"2024-12-06T02:18:44.934226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cleaned_test_df['Premium Amount'] = test_predictions\nsubmission_df = cleaned_test_df['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:18:52.334355Z","iopub.execute_input":"2024-12-06T02:18:52.335153Z","iopub.status.idle":"2024-12-06T02:18:52.340767Z","shell.execute_reply.started":"2024-12-06T02:18:52.33512Z","shell.execute_reply":"2024-12-06T02:18:52.339801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:18:54.13836Z","iopub.execute_input":"2024-12-06T02:18:54.138758Z","iopub.status.idle":"2024-12-06T02:18:54.145974Z","shell.execute_reply.started":"2024-12-06T02:18:54.138725Z","shell.execute_reply":"2024-12-06T02:18:54.145047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df.to_csv('submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T04:31:34.324783Z","iopub.execute_input":"2024-12-05T04:31:34.325132Z","iopub.status.idle":"2024-12-05T04:31:35.674107Z","shell.execute_reply.started":"2024-12-05T04:31:34.325103Z","shell.execute_reply":"2024-12-05T04:31:35.673389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # We will go wtih Random Forest Regressor for now\n\n# # importing required libraries\n\n\n# # Split data into training and testing sets\n# X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n\n# # Instantiate the regressor\n# rf_regressor = RandomForestRegressor(n_estimators=100, random_state=42)\n\n# # Fit the model to the training data\n# rf_regressor.fit(X_train, y_train)\n\n# # Make predictions on the test set\n# y_pred = rf_regressor.predict(X_test)\n\n# # Evaluate the model\n\n# y_pred = np.maximum(0, y_pred)  # Ensure no negative predictions \n# y_test = np.maximum(0, y_test)\n\n# rmsle = np.sqrt(mean_squared_log_error(rmslet, y_pred))\n# r2 = r2_score(y_test, y_pred)\n# # Print the evaluation metrics\n# print(f' Root Mean Squared Error: {rmsle}')\n# print(f'R^2 Score: {r2}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T04:31:35.675393Z","iopub.execute_input":"2024-12-05T04:31:35.675675Z","iopub.status.idle":"2024-12-05T04:31:35.679777Z","shell.execute_reply.started":"2024-12-05T04:31:35.67565Z","shell.execute_reply":"2024-12-05T04:31:35.678936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}