{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:10.79711Z","iopub.execute_input":"2024-12-31T14:02:10.797411Z","iopub.status.idle":"2024-12-31T14:02:11.226449Z","shell.execute_reply.started":"2024-12-31T14:02:10.797373Z","shell.execute_reply":"2024-12-31T14:02:11.225173Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### In this notebook, I outline my approach to solving a regression problem using the insurance dataset for a Playground competition. The dataset provides valuable information that I leveraged to build a regression model, and throughout the notebook, I walk through the steps I took, from data exploration to model evaluation.","metadata":{}},{"cell_type":"markdown","source":"the source is \"https://www.kaggle.com/competitions/playground-series-s4e12\"","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:11.227368Z","iopub.execute_input":"2024-12-31T14:02:11.227822Z","iopub.status.idle":"2024-12-31T14:02:12.720104Z","shell.execute_reply.started":"2024-12-31T14:02:11.227793Z","shell.execute_reply":"2024-12-31T14:02:12.719121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndf_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:12.721027Z","iopub.execute_input":"2024-12-31T14:02:12.721608Z","iopub.status.idle":"2024-12-31T14:02:23.750903Z","shell.execute_reply.started":"2024-12-31T14:02:12.721559Z","shell.execute_reply":"2024-12-31T14:02:23.750019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.duplicated().any()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:23.752858Z","iopub.execute_input":"2024-12-31T14:02:23.753184Z","iopub.status.idle":"2024-12-31T14:02:25.458605Z","shell.execute_reply.started":"2024-12-31T14:02:23.753158Z","shell.execute_reply":"2024-12-31T14:02:25.457588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:25.460318Z","iopub.execute_input":"2024-12-31T14:02:25.460638Z","iopub.status.idle":"2024-12-31T14:02:25.468558Z","shell.execute_reply.started":"2024-12-31T14:02:25.460601Z","shell.execute_reply":"2024-12-31T14:02:25.467407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:25.46969Z","iopub.execute_input":"2024-12-31T14:02:25.470065Z","iopub.status.idle":"2024-12-31T14:02:25.518885Z","shell.execute_reply.started":"2024-12-31T14:02:25.470018Z","shell.execute_reply":"2024-12-31T14:02:25.517845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:25.519858Z","iopub.execute_input":"2024-12-31T14:02:25.520182Z","iopub.status.idle":"2024-12-31T14:02:26.161649Z","shell.execute_reply.started":"2024-12-31T14:02:25.520156Z","shell.execute_reply":"2024-12-31T14:02:26.160411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# EDA\n\n# Firstly, a lot of null values for occupation and previous claims.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:26.162805Z","iopub.execute_input":"2024-12-31T14:02:26.163209Z","iopub.status.idle":"2024-12-31T14:02:26.167419Z","shell.execute_reply.started":"2024-12-31T14:02:26.163171Z","shell.execute_reply":"2024-12-31T14:02:26.166101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.boxplot(x=df_train['Occupation'],y=df_train['Premium Amount'])\n# it seems, premium amount barely depend on 'Occupation'\n# this would be tested using ANOVA technique\n# If it approve the claim, this column can be discarded\n# But, other option is to create new column called NaN which will correspond to missing values for Occupation","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:26.168782Z","iopub.execute_input":"2024-12-31T14:02:26.1691Z","iopub.status.idle":"2024-12-31T14:02:27.089342Z","shell.execute_reply.started":"2024-12-31T14:02:26.169068Z","shell.execute_reply":"2024-12-31T14:02:27.088225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.boxplot(x=df_train['Previous Claims'],y=df_train['Premium Amount'])\n# unlike the previous case now, it is evident that 'Previous claims' significantly affect 'Premium Amount'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:27.090345Z","iopub.execute_input":"2024-12-31T14:02:27.090671Z","iopub.status.idle":"2024-12-31T14:02:27.684023Z","shell.execute_reply.started":"2024-12-31T14:02:27.090645Z","shell.execute_reply":"2024-12-31T14:02:27.682694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.hist(df_train['Annual Income']);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:27.685313Z","iopub.execute_input":"2024-12-31T14:02:27.685755Z","iopub.status.idle":"2024-12-31T14:02:27.994371Z","shell.execute_reply.started":"2024-12-31T14:02:27.685701Z","shell.execute_reply":"2024-12-31T14:02:27.993313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.hist(df_train['Premium Amount']);\n# Target ariable and 'Annual Income' has non normal distribution, hence, I am going to implement log transform later","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:27.995365Z","iopub.execute_input":"2024-12-31T14:02:27.995612Z","iopub.status.idle":"2024-12-31T14:02:28.275676Z","shell.execute_reply.started":"2024-12-31T14:02:27.99559Z","shell.execute_reply":"2024-12-31T14:02:28.274249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get rid of outliers in 'Previous Claims'\ndf_train['Previous Claims'] = df_train['Previous Claims'].apply(lambda x: min(x, 7))\ndf_test['Previous Claims'] = df_test['Previous Claims'].apply(lambda x: min(x, 7))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:28.279267Z","iopub.execute_input":"2024-12-31T14:02:28.279571Z","iopub.status.idle":"2024-12-31T14:02:29.056452Z","shell.execute_reply.started":"2024-12-31T14:02:28.279547Z","shell.execute_reply":"2024-12-31T14:02:29.055285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_policy_date_features(df, date_column):\n    \"\"\"\n    Extracts year and month from a datetime column and creates new columns.\n\n    Parameters:\n    df (pd.DataFrame): The input DataFrame.\n    date_column (str): The name of the datetime column to extract features from.\n\n    Returns:\n    pd.DataFrame: The DataFrame with new columns for year and month.\n    \"\"\"\n    # Ensure the date column is in datetime format\n    df[date_column] = pd.to_datetime(df[date_column], errors='coerce')\n    \n    # Create new columns for year and month\n    df['Policy Year'] = df[date_column].dt.year\n    df['Policy Month'] = df[date_column].dt.month\n\n    return df\n\n\ndf_train = extract_policy_date_features(df_train, 'Policy Start Date')\ndf_test = extract_policy_date_features(df_test, 'Policy Start Date')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:29.058143Z","iopub.execute_input":"2024-12-31T14:02:29.058538Z","iopub.status.idle":"2024-12-31T14:02:30.009885Z","shell.execute_reply.started":"2024-12-31T14:02:29.058507Z","shell.execute_reply":"2024-12-31T14:02:30.008859Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\ndef impute_numerical_columns(df, num_columns):\n    \"\"\"\n    This function imputes missing values (NaN) in numerical columns of a DataFrame using the median strategy.\n    \n    Parameters:\n    - df: DataFrame with missing values in numerical columns.\n    - num_columns: List of numerical column names to apply the median imputation.\n\n    Returns:\n    - A new DataFrame with the numerical columns imputed (NaN values replaced with the median).\n    \"\"\"\n    # Create a copy of the original DataFrame to preserve it\n    df_copy = df.copy()\n\n    # Initialize the SimpleImputer with median strategy for imputation\n    imputer = SimpleImputer(strategy='median')\n\n    # Apply the imputer to the numerical columns\n    df_copy[num_columns] = imputer.fit_transform(df_copy[num_columns])\n\n    return df_copy\n\n# Example usage:\nnum_columns = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', \n               'Vehicle Age', 'Credit Score', 'Insurance Duration']\n\n# Assuming df_train is your original DataFrame\nprocessed_df_train = impute_numerical_columns(df_train, num_columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:30.01083Z","iopub.execute_input":"2024-12-31T14:02:30.011125Z","iopub.status.idle":"2024-12-31T14:02:32.740026Z","shell.execute_reply.started":"2024-12-31T14:02:30.0111Z","shell.execute_reply":"2024-12-31T14:02:32.739094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\ndef one_hot_encode_categorical(df, categorical_columns):\n    \"\"\"\n    This function performs One-Hot Encoding on categorical columns and returns a DataFrame with the encoded columns.\n\n    Parameters:\n    - df: DataFrame with categorical columns to be encoded.\n    - categorical_columns: List of categorical column names to apply One-Hot Encoding.\n\n    Returns:\n    - A DataFrame with one-hot encoded categorical columns and the rest of the original columns unchanged.\n    \"\"\"\n    # Create a copy of the original DataFrame to preserve it\n    df_copy = df.copy()\n\n    # Initialize OneHotEncoder\n    encoder = OneHotEncoder(drop='first', sparse=False)\n\n    # Perform One-Hot Encoding\n    encoded_values = encoder.fit_transform(df_copy[categorical_columns])\n\n    # Create a DataFrame with the encoded columns\n    encoded_df = pd.DataFrame(encoded_values, columns=encoder.get_feature_names_out(categorical_columns))\n\n    # Drop original categorical columns and concatenate the encoded ones\n    df_copy = df_copy.drop(columns=categorical_columns)\n    df_copy = pd.concat([df_copy, encoded_df], axis=1)\n\n    return df_copy\n\n\ncategorical_columns = ['Gender', 'Marital Status', 'Education Level', 'Occupation','Customer Feedback',\n                      'Location','Policy Type','Smoking Status','Exercise Frequency','Property Type','Previous Claims',\n                       'Policy Year'\n                      ]\n\n\nprocessed_df_train_encoded = one_hot_encode_categorical(processed_df_train, categorical_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:32.741177Z","iopub.execute_input":"2024-12-31T14:02:32.741509Z","iopub.status.idle":"2024-12-31T14:02:38.943303Z","shell.execute_reply.started":"2024-12-31T14:02:32.74148Z","shell.execute_reply":"2024-12-31T14:02:38.942153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"processed_df_train_encoded = processed_df_train_encoded.drop(columns=['id','Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:38.944448Z","iopub.execute_input":"2024-12-31T14:02:38.944758Z","iopub.status.idle":"2024-12-31T14:02:39.091246Z","shell.execute_reply.started":"2024-12-31T14:02:38.944722Z","shell.execute_reply":"2024-12-31T14:02:39.090093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"processed_df_test = impute_numerical_columns(df_test, num_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:39.092193Z","iopub.execute_input":"2024-12-31T14:02:39.092462Z","iopub.status.idle":"2024-12-31T14:02:40.496495Z","shell.execute_reply.started":"2024-12-31T14:02:39.092441Z","shell.execute_reply":"2024-12-31T14:02:40.495443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"processed_df_test_encoded = one_hot_encode_categorical(processed_df_test, categorical_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:40.497395Z","iopub.execute_input":"2024-12-31T14:02:40.497789Z","iopub.status.idle":"2024-12-31T14:02:44.567824Z","shell.execute_reply.started":"2024-12-31T14:02:40.497749Z","shell.execute_reply":"2024-12-31T14:02:44.566945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"id_ans=processed_df_test_encoded['id']\nprocessed_df_test_encoded = processed_df_test_encoded.drop(columns=['id','Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:44.568673Z","iopub.execute_input":"2024-12-31T14:02:44.56897Z","iopub.status.idle":"2024-12-31T14:02:44.664046Z","shell.execute_reply.started":"2024-12-31T14:02:44.568946Z","shell.execute_reply":"2024-12-31T14:02:44.662932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"processed_df_train_encoded.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:44.665153Z","iopub.execute_input":"2024-12-31T14:02:44.66557Z","iopub.status.idle":"2024-12-31T14:02:44.782232Z","shell.execute_reply.started":"2024-12-31T14:02:44.665519Z","shell.execute_reply":"2024-12-31T14:02:44.781048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def log_transform_annual_income(df, column_name='Annual Income'):\n    \"\"\"\n    Apply log transformation to the specified column of the DataFrame.\n    \n    Parameters:\n    - df: The input DataFrame.\n    - column_name: The column to apply the log transformation (default is 'Annual Income').\n    \n    Returns:\n    - df: The DataFrame with the log-transformed column.\n    \"\"\"\n    # Apply log transformation (log(y + 1) to avoid issues with zero or negative values)\n    df[column_name] = np.log(df[column_name] + 1)\n    \n    return df\n\n# Example usage with your df_train DataFrame\nprocessed_df_train_encoded = log_transform_annual_income(processed_df_train_encoded)\nprocessed_df_test_encoded = log_transform_annual_income(processed_df_test_encoded)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:44.783288Z","iopub.execute_input":"2024-12-31T14:02:44.783697Z","iopub.status.idle":"2024-12-31T14:02:44.817579Z","shell.execute_reply.started":"2024-12-31T14:02:44.783658Z","shell.execute_reply":"2024-12-31T14:02:44.816369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.metrics import make_scorer\nimport numpy as np\nfrom xgboost import XGBRegressor\n\n# Step 1: Prepare the data\ntarget_column = 'Premium Amount'\nX = processed_df_train_encoded.drop(columns=[target_column])  # Features (all columns except 'Premium Amount')\ny = processed_df_train_encoded[target_column]  # Target ('Premium Amount')\n\n# Log scale the target variable\ny_log = np.log(y + 1)  # Apply log transformation to the target (y + 1 to avoid log(0))\n\n# Split the data into training and testing sets (80% train, 20% test)\nX_train, X_test, y_train, y_test = train_test_split(X, y_log, test_size=0.3, random_state=42)\n\n# Define RMSLE as a scorer for GridSearchCV\ndef rmsle(y_true, y_pred):\n    return np.sqrt(np.mean((np.log(y_true + 1) - np.log(y_pred + 1)) ** 2))\n\nrmsle_scorer = make_scorer(rmsle, greater_is_better=False)\n\n# Step 2: Define the model and parameter grid\nxgb_model = XGBRegressor(random_state=42)\n\nparam_grid = {\n    'n_estimators': [80,100,120],           # Number of trees\n    'max_depth': [8,10,12],                # Maximum depth of a tree\n    'learning_rate': [0.08,0.1,0.12],       # Step size shrinkage\n    'min_child_weight': [120,150,180]        # Minimum sum of instance weight needed in a child\n}\n\n# Step 3: Perform grid search with cross-validation\ngrid_search = GridSearchCV(\n    estimator=xgb_model,\n    param_grid=param_grid,\n    scoring='neg_root_mean_squared_error', #rmsle_scorer,  # Use RMSLE as the metric\n    cv=3,                  # 5-fold cross-validation\n    verbose=1,\n    n_jobs=-1              # Use all available cores\n)\n\ngrid_search.fit(X_train, y_train)\n\n\n# Step 4: Output the best parameters and score\nprint(\"Best Parameters:\", grid_search.best_params_)\nprint(\"Best RMSLE Score:\", -grid_search.best_score_)  # Negate because RMSLE scorer is negative\n\n# Step 5: Evaluate the best model on the test set\nbest_model = grid_search.best_estimator_\ny_pred_log = best_model.predict(X_test)  # Predictions in log scale\n\n# Reverse the log transformation to get predictions back to original scale\ny_pred = np.exp(y_pred_log) - 1  # Reverse the log transformation\n\n# Compute RMSLE on the test set (after reversing log scaling)\ntest_rmsle = rmsle(np.exp(y_test) - 1, y_pred)  # Reverse the log transformation for y_test as well\n\nprint(f\"Test RMSLE with Best Model: {test_rmsle}\")\n\n# Make predictions on the test dataset\ny_test_pred_log = best_model.predict(processed_df_test_encoded)  # Predictions in log scale\ny_test_pred = np.exp(y_test_pred_log) - 1  # Reverse the log transformation","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:02:44.818551Z","iopub.execute_input":"2024-12-31T14:02:44.818846Z","iopub.status.idle":"2024-12-31T14:28:58.967228Z","shell.execute_reply.started":"2024-12-31T14:02:44.818821Z","shell.execute_reply":"2024-12-31T14:28:58.965231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import plot_importance\nplot_importance(best_model, max_num_features=10) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:28:58.968736Z","iopub.execute_input":"2024-12-31T14:28:58.969126Z","iopub.status.idle":"2024-12-31T14:28:59.334446Z","shell.execute_reply.started":"2024-12-31T14:28:58.969095Z","shell.execute_reply":"2024-12-31T14:28:59.333321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result_df = pd.DataFrame({\n    'id': id_ans,\n    'Premium Amount': y_test_pred\n})\nresult_df.to_csv('results_XGB.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:28:59.335748Z","iopub.execute_input":"2024-12-31T14:28:59.336107Z","iopub.status.idle":"2024-12-31T14:29:00.509446Z","shell.execute_reply.started":"2024-12-31T14:28:59.336076Z","shell.execute_reply":"2024-12-31T14:29:00.508284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.metrics import make_scorer\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense\nfrom tensorflow.keras.optimizers import Adam\nfrom sklearn.base import BaseEstimator, RegressorMixin\n\n# Step 1: Prepare the data\ntarget_column = 'Premium Amount'\nX = processed_df_train_encoded.drop(columns=[target_column])  # Features (all columns except 'Premium Amount')\ny = processed_df_train_encoded[target_column]  # Target ('Premium Amount')\n\n# Log scale the target variable\ny_log = np.log(y + 1)  # Apply log transformation to the target (y + 1 to avoid log(0))\n\n# Split the data into training and testing sets (80% train, 20% test)\nX_train, X_test, y_train, y_test = train_test_split(X, y_log, test_size=0.3, random_state=42)\n\n# Define RMSLE as a scorer for GridSearchCV\ndef rmsle(y_true, y_pred):\n    return np.sqrt(np.mean((np.log(y_true + 1) - np.log(y_pred + 1)) ** 2))\n\nrmsle_scorer = make_scorer(rmsle, greater_is_better=False)\n\n# Step 2: Define the neural network model function\ndef create_model(learning_rate=0.001, layers=3, neurons=64):\n    model = Sequential()\n    model.add(Dense(neurons, input_dim=X_train.shape[1], activation='relu'))  # First layer\n\n    for _ in range(layers - 1):  # Add hidden layers\n        model.add(Dense(neurons, activation='relu'))\n\n    model.add(Dense(1))  # Output layer for regression\n\n    # Compile the model with Adam optimizer and MSE loss\n    model.compile(optimizer=Adam(learning_rate=learning_rate), loss='mean_squared_error')\n\n    return model\n\n# Custom wrapper for Keras model to work with scikit-learn GridSearchCV\nclass KerasRegressorCustom(BaseEstimator, RegressorMixin):\n    def __init__(self, learning_rate=0.001, layers=3, neurons=64, epochs=50, batch_size=32):\n        self.learning_rate = learning_rate\n        self.layers = layers\n        self.neurons = neurons\n        self.epochs = epochs\n        self.batch_size = batch_size\n        self.model = None\n\n    def fit(self, X, y):\n        # Create and train the model\n        self.model = create_model(self.learning_rate, self.layers, self.neurons)\n        self.model.fit(X, y, epochs=self.epochs, batch_size=self.batch_size, verbose=0)\n        return self\n\n    def predict(self, X):\n        return self.model.predict(X).flatten()  # Return a flat array for regression\n\n# Step 3: Define the parameter grid for grid search\nparam_grid = {\n    'learning_rate': [0.01,0.1],  # Learning rate for the optimizer\n    'layers': [2,3],                  # Number of hidden layers\n    'neurons': [32,64],             # Number of neurons in each hidden layer\n    'batch_size': [32,64],               # Batch size for training\n    'epochs': [50]                   # Number of epochs for training\n}\n\n# Step 4: Initialize the custom KerasRegressor\nmodel = KerasRegressorCustom()\n\n# Step 5: Perform grid search with cross-validation\ngrid_search = GridSearchCV(\n    estimator=model,\n    param_grid=param_grid,\n    scoring='neg_root_mean_squared_error',  # Use RMSLE as the metric\n    cv=3,                  # 3-fold cross-validation\n    verbose=1,\n    n_jobs=-1              # Use all available cores\n)\n\ngrid_search.fit(X_train, y_train)\n\n# Step 6: Output the best parameters and score\nprint(\"Best Parameters:\", grid_search.best_params_)\nprint(\"Best RMSLE Score:\", -grid_search.best_score_)  # Negate because RMSLE scorer is negative\n\n# Step 7: Evaluate the best model on the test set\nbest_model = grid_search.best_estimator_\ny_pred_log = best_model.predict(X_test)  # Predictions in log scale\n\n# Reverse the log transformation to get predictions back to original scale\ny_pred = np.exp(y_pred_log) - 1  # Reverse the log transformation\n\n# Compute RMSLE on the test set (after reversing log scaling)\ntest_rmsle = rmsle(np.exp(y_test) - 1, y_pred)  # Reverse the log transformation for y_test as well\n\nprint(f\"Test RMSLE with Best Model: {test_rmsle}\")\n\n# Step 8: Make predictions on the test dataset\ny_test_pred_log = best_model.predict(processed_df_test_encoded)  # Predictions in log scale\ny_test_pred = np.exp(y_test_pred_log) - 1  # Reverse the log transformation","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:29:00.510585Z","iopub.execute_input":"2024-12-31T14:29:00.510892Z","iopub.status.idle":"2024-12-31T15:24:18.168231Z","shell.execute_reply.started":"2024-12-31T14:29:00.510868Z","shell.execute_reply":"2024-12-31T15:24:18.167021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result_df = pd.DataFrame({\n    'id': id_ans,\n    'Premium Amount': y_test_pred\n})\nresult_df.to_csv('results_NN.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T15:24:18.169753Z","iopub.execute_input":"2024-12-31T15:24:18.17062Z","iopub.status.idle":"2024-12-31T15:24:19.298927Z","shell.execute_reply.started":"2024-12-31T15:24:18.170583Z","shell.execute_reply":"2024-12-31T15:24:19.297572Z"}},"outputs":[],"execution_count":null}]}