{"metadata":{"kernelspec":{"display_name":".venv","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.13.1"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 1. Loading the dataset","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport optuna\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport re\nimport warnings\n\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:30:49.849905Z","iopub.execute_input":"2024-12-31T01:30:49.850285Z","iopub.status.idle":"2024-12-31T01:30:49.856404Z","shell.execute_reply.started":"2024-12-31T01:30:49.850257Z","shell.execute_reply":"2024-12-31T01:30:49.854982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)\ndf_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndf_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\ndf_sample_submission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\nprint(f'''\nTraining set info:\\n{df_train.info()}\nTest set info:\\n{df_test.info()}\nTraining set shape: {df_train.shape}  \nTest set shape: {df_test.shape} \n''')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:30:49.86116Z","iopub.execute_input":"2024-12-31T01:30:49.861537Z","iopub.status.idle":"2024-12-31T01:30:59.74612Z","shell.execute_reply.started":"2024-12-31T01:30:49.861505Z","shell.execute_reply":"2024-12-31T01:30:59.744729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:30:59.747566Z","iopub.execute_input":"2024-12-31T01:30:59.748006Z","iopub.status.idle":"2024-12-31T01:30:59.777004Z","shell.execute_reply.started":"2024-12-31T01:30:59.74796Z","shell.execute_reply":"2024-12-31T01:30:59.775727Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Data preprocessing\n\n### 2.1 Distribuation of categorical and continuous data","metadata":{}},{"cell_type":"code","source":"# Let's split the features into continuous and categorical\ncontinuous_columns = [key for key in df_train.keys() if df_train[key].dtype in (\"int64\", \"float64\")]\ncategorical_columns = [key for key in df_train.keys() if df_train[key].dtype == \"object\"]\ncategorical_columns.remove('Policy Start Date') # Need to be transformed later\nprint(f\"Continuous : {len(continuous_columns)}, Categorical : {len(categorical_columns)}\")\nprint(f\"\"\"\nContinuous  features: {continuous_columns},\\n\nCategorical features: {categorical_columns}\n      \"\"\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:30:59.77889Z","iopub.execute_input":"2024-12-31T01:30:59.779286Z","iopub.status.idle":"2024-12-31T01:30:59.786343Z","shell.execute_reply.started":"2024-12-31T01:30:59.779255Z","shell.execute_reply":"2024-12-31T01:30:59.785324Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Our target varible is **Premium Amount** and it is a continuous feature. This task is a regression task meaning based on all the other feature we want to predict the how much will the PRemium Amount be paid. ","metadata":{}},{"cell_type":"code","source":"print(f'''\nNaN values in each feature and target varible.\\n\nTraining set:\\n\n{df_train.isna().sum()}      \n================================================\nTest set:\\n\n{df_test.isna().sum()}      \n''')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:30:59.788061Z","iopub.execute_input":"2024-12-31T01:30:59.78844Z","iopub.status.idle":"2024-12-31T01:31:00.830831Z","shell.execute_reply.started":"2024-12-31T01:30:59.788409Z","shell.execute_reply":"2024-12-31T01:31:00.829453Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Since I am using tree based models that can nativly handle NaN values I will leave them. Also thanks to insights from Chris Deotte, who observed the following: \"NaN values are strongly correlated with the target, Premium Amount. This means that NAN contains additional information and we do not want to impute it and hide NANs from our models.\"\n","metadata":{}},{"cell_type":"code","source":"for col in continuous_columns:\n    plt.figure(figsize=(8, 6))\n    sns.histplot(df_train[col], kde=True, color='skyblue')\n    plt.title(f'Distribution of {col}', fontsize=15)\n    plt.xlabel(col, fontsize=12)\n    plt.ylabel('Frequency', fontsize=12)\n    plt.show()\nfor col in categorical_columns:\n    plt.figure(figsize=(10, 6))\n    \n    # Count plot (bar plot showing frequency of each category)\n    sns.countplot(x=col, hue=col, data=df_train, palette='viridis')\n    \n    plt.title(f'Distribution of {col}', fontsize=15)\n    plt.xlabel(col, fontsize=12)\n    plt.ylabel('Count', fontsize=12)\n    plt.tight_layout() \n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:31:00.832012Z","iopub.execute_input":"2024-12-31T01:31:00.832402Z","iopub.status.idle":"2024-12-31T01:32:11.598565Z","shell.execute_reply.started":"2024-12-31T01:31:00.832369Z","shell.execute_reply":"2024-12-31T01:32:11.597552Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's transform Policy Start Date into something useful and create new features.\nIdeas for new features:\n\n1. Ratio between Annual Income and Number of Dependents\n2. Ratio between Annual Income and Previous Claims\n3. Ratio between Previous Claims and Insurance Duration\n4. Ratio between Previous Claims and Credit Score\n5. Ratio between Age and Credit Score\n6. Ratio between Education Level and Credit Score\n\n\nLet's also calculate the skew from our continuous values to select features that reguire log transformation","metadata":{}},{"cell_type":"code","source":"df_train[continuous_columns].skew().sort_values(ascending=False).apply(lambda x: format(x, 'f'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:11.599531Z","iopub.execute_input":"2024-12-31T01:32:11.599815Z","iopub.status.idle":"2024-12-31T01:32:11.814009Z","shell.execute_reply.started":"2024-12-31T01:32:11.599791Z","shell.execute_reply":"2024-12-31T01:32:11.813007Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There 3 variables (Annual Income, Premium Amount, Previous Claims) are strongly right skewed and one (Health Score) that is lightly right skewed. We will apply log transformation for all of them,\n","metadata":{}},{"cell_type":"markdown","source":"### 2.2 Handling NaN values ","metadata":{}},{"cell_type":"markdown","source":"To do that and create new features we need to first encode our features","metadata":{}},{"cell_type":"code","source":"continuous_columns_01 = continuous_columns.copy()\ncontinuous_columns_01.remove('Premium Amount')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:11.815054Z","iopub.execute_input":"2024-12-31T01:32:11.815346Z","iopub.status.idle":"2024-12-31T01:32:11.820179Z","shell.execute_reply.started":"2024-12-31T01:32:11.81532Z","shell.execute_reply":"2024-12-31T01:32:11.8188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df_train[continuous_columns].min().apply(lambda x: format(x, 'f')), '\\n', df_test[continuous_columns_01].min().apply(lambda x: format(x, 'f')))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:11.82305Z","iopub.execute_input":"2024-12-31T01:32:11.823378Z","iopub.status.idle":"2024-12-31T01:32:12.02328Z","shell.execute_reply.started":"2024-12-31T01:32:11.823352Z","shell.execute_reply":"2024-12-31T01:32:12.022222Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Since there are no negative values we will impute NaNs with -1.","metadata":{}},{"cell_type":"code","source":"df_train.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:12.02533Z","iopub.execute_input":"2024-12-31T01:32:12.025791Z","iopub.status.idle":"2024-12-31T01:32:12.658902Z","shell.execute_reply.started":"2024-12-31T01:32:12.025755Z","shell.execute_reply":"2024-12-31T01:32:12.657659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fill NaN values for continuous columns with -1\ndf_train[continuous_columns] = df_train[continuous_columns].fillna(-1)\ndf_test[continuous_columns_01] = df_test[continuous_columns_01].fillna(-1)\n\n# Fill NaN values for categorical columns with np.nan \ndf_train[categorical_columns] = df_train[categorical_columns].fillna(np.nan)\ndf_test[categorical_columns] = df_test[categorical_columns].fillna(np.nan)\n\nprint(df_train.isnull().sum())\nprint(df_test.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:12.660237Z","iopub.execute_input":"2024-12-31T01:32:12.660621Z","iopub.status.idle":"2024-12-31T01:32:16.037066Z","shell.execute_reply.started":"2024-12-31T01:32:12.660523Z","shell.execute_reply":"2024-12-31T01:32:16.035908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nominal_features = ['Gender', 'Marital Status', 'Occupation', 'Location', 'Smoking Status', 'Property Type']\nordinal_features = ['Education Level', 'Policy Type', 'Customer Feedback', 'Exercise Frequency']\n# Define the order for ordinal encoding\nordinal_categories = [\n    ['High School', \"Bachelor's\", \"Master's\", 'PhD'],  # Education Level\n    ['Basic', 'Comprehensive', 'Premium'],            # Policy Type\n    ['Poor', 'Average', 'Good'],                       # Customer Feedback\n    ['Rarely', 'Monthly', 'Weekly', 'Daily'],          # Exercise Frequency\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:16.038436Z","iopub.execute_input":"2024-12-31T01:32:16.038824Z","iopub.status.idle":"2024-12-31T01:32:16.045136Z","shell.execute_reply.started":"2024-12-31T01:32:16.038786Z","shell.execute_reply":"2024-12-31T01:32:16.043785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\n\nordinal_encoder = OrdinalEncoder(categories=ordinal_categories, handle_unknown='use_encoded_value', \n                                unknown_value=np.nan, encoded_missing_value=np.nan)\n\nordinal_encoded = ordinal_encoder.fit_transform(df_train[ordinal_features])\n\nordinal_encoded_df = pd.DataFrame(ordinal_encoded, columns=ordinal_features)\n\ndf_train[ordinal_features] = ordinal_encoded_df\n\ndf_train.head(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:16.046353Z","iopub.execute_input":"2024-12-31T01:32:16.046742Z","iopub.status.idle":"2024-12-31T01:32:17.46611Z","shell.execute_reply.started":"2024-12-31T01:32:16.046708Z","shell.execute_reply":"2024-12-31T01:32:17.464898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\nohe = OneHotEncoder(handle_unknown='error', sparse_output=False).set_output(transform='pandas')\nohetrans = ohe.fit_transform(df_train[nominal_features])\ndf_train = pd.concat([df_train, ohetrans], axis=1).drop(columns=nominal_features)\ndf_train.head(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:17.466895Z","iopub.execute_input":"2024-12-31T01:32:17.467201Z","iopub.status.idle":"2024-12-31T01:32:21.100988Z","shell.execute_reply.started":"2024-12-31T01:32:17.467176Z","shell.execute_reply":"2024-12-31T01:32:21.100073Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 2.3 Feature Engineering","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA\n\nhealth_features = df_train[['Health Score', 'Smoking Status_No', 'Smoking Status_Yes', 'Exercise Frequency']]\npca = PCA(n_components=1)\ndf_train['PCA_Health'] = pca.fit_transform(health_features)\ndf_train.head(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:21.102016Z","iopub.execute_input":"2024-12-31T01:32:21.102281Z","iopub.status.idle":"2024-12-31T01:32:22.669131Z","shell.execute_reply.started":"2024-12-31T01:32:21.102259Z","shell.execute_reply":"2024-12-31T01:32:22.668138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from datetime import timedelta\n\n\ndef feature_engineering(df: pd.DataFrame) -> pd.DataFrame:\n    \"\"\"\n    Perform feature engineering on the given DataFrame.\n    \n    This function transforms the 'Policy Start Date' into Year, Month, Day and cyclic features\n    and creates new features based on the specified ratios. \n    \n    Parameters:\n    df (pd.DataFrame): The input DataFrame containing the relevant features.\n    \n    Returns:\n    pd.DataFrame: The DataFrame with new features added.\n    \"\"\"\n    \n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    \n    # Transform 'Policy Start Date' into various date-related features\n    df['Year'] = df['Policy Start Date'].dt.year\n    df['Day'] = df['Policy Start Date'].dt.day\n    df['Month'] = df['Policy Start Date'].dt.month\n    df['Day_of_year'] = df['Policy Start Date'].dt.dayofyear\n    df['Day_of_week'] = df['Policy Start Date'].dt.weekday\n    df['Sin_day_of_week'] = np.sin(2 * np.pi * df['Day_of_week'] / 7)\n    df['Cos_day_of_week'] = np.cos(2 * np.pi * df['Day_of_week'] / 7)\n    df = df.drop(columns='Policy Start Date')\n    # Create new features based on the specified ratios\n    df['Income_to_Dependants'] = df['Annual Income'] / np.maximum(df['Number of Dependents'], 1)\n    df['Income_to_Claims'] = df['Annual Income'] / np.maximum(df['Previous Claims'], 1)\n    df['Claims_to_Duration'] = df['Previous Claims'] / np.maximum(df['Insurance Duration'], 1)\n    df['Claims_to_Credit'] = df['Previous Claims'] / np.maximum(df['Credit Score'], 1)\n    df['Claims_to_Income'] = df['Previous Claims'] / np.maximum(df['Annual Income'], 1)\n    df['Dependents_to_Duration'] = df['Number of Dependents'] / np.maximum(df['Insurance Duration'], 1)\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:22.670115Z","iopub.execute_input":"2024-12-31T01:32:22.670392Z","iopub.status.idle":"2024-12-31T01:32:22.678781Z","shell.execute_reply.started":"2024-12-31T01:32:22.670366Z","shell.execute_reply":"2024-12-31T01:32:22.677335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = feature_engineering(df_train)\n\ndf_train.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:22.679872Z","iopub.execute_input":"2024-12-31T01:32:22.680425Z","iopub.status.idle":"2024-12-31T01:32:23.788222Z","shell.execute_reply.started":"2024-12-31T01:32:22.680384Z","shell.execute_reply":"2024-12-31T01:32:23.787288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Let's do the same with test\nordinal_encoded_test = ordinal_encoder.fit_transform(df_test[ordinal_features])\n\nordinal_encoded_df_test = pd.DataFrame(ordinal_encoded_test, columns=ordinal_features)\n\ndf_test[ordinal_features] = ordinal_encoded_df_test\n\ndf_test.head(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:23.789187Z","iopub.execute_input":"2024-12-31T01:32:23.789456Z","iopub.status.idle":"2024-12-31T01:32:24.654246Z","shell.execute_reply.started":"2024-12-31T01:32:23.789425Z","shell.execute_reply":"2024-12-31T01:32:24.653121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ohetrans = ohe.fit_transform(df_test[nominal_features])\ndf_test = pd.concat([df_test, ohetrans], axis=1).drop(columns=nominal_features)\ndf_test.head(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:24.655071Z","iopub.execute_input":"2024-12-31T01:32:24.655365Z","iopub.status.idle":"2024-12-31T01:32:27.025511Z","shell.execute_reply.started":"2024-12-31T01:32:24.65534Z","shell.execute_reply":"2024-12-31T01:32:27.024574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"health_features = df_test[['Health Score', 'Smoking Status_No', 'Smoking Status_Yes', 'Exercise Frequency']]\ndf_test['PCA_Health'] = pca.fit_transform(health_features)\ndf_test.head(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:27.026451Z","iopub.execute_input":"2024-12-31T01:32:27.026718Z","iopub.status.idle":"2024-12-31T01:32:27.810555Z","shell.execute_reply.started":"2024-12-31T01:32:27.026695Z","shell.execute_reply":"2024-12-31T01:32:27.808262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test = feature_engineering(df_test)\n\ndf_test.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:27.811659Z","iopub.execute_input":"2024-12-31T01:32:27.812133Z","iopub.status.idle":"2024-12-31T01:32:28.56536Z","shell.execute_reply.started":"2024-12-31T01:32:27.812077Z","shell.execute_reply":"2024-12-31T01:32:28.563886Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Since there are features that are right skewed. We will take the log of them","metadata":{}},{"cell_type":"code","source":"skewed_features = ['Annual Income', 'Previous Claims', 'Health Score']\n\nfor feature in skewed_features:\n    df_train[feature] =  np.log1p(df_train[feature])\n    df_test[feature] =  np.log1p(df_test[feature])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:28.56664Z","iopub.execute_input":"2024-12-31T01:32:28.567086Z","iopub.status.idle":"2024-12-31T01:32:28.703867Z","shell.execute_reply.started":"2024-12-31T01:32:28.567032Z","shell.execute_reply":"2024-12-31T01:32:28.70298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:28.704875Z","iopub.execute_input":"2024-12-31T01:32:28.705281Z","iopub.status.idle":"2024-12-31T01:32:28.738525Z","shell.execute_reply.started":"2024-12-31T01:32:28.70524Z","shell.execute_reply":"2024-12-31T01:32:28.737191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:28.743669Z","iopub.execute_input":"2024-12-31T01:32:28.744061Z","iopub.status.idle":"2024-12-31T01:32:28.756828Z","shell.execute_reply.started":"2024-12-31T01:32:28.744029Z","shell.execute_reply":"2024-12-31T01:32:28.755349Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 2.4 Feature Selection","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX = df_train.drop(columns=['Premium Amount', 'id'])  # Features\ny_log = np.log1p(df_train['Premium Amount'])\nX_train, X_valid, y_train, y_valid = train_test_split(X, y_log, test_size=0.2,  shuffle=True, random_state=42)\nprint(f'''\nX_train shape: {X_train.shape}\nX_valid shape: {X_valid.shape}\ny_train shape: {y_train.shape}\ny_valid shape: {y_valid.shape} \n''')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:28.758532Z","iopub.execute_input":"2024-12-31T01:32:28.758923Z","iopub.status.idle":"2024-12-31T01:32:30.010809Z","shell.execute_reply.started":"2024-12-31T01:32:28.758891Z","shell.execute_reply":"2024-12-31T01:32:30.00956Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lightgbm import LGBMRegressor\n\nmodel_lgb_features = LGBMRegressor(n_jobs=-1, random_state=42)\n\n# Fit the model\nmodel_lgb_features.fit(\n    X_train, y_train,\n    eval_set=[(X_valid, y_valid)],\n    eval_metric='rmse',\n)\n\n# Get feature importances\nfeature_importance = pd.DataFrame({\n    'Feature': X_train.columns,  \n    'Importance': model_lgb_features.feature_importances_ \n}).sort_values(by='Importance', ascending=False)\n\nprint(feature_importance)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:30.011797Z","iopub.execute_input":"2024-12-31T01:32:30.012153Z","iopub.status.idle":"2024-12-31T01:32:40.326412Z","shell.execute_reply.started":"2024-12-31T01:32:30.012124Z","shell.execute_reply":"2024-12-31T01:32:40.325229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(5, 10))\nplt.barh(feature_importance['Feature'], feature_importance['Importance'])\nplt.gca().invert_yaxis()\nplt.xlabel(\"Importance\")\nplt.title(\"Feature Importance from CatBoost\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:40.327226Z","iopub.execute_input":"2024-12-31T01:32:40.327878Z","iopub.status.idle":"2024-12-31T01:32:40.922094Z","shell.execute_reply.started":"2024-12-31T01:32:40.327845Z","shell.execute_reply":"2024-12-31T01:32:40.920916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_importance['Feature'][:20]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:32:40.923401Z","iopub.execute_input":"2024-12-31T01:32:40.923825Z","iopub.status.idle":"2024-12-31T01:32:40.931671Z","shell.execute_reply.started":"2024-12-31T01:32:40.923761Z","shell.execute_reply":"2024-12-31T01:32:40.9305Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Model Training","metadata":{}},{"cell_type":"code","source":"from lightgbm import LGBMRegressor\nfrom sklearn.metrics import mean_squared_error\n\n# List of number feature\nk_values = [2, 3, 4, 5, 10, 20, 25, 30]\n\n# Dictionary to store results\nresults = {}\n\n# Find optimal (or best) number of features\nfor k in k_values:\n    # Select top k features\n    top_k_features = feature_importance['Feature'][:k]\n    \n    # Subset training and validation data\n    X_train_k = X_train[top_k_features]\n    X_valid_k = X_valid[top_k_features]\n    \n    model_lgb = LGBMRegressor(num_leaves=32, max_depth=5, n_estimators=250, reg_alpha= 0.5, random_state=42, n_jobs=-1)\n    \n    # Fit the model\n    model_lgb.fit(\n        X_train_k, y_train,\n        eval_set=[(X_valid_k, y_valid)],\n        eval_metric='rmse',\n    )\n    \n    # use root_mean_squared_error locally\n    # on kaggle it doesn't work for some reason\n    y_valid_pred = model_lgb.predict(X_valid_k)\n    rmsle = np.sqrt(mean_squared_error(y_valid, y_valid_pred))\n    \n    # Store the results\n    results[k] = rmsle\n    print(f\"Top {k} features RMSLE: {rmsle}\")\n\n# Display final results\nprint(\"\\nSummary of RMSE for different k values:\")\nfor k, rmsle in results.items():\n    print(f\"Top {k} features: RMSLE = {rmsle}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:34:17.31627Z","iopub.execute_input":"2024-12-31T01:34:17.316705Z","iopub.status.idle":"2024-12-31T01:35:52.90544Z","shell.execute_reply.started":"2024-12-31T01:34:17.316671Z","shell.execute_reply":"2024-12-31T01:35:52.901777Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Fine Tuning Model's hyperparameters","metadata":{}},{"cell_type":"code","source":"import optuna\n\n# Define the objective function for Optuna\ndef objective(trial):\n    # Suggest hyperparameters\n    num_leaves = trial.suggest_int(\"num_leaves\", 8, 100)\n    max_depth = trial.suggest_int(\"max_depth\", 3, 10)\n    n_estimators = trial.suggest_int(\"n_estimators\", 100, 500)\n    reg_alpha = trial.suggest_loguniform(\"reg_alpha\", 1e-3, 10.0)\n    reg_lambda = trial.suggest_loguniform(\"reg_lambda\", 1e-3, 10.0)\n    learning_rate = trial.suggest_loguniform(\"learning_rate\", 1e-3, 0.1)\n\n    # Select top k features (adjust k as needed or make it a parameter)\n    k = 30\n    top_k_features = feature_importance['Feature'][:k]\n    X_train_k = X_train[top_k_features]\n    X_valid_k = X_valid[top_k_features]\n\n    # Initialize the model\n    model = LGBMRegressor(\n        num_leaves=num_leaves,\n        max_depth=max_depth,\n        n_estimators=n_estimators,\n        reg_alpha=reg_alpha,\n        reg_lambda=reg_lambda,\n        learning_rate=learning_rate,\n        random_state=42,\n        n_jobs=-1\n    )\n\n    # Train the model\n    model.fit(\n        X_train_k, y_train,\n        eval_set=[(X_valid_k, y_valid)],\n        eval_metric=\"rmse\",\n    )\n\n    # Make predictions and compute RMSE\n    y_valid_pred = model.predict(X_valid_k)\n    rmsle = np.sqrt(mean_squared_error(y_valid, y_valid_pred))\n\n    return rmsle\n\n# Create an Optuna study and optimize\nstudy = optuna.create_study(direction=\"minimize\")\nstudy.optimize(objective, n_trials=50, show_progress_bar=True)\n\n# Display the best hyperparameters\nprint(\"Best hyperparameters:\")\nprint(study.best_params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:36:57.160798Z","iopub.execute_input":"2024-12-31T01:36:57.161287Z","iopub.status.idle":"2024-12-31T02:00:26.901034Z","shell.execute_reply.started":"2024-12-31T01:36:57.161231Z","shell.execute_reply":"2024-12-31T02:00:26.900038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the final model using the best parameters\nbest_params = study.best_params\n\nfinal_model = LGBMRegressor(\n    num_leaves=best_params[\"num_leaves\"],\n    max_depth=best_params[\"max_depth\"],\n    n_estimators=best_params[\"n_estimators\"],\n    reg_alpha=best_params[\"reg_alpha\"],\n    reg_lambda=best_params[\"reg_lambda\"],\n    learning_rate=best_params[\"learning_rate\"],\n    random_state=42,\n    n_jobs=-1\n)\n\n# Train with the final set of features\nk = 30\nfinal_top_k_features = feature_importance['Feature'][:k]\nX_train_k = X_train[final_top_k_features]\nX_valid_k = X_valid[final_top_k_features]\n\nfinal_model.fit(\n    X_train_k, y_train,\n    eval_set=[(X_valid_k, y_valid)],\n    eval_metric=\"rmse\"\n)\n\n# Evaluate final model\nfinal_predictions = final_model.predict(X_valid_k)\nfinal_rmsle = np.sqrt(mean_squared_error(y_valid, final_predictions))\nprint(f\"Final model RMSLE: {final_rmsle}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:01:00.444351Z","iopub.execute_input":"2024-12-31T02:01:00.44467Z","iopub.status.idle":"2024-12-31T02:01:33.955828Z","shell.execute_reply.started":"2024-12-31T02:01:00.444643Z","shell.execute_reply":"2024-12-31T02:01:33.95455Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model Testing","metadata":{}},{"cell_type":"code","source":"# Predict using the model\npredictions_log = final_model.predict(df_test[final_top_k_features])\npredictions = np.expm1(predictions_log)  # Reverse log transformation\npredictions[:2]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:37.370893Z","iopub.execute_input":"2024-12-31T02:02:37.371283Z","iopub.status.idle":"2024-12-31T02:02:45.875345Z","shell.execute_reply.started":"2024-12-31T02:02:37.371252Z","shell.execute_reply":"2024-12-31T02:02:45.874044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save predictions\nres = pd.DataFrame({\n    \"id\": df_test[\"id\"],\n    \"Premium Amount\": predictions\n})\nres = res.set_index(\"id\")\nres.to_csv(\"Submission03.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:04:00.754621Z","iopub.execute_input":"2024-12-31T02:04:00.754998Z","iopub.status.idle":"2024-12-31T02:04:02.59292Z","shell.execute_reply.started":"2024-12-31T02:04:00.754967Z","shell.execute_reply":"2024-12-31T02:04:02.591761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}