{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Regression of Insurance Premiums\n\nAnalyzing insurance data to then predict premium amount.","metadata":{}},{"cell_type":"markdown","source":"# Load in Necessary Packages","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport scipy.stats as ss\nimport statsmodels.formula.api as smf\nimport statsmodels.genmod as sm_gm\nimport statsmodels.api as sm\n\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OneHotEncoder, OrdinalEncoder, MinMaxScaler\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import mean_squared_log_error","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-31T00:18:52.867705Z","iopub.execute_input":"2024-12-31T00:18:52.868148Z","iopub.status.idle":"2024-12-31T00:18:56.261146Z","shell.execute_reply.started":"2024-12-31T00:18:52.86811Z","shell.execute_reply":"2024-12-31T00:18:56.259967Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load in Data and Initial Exploration","metadata":{}},{"cell_type":"markdown","source":"## Variable Information\n```\nColumn               | Type | Description\n------               | ---- | -----------\nid                   | NUM  | ID Number\nAge                  | NUM  | Age of the individual (Numerical, in years)\nGender               | CAT  | Gender (Categorical: Male, Female)\nAnnual Income        | NUM  | Annual income in USD of the insured individual (Numerical, skewed)\nMarital Status       | CAT  | Marital Status (Categorical: Married, Divorced, Single)\nNumber of Dependents | NUM  | Number of Dependents (Numerical)\nEducation Level      | CAT  | Highest Level of Education Attained (Categorical: High School, Bachelors's, Master's, PhD)\nOccupation           | CAT  | Occupation (Self-Employed, Employed, Unemployed)\nHealth Score         | NUM  | Score representing health status of individual (Numerical, skewed)\nLocation             | CAT  | Location type (Urban, Rural, Suburban)\nPolicy Type          | CAT  | Type of insurance policy (Categorical: Basic, Comprehensive, Premium)\nPrevious Claims      | NUM  | Number of previous claims made (Numerical, with outliers)\nVehicle Age          | NUM  | Age of the vehicle insured (Numerical)\nCredit Score         | NUM  | Credit score of the insured individual (Numerical)\nInsurance Duration   | NUM  | Duration of the insurance policy (Numerical, in years)\nPolicy Start Date    | DATE | Start date of the insurance policy (Text, improperly formatted)\nCustomer Feedback    | TEXT | Short feedback comments from customers (Text)\nSmoking Status       | CAT  | Smoking status of the insured individual (Categorical: Yes, No)\nExercise Frequency   | CAT  | Frequency of exercise (Categorical: Daily, Weekly, Monthly, Rarely)\nProperty Type        | CAT  | Type of property owned (Categorical: House, Apartment, Condo)\nPremium Amount       | NUM  | Target variable representing the insurance premium amount (Numerical, skewed)\n ```","metadata":{}},{"cell_type":"markdown","source":"## Initial Exploration\nLooking at missing data and the distributions of the variables.","metadata":{}},{"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv', \n                      parse_dates = ['Policy Start Date'])\ndf_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv', \n                       parse_dates = ['Policy Start Date'])\n\nprint('Size of Data:')\nprint('Training Set:', f\"{df_train.shape[0]:,} rows and {df_train.shape[1]} variables\")\nprint('Testing Set: ', f\"{df_test.shape[0]:,} rows and {df_test.shape[1]} variables\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T00:18:56.263015Z","iopub.execute_input":"2024-12-31T00:18:56.263535Z","iopub.status.idle":"2024-12-31T00:19:09.183049Z","shell.execute_reply.started":"2024-12-31T00:18:56.2635Z","shell.execute_reply":"2024-12-31T00:19:09.181719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Missing Data:')\nmissing_data_train = df_train.isna().sum(axis = 1).sum()\nmissing_data_test = df_test.isna().sum(axis = 1).sum()\nprint('Training Set:' , f\"{missing_data_train:,} out of {df_train.size:,} data points\", '|', 'Percent of data values missing:', f\"{missing_data_train / df_train.size * 100:.2f}%\")\nprint('Testing Set: ' , f\"{missing_data_test:,} out of {df_test.size:,} data points\", '  |', 'Percent of data values missing:', f\"{missing_data_test / df_test.size * 100:.2f}%\")","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T00:17:19.329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows = 1, ncols = 2, figsize = (15, 5))\nf1 = sns.heatmap(df_train.sample(frac = 0.05).isna(), ax = ax.flatten()[0])\nf2 = sns.heatmap(df_test.sample(frac = 0.05).isna(), ax = ax.flatten()[1])\nf1.set_title('Training Set')\nf2.set_title('Testing Set')\nf1.set(yticklabels=[])\nf2.set(yticklabels=[])\nf1.tick_params(left = False)\nf2.tick_params(left = False)\nplt.suptitle('Visualization of Missing Data per Column');","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T00:17:19.329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.concat([df_train.isna().sum(axis = 0), \n           df_test.isna().sum(axis = 0),\n           round(df_train.isna().sum(axis = 0) / df_train.shape[0] * 100, 1), \n           round(df_test.isna().sum(axis = 0) / df_test.shape[0] * 100, 1)], \n          axis = 1).rename(columns = {0:'Training', 1:'Testing', 2:'Train %', 3:'Test %'})","metadata":{"execution":{"execution_failed":"2024-12-31T00:17:19.329Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Both the visualization and the table show that the columns `Occupation` and `Previous Claims` have a fair amound of missing rows of data at about 30% of all observations. To a lesser extent, `Number of Dependents`, `Health Score`, `Credit Score`, and `Customer Feedback` have noticeably missing data at between 6% and 11% of all rows.","metadata":{}},{"cell_type":"code","source":"num_cols = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', \n            'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration', 'Premium Amount']\nwith pd.option_context('display.float_format', lambda x: f'{x:,.3f}'):\n    display(pd.concat([df_train[num_cols].describe(), \n                       df_train[num_cols].agg(func = ['median'], axis = 0)]).T)","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T00:17:19.329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train[['Gender', 'Marital Status', 'Education Level', 'Occupation',  'Location', 'Policy Type', \n          'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type']].describe().T","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T00:17:19.329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in ['Gender', 'Marital Status', 'Education Level', 'Occupation',  'Location', 'Policy Type', \n            'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type']:\n    print(df_train[col].value_counts(normalize = True))\n    print('-------------')","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T00:17:19.329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(2, 5, figsize = (16, 4))\naxes = ax.flatten()\nfor i, col in enumerate(['Gender', 'Marital Status', 'Education Level', 'Occupation',  'Location', 'Policy Type', \n            'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type']):\n    sns.countplot(x = col, data = df_train, ax = axes[i])\n    bars = [x.get_height() for x in axes[i].containers[0]]\n    bars_total = sum(bars)\n    axes[i].bar_label(axes[i].containers[0], labels = [f'{x / bars_total:.2%}' for x in bars], padding = -15)\nplt.suptitle('Countplots of Categorical Variables with Proportions')\nplt.tight_layout();","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T00:17:19.33Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(3, 3, figsize = (16, 6))\naxes = ax.flatten()\nfor i, col in enumerate(['Gender', 'Marital Status', 'Education Level', 'Occupation',  'Location', 'Policy Type', \n            'Smoking Status', 'Exercise Frequency', 'Property Type']):\n    sns.boxplot(x = 'Premium Amount', y = col, data = df_train, ax = axes[i])\nplt.suptitle('Distributions of \"Premium Amount\" by Categorical Variables')\nplt.tight_layout();","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T00:17:19.33Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* Overall, each of the categorical variables have nearly equal proportions for their categories.\n* For `Education Level`, there is slightly less individuals with a High School education level compared with the other levels.\n* There are slightly fewer reported indivdiuals who are Unemployed compared with Self-Employed and Employed\n* The distribution of `Premium Amount` does not appear any different between any categorical variable's categories.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(3, 3, figsize = (15, 8))\naxes = ax.flatten()\nprint()\nfor i, col in enumerate(['Age', 'Annual Income', 'Number of Dependents', 'Health Score', \n                         'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration', 'Premium Amount']):\n    col_max, col_min = df_train[col].agg(func = ['max', 'min'], axis = 0)\n    num_bins = int(col_max - col_min + 1)\n    if num_bins > 100:\n        num_bins = 50\n    df_train[col].plot(kind = 'hist', ec = 'w', bins = num_bins, ax = axes[i], title = col)\n    \n    col_mean, col_median = df_train[col].agg(func = ['mean', 'median'], axis = 0)\n    line_mean = axes[i].axvline(x = col_mean, color = 'red', ls = '-.')\n    line_median = axes[i].axvline(x = col_median, color = 'purple', ls = '--')\nplt.legend([line_mean, line_median], ['Mean', 'Median'])\nplt.suptitle('Histograms of Numerical Data')\nplt.tight_layout();","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T00:17:19.33Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* `Age`, `Number of Dependents`, `Vehicle Age`, and `Insurance Duration` are fairly uniform.\n* `Annual Income` and `Premium Amount` are both greatly skewed to the right, with `Premium Amount` having a lot of zero values.\n* `Previous Claims` is skewed to the left, with many individuals having had between zero and three claims.\n* `Health Score` is unimodal with a very slight skew to the right with a mean of about 15 and a standard deviation of about 12.\n* `Credit Score` is mostly uniform with a slight dip in the amount of individuals who have a credit score lower than 400.","metadata":{}},{"cell_type":"markdown","source":"## Data Cleaning To Do\n1. Rename columns to standarize.\n2. Reformat the `Policy Start Date` column, extracting year, month, and day of month information.\n3. Impute or remove certain columns in order to deal with missing values. (Mean/Median/Frequency Imputation, Random Sample Imputation, or Multiple Imputation/IterativeImputer)\n4. Log transform Annual Income and Premium Amount to reduce skewness.","metadata":{}},{"cell_type":"markdown","source":"# Data Cleaning","metadata":{}},{"cell_type":"markdown","source":"## Rename Columns","metadata":{}},{"cell_type":"code","source":"old_col_names = ['id', 'Age', 'Gender', 'Annual Income', 'Marital Status', 'Number of Dependents', 'Education Level', 'Occupation', 'Health Score',\n                 'Location', 'Policy Type', 'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration', 'Policy Start Date',\n                 'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type', 'Premium Amount']\nnew_col_names = ['id', 'age', 'gender', 'annual_income', 'marital_status', 'number_dependents', 'education_level', 'occupation', 'health_score',\n                 'location', 'policy_type', 'previous_claims', 'vehicle_age', 'credit_score', 'insurance_duration', 'policy_start_date',\n                 'customer_feedback', 'smoking_status', 'exercise_frequency', 'property_type', 'premium_amount']\ndict_columns_rename = dict(zip(old_col_names, new_col_names))\n\ndf_train = df_train.rename(columns = dict_columns_rename, inplace = False)\ndf_test = df_test.rename(columns = dict_columns_rename, inplace = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T00:19:09.185492Z","iopub.execute_input":"2024-12-31T00:19:09.186048Z","iopub.status.idle":"2024-12-31T00:19:09.475173Z","shell.execute_reply.started":"2024-12-31T00:19:09.185999Z","shell.execute_reply":"2024-12-31T00:19:09.473709Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Reformat Policy Start Date","metadata":{}},{"cell_type":"code","source":"def get_date_info(df):\n    df_new = df.copy()\n\n    df_new['policy_start_year'] = df_new['policy_start_date'].dt.year\n    df_new['policy_start_month'] = df_new['policy_start_date'].dt.month\n    df_new['policy_start_day'] = df_new['policy_start_date'].dt.day\n\n    return df_new\n\ndf_train = get_date_info(df_train)\ndf_test = get_date_info(df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T00:19:09.476838Z","iopub.execute_input":"2024-12-31T00:19:09.477253Z","iopub.status.idle":"2024-12-31T00:19:10.139491Z","shell.execute_reply.started":"2024-12-31T00:19:09.477214Z","shell.execute_reply":"2024-12-31T00:19:10.138212Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Impute Missing Data\nUsing median values for numerical data and creating a separate category for missing data for categorical variables.\n\nThis less complex way of imputation is used, however since many of the variables appear uniformily distributed, Random Sample Imputation or Multiple Imputation could also be used so as to not disrupt these distributions.","metadata":{}},{"cell_type":"code","source":"id_cols = ['id']\nnum_cols = ['age', 'annual_income', 'number_dependents', 'health_score', 'previous_claims', 'vehicle_age', \n            'credit_score', 'insurance_duration']\ncat_cols = ['gender', 'marital_status', 'education_level', 'occupation', 'location', \n            'policy_type', 'customer_feedback', 'smoking_status', 'exercise_frequency', 'property_type']\npolicy_date_cols = ['policy_start_year', 'policy_start_month', 'policy_start_day']\ntarget_col = ['premium_amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T00:19:10.141693Z","iopub.execute_input":"2024-12-31T00:19:10.14253Z","iopub.status.idle":"2024-12-31T00:19:10.148797Z","shell.execute_reply.started":"2024-12-31T00:19:10.142487Z","shell.execute_reply":"2024-12-31T00:19:10.147366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Processing using pipelines with impute and encoding\n# Setting up both onehot/dummy encoding and ordinal encoding\n\n# Separating the target variable from imputation\ndf_train_target = df_train['premium_amount'].copy()\n\nnumerical_impute = SimpleImputer(strategy = \"median\")\n\ncat_onehot_impute = Pipeline(steps = [\n    ('imputer', SimpleImputer(strategy = \"constant\", fill_value = \"No-Data\")), \n    ('onehot', OneHotEncoder(sparse_output = False, drop = 'if_binary', handle_unknown = 'ignore'))])\n\ncat_ordinal_impute = Pipeline(steps = [\n    ('imputer', SimpleImputer(strategy = \"constant\", fill_value = \"No-Data\")), \n    ('ordinal', OrdinalEncoder())])\n\n# Preprocessor for Onehot Encoding\npreprocessor_onehot = ColumnTransformer(transformers = [\n    ('id_col', 'passthrough', id_cols), \n    ('numerical', numerical_impute, num_cols), \n    ('onehot', cat_onehot_impute, cat_cols), \n    ('date_cols', 'passthrough', policy_date_cols)])\n\n# Preprocessor for Ordinal and Label Encoding\npreprocessor_ordinal = ColumnTransformer(transformers = [\n    ('id_col', 'passthrough', id_cols), \n    ('numerical', numerical_impute, num_cols), \n    ('ordinal', cat_ordinal_impute, cat_cols), \n    ('date_cols', 'passthrough', policy_date_cols)])\n\ndf_train_oh = preprocessor_onehot.fit_transform(df_train)\ndf_train_oh = pd.DataFrame(df_train_oh, columns = pd.Series(preprocessor_onehot.get_feature_names_out()).str.split('__').str[1])\n\ndf_test_oh = preprocessor_onehot.fit_transform(df_test)\ndf_test_oh = pd.DataFrame(df_test_oh, columns = pd.Series(preprocessor_onehot.get_feature_names_out()).str.split('__').str[1])\n\ndf_train_or = preprocessor_ordinal.fit_transform(df_train)\ndf_train_or = pd.DataFrame(df_train_or, columns = pd.Series(preprocessor_ordinal.get_feature_names_out()).str.split('__').str[1])\n\ndf_test_or = preprocessor_ordinal.fit_transform(df_test)\ndf_test_or = pd.DataFrame(df_test_or, columns = pd.Series(preprocessor_ordinal.get_feature_names_out()).str.split('__').str[1])\n\n# Adding the Target variable back to the training data\ndf_train_oh = pd.concat([df_train_oh, df_train_target], axis = 1)\ndf_train_or = pd.concat([df_train_or, df_train_target], axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T00:19:10.150494Z","iopub.execute_input":"2024-12-31T00:19:10.151061Z","iopub.status.idle":"2024-12-31T00:19:34.521081Z","shell.execute_reply.started":"2024-12-31T00:19:10.151005Z","shell.execute_reply":"2024-12-31T00:19:34.519882Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Log Transforming Annual Income and Premium Amount\nThe distributions appeared skewed. Log transforming helps to transform the variables into something that is more normally distributed","metadata":{}},{"cell_type":"code","source":"df_train_oh['annual_income_log'] = np.log1p(df_train_oh['annual_income'])\ndf_test_oh['annual_income_log'] = np.log1p(df_test_oh['annual_income'])\n\ndf_train_oh['premium_amount_log'] = np.log1p(df_train_oh['premium_amount'])\n\n\ndf_train_or['annual_income_log'] = np.log1p(df_train_or['annual_income'])\ndf_test_or['annual_income_log'] = np.log1p(df_test_or['annual_income'])\n\ndf_train_or['premium_amount_log'] = np.log1p(df_train_or['premium_amount'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T00:19:34.522465Z","iopub.execute_input":"2024-12-31T00:19:34.522875Z","iopub.status.idle":"2024-12-31T00:19:34.761913Z","shell.execute_reply.started":"2024-12-31T00:19:34.52277Z","shell.execute_reply":"2024-12-31T00:19:34.760479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(2, 2, figsize = (10, 4))\naxes = ax.flatten()\nfor i, col in enumerate(['annual_income', 'annual_income_log', 'premium_amount', 'premium_amount_log']):\n    df_train_oh[col].plot(kind = 'hist', ec = 'w', bins = 50, ax = axes[i], title = col)\n\nplt.suptitle('Histograms of Numerical Data')\nplt.tight_layout();","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T00:17:19.331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = MinMaxScaler(feature_range = (0, 1))\n\ndf_train_oh[num_cols + policy_date_cols] = scaler.fit_transform(df_train_oh[num_cols + policy_date_cols])\ndf_test_oh[num_cols + policy_date_cols] = scaler.fit_transform(df_test_oh[num_cols + policy_date_cols])\n\ndf_train_or[num_cols + policy_date_cols] = scaler.fit_transform(df_train_or[num_cols + policy_date_cols])\ndf_test_or[num_cols + policy_date_cols] = scaler.fit_transform(df_test_or[num_cols + policy_date_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T00:19:34.764359Z","iopub.execute_input":"2024-12-31T00:19:34.764742Z","iopub.status.idle":"2024-12-31T00:19:35.771322Z","shell.execute_reply.started":"2024-12-31T00:19:34.76468Z","shell.execute_reply":"2024-12-31T00:19:35.770036Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploration of Data","metadata":{}},{"cell_type":"code","source":"df_train_or.corr().style.background_gradient()","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T00:17:19.331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize = (16, 7))\naxes = ax.flatten()\nsns.heatmap(df_train_oh.corr(), cmap = 'PuOr', vmin = -1, vmax = 1, ax = axes[0])\naxes[0].set_title('OneHot Encoded Data')\nsns.heatmap(df_train_or.corr(), cmap = 'PuOr', vmin = -1, vmax = 1, ax = axes[1])\naxes[1].set_title('Ordinal and Label Encoded Data')\nplt.tight_layout();","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T00:17:19.331Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Initial Data Modeling\n1. Multiple Linear Regression","metadata":{}},{"cell_type":"code","source":"expl_cols_or = ['age', 'number_dependents', 'health_score', \n               'previous_claims', 'vehicle_age', 'credit_score', 'insurance_duration', \n               'gender', 'marital_status', 'education_level', 'occupation', 'location', \n               'policy_type', 'customer_feedback', 'smoking_status', \n               'exercise_frequency', 'property_type', 'policy_start_year', \n               'policy_start_month', 'policy_start_day']\nexpl_cols_oh = [\n    'age', 'number_dependents', 'health_score', 'previous_claims', 'vehicle_age', \n    'credit_score', 'insurance_duration', 'gender_Male', 'marital_status_Divorced', 'marital_status_Married', \n    'marital_status_No-Data', 'marital_status_Single', \"education_level_Bachelor's\", 'education_level_High School', \"education_level_Master's\", \n    'education_level_PhD', 'occupation_Employed', 'occupation_No-Data', 'occupation_Self-Employed', 'occupation_Unemployed', \n    'location_Rural', 'location_Suburban', 'location_Urban', 'policy_type_Basic', 'policy_type_Comprehensive', \n    'policy_type_Premium', 'customer_feedback_Average', 'customer_feedback_Good', 'customer_feedback_No-Data', 'customer_feedback_Poor', \n    'smoking_status_Yes', 'exercise_frequency_Daily', 'exercise_frequency_Monthly', 'exercise_frequency_Rarely', 'exercise_frequency_Weekly', \n    'property_type_Apartment', 'property_type_Condo', 'property_type_House', 'policy_start_year', 'policy_start_month', \n    'policy_start_day'\n    ]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T00:19:35.772609Z","iopub.execute_input":"2024-12-31T00:19:35.772972Z","iopub.status.idle":"2024-12-31T00:19:35.780497Z","shell.execute_reply.started":"2024-12-31T00:19:35.772938Z","shell.execute_reply":"2024-12-31T00:19:35.779035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def test_log_models(explanatory, dependent, df, is_premium_log):\n    ex = df[explanatory].copy()\n    X_1 = sm.add_constant(ex)\n\n    y_1 = df[dependent].copy()\n    model_1 = sm.OLS(exog = X_1, endog = y_1)\n    res_1 = model_1.fit()\n\n    pred_in_sample_1 = res_1.predict(X_1)\n    if is_premium_log:\n        pred_in_sample_1 = np.expm1(pred_in_sample_1)\n    rmsle_score_1 = mean_squared_log_error(y_true = df['premium_amount'], y_pred = pred_in_sample_1, squared = False)\n    aic_1 = res_1.aic\n    bic_1 = res_1.bic\n\n    return (rmsle_score_1, aic_1, bic_1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T00:19:35.782327Z","iopub.execute_input":"2024-12-31T00:19:35.782858Z","iopub.status.idle":"2024-12-31T00:19:35.802104Z","shell.execute_reply.started":"2024-12-31T00:19:35.782776Z","shell.execute_reply":"2024-12-31T00:19:35.800704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# no log\nrscore, aic_score, bic_score = test_log_models(explanatory = expl_cols_oh + ['annual_income'], dependent = ['premium_amount'], df = df_train_oh, is_premium_log = False)\nprint('No Log:\\n', 'Score:', rscore, 'AIC:', aic_score, 'BIC:', bic_score)\n\n# Log income only\nrscore, aic_score, bic_score = test_log_models(explanatory = expl_cols_oh + ['annual_income_log'], dependent = ['premium_amount'], df = df_train_oh, is_premium_log = False)\nprint('Log Income Only:\\n', 'Score:', rscore, 'AIC:', aic_score, 'BIC:', bic_score)\n\n# log premium only\nrscore, aic_score, bic_score = test_log_models(explanatory = expl_cols_oh + ['annual_income'], dependent = ['premium_amount_log'], df = df_train_oh, is_premium_log = True)\nprint('Log Premium Only:\\n', 'Score:', rscore, 'AIC:', aic_score, 'BIC:', bic_score)\n\n# log both\nrscore, aic_score, bic_score = test_log_models(explanatory = expl_cols_oh + ['annual_income_log'], dependent = ['premium_amount_log'], df = df_train_oh, is_premium_log = True)\nprint('Log Both:\\n', 'Score:', rscore, 'AIC:', aic_score, 'BIC:', bic_score)","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T00:17:19.332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# no log\nrscore, aic_score, bic_score = test_log_models(explanatory = expl_cols_or + ['annual_income'], dependent = ['premium_amount'], df = df_train_or, is_premium_log = False)\nprint('No Log:\\n', 'Score:', rscore, 'AIC:', aic_score, 'BIC:', bic_score)\n\n# Log income only\nrscore, aic_score, bic_score = test_log_models(explanatory = expl_cols_or + ['annual_income_log'], dependent = ['premium_amount'], df = df_train_or, is_premium_log = False)\nprint('Log Income Only:\\n', 'Score:', rscore, 'AIC:', aic_score, 'BIC:', bic_score)\n\n# log premium only\nrscore, aic_score, bic_score = test_log_models(explanatory = expl_cols_or + ['annual_income'], dependent = ['premium_amount_log'], df = df_train_or, is_premium_log = True)\nprint('Log Premium Only:\\n', 'Score:', rscore, 'AIC:', aic_score, 'BIC:', bic_score)\n\n# log both\nrscore, aic_score, bic_score = test_log_models(explanatory = expl_cols_or + ['annual_income_log'], dependent = ['premium_amount_log'], df = df_train_or, is_premium_log = True)\nprint('Log Both:\\n', 'Score:', rscore, 'AIC:', aic_score, 'BIC:', bic_score)","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T00:17:19.332Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Using AIC, BIC, and RMSLE for comparisons, between the models using categorical data that was onehot encoded and the models using categorical data that was label encoded, the models with onehot encoded data overall had slightly lower AIC, BIC, and RMSLE scores. When comparing linear models, the better performing model should have a lower AIC or BIC score. As such, for further modeling only the onehot encoded data will be used.\n\nThere are four different models being compared with either onehot encoded or ordinal and label encoded data: 1. Where no data is log transformed, 2. Where only `Annual Income` is log transformed, 3. Where only the target variable `Premium Amount` is log transformed, and 4. Where both `Annual Income` and `Premium Amount` are log transformed.\n\nBetween no transformations and logging only `Annual Income`, the later model performed better. Additionally, the model where only `Premium Amount` was logged performed better over either no logging or where only `Annual Income` was logged. However, The model where both variables were logged performed worse than just logging `Premium Amount`. This is all just under the context of the baseline Linear Regresion model using the OLS method, however it does look like an indication to use a model where just the target variable `Premium Amount` is logged.","metadata":{}},{"cell_type":"code","source":"id_col = ['id']\nexplanatory_cols = [\n    'age', 'annual_income', 'number_dependents', 'health_score', 'previous_claims', 'vehicle_age', \n    'credit_score', 'insurance_duration', 'gender_Male', 'marital_status_Divorced', 'marital_status_Married', \n    'marital_status_No-Data', 'marital_status_Single', \"education_level_Bachelor's\", 'education_level_High School', \"education_level_Master's\", \n    'education_level_PhD', 'occupation_Employed', 'occupation_No-Data', 'occupation_Self-Employed', 'occupation_Unemployed', \n    'location_Rural', 'location_Suburban', 'location_Urban', 'policy_type_Basic', 'policy_type_Comprehensive', \n    'policy_type_Premium', 'customer_feedback_Average', 'customer_feedback_Good', 'customer_feedback_No-Data', 'customer_feedback_Poor', \n    'smoking_status_Yes', 'exercise_frequency_Daily', 'exercise_frequency_Monthly', 'exercise_frequency_Rarely', 'exercise_frequency_Weekly', \n    'property_type_Apartment', 'property_type_Condo', 'property_type_House', 'policy_start_year', 'policy_start_month', \n    'policy_start_day']\ntarget_col = 'premium_amount_log'\n\ntest_id = df_test_oh[id_col].copy()\n\ndata = df_train_oh[explanatory_cols].copy()\ndata_test = df_test_oh[explanatory_cols].copy()\n\ny = df_train_oh[target_col].copy()\n\ny_not_logged = df_train_oh['premium_amount'].copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T00:19:35.803506Z","iopub.execute_input":"2024-12-31T00:19:35.804008Z","iopub.status.idle":"2024-12-31T00:19:37.27611Z","shell.execute_reply.started":"2024-12-31T00:19:35.803954Z","shell.execute_reply":"2024-12-31T00:19:37.274812Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The linear model below is not using logged `Premium Amount` for the sake of easier explainability when considering how the independent variables affect the target variable. All other future models will used logged `Premium Amount`.","metadata":{}},{"cell_type":"code","source":"X = sm.add_constant(data)\n\nmodel = sm.OLS(exog = X, endog = y_not_logged)\nmodel_fit = model.fit()\nres = model_fit.resid","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T00:17:19.332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(model_fit.summary())","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T00:17:19.332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_in_sample = model_fit.predict(X)\n\nprint('In Sample RMSLE:', mean_squared_log_error(y_true = y_not_logged, y_pred = pred_in_sample, squared = False))","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T00:17:19.332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# These are all of the explanatory variables as before with the exception of variables that had a p-value that was greater than the standard 0.05 threshold for statistical significance.\nexplanatory_cols_2 = [\n    'age', 'health_score', 'previous_claims', \n    'credit_score', 'marital_status_Divorced', 'marital_status_Married', \n    'marital_status_No-Data', 'marital_status_Single', \"education_level_Bachelor's\", 'education_level_High School', \"education_level_Master's\", \n    'education_level_PhD', 'occupation_Employed', 'occupation_No-Data', 'occupation_Self-Employed', 'occupation_Unemployed', \n    'location_Rural', 'location_Suburban', 'location_Urban', 'policy_type_Basic', 'policy_type_Comprehensive', \n    'policy_type_Premium', 'customer_feedback_Average', 'customer_feedback_Good', 'customer_feedback_No-Data', 'customer_feedback_Poor', \n    'exercise_frequency_Daily', 'exercise_frequency_Monthly', 'exercise_frequency_Rarely', 'exercise_frequency_Weekly', \n    'property_type_Apartment', 'property_type_Condo', 'property_type_House', 'policy_start_year', 'policy_start_month']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T00:19:37.27748Z","iopub.execute_input":"2024-12-31T00:19:37.277841Z","iopub.status.idle":"2024-12-31T00:19:37.284375Z","shell.execute_reply.started":"2024-12-31T00:19:37.277809Z","shell.execute_reply":"2024-12-31T00:19:37.28314Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Conducting Additional Baseline Modelling\nUsing different types of regression models and comparing performance.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV, KFold, train_test_split, RandomizedSearchCV\n\nfrom sklearn.linear_model import LinearRegression, LassoCV, RidgeCV, Ridge\nfrom sklearn.ensemble import AdaBoostRegressor, BaggingRegressor, GradientBoostingRegressor, RandomForestRegressor\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.mixture import GaussianMixture\n\nfrom xgboost import XGBRegressor\n\nfrom sklearn.metrics import r2_score, explained_variance_score, mean_squared_log_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T00:19:37.286114Z","iopub.execute_input":"2024-12-31T00:19:37.286535Z","iopub.status.idle":"2024-12-31T00:19:37.712434Z","shell.execute_reply.started":"2024-12-31T00:19:37.2865Z","shell.execute_reply":"2024-12-31T00:19:37.710896Z"},"_kg_hide-input":false},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_test_scoring(model, name, x_train, x_valid, y_train, y_valid, is_logged = True)->tuple:\n    model.fit(x_train, y_train)\n    \n    train_predict = model.predict(x_train)\n    valid_predict = model.predict(x_valid)\n\n    if is_logged: # Reverts logged variables back to unlogged ones\n        y_train = np.expm1(y_train)\n        y_valid = np.expm1(y_valid)\n        train_predict = np.expm1(train_predict)\n        valid_predict = np.expm1(valid_predict)\n    \n    RMSLE_2 = mean_squared_log_error(y_train, train_predict, squared=False)\n    r2_2 = r2_score(y_train, train_predict)\n    evs_2 = explained_variance_score(y_train, train_predict)\n    \n    RMSLE = mean_squared_log_error(y_valid, valid_predict, squared=False)\n    r2 = r2_score(y_valid, valid_predict)\n    evs = explained_variance_score(y_valid, valid_predict)\n        \n    scores_test = pd.Series([RMSLE_2, r2_2, evs_2], index = ['RMSLE', 'r2', 'evs'], name = name + '_test')\n    scores_valid = pd.Series([RMSLE, r2, evs], index = ['RMSLE', 'r2', 'evs'], name = name + '_valid')\n    return scores_test, scores_valid","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T00:19:37.889327Z","iopub.execute_input":"2024-12-31T00:19:37.889776Z","iopub.status.idle":"2024-12-31T00:19:37.912691Z","shell.execute_reply.started":"2024-12-31T00:19:37.889735Z","shell.execute_reply":"2024-12-31T00:19:37.911189Z"},"_kg_hide-input":false},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(data, y, test_size = 0.20, random_state = 123)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T00:19:37.914736Z","iopub.execute_input":"2024-12-31T00:19:37.915321Z","iopub.status.idle":"2024-12-31T00:19:39.10774Z","shell.execute_reply.started":"2024-12-31T00:19:37.915266Z","shell.execute_reply":"2024-12-31T00:19:39.106377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mod_lin = LinearRegression()\nmod_lin.fit(X_train, y_train)\n\ny_pred_lin = mod_lin.predict(X_val)\ny_pred_lin = np.expm1(y_pred_lin)\n\nmean_squared_log_error(y_true = np.expm1(y_val), y_pred = y_pred_lin, squared = False)","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T00:17:19.333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mod_lin = LinearRegression()\nmod_lasso = LassoCV(n_jobs=-1, random_state=0)\nmod_ridge = RidgeCV(scoring = 'neg_root_mean_squared_error')\nmod_ada = AdaBoostRegressor(n_estimators = 30)\nmod_rf = RandomForestRegressor(n_estimators = 75, max_depth = 10, n_jobs = -1)\nmod_dt = DecisionTreeRegressor(max_depth = 20, random_state = 0)\nmod_xgb = XGBRegressor(random_state = 0)\nmod_gmm = GaussianMixture(random_state = 0)\n\nmodels = dict(zip(['Linear', 'LassoCV', 'RidgeCV', 'AdaBoost', 'RF', 'DecisionTree', 'XGB', 'GMM'], \n                  [mod_lin, mod_lasso, mod_ridge, mod_ada, mod_rf, mod_dt, mod_xgb, mod_gmm]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:32:49.910699Z","iopub.execute_input":"2024-12-31T02:32:49.911228Z","iopub.status.idle":"2024-12-31T02:32:49.918394Z","shell.execute_reply.started":"2024-12-31T02:32:49.911189Z","shell.execute_reply":"2024-12-31T02:32:49.917192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_models = []\nfor name, model in models.items():\n    t_test, t_valid = train_test_scoring(model, name, X_train, X_val, y_train, y_val)\n    \n    test_models.append(t_test)\n    test_models.append(t_valid)\n    print(f'Fitted {name} model')\n    \ntm_df = pd.DataFrame(test_models)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:32:52.908432Z","iopub.execute_input":"2024-12-31T02:32:52.908995Z","iopub.status.idle":"2024-12-31T02:33:22.837352Z","shell.execute_reply.started":"2024-12-31T02:32:52.908949Z","shell.execute_reply":"2024-12-31T02:33:22.835905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tm_df.style.background_gradient(cmap = 'tab20', axis = 0, gmap = [1, 2, 4, 5, 7, 8, 10, 11, 13, 14, 15, 16, 1, 2, 4, 5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:33:39.007449Z","iopub.execute_input":"2024-12-31T02:33:39.008179Z","iopub.status.idle":"2024-12-31T02:33:39.034193Z","shell.execute_reply.started":"2024-12-31T02:33:39.008136Z","shell.execute_reply":"2024-12-31T02:33:39.032887Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"When considering the main Root Mean Squared Logarithmic Error metric for the baseline models, RandomForest and XGBoost tree-based models do the best. This is followed by Linear Regression, Lasso, and Ridge Regression models. The AdaBoost and the simple DecisionTree models models perform even worse, with the single decision tree model considerably overfitting on the training set. \n\nFinally, the Gaussian Mixture model performs the worst by a large margin. This makes sense as the model assumes the data comes from gaussian distributions, yet earlier visualization and analysis shows that many of the variables do not appear normally distributed. \n\nThe `r2` score is the coefficient of determination, which describes the proportion of the variance in the dependent variable `Premium Amount` that is explained by the independent variables. The `evs` score is the explained variance score which is similar to $R^2$, where a value of 1.0 indicates the best possible score with lower values being worse.\n\nGiven the baseline models' performances, RandomForest and XGBRegressor models will be used going foreward along with a simpler Ridge regularized linear regression model.","metadata":{}},{"cell_type":"markdown","source":"# Final Data Modeling","metadata":{}},{"cell_type":"markdown","source":"## XGBoost Model","metadata":{}},{"cell_type":"code","source":"# The comments are the best performing values from various runs with differing value ranges for parameters\n# Each Parameter in the parameter grid was run seperately, with the other parameter being commented out.\nparam_grid_xgb = {\n    'learning_rate':np.arange(0.14, 0.18, 0.01), # 0.15, #0.16\n    'n_estimators':np.arange(50, 140, 25) # 100, # 50, # 50\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:53:52.284233Z","iopub.execute_input":"2024-12-31T01:53:52.284762Z","iopub.status.idle":"2024-12-31T01:53:52.291399Z","shell.execute_reply.started":"2024-12-31T01:53:52.284718Z","shell.execute_reply":"2024-12-31T01:53:52.289859Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#mod_xgb_1 = XGBRegressor(objective = \"reg:squaredlogerror\", random_state = 0)\n\n#gs_xgb = GridSearchCV(estimator = mod_xgb_1, param_grid = param_grid_xgb, scoring = 'neg_mean_squared_log_error', n_jobs = - 1, cv = 3, verbose = 2)\n\n#gs_xgb.fit(X_train, y_train)\n#print('Best Parameters:', gs_xgb.best_params_)\n#print('Best Cross-Validation Score:', gs_xgb.best_score_)\n\n#y_pred_xgb = gs_xgb.predict(X_val)\n#y_pred_xgb_e = np.expm1(y_pred_xgb)\n    \n#print('RMSLE:', mean_squared_log_error(np.expm1(y_val), y_pred_xgb_e, squared = False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T01:54:25.679352Z","iopub.execute_input":"2024-12-31T01:54:25.679877Z","iopub.status.idle":"2024-12-31T01:59:53.993727Z","shell.execute_reply.started":"2024-12-31T01:54:25.679839Z","shell.execute_reply":"2024-12-31T01:59:53.991988Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"XGBoost Results:\n\nBest Parameters: {'learning_rate': 0.15000000000000002, 'n_estimators': 100}\n\nBest Cross-Validation Score: -0.02499191690013004\n\nRMSLE: 1.047792148462716","metadata":{}},{"cell_type":"markdown","source":"## Linear Regression with Ridge Regularization (L2)","metadata":{}},{"cell_type":"code","source":"param_grid_ridge = {\n    'alpha':[0.1, 1.0, 10.0], # 0.1\n    'solver':['svd', 'lsqr', 'saga'] # lsqr\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:24:47.914284Z","iopub.execute_input":"2024-12-31T02:24:47.914926Z","iopub.status.idle":"2024-12-31T02:24:47.921822Z","shell.execute_reply.started":"2024-12-31T02:24:47.914881Z","shell.execute_reply":"2024-12-31T02:24:47.920214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#mod_ridge_1 = Ridge(random_state = 0)\n\n#gs_ridge = GridSearchCV(estimator = mod_ridge_1, \n#                        param_grid = param_grid_ridge, \n#                        scoring = 'neg_mean_squared_log_error', \n#                        n_jobs = - 1, \n#                        verbose = 2)\n\n#gs_ridge.fit(X_train, y_train)\n#print('Best Parameters:', gs_ridge.best_params_)\n#print('Best Cross-Validation Score:', gs_ridge.best_score_)\n\n#y_pred_ridge = gs_ridge.predict(X_val)\n#y_pred_ridge_e = np.expm1(y_pred_ridge)\n    \n#print('RMSLE:', mean_squared_log_error(np.expm1(y_val), y_pred_ridge_e, squared = False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:24:48.572827Z","iopub.execute_input":"2024-12-31T02:24:48.573332Z","iopub.status.idle":"2024-12-31T02:27:18.009525Z","shell.execute_reply.started":"2024-12-31T02:24:48.573291Z","shell.execute_reply":"2024-12-31T02:27:18.005638Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Ridge Regression Results:\n\nBest Parameters: {'alpha': 0.1, 'solver': 'lsqr'}\n\nBest Cross-Validation Score: -0.027033998241263523\n\nRMSLE: 1.0851119683365593","metadata":{}},{"cell_type":"markdown","source":"## RandomForest Model","metadata":{}},{"cell_type":"code","source":"# The comments are the best performing values from various runs with differing value ranges for parameters\n# Each Parameter in the parameter grid was run seperately, with the other parameter being commented out.\nparam_grid_rf = {\n    'n_estimators':[50, 75, 100, 200, 500, 1000],\n    'max_features':[1.0, 'sqrt'],\n    'ccp_alpha':[0.0, 0.001, 0.002, 0.005, 0.01, 0.02, 0.03, 0.05], \n    'max_samples':[0.4, 0.5, 0.75, 0.80]\n}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#mod_rf_1 = RandomForestRegressor(max_depth = 10, criterion = 'squared_error', random_state = 0)\n\n#gs_rf = GridSearchCV(estimator = mod_rf_1, param_grid = param_grid_rf, scoring = 'neg_mean_squared_log_error', n_jobs = - 1, cv = 3)\n#print(3 * )\n\n#gs_xgb.fit(X_train, y_train)\n#print('Best Parameters:', gs_xgb.best_params_)\n#print('Best Cross-Validation Score:', gs_xgb.best_score_)\n\n#y_pred_xgb = gs_xgb.predict(X_val)\n#y_pred_xgb_e = np.expm1(y_pred_xgb)\n    \n#print('RMSLE:', mean_squared_log_error(np.expm1(y_val), y_pred_xgb_e, squared = False))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Creating Submission Output","metadata":{}},{"cell_type":"code","source":"mod_xgb_2 = XGBRegressor(learning_rate = 0.15, \n                         n_estimators = 100, \n                         max_depth = 5, \n                         subsample = 0.75, \n                         reg_lambda = 5, \n                         objective = \"reg:squaredlogerror\", \n                         random_state = 0)\n\nmod_xgb_2.fit(data, y)\n\ny_pred_xgb_2 = mod_xgb_2.predict(data_test)\ny_pred_xgb_2 = np.expm1(y_pred_xgb_2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T05:57:48.154655Z","iopub.execute_input":"2024-12-31T05:57:48.155298Z","iopub.status.idle":"2024-12-31T05:58:03.784765Z","shell.execute_reply.started":"2024-12-31T05:57:48.155252Z","shell.execute_reply":"2024-12-31T05:58:03.783543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_id = test_id.astype(int).copy()\ndf_submission = pd.concat([test_id, pd.Series(y_pred_xgb_2, name = 'Premium Amount')], axis = 1)\ndf_submission.to_csv('submission.csv', index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:05:01.287218Z","iopub.execute_input":"2024-12-31T06:05:01.288076Z","iopub.status.idle":"2024-12-31T06:05:02.547477Z","shell.execute_reply.started":"2024-12-31T06:05:01.288006Z","shell.execute_reply":"2024-12-31T06:05:02.5459Z"}},"outputs":[],"execution_count":null}]}