{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30839,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Insurance Premium Analysis and Regression","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom scipy import stats\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport math\n\nfrom datetime import datetime\n\nfrom itertools import combinations\n\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split, RandomizedSearchCV\nfrom sklearn.feature_selection import SelectKBest, f_classif\n\nfrom sklearn.ensemble import RandomForestRegressor\n\nfrom sklearn.metrics import r2_score, mean_squared_log_error\n\nfrom xgboost import XGBRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:30:30.46206Z","iopub.execute_input":"2025-01-21T05:30:30.462629Z","iopub.status.idle":"2025-01-21T05:30:30.470974Z","shell.execute_reply.started":"2025-01-21T05:30:30.462568Z","shell.execute_reply":"2025-01-21T05:30:30.469607Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Initial Look at the Data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\n\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:30:30.472877Z","iopub.execute_input":"2025-01-21T05:30:30.47339Z","iopub.status.idle":"2025-01-21T05:30:36.463331Z","shell.execute_reply.started":"2025-01-21T05:30:30.473338Z","shell.execute_reply":"2025-01-21T05:30:36.462314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:38:14.775417Z","iopub.execute_input":"2025-01-21T05:38:14.775946Z","iopub.status.idle":"2025-01-21T05:38:18.623021Z","shell.execute_reply.started":"2025-01-21T05:38:14.775908Z","shell.execute_reply":"2025-01-21T05:38:18.621848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:30:41.505613Z","iopub.execute_input":"2025-01-21T05:30:41.506056Z","iopub.status.idle":"2025-01-21T05:30:42.172992Z","shell.execute_reply.started":"2025-01-21T05:30:41.506Z","shell.execute_reply":"2025-01-21T05:30:42.172159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"Premium Amount\"].plot(kind = \"hist\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:30:42.174471Z","iopub.execute_input":"2025-01-21T05:30:42.174816Z","iopub.status.idle":"2025-01-21T05:30:42.800384Z","shell.execute_reply.started":"2025-01-21T05:30:42.174787Z","shell.execute_reply":"2025-01-21T05:30:42.7994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in df:\n    print(col)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:30:42.801368Z","iopub.execute_input":"2025-01-21T05:30:42.801725Z","iopub.status.idle":"2025-01-21T05:30:42.81074Z","shell.execute_reply.started":"2025-01-21T05:30:42.801668Z","shell.execute_reply":"2025-01-21T05:30:42.809616Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Cleaning","metadata":{}},{"cell_type":"code","source":"all_df = df.columns.to_list()\nint_df = list(df.select_dtypes(include=['float', 'int' ]).columns)\nchar_df = list(df.select_dtypes(include=['object' ]).columns)\n\nint_df.remove('id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:30:42.811882Z","iopub.execute_input":"2025-01-21T05:30:42.812187Z","iopub.status.idle":"2025-01-21T05:30:43.029644Z","shell.execute_reply.started":"2025-01-21T05:30:42.81216Z","shell.execute_reply":"2025-01-21T05:30:43.028445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[df[\"Annual Income\"].isna()]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:30:43.030786Z","iopub.execute_input":"2025-01-21T05:30:43.031178Z","iopub.status.idle":"2025-01-21T05:30:43.118738Z","shell.execute_reply.started":"2025-01-21T05:30:43.031138Z","shell.execute_reply":"2025-01-21T05:30:43.117561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isna().sum()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:30:43.122071Z","iopub.execute_input":"2025-01-21T05:30:43.122362Z","iopub.status.idle":"2025-01-21T05:30:43.816294Z","shell.execute_reply.started":"2025-01-21T05:30:43.122336Z","shell.execute_reply":"2025-01-21T05:30:43.815059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Noted that all categorical columns could have thier missing values preserved so we will include a \"Blank\" value in all. \n#We also have one NA value in \"Insurance Duration\" which we will give to the 1 category\n\nfor col in char_df:\n    df[col] = df[col].fillna(\"Blank\")\n    print(col, \"has this many NA values\", df[col].isna().sum())\n\nfor col in char_df:\n    df_test[col] = df_test[col].fillna(\"Blank\")\n    print(col, \"has this many NA values\", df_test[col].isna().sum())\n\ndf[\"Insurance Duration\"] = df[\"Insurance Duration\"].fillna(1.0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:30:43.818921Z","iopub.execute_input":"2025-01-21T05:30:43.819269Z","iopub.status.idle":"2025-01-21T05:30:47.127302Z","shell.execute_reply.started":"2025-01-21T05:30:43.819238Z","shell.execute_reply":"2025-01-21T05:30:47.125973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#For the numerical data, we will impute the missing data with the Median\n\nfor col in int_df:\n    median = df[col].median()\n    if df[col].isna().sum() > 1:\n        df[col] = df[col].fillna(median)\n        print(col, \"has this many NA values:\", df[col].isna().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:30:47.128572Z","iopub.execute_input":"2025-01-21T05:30:47.128977Z","iopub.status.idle":"2025-01-21T05:30:47.47572Z","shell.execute_reply.started":"2025-01-21T05:30:47.128936Z","shell.execute_reply":"2025-01-21T05:30:47.474689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:30:47.476573Z","iopub.execute_input":"2025-01-21T05:30:47.476844Z","iopub.status.idle":"2025-01-21T05:30:48.176787Z","shell.execute_reply.started":"2025-01-21T05:30:47.47682Z","shell.execute_reply":"2025-01-21T05:30:48.175647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Annual Income is skewed - we can fix this with taking the log instead\ndf[\"Annual Income\"].plot(kind=\"hist\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:30:48.178196Z","iopub.execute_input":"2025-01-21T05:30:48.178608Z","iopub.status.idle":"2025-01-21T05:30:48.544669Z","shell.execute_reply.started":"2025-01-21T05:30:48.178568Z","shell.execute_reply":"2025-01-21T05:30:48.543559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"Annual Income\"] = np.log1p(df[\"Annual Income\"])\ndf_test[\"Annual Income\"] = np.log1p(df[\"Annual Income\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:30:48.545813Z","iopub.execute_input":"2025-01-21T05:30:48.546116Z","iopub.status.idle":"2025-01-21T05:30:48.57679Z","shell.execute_reply.started":"2025-01-21T05:30:48.546088Z","shell.execute_reply":"2025-01-21T05:30:48.575628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"Annual Income\"].plot(kind=\"hist\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:30:48.57773Z","iopub.execute_input":"2025-01-21T05:30:48.578034Z","iopub.status.idle":"2025-01-21T05:30:48.918821Z","shell.execute_reply.started":"2025-01-21T05:30:48.578007Z","shell.execute_reply":"2025-01-21T05:30:48.917824Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Extraction and Selection","metadata":{}},{"cell_type":"code","source":"temp_df = df[[\"Age\", \"Annual Income\", \"Number of Dependents\", \"Health Score\", \"Vehicle Age\",\"Credit Score\", \"Insurance Duration\"]].copy()\n\ntemp_df.dropna(axis = 0, inplace = True)\n\ntemp_df.corrwith(df[\"Premium Amount\"]).sort_values().plot(kind= \"barh\", figsize = (6,6))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:30:48.919865Z","iopub.execute_input":"2025-01-21T05:30:48.920142Z","iopub.status.idle":"2025-01-21T05:30:49.4362Z","shell.execute_reply.started":"2025-01-21T05:30:48.920117Z","shell.execute_reply":"2025-01-21T05:30:49.434504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for val in char_df:\n    if (len(df[val].value_counts()) < 40):\n        print(df[val].value_counts())\n    else:\n        print(val, \" Is quite large...\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:30:49.437528Z","iopub.execute_input":"2025-01-21T05:30:49.437865Z","iopub.status.idle":"2025-01-21T05:30:51.790869Z","shell.execute_reply.started":"2025-01-21T05:30:49.437837Z","shell.execute_reply":"2025-01-21T05:30:51.789582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#The Policy Start Date column is tricky as there are so many differnt values. There are a couple of routes we can take:\n\n## 1. We can work with the dates as is and put stress on our models\n## 2. We can bin the dates into Year or Month_Year, losing some of the specificity to gain computation efficiency and model accuracy\n\ndf[\"Policy Start Date\"] = pd.to_datetime(df[\"Policy Start Date\"]).dt.strftime('%Y')\ndf_test[\"Policy Start Date\"] = pd.to_datetime(df[\"Policy Start Date\"]).dt.strftime('%Y')\n\nprint(df.head())\n\nprint(df[\"Policy Start Date\"].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:30:51.791671Z","iopub.execute_input":"2025-01-21T05:30:51.791959Z","iopub.status.idle":"2025-01-21T05:31:03.546862Z","shell.execute_reply.started":"2025-01-21T05:30:51.791933Z","shell.execute_reply":"2025-01-21T05:31:03.545621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"temp_char = char_df.copy()\n\ndummy_df = pd.get_dummies(df[temp_char],drop_first = True)\ndummy_test = pd.get_dummies(df_test[temp_char], drop_first = True)\n\ndummy_df.corrwith(df[\"Premium Amount\"]).sort_values().plot(kind= \"barh\", figsize = (10,25))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:31:03.547974Z","iopub.execute_input":"2025-01-21T05:31:03.548323Z","iopub.status.idle":"2025-01-21T05:31:08.617798Z","shell.execute_reply.started":"2025-01-21T05:31:03.548293Z","shell.execute_reply":"2025-01-21T05:31:08.616501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dummy_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:31:08.618856Z","iopub.execute_input":"2025-01-21T05:31:08.619226Z","iopub.status.idle":"2025-01-21T05:31:08.71802Z","shell.execute_reply.started":"2025-01-21T05:31:08.619183Z","shell.execute_reply":"2025-01-21T05:31:08.716977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dummy_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:31:08.718916Z","iopub.execute_input":"2025-01-21T05:31:08.719188Z","iopub.status.idle":"2025-01-21T05:31:08.817735Z","shell.execute_reply.started":"2025-01-21T05:31:08.719164Z","shell.execute_reply":"2025-01-21T05:31:08.816398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_pairwise_columns(df):\n    # Check if all columns are of type bool\n    if not all(df.dtypes.apply(lambda x: x == 'bool')):\n        raise ValueError(\"All columns in the DataFrame must be of type bool.\")\n    \n    # Create pairwise combinations of columns\n    column_pairs = list(combinations(df.columns, 2))\n    \n    # Use a dictionary to store new columns\n    pairwise_data = {\n        f\"{col1}_AND_{col2}\": np.logical_and(df[col1], df[col2]) \n        for col1, col2 in column_pairs\n    }\n    \n    # Add all new columns to the DataFrame at once\n    df = pd.concat([df, pd.DataFrame(pairwise_data, index = df.index)], axis=1)\n    \n    return df\n\n\n\nnew = create_pairwise_columns(dummy_df)\nnew_test = create_pairwise_columns(dummy_test)\n\nnew\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:31:08.818948Z","iopub.execute_input":"2025-01-21T05:31:08.819264Z","iopub.status.idle":"2025-01-21T05:31:10.974239Z","shell.execute_reply.started":"2025-01-21T05:31:08.819235Z","shell.execute_reply":"2025-01-21T05:31:10.973024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corrnew = new.corrwith(df[\"Premium Amount\"])\n\ncorrnew.dropna(inplace= True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:31:10.975384Z","iopub.execute_input":"2025-01-21T05:31:10.975823Z","iopub.status.idle":"2025-01-21T05:31:20.387764Z","shell.execute_reply.started":"2025-01-21T05:31:10.975772Z","shell.execute_reply":"2025-01-21T05:31:20.386573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corrnew1 = corrnew.sort_values(ascending = False, key = abs)[:80]\n\ncorrnew1 = corrnew1.sort_values()\n\nax = corrnew1.plot(kind= \"barh\", figsize = (10,35))\n\nplt.axvline(0.005, c='red', linestyle='--')\nplt.axvline(-0.005, c='red', linestyle='--')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:31:20.388969Z","iopub.execute_input":"2025-01-21T05:31:20.389379Z","iopub.status.idle":"2025-01-21T05:31:21.711774Z","shell.execute_reply.started":"2025-01-21T05:31:20.389325Z","shell.execute_reply":"2025-01-21T05:31:21.710388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def find_relevant(series, threshold):\n    relevant_columns = []\n    for col in series.index:\n        if abs(series[col]) > threshold and \"_AND_\" in col:\n            relevant_columns.append(col) \n    return relevant_columns\n\nnewtest = find_relevant(corrnew, 0.005) \nprint(\"\\n\\nRelevant columns:\", len(newtest))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:31:21.712846Z","iopub.execute_input":"2025-01-21T05:31:21.713157Z","iopub.status.idle":"2025-01-21T05:31:21.722537Z","shell.execute_reply.started":"2025-01-21T05:31:21.71313Z","shell.execute_reply":"2025-01-21T05:31:21.720837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.concat([df, new[newtest]], axis = 1)\ndf_test = pd.concat([df_test, new_test[newtest]], axis = 1)\n\nchar_df.extend(newtest)\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:31:21.723944Z","iopub.execute_input":"2025-01-21T05:31:21.724343Z","iopub.status.idle":"2025-01-21T05:31:23.499798Z","shell.execute_reply.started":"2025-01-21T05:31:21.724307Z","shell.execute_reply":"2025-01-21T05:31:23.498766Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model Selection","metadata":{}},{"cell_type":"code","source":"ohencode_df = np.array(char_df)\nnum_df = ('Age', 'Annual Income', 'Number of Dependents', 'Health Score', 'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration')\n\ncat_imputer = SimpleImputer(strategy= \"constant\", fill_value = \"Blank\")\nnum_imputer = SimpleImputer(strategy = \"median\")\n\nnum_scaler = StandardScaler()\n\nnum_transformer = Pipeline(steps = [('impute', num_imputer)])\nnum_transformer_s = Pipeline(steps = [('impute', num_imputer), ('scale', num_scaler)])\nohcat_transformer = Pipeline( steps = [('impute', cat_imputer), ('encode', OneHotEncoder(handle_unknown='ignore'))])\n\npreprocessor = ColumnTransformer(transformers = [(\"oh_cat\", ohcat_transformer, ohencode_df), (\"num\", num_transformer_s, num_df)])\n\nfeature_selection = SelectKBest(score_func=f_classif, k=15)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:31:23.503932Z","iopub.execute_input":"2025-01-21T05:31:23.504253Z","iopub.status.idle":"2025-01-21T05:31:23.51107Z","shell.execute_reply.started":"2025-01-21T05:31:23.504227Z","shell.execute_reply":"2025-01-21T05:31:23.509642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df.drop(['Premium Amount'], axis = 1)\ny = df['Premium Amount']\n\nX_train, X_valid, y_train, y_valid = train_test_split(X, y, train_size=0.8, test_size=0.2, random_state = 0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:31:23.512892Z","iopub.execute_input":"2025-01-21T05:31:23.513218Z","iopub.status.idle":"2025-01-21T05:31:25.07281Z","shell.execute_reply.started":"2025-01-21T05:31:23.51319Z","shell.execute_reply":"2025-01-21T05:31:25.071653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = XGBRegressor(learning_rate = np.float64(0.02028633586934382), max_depth = 9, n_estimators = 161, subsample = np.float64(0.6770233310419052), random_state = 0)\n\n#We need to take the log of our y variable as there are issues with negative values in the output\ny_train_log = np.log1p(y_train)\n\nxgb_pipeline = Pipeline(steps = [('preprocessor', preprocessor),\n                                 ('feature_selection', feature_selection),\n                                 ('model', model)]\n                        )\n\nxgb_pipeline.fit(X_train, y_train_log)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:31:25.074488Z","iopub.execute_input":"2025-01-21T05:31:25.07495Z","iopub.status.idle":"2025-01-21T05:32:14.64707Z","shell.execute_reply.started":"2025-01-21T05:31:25.07491Z","shell.execute_reply":"2025-01-21T05:32:14.645934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_log = xgb_pipeline.predict(X_valid)\n\n#We then unpack the log to get the true pred\ny_pred = np.expm1(y_pred_log)\n\n\nprint(f'Root Mean Squared Log Error: {math.sqrt(mean_squared_log_error(y_valid, y_pred))}')\n\n# Without F_stat feature selection: Root Mean Squared Log Error: 1.0503561746556593\n# With F_stat feature selection: Root Mean Squared Log Error: 1.049335138846521\n# With F_stat feature selection and best params: Root Mean Squared Log Error: 1.0486716732014243\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:32:14.64799Z","iopub.execute_input":"2025-01-21T05:32:14.648278Z","iopub.status.idle":"2025-01-21T05:32:18.302165Z","shell.execute_reply.started":"2025-01-21T05:32:14.648251Z","shell.execute_reply":"2025-01-21T05:32:18.300692Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### The following is more model selection and parameter tuning","metadata":{}},{"cell_type":"code","source":"\n# X_train_prepped = pd.DataFrame(preprocessor.fit_transform(X_train), columns= preprocessor.get_feature_names_out())\n\n# X_train_fet = pd.DataFrame(feature_selection.fit_transform(X_train_prepped, y_train), columns=feature_selection.get_feature_names_out())\n\n# X_train_sample = X_train_fet[:100000]\n# y_train_sample = y_train[:100000]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:32:18.303614Z","iopub.execute_input":"2025-01-21T05:32:18.30402Z","iopub.status.idle":"2025-01-21T05:32:18.308018Z","shell.execute_reply.started":"2025-01-21T05:32:18.303974Z","shell.execute_reply":"2025-01-21T05:32:18.306889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# param_grid = {\n#     'max_depth': stats.randint(3, 10),\n#     'learning_rate': stats.uniform(0.01, 0.1),\n#     'subsample': stats.uniform(0.5, 0.5),\n#     'n_estimators':stats.randint(50, 1000)\n#}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:32:18.309168Z","iopub.execute_input":"2025-01-21T05:32:18.309465Z","iopub.status.idle":"2025-01-21T05:32:18.334418Z","shell.execute_reply.started":"2025-01-21T05:32:18.309439Z","shell.execute_reply":"2025-01-21T05:32:18.33316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# param_search = RandomizedSearchCV(\n#     estimator = XGBRegressor(random_state = 0),\n#     param_distributions = param_grid,\n#     n_iter=50,\n#     cv=5,\n#     scoring= 'neg_root_mean_squared_log_error',\n#     n_jobs =-1, #use all cores\n#     verbose=1,\n#     return_train_score=True,\n#     random_state=0\n#)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:32:18.335564Z","iopub.execute_input":"2025-01-21T05:32:18.336Z","iopub.status.idle":"2025-01-21T05:32:18.353249Z","shell.execute_reply.started":"2025-01-21T05:32:18.335957Z","shell.execute_reply":"2025-01-21T05:32:18.352149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# param_search.fit(X_train_sample, y_train_sample)\n\n# best_params=param_search.best_params_\n# best_score=param_search.best_score_\n\n# print(\"Best Params\", best_params)\n# print(\"Best Score\", best_score)\n\n###Best Params {'learning_rate': np.float64(0.02028633586934382), 'max_depth': 9, 'n_estimators': 161, 'subsample': np.float64(0.6770233310419052)}\n###Best Score -1.1382992555441165","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:32:18.354335Z","iopub.execute_input":"2025-01-21T05:32:18.354739Z","iopub.status.idle":"2025-01-21T05:32:18.373149Z","shell.execute_reply.started":"2025-01-21T05:32:18.354707Z","shell.execute_reply":"2025-01-21T05:32:18.371812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model = RandomForestRegressor(n_estimators = 250, n_jobs = -1)\n\n# my_pipeline = Pipeline(steps = [('preprocessor', preprocessor), \n#                                 ('feature_selection', feature_selection), \n#                                 ('model', model)]\n#                                 )\n\n# my_pipeline.fit(X_train, y_train)\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:32:18.373979Z","iopub.execute_input":"2025-01-21T05:32:18.374258Z","iopub.status.idle":"2025-01-21T05:32:18.393957Z","shell.execute_reply.started":"2025-01-21T05:32:18.374234Z","shell.execute_reply":"2025-01-21T05:32:18.392532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# y_pred = my_pipeline.predict(X_valid)\n\n# print(f'Root Mean Squared Log Error: {math.sqrt(mean_squared_log_error(y_valid, y_pred))}')\n\n# \"Root Mean Squared Log Error: 1.1447432853335768\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:32:18.395144Z","iopub.execute_input":"2025-01-21T05:32:18.395445Z","iopub.status.idle":"2025-01-21T05:32:18.412336Z","shell.execute_reply.started":"2025-01-21T05:32:18.39542Z","shell.execute_reply":"2025-01-21T05:32:18.411187Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Final Model Submission","metadata":{}},{"cell_type":"code","source":"final_pred_log = xgb_pipeline.predict(df_test)\nfinal_pred = np.expm1(final_pred_log)\ndf_test[\"Premium Amount\"] = final_pred.astype(float).round(3)\ndf_test[\"Premium Amount\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:32:18.41359Z","iopub.execute_input":"2025-01-21T05:32:18.414022Z","iopub.status.idle":"2025-01-21T05:32:37.37357Z","shell.execute_reply.started":"2025-01-21T05:32:18.413992Z","shell.execute_reply":"2025-01-21T05:32:37.372559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.DataFrame(final_pred).plot(kind='hist')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:32:37.374598Z","iopub.execute_input":"2025-01-21T05:32:37.374971Z","iopub.status.idle":"2025-01-21T05:32:37.71701Z","shell.execute_reply.started":"2025-01-21T05:32:37.374944Z","shell.execute_reply":"2025-01-21T05:32:37.7158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = df_test[[\"id\", \"Premium Amount\"]]\nsubmission.set_index(\"id\")\nsubmission.to_csv('submission.csv', index = False)\nsubmission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T05:33:09.643306Z","iopub.execute_input":"2025-01-21T05:33:09.643769Z","iopub.status.idle":"2025-01-21T05:33:11.276196Z","shell.execute_reply.started":"2025-01-21T05:33:09.643736Z","shell.execute_reply":"2025-01-21T05:33:11.275247Z"}},"outputs":[],"execution_count":null}]}