{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np\nimport pandas as pd\n\n# input files\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:49:42.287945Z","iopub.execute_input":"2024-12-31T20:49:42.288993Z","iopub.status.idle":"2024-12-31T20:49:42.298377Z","shell.execute_reply.started":"2024-12-31T20:49:42.28895Z","shell.execute_reply":"2024-12-31T20:49:42.297282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import RandomizedSearchCV\nimport numpy as np\n\n\n#train and test dataset.\ntrain_df =pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest_df= pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\n\n# printing few rows of data only.\nprint(\"train Data:\")\nprint(train_df.head(2))\n\nprint(\"\\ntest Data:\")\nprint(test_df.head(2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:49:43.878023Z","iopub.execute_input":"2024-12-31T20:49:43.879084Z","iopub.status.idle":"2024-12-31T20:49:50.184682Z","shell.execute_reply.started":"2024-12-31T20:49:43.879014Z","shell.execute_reply":"2024-12-31T20:49:50.183536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# data types\nprint(train_df.info())\n\n# Checking for missing values\nprint(train_df.isnull().sum())\n# Summary\nprint(train_df.describe())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:49:54.600558Z","iopub.execute_input":"2024-12-31T20:49:54.600974Z","iopub.status.idle":"2024-12-31T20:49:56.5337Z","shell.execute_reply.started":"2024-12-31T20:49:54.600936Z","shell.execute_reply":"2024-12-31T20:49:56.532527Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# numerical columns\nnumerical_cols = train_df.select_dtypes(include=[np.number]).columns.tolist()\n# correlation with the target variable and sort the values\ncorr_target = train_df[numerical_cols].corrwith(train_df['Premium Amount']).sort_values(ascending=False)\n\nprint(\"Correlation with Premium Amount\")\nprint(corr_target)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:50:01.899428Z","iopub.execute_input":"2024-12-31T20:50:01.899821Z","iopub.status.idle":"2024-12-31T20:50:02.213481Z","shell.execute_reply.started":"2024-12-31T20:50:01.89978Z","shell.execute_reply":"2024-12-31T20:50:02.2125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\" \ncreating new features for the model like Dependents_per_Age, Health_Score_Vehicle_Age_Interaction, \nPrevious_Claims_to_Duration and Income_to_Credit_Score. new features can give us more insights into the data.\n\"\"\"\ntrain_df['Dependents_per_Age']=train_df['Number of Dependents']/(train_df['Age']+ 1)\ntest_df['Dependents_per_Age']=test_df['Number of Dependents']/(test_df['Age'] +1)\ntrain_df['Health_Score_Vehicle_Age_Interaction']=train_df['Health Score']* train_df['Vehicle Age']\ntest_df['Health_Score_Vehicle_Age_Interaction']=test_df['Health Score']*test_df['Vehicle Age']\n\ntrain_df['Previous_Claims_to_Duration']= train_df['Previous Claims']/ (train_df['Insurance Duration'] +1)\ntest_df['Previous_Claims_to_Duration']=test_df['Previous Claims']/ (test_df['Insurance Duration']+1)\n\ntrain_df['Income_to_Credit_Score']=train_df['Annual Income']/(train_df['Credit Score']+1)\ntest_df['Income_to_Credit_Score']=test_df['Annual Income']/(test_df['Credit Score']+1)\n\n\n# after the new features we can check the new correlation with the target var.\nnumerical_cols = train_df.select_dtypes(include=['int64','float64']).columns.tolist()\ncorr_target = train_df[numerical_cols].corrwith(train_df['Premium Amount']).sort_values(ascending=False)\n\nprint(\"Updated correlation with target variable premium amount: \")\nprint(corr_target)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T21:29:22.355781Z","iopub.execute_input":"2024-12-31T21:29:22.356267Z","iopub.status.idle":"2024-12-31T21:29:22.875864Z","shell.execute_reply.started":"2024-12-31T21:29:22.356223Z","shell.execute_reply":"2024-12-31T21:29:22.874328Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"After creating new features, we observe some correlations. The highest correlation is with Previous Claims, followed by Previous_Claims_to_Duration. Many correlations are weak, and some are negative. These features should be further dropped as they are unlikely to contribute effectively to the model.","metadata":{}},{"cell_type":"markdown","source":"The code below cleans the basic data, handles missing values, performs basic imputation, and applies label encoding and one-hot encoding techniques for model training purposes.","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import RandomizedSearchCV\nimport numpy as np\n\ntrain_df.drop(columns=['id', 'Policy Start Date'], inplace=True)\ntest_df.drop(columns=['id', 'Policy Start Date'], inplace=True)\n\nnumerical_cols = train_df.select_dtypes(include=['float64', 'int64']).columns.tolist()\ncategorical_cols = train_df.select_dtypes(include=['object']).columns.tolist()\n\n# Remove 'Premium Amount' from the cols\nnumerical_cols = [col for col in numerical_cols if col != 'Premium Amount']\ncategorical_cols = [col for col in categorical_cols if col != 'Premium Amount']\n\nimputer_num = SimpleImputer(strategy='median')\nimputer_cat = SimpleImputer(strategy='most_frequent')\n\n# Apply imputation for numerical columns (In-place) for both train_df and test_df\ntrain_df[numerical_cols] = imputer_num.fit_transform(train_df[numerical_cols])\ntest_df[numerical_cols] = imputer_num.transform(test_df[numerical_cols])\n\ntrain_df[categorical_cols] = imputer_cat.fit_transform(train_df[categorical_cols])\ntest_df[categorical_cols] = imputer_cat.transform(test_df[categorical_cols])\n\n# using labelencoder to normalize the labels.\nordinal_cols = ['Education Level']\nle = LabelEncoder()\n\n# applying label encoder for model.\ntrain_df[ordinal_cols] = train_df[ordinal_cols].apply(le.fit_transform) #cat variables\ntest_df[ordinal_cols] = test_df[ordinal_cols].apply(le.transform)\n\n#encoding\ntrain_df = pd.get_dummies(train_df, \n                          columns=[col for col in categorical_cols if col not in ordinal_cols],drop_first=True)\n\ntest_df = pd.get_dummies(test_df, \n                          columns=[col for col in categorical_cols if col not in ordinal_cols],drop_first=True)\n\n# test dataset and training set has same set of columns.\ntest_df = test_df.reindex(columns=train_df.columns, fill_value=0)\n\nX = train_df.drop(columns=['Premium Amount'])\ny = train_df['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:53:52.605283Z","iopub.execute_input":"2024-12-31T20:53:52.606233Z","iopub.status.idle":"2024-12-31T20:54:00.489037Z","shell.execute_reply.started":"2024-12-31T20:53:52.606186Z","shell.execute_reply":"2024-12-31T20:54:00.487882Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train,X_val,y_train,y_val = train_test_split(X,y_log,test_size=0.2,random_state=42)\n\n# Few of the features have less correlation with target var so we can drop those var.\nlow_corr_features = ['id','Number of Dependents','Age',\n                     'Vehicle Age','Insurance Duration',\n                     'Annual Income','Credit Score']\nX_train=X_train.drop(columns=low_corr_features,errors='ignore')\nX_val=X_val.drop(columns=low_corr_features,errors='ignore')\n\n\n\"\"\"\nUsing random forest model, \n1. n_jobs=-1 is required for parallel processing to use all the CPU cores\n2. n_estimartors for now is 100 trees\n\"\"\"\nmodel = RandomForestRegressor(n_estimators=100, random_state=42, n_jobs=-1)\n\n# Train the model\nmodel.fit(X_train, y_train)\n\n# Predict on the validation set\ny_pred = model.predict(X_val)\n\n# Clip the predicted values to avoid very large values\ny_pred = np.clip(y_pred, -1e10, 1e10)\n\n# Inverse the log transformation for predictions to get results on the original scale\ny_pred = np.expm1(y_pred)\n\n# Calculate the RMSLE\nfrom sklearn.metrics import mean_squared_log_error\nrmsle = np.sqrt(mean_squared_log_error(np.expm1(y_val), y_pred))\n\nprint(f\"RMSLE: {rmsle}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:54:32.97804Z","iopub.execute_input":"2024-12-31T20:54:32.978909Z","iopub.status.idle":"2024-12-31T21:06:12.10216Z","shell.execute_reply.started":"2024-12-31T20:54:32.978863Z","shell.execute_reply":"2024-12-31T21:06:12.100937Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The code below uses RandomizedSearchCV to find the best parameters for the model, which can then be used for training. The parameters, number of iterations, and folds can be modified to observe different outputs and optimize the runtime.","metadata":{}},{"cell_type":"code","source":"# get the best parameters for immproved rmsle\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nimport numpy as np\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_log_error\nfrom sklearn.model_selection import RandomizedSearchCV, train_test_split\n\n# below parameters to run a bit faster. can be adjusted in the future.\nparam_grid = { 'n_estimators': [50, 100], 'max_depth': [10, 20], 'min_samples_split': [2, 5],\n    'max_features': ['sqrt'], 'min_samples_leaf': [1, 2]\n}\n\n# Initialize Random Forest\nreg = RandomForestRegressor(random_state=42)\n\n# RandomizedSearchCV with reduced iterations and folds. n_jobs to ensure parallel processing.\ncv_search = RandomizedSearchCV(reg,param_grid,n_iter=10,cv=3,random_state=42,n_jobs=-1,verbose=2)\ncv_search.fit(X_train, y_train)\n\n# improved parameters\nprint(\"New Parameters: \", cv_search.best_params_)\nbest_rf = cv_search.best_estimator_\n\ny_pred =best_rf.predict(X_val)\ny_pred =np.expm1(y_pred)\n\n# Compute RMSLE using original scale\nrmsle = np.sqrt(mean_squared_log_error(np.expm1(y_val), y_pred))\nprint(f\"Improved RMSLE is = {rmsle}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T21:07:16.278834Z","iopub.execute_input":"2024-12-31T21:07:16.279993Z","iopub.status.idle":"2024-12-31T21:29:10.227394Z","shell.execute_reply.started":"2024-12-31T21:07:16.279942Z","shell.execute_reply":"2024-12-31T21:29:10.226027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# after getting best parameters we can use them and get the final rmsle\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# Initialize the final model with better parameters\nfinal_model = RandomForestRegressor(\n  n_estimators= 100, min_samples_split= 2, \n  min_samples_leaf= 1, max_features= 'sqrt', max_depth= 10\n)\n\n# Train the model\nfinal_model.fit(X_train, y_train)\n\n# Predict on validation set\ny_pred=final_model.predict(X_val)\n\n# Transform predictions back to the original scale\ny_pred = np.expm1(y_pred)\n\n# Evaluate RMSLE\nfrom sklearn.metrics import mean_squared_log_error\nrmsle = np.sqrt(mean_squared_log_error(np.expm1(y_val), y_pred))\nprint(f\"Final Model RMSLE: {rmsle}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T18:09:03.396639Z","iopub.execute_input":"2024-12-31T18:09:03.397029Z","iopub.status.idle":"2024-12-31T18:11:02.671512Z","shell.execute_reply.started":"2024-12-31T18:09:03.396996Z","shell.execute_reply":"2024-12-31T18:11:02.670417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import train_test_split\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import mean_squared_log_error\nfrom sklearn.ensemble import StackingRegressor\n\n# We can use some popular base models with adjustable parameters.\nmodels = [\n    ('rf', RandomForestRegressor( n_estimators= 100, min_samples_split= 2, min_samples_leaf= 1,\n                                 max_features= 'sqrt', max_depth= 10, n_jobs=-1)),\n    ('gb', GradientBoostingRegressor(n_estimators=50, learning_rate=0.05, max_depth=3)),\n    ('xgb', XGBRegressor(n_estimators=50, learning_rate=0.1, max_depth=3, n_jobs=-1))\n]\n\n\"\"\"\"meta model is used to combine the predictions from all the base models \n(rf, gb, and xgb). The predictions from each base model serve as inputs for the meta model,\nwhich makes the final prediction.\"\n\"\"\"\nmeta_model =Ridge()\n\n# Initialize\nstack_model = StackingRegressor(estimators=models,final_estimator=meta_model)\n\n# Fit the model with features\nstack_model.fit(X_train, y_train)\n\n# Predict the target variable\ny_pred = stack_model.predict(X_val)\n\n\nstacked_rmsle = np.sqrt(mean_squared_log_error(y_val, y_pred))\nprint(f\"RMSLE for the Stacking model is = {stacked_rmsle}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T21:29:38.86457Z","iopub.execute_input":"2024-12-31T21:29:38.865089Z","iopub.status.idle":"2024-12-31T21:48:53.737263Z","shell.execute_reply.started":"2024-12-31T21:29:38.86502Z","shell.execute_reply":"2024-12-31T21:48:53.732539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = test_df.drop(columns=['Premium Amount'],errors='ignore')\nX_test = X_test.reindex(columns=X_train.columns,fill_value=0)\ny_test_pred_log = stacking_model.predict(X_test)\n\n# Transform back to original scale\ny_test_pred = np.expm1(y_test_pred_log)\nupdated_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')['id']\n\npred_final_df = pd.DataFrame({'id': updated_test,\n    'Premium Amount': y_test_pred\n})\n\n# Print the first few rows of the predictions\nprint(\"Predictions with IDs:\")\nprint(pred_final_df.head(5))\n\n# write to the working dir\npred_final_df.to_csv('/kaggle/working/submission.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T21:52:10.208366Z","iopub.execute_input":"2024-12-31T21:52:10.209823Z","iopub.status.idle":"2024-12-31T21:52:21.324961Z","shell.execute_reply.started":"2024-12-31T21:52:10.209772Z","shell.execute_reply":"2024-12-31T21:52:21.323544Z"}},"outputs":[],"execution_count":null}]}