{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:30.672697Z","iopub.execute_input":"2024-12-06T19:37:30.673044Z","iopub.status.idle":"2024-12-06T19:37:31.035969Z","shell.execute_reply.started":"2024-12-06T19:37:30.673015Z","shell.execute_reply":"2024-12-06T19:37:31.034996Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# IMPORTING DATA","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\nsample_submission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:31.037452Z","iopub.execute_input":"2024-12-06T19:37:31.037882Z","iopub.status.idle":"2024-12-06T19:37:40.74881Z","shell.execute_reply.started":"2024-12-06T19:37:31.037851Z","shell.execute_reply":"2024-12-06T19:37:40.747898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# data_dir = 'playground-series-s4e12'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:40.750341Z","iopub.execute_input":"2024-12-06T19:37:40.750667Z","iopub.status.idle":"2024-12-06T19:37:40.754716Z","shell.execute_reply.started":"2024-12-06T19:37:40.750636Z","shell.execute_reply":"2024-12-06T19:37:40.753794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape , test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:40.755882Z","iopub.execute_input":"2024-12-06T19:37:40.756161Z","iopub.status.idle":"2024-12-06T19:37:40.772248Z","shell.execute_reply.started":"2024-12-06T19:37:40.756133Z","shell.execute_reply":"2024-12-06T19:37:40.77127Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# FEATURE ENGINEERING","metadata":{}},{"cell_type":"code","source":"def null_percentage(df):\n    per = ((df.isnull().sum()/len(df))*100).round(2)\n    return per\nprint('Null values in Train Data')\nprint(null_percentage(train))\n\nprint('Null values in Test Data')\nprint(null_percentage(test))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:40.774122Z","iopub.execute_input":"2024-12-06T19:37:40.774415Z","iopub.status.idle":"2024-12-06T19:37:41.790242Z","shell.execute_reply.started":"2024-12-06T19:37:40.774386Z","shell.execute_reply":"2024-12-06T19:37:41.789263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'{train.info()}')\nprint(f'{test.info()}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:41.791606Z","iopub.execute_input":"2024-12-06T19:37:41.792312Z","iopub.status.idle":"2024-12-06T19:37:42.829275Z","shell.execute_reply.started":"2024-12-06T19:37:41.792267Z","shell.execute_reply":"2024-12-06T19:37:42.828241Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2.1 Handling NULL Values","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ndef train_test_null_handle(train, test):\n    # Handling Numerical Columns \n    numerical_cols_train = train.select_dtypes(include=['float64'])\n    numerical_cols_test = test.select_dtypes(include=['float64'])\n    \n    for col in numerical_cols_train:\n        if train[col].isnull().any():\n            train[col].fillna(train[col].mean(), inplace=True)\n    \n    for col in numerical_cols_test:\n        if test[col].isnull().any():\n            test[col].fillna(test[col].mean(), inplace=True)\n\n    # Handling Categorical Columns\n    categorical_cols_train = train.select_dtypes(include=['object'])\n    categorical_cols_test = test.select_dtypes(include=['object'])\n\n    for col in categorical_cols_train:\n        if train[col].isnull().any():\n            train[col].fillna(train[col].mode()[0], inplace=True)\n    \n    for col in categorical_cols_test:\n        if test[col].isnull().any():\n            test[col].fillna(test[col].mode()[0], inplace=True)\n            \n    return train, test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:42.830477Z","iopub.execute_input":"2024-12-06T19:37:42.830821Z","iopub.status.idle":"2024-12-06T19:37:42.838462Z","shell.execute_reply.started":"2024-12-06T19:37:42.83079Z","shell.execute_reply":"2024-12-06T19:37:42.837552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train,test = train_test_null_handle(train, test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:42.839402Z","iopub.execute_input":"2024-12-06T19:37:42.839767Z","iopub.status.idle":"2024-12-06T19:37:44.963054Z","shell.execute_reply.started":"2024-12-06T19:37:42.839727Z","shell.execute_reply":"2024-12-06T19:37:44.961967Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(null_percentage(train))\nprint(null_percentage(test))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:44.964223Z","iopub.execute_input":"2024-12-06T19:37:44.964508Z","iopub.status.idle":"2024-12-06T19:37:45.972886Z","shell.execute_reply.started":"2024-12-06T19:37:44.964461Z","shell.execute_reply":"2024-12-06T19:37:45.971918Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2.2 Numerical and Categorical Columns","metadata":{}},{"cell_type":"code","source":"numerical_columns = train.select_dtypes(include=['float64', 'int64']).columns\ncategorical_columns = train.select_dtypes(include=['object']).columns\n\n\nprint(\"Numerical Columns:\")\nprint(numerical_columns)\n\nprint(\"\\nCategorical Columns:\")\nprint(categorical_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:45.974127Z","iopub.execute_input":"2024-12-06T19:37:45.974465Z","iopub.status.idle":"2024-12-06T19:37:46.14598Z","shell.execute_reply.started":"2024-12-06T19:37:45.974435Z","shell.execute_reply":"2024-12-06T19:37:46.144948Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2.3 Extracting Date and Time from Policy Start Date Column","metadata":{}},{"cell_type":"code","source":"def extract_date_time(df,datetime_column,drop_original = True):\n    df[datetime_column] = pd.to_datetime(df[datetime_column],errors = 'coerce')\n\n    df['year'] = df[datetime_column].dt.year\n    df['month'] = df[datetime_column].dt.month\n    df['day'] = df[datetime_column].dt.day\n    df['dayofweek'] = df[datetime_column].dt.dayofweek\n\n    if drop_original:\n        df.drop(datetime_column,axis = 1,inplace = True)\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:46.148982Z","iopub.execute_input":"2024-12-06T19:37:46.149275Z","iopub.status.idle":"2024-12-06T19:37:46.154859Z","shell.execute_reply.started":"2024-12-06T19:37:46.149246Z","shell.execute_reply":"2024-12-06T19:37:46.15391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = extract_date_time(train,datetime_column='Policy Start Date',drop_original=True)\ntest = extract_date_time(test,datetime_column='Policy Start Date',drop_original=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:46.155986Z","iopub.execute_input":"2024-12-06T19:37:46.156252Z","iopub.status.idle":"2024-12-06T19:37:47.535009Z","shell.execute_reply.started":"2024-12-06T19:37:46.156226Z","shell.execute_reply":"2024-12-06T19:37:47.533775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:47.536444Z","iopub.execute_input":"2024-12-06T19:37:47.537188Z","iopub.status.idle":"2024-12-06T19:37:47.562224Z","shell.execute_reply.started":"2024-12-06T19:37:47.53714Z","shell.execute_reply":"2024-12-06T19:37:47.561104Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Label Encoding ","metadata":{}},{"cell_type":"code","source":"categorical_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:47.563315Z","iopub.execute_input":"2024-12-06T19:37:47.563666Z","iopub.status.idle":"2024-12-06T19:37:47.570221Z","shell.execute_reply.started":"2024-12-06T19:37:47.563631Z","shell.execute_reply":"2024-12-06T19:37:47.569255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat = ['Gender', 'Marital Status', 'Education Level', \n       'Occupation', 'Location', 'Policy Type', \n       'Customer Feedback','Smoking Status', \n       'Exercise Frequency', 'Property Type']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:47.571373Z","iopub.execute_input":"2024-12-06T19:37:47.571731Z","iopub.status.idle":"2024-12-06T19:37:47.583585Z","shell.execute_reply.started":"2024-12-06T19:37:47.571701Z","shell.execute_reply":"2024-12-06T19:37:47.582664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n\ndef encoding(train,test,cat):\n    for col in cat:\n        label = LabelEncoder()\n        train[col] = label.fit_transform(train[col])\n        test[col] = label.transform(test[col])\n    return train,test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:47.584818Z","iopub.execute_input":"2024-12-06T19:37:47.585182Z","iopub.status.idle":"2024-12-06T19:37:48.05708Z","shell.execute_reply.started":"2024-12-06T19:37:47.585151Z","shell.execute_reply":"2024-12-06T19:37:48.056287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train,test = encoding(train,test,cat)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:48.058232Z","iopub.execute_input":"2024-12-06T19:37:48.058689Z","iopub.status.idle":"2024-12-06T19:37:51.605571Z","shell.execute_reply.started":"2024-12-06T19:37:48.058659Z","shell.execute_reply":"2024-12-06T19:37:51.604702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:51.607018Z","iopub.execute_input":"2024-12-06T19:37:51.607711Z","iopub.status.idle":"2024-12-06T19:37:51.970291Z","shell.execute_reply.started":"2024-12-06T19:37:51.607665Z","shell.execute_reply":"2024-12-06T19:37:51.969346Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train Test Split","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = train.drop(columns = ['Premium Amount'])\ny = train['Premium Amount']\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\nprint(\"X_train shape:\", X_train.shape)\nprint(\"X_test shape:\", X_test.shape)\nprint(\"y_train shape:\", y_train.shape)\n\nprint(\"y_test shape:\", y_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:51.971706Z","iopub.execute_input":"2024-12-06T19:37:51.972523Z","iopub.status.idle":"2024-12-06T19:37:52.684358Z","shell.execute_reply.started":"2024-12-06T19:37:51.972455Z","shell.execute_reply":"2024-12-06T19:37:52.683222Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Hyperparameter Tuning","metadata":{}},{"cell_type":"code","source":"# def objective(trial):\n#     params = {\n#         'objective':'reg:squarederror',\n#         'eval_metric':'rmse',\n#         'booster':trial.suggest_categorical('booster',['gbtree','dart']),\n#         'eta':trial.suggest_float('eta',0.01,0.3,log = True), #Learning Rate \n#         'max_depth':trial.suggest_int('max_depth',3,10),\n#         'min_child_weight':trial.suggest_int('min_child_weight',1,10),\n#         'subsample':trial.suggest_float('subsample',0.5,1.0),\n#         \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.5, 1.0),\n#         \"lambda\": trial.suggest_float(\"lambda\", 1e-3, 10.0, log=True),\n#         \"alpha\": trial.suggest_float(\"alpha\", 1e-3, 10.0, log=True),\n#     }\n#     # Train the Model\n#     model_xgb_optuna = xgb.train(\n#         params,\n#         d_train,\n#         num_boost_round=1000,\n#         evals = [(d_val,'test')],\n#         early_stopping_rounds = 50,\n#         verbose_eval = False\n#     )\n\n#     # Predict and Evaluate \n#     preds = model_xgb_optuna.predict(d_val)\n#     rmse = mean_squared_error(y_val,preds,squared=False)\n#     return rmse\n\n# # Run Optuna Optimization \n# study = optuna.create_study(direction = 'minimize')\n# study.optimize(objective,n_trials=5)\n\n# # Best parameters\n# print(\"Best parameters:\", study.best_params)\n\n# # Best trial score\n# print(\"Best RMSE:\", study.best_value)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:52.685889Z","iopub.execute_input":"2024-12-06T19:37:52.686315Z","iopub.status.idle":"2024-12-06T19:37:52.691912Z","shell.execute_reply.started":"2024-12-06T19:37:52.686271Z","shell.execute_reply":"2024-12-06T19:37:52.690765Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# best_params = study.best_params\n# for param,value in best_params.items():\n#     print(f'{param}:{value}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:52.693128Z","iopub.execute_input":"2024-12-06T19:37:52.693424Z","iopub.status.idle":"2024-12-06T19:37:52.705475Z","shell.execute_reply.started":"2024-12-06T19:37:52.693396Z","shell.execute_reply":"2024-12-06T19:37:52.704679Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Found using optuna\n\nparams = {\n    \"booster\": \"gbtree\", \n    \"eta\": 0.025983889617311084,\n    \"max_depth\": 7,  \n    \"min_child_weight\": 1,  \n    \"subsample\": 0.9382223060754771,  \n    \"colsample_bytree\": 0.7344288125566767, \n    \"lambda\": 1.9378024080485816,  \n    \"alpha\": 0.004391134889789325, \n    \"objective\": \"reg:squarederror\", \n    \"eval_metric\": \"rmse\", \n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:52.706642Z","iopub.execute_input":"2024-12-06T19:37:52.706893Z","iopub.status.idle":"2024-12-06T19:37:52.717669Z","shell.execute_reply.started":"2024-12-06T19:37:52.706868Z","shell.execute_reply":"2024-12-06T19:37:52.71682Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"import xgboost as xgb\nfrom sklearn.metrics import mean_squared_error\n\ndtrain = xgb.DMatrix(data=X_train, label=y_train)\ndtest = xgb.DMatrix(data=X_test, label=y_test)\n\n\n\nmodel = xgb.train(\n    params=params,\n    dtrain=dtrain,\n    num_boost_round=1000,\n    evals=[(dtrain, \"train\"), (dtest, \"eval\")],\n    early_stopping_rounds=12,\n    verbose_eval=10,  # Display evaluation metrics every 10 rounds\n)\n\ny_pred = model.predict(dtest)\n\nrmse = mean_squared_error(y_test, y_pred, squared=False)\nprint(f\"RMSE on test data: {rmse}\")\n\n# model.save_model(\"xgb_regression_model.json\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:37:52.718834Z","iopub.execute_input":"2024-12-06T19:37:52.719132Z","iopub.status.idle":"2024-12-06T19:38:30.165624Z","shell.execute_reply.started":"2024-12-06T19:37:52.719104Z","shell.execute_reply":"2024-12-06T19:38:30.164505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# preds = model.predict(test)  # Predicting the target values\n# submission = pd.DataFrame({'id': test.id, 'Premium Amount': preds})  # Creating a DataFrame for submission\n# submission.to_csv('submission_by_model.csv', index=False)  # Exporting to CSV","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T19:40:41.364367Z","iopub.execute_input":"2024-12-06T19:40:41.365245Z","iopub.status.idle":"2024-12-06T19:40:41.36918Z","shell.execute_reply.started":"2024-12-06T19:40:41.365201Z","shell.execute_reply":"2024-12-06T19:40:41.368169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}