{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-06T09:52:15.222964Z","iopub.execute_input":"2024-12-06T09:52:15.223326Z","iopub.status.idle":"2024-12-06T09:52:15.620066Z","shell.execute_reply.started":"2024-12-06T09:52:15.223294Z","shell.execute_reply":"2024-12-06T09:52:15.618928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndata_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T09:52:17.305654Z","iopub.execute_input":"2024-12-06T09:52:17.306218Z","iopub.status.idle":"2024-12-06T09:52:28.645084Z","shell.execute_reply.started":"2024-12-06T09:52:17.30618Z","shell.execute_reply":"2024-12-06T09:52:28.64392Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{}},{"cell_type":"code","source":"def datetime_feature(data):\n    date_time = pd.to_datetime(data['Policy Start Date'])\n    data['year'] = date_time.dt.year\n    data['month'] = date_time.dt.month\n    data['day'] = date_time.dt.day\n    data['month_sin'] = np.sin(2 * np.pi * data['month'] / 12)\n    data['month_cos'] = np.cos(2 * np.pi * data['month'] / 12)\n    data['day_sin'] = np.sin(2 * np.pi * data['day'] / 31)\n    data['day_cos'] = np.cos(2 * np.pi * data['day'] / 31)\n    data['year'] = data['year'].astype('category')\n    data.drop(columns=['Policy Start Date', 'month', 'day'], axis=1, inplace=True)\n    return data\n\ndef missing_filler(data):\n    for i in data.columns:\n        fill = 0\n        if data[i].dtypes == 'object':\n            fill = data[i].value_counts().idxmax()\n        else:\n            fill = data[i].mean()\n        data[i] = data[i].fillna(fill)\n    return data\n\ndef cat_maker(data, cat_cols):\n    for i in cat_cols:\n        data[i] = data[i].astype('category')\n    return data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T09:52:37.002563Z","iopub.execute_input":"2024-12-06T09:52:37.002983Z","iopub.status.idle":"2024-12-06T09:52:37.012792Z","shell.execute_reply.started":"2024-12-06T09:52:37.002948Z","shell.execute_reply":"2024-12-06T09:52:37.011468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train = missing_filler(data_train)\ndata_train = datetime_feature(data_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T09:52:38.931162Z","iopub.execute_input":"2024-12-06T09:52:38.93155Z","iopub.status.idle":"2024-12-06T09:52:42.48812Z","shell.execute_reply.started":"2024-12-06T09:52:38.931516Z","shell.execute_reply":"2024-12-06T09:52:42.487065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:26:51.077233Z","iopub.execute_input":"2024-12-06T01:26:51.077621Z","iopub.status.idle":"2024-12-06T01:26:51.110561Z","shell.execute_reply.started":"2024-12-06T01:26:51.077588Z","shell.execute_reply":"2024-12-06T01:26:51.109137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_cols = ['Gender', 'Marital Status', 'Education Level', 'Occupation',\n       'Location', 'Policy Type', 'Smoking Status', 'Exercise Frequency', 'Property Type', 'Customer Feedback']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T09:52:43.438708Z","iopub.execute_input":"2024-12-06T09:52:43.43908Z","iopub.status.idle":"2024-12-06T09:52:43.444345Z","shell.execute_reply.started":"2024-12-06T09:52:43.43905Z","shell.execute_reply":"2024-12-06T09:52:43.443137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train = cat_maker(data_train, cat_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T09:52:45.355187Z","iopub.execute_input":"2024-12-06T09:52:45.355567Z","iopub.status.idle":"2024-12-06T09:52:46.095695Z","shell.execute_reply.started":"2024-12-06T09:52:45.355534Z","shell.execute_reply":"2024-12-06T09:52:46.094525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:26:59.419612Z","iopub.execute_input":"2024-12-06T01:26:59.420604Z","iopub.status.idle":"2024-12-06T01:26:59.428977Z","shell.execute_reply.started":"2024-12-06T01:26:59.420551Z","shell.execute_reply":"2024-12-06T01:26:59.427724Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Modelling","metadata":{}},{"cell_type":"code","source":"import optuna\nimport xgboost as xgb\nfrom sklearn.model_selection import KFold, train_test_split\nfrom sklearn.metrics import mean_squared_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T09:55:20.757784Z","iopub.execute_input":"2024-12-06T09:55:20.758178Z","iopub.status.idle":"2024-12-06T09:55:20.763535Z","shell.execute_reply.started":"2024-12-06T09:55:20.758144Z","shell.execute_reply":"2024-12-06T09:55:20.762355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = data_train['Premium Amount']\nX = data_train.drop(columns=['id', 'Premium Amount'], axis=1)\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T09:52:53.672873Z","iopub.execute_input":"2024-12-06T09:52:53.67342Z","iopub.status.idle":"2024-12-06T09:52:54.130221Z","shell.execute_reply.started":"2024-12-06T09:52:53.673384Z","shell.execute_reply":"2024-12-06T09:52:54.129022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def objective(trial):\n    param = {\n        \"objective\": \"reg:squarederror\",\n        \"booster\": \"gbtree\",\n        \"n_estimators\": trial.suggest_int(\"n_estimators\", 100, 500),\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.01, 0.3),\n        \"max_depth\": trial.suggest_int(\"max_depth\", 3, 12),\n        \"min_child_weight\": trial.suggest_int(\"min_child_weight\", 1, 10),\n        \"subsample\": trial.suggest_float(\"subsample\", 0.6, 1.0),\n        \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.6, 1.0),\n        \"random_state\": 1\n    }\n    \n    kf = KFold(n_splits=5, shuffle=True, random_state=1)\n    cv = []\n    for train_index, valid_index in kf.split(X_train):\n        X_train_cv, X_valid = X_train.iloc[train_index], X_train.iloc[valid_index]\n        #log transformation\n        y_train_cv, y_valid = np.log1p(y_train).iloc[train_index], np.log1p(y_train).iloc[valid_index]\n        model = xgb.XGBRegressor(**param, enable_categorical=True, eval_metric='rmse')\n        \n        model.fit(X_train_cv, y_train_cv, verbose=False)\n        preds = model.predict(X_valid)\n        #gives a close result to RMSLE\n        error = np.sqrt(mean_squared_error(y_valid, preds))\n        cv.append(error)\n    return np.mean(cv)\n\nstudy = optuna.create_study(direction=\"minimize\")\nstudy.optimize(objective, n_trials=50)\n\nprint(\"Best Parameters:\", study.best_params)\nprint(\"Best RMSLE:\", study.best_value)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:09:32.328613Z","iopub.execute_input":"2024-12-06T04:09:32.329879Z","iopub.status.idle":"2024-12-06T05:40:28.128048Z","shell.execute_reply.started":"2024-12-06T04:09:32.32983Z","shell.execute_reply":"2024-12-06T05:40:28.12497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train the whole stuffs\nbest_param = {'n_estimators': 221, 'learning_rate': 0.028812123231198472, 'max_depth': 8, 'min_child_weight': 10, 'subsample': 0.7943663761611918, 'colsample_bytree': 0.827901646829519}\nmodel = xgb.XGBRegressor(**best_param, enable_categorical=True, eval_metric='rmse')\nmodel.fit(X_train, np.log1p(y_train), verbose = False)\npreds = model.predict(X_test)\nscore = np.sqrt(mean_squared_error(np.log1p(y_test), preds))\nprint(score)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T09:56:20.080354Z","iopub.execute_input":"2024-12-06T09:56:20.080743Z","iopub.status.idle":"2024-12-06T09:56:46.159741Z","shell.execute_reply.started":"2024-12-06T09:56:20.08071Z","shell.execute_reply":"2024-12-06T09:56:46.156996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#final train\nmodel = xgb.XGBRegressor(**best_param, enable_categorical=True, eval_metric='rmse')\nmodel.fit(X, np.log1p(y), verbose = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T09:57:49.28554Z","iopub.execute_input":"2024-12-06T09:57:49.285953Z","iopub.status.idle":"2024-12-06T09:58:19.592521Z","shell.execute_reply.started":"2024-12-06T09:57:49.285919Z","shell.execute_reply":"2024-12-06T09:58:19.59047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#transform test data\ndata_test = missing_filler(data_test)\ndata_test = datetime_feature(data_test)\ndata_test = cat_maker(data_test, cat_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T09:58:28.264069Z","iopub.execute_input":"2024-12-06T09:58:28.264453Z","iopub.status.idle":"2024-12-06T09:58:31.018388Z","shell.execute_reply.started":"2024-12-06T09:58:28.264422Z","shell.execute_reply":"2024-12-06T09:58:31.017486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds = model.predict(data_test.drop(columns=['id']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T09:58:39.312329Z","iopub.execute_input":"2024-12-06T09:58:39.313196Z","iopub.status.idle":"2024-12-06T09:58:44.225378Z","shell.execute_reply.started":"2024-12-06T09:58:39.313151Z","shell.execute_reply":"2024-12-06T09:58:44.221725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_res = pd.DataFrame({'id':data_test['id'], 'Premium Amount':np.expm1(preds)})\ndf_res.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T09:58:52.357112Z","iopub.execute_input":"2024-12-06T09:58:52.357515Z","iopub.status.idle":"2024-12-06T09:58:53.554215Z","shell.execute_reply.started":"2024-12-06T09:58:52.357478Z","shell.execute_reply":"2024-12-06T09:58:53.553192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_res.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:57:47.762032Z","iopub.execute_input":"2024-12-05T07:57:47.762529Z","iopub.status.idle":"2024-12-05T07:57:47.777413Z","shell.execute_reply.started":"2024-12-05T07:57:47.762484Z","shell.execute_reply":"2024-12-05T07:57:47.776297Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"More to come hopefully. Any suggestions are welcome.","metadata":{}}]}