{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":10102872,"sourceType":"datasetVersion","datasetId":6231534}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <p style=\"background-color:Skyblue; font-family:'Orbitron', sans-serif; color:#FFFFFF; font-size:140%; text-align:center; border: 2px solid black; border-radius:15px; padding: 15px; box-shadow: 5px 5px 20px rgba(0, 0, 0, 0.5); font-weight: bold; letter-spacing: 1px;\">Regression With Insurance Data | LGBM</p>\r\n","metadata":{}},{"cell_type":"code","source":"%%time\n\nimport numpy as np\nimport polars as pl\nimport pandas as pd\n\nfrom sklearn.base import clone\nimport optuna\nimport os\n\nfrom tqdm import tqdm\nimport category_encoders as ce\nfrom IPython.display import clear_output\n\nfrom sklearn.decomposition import TruncatedSVD\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n!pip install -qq pytorch_tabnet\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nimport lightgbm as lgb\nfrom lightgbm import early_stopping  \nfrom catboost import CatBoostRegressor, CatBoostClassifier, Pool\nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\n\nSEED = 42\nn_splits = 10\n\n!pip install -q lifelines","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-12T23:44:56.75616Z","iopub.execute_input":"2024-12-12T23:44:56.756822Z","iopub.status.idle":"2024-12-12T23:45:20.715505Z","shell.execute_reply.started":"2024-12-12T23:44:56.756782Z","shell.execute_reply":"2024-12-12T23:45:20.714193Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <p style=\"background-color:Skyblue; font-family:'Orbitron', sans-serif; color:#FFFFFF; font-size:140%; text-align:center; border: 2px solid black; border-radius:15px; padding: 15px; box-shadow: 5px 5px 20px rgba(0, 0, 0, 0.5); font-weight: bold; letter-spacing: 1px;\">Load Data</p>","metadata":{}},{"cell_type":"code","source":"%%time\n\n!git clone https://github.com/muhammadabdullah0303/AbdML\n\nimport sys\nsys.path.append('/kaggle/working/repository')\n\nfrom AbdML.main import AbdBase\n\ntrain = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\nsample = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\nlgb_off = pd.read_csv('/kaggle/input/regression-with-an-insurance-dataset-offpreds/lgb_off.csv')\n\ntrain.drop('id', axis=1, inplace=True)\ntest.drop('id', axis=1, inplace=True) \n\ndef date(Df):\n\n    Df['Policy Start Date'] = pd.to_datetime(Df['Policy Start Date'])\n    Df['Year'] = Df['Policy Start Date'].dt.year\n    Df['Day'] = Df['Policy Start Date'].dt.day\n    Df['Month'] = Df['Policy Start Date'].dt.month\n    Df['Month_name'] = Df['Policy Start Date'].dt.month_name()\n    Df['Day_of_week'] = Df['Policy Start Date'].dt.day_name()\n    Df['Week'] = Df['Policy Start Date'].dt.isocalendar().week\n    Df['Year_sin'] = np.sin(2 * np.pi * Df['Year'])\n    Df['Year_cos'] = np.cos(2 * np.pi * Df['Year'])\n    min_year = Df['Year'].min()\n    max_year = Df['Year'].max()\n    Df['Year_sin'] = np.sin(2 * np.pi * (Df['Year'] - min_year) / (max_year - min_year))\n    Df['Year_cos'] = np.cos(2 * np.pi * (Df['Year'] - min_year) / (max_year - min_year))\n    Df['Month_sin'] = np.sin(2 * np.pi * Df['Month'] / 12) \n    Df['Month_cos'] = np.cos(2 * np.pi * Df['Month'] / 12)\n    Df['Day_sin'] = np.sin(2 * np.pi * Df['Day'] / 31)  \n    Df['Day_cos'] = np.cos(2 * np.pi * Df['Day'] / 31)\n    Df['Group']=(Df['Year']-2020)*48+Df['Month']*4+Df['Day']//7\n    \n    Df.drop('Policy Start Date', axis=1, inplace=True)\n\n    Df['contract length'] = pd.cut(\n        Df[\"Insurance Duration\"].fillna(99),  \n        bins=[-float('inf'), 1, 3, float('inf')],  \n        labels=[0, 1, 2]  \n    ).astype(int)\n\n    return Df\n\ntrain = date(train)\ntest = date(test)\n\ncat_c = [col for col in train.columns if train[col].dtype == 'object']\n\ndef update(df):\n    \n    global cat_c\n\n    for c in cat_c:\n        df[c] = df[c].fillna('None').astype('category')\n                \n    return df\n\ntrain = update(train)\ntest = update(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T23:46:03.67975Z","iopub.execute_input":"2024-12-12T23:46:03.680743Z","iopub.status.idle":"2024-12-12T23:46:19.495505Z","shell.execute_reply.started":"2024-12-12T23:46:03.680699Z","shell.execute_reply":"2024-12-12T23:46:19.493945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T23:46:19.498089Z","iopub.execute_input":"2024-12-12T23:46:19.498624Z","iopub.status.idle":"2024-12-12T23:46:19.533452Z","shell.execute_reply.started":"2024-12-12T23:46:19.49857Z","shell.execute_reply":"2024-12-12T23:46:19.53221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T23:46:19.535496Z","iopub.execute_input":"2024-12-12T23:46:19.536007Z","iopub.status.idle":"2024-12-12T23:46:19.576173Z","shell.execute_reply.started":"2024-12-12T23:46:19.535954Z","shell.execute_reply":"2024-12-12T23:46:19.574971Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <p style=\"background-color:Skyblue; font-family:'Orbitron', sans-serif; color:#FFFFFF; font-size:140%; text-align:center; border: 2px solid black; border-radius:15px; padding: 15px; box-shadow: 5px 5px 20px rgba(0, 0, 0, 0.5); font-weight: bold; letter-spacing: 1px;\">LGBM Models</p>","metadata":{}},{"cell_type":"code","source":"%%time\n\nSEED = 42\nn_splits = 10\ncat_c = [col for col in base.X_train.columns if base.X_train[col].dtype == 'category']\n\n\nbase = AbdBase(train_data=train, test_data=test, target_column='Premium Amount',gpu=False,\n                 problem_type=\"regression\", metric=\"rmsle\", seed=SEED,\n                 n_splits=n_splits,early_stop=True,num_classes=0,cat_features = cat_c,\n                 fold_type='RKF')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:43:48.976102Z","iopub.execute_input":"2024-12-13T01:43:48.97693Z","iopub.status.idle":"2024-12-13T01:43:49.089239Z","shell.execute_reply.started":"2024-12-13T01:43:48.976891Z","shell.execute_reply":"2024-12-13T01:43:49.088014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\nParams = {'n_estimators': 200,\"n_jobs\":-1}\n\nresults = base.Train_ML(Params,'LGBM',e_stop=40, y_log=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:54:38.418382Z","iopub.execute_input":"2024-12-13T00:54:38.41908Z","iopub.status.idle":"2024-12-13T01:00:43.24743Z","shell.execute_reply.started":"2024-12-13T00:54:38.419038Z","shell.execute_reply":"2024-12-13T01:00:43.24606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\nbase.X_train['results'] = results[0]\nbase.X_test['results'] = results[1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:33:20.395664Z","iopub.execute_input":"2024-12-13T01:33:20.396792Z","iopub.status.idle":"2024-12-13T01:33:20.406614Z","shell.execute_reply.started":"2024-12-13T01:33:20.396741Z","shell.execute_reply":"2024-12-13T01:33:20.405291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\nParams1 = {'learning_rate': 0.08692991511139551, 'num_leaves': 85, 'max_depth': 15, 'min_data_in_leaf': 95,\n          'feature_fraction': 0.7567559292276751, 'bagging_fraction': 0.9472874885021447, 'bagging_freq': 1,\n           'min_child_weight': 1, 'scale_pos_weight': 4,'n_estimators': 200}\n\nresults1 = base.Train_ML(Params1,'LGBM',e_stop=40, y_log=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:33:21.944052Z","iopub.execute_input":"2024-12-13T01:33:21.944506Z","iopub.status.idle":"2024-12-13T01:37:17.605598Z","shell.execute_reply.started":"2024-12-13T01:33:21.944467Z","shell.execute_reply":"2024-12-13T01:37:17.604199Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <p style=\"background-color:Skyblue; font-family:'Orbitron', sans-serif; color:#FFFFFF; font-size:140%; text-align:center; border: 2px solid black; border-radius:15px; padding: 15px; box-shadow: 5px 5px 20px rgba(0, 0, 0, 0.5); font-weight: bold; letter-spacing: 1px;\">Submission</p>","metadata":{}},{"cell_type":"code","source":"%%time\n\nxTest = results[1]\nxTest1 = results1[1] \n\nsample['Premium Amount'] = xTest * 0.53348895 + xTest1 * 0.46651105\n\nsample.to_csv('submission.csv', index = False)\nsample.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:48:38.376013Z","iopub.execute_input":"2024-12-13T01:48:38.376495Z","iopub.status.idle":"2024-12-13T01:48:40.077343Z","shell.execute_reply.started":"2024-12-13T01:48:38.376456Z","shell.execute_reply":"2024-12-13T01:48:40.076006Z"}},"outputs":[],"execution_count":null}]}