{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import accuracy_score, mean_squared_error\nimport warnings\n\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:57:00.412326Z","iopub.execute_input":"2024-12-07T10:57:00.41358Z","iopub.status.idle":"2024-12-07T10:57:00.889198Z","shell.execute_reply.started":"2024-12-07T10:57:00.413511Z","shell.execute_reply":"2024-12-07T10:57:00.888153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv', index_col='id', engine='pyarrow')\ntest_df = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv', index_col='id', engine='pyarrow')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:57:02.101292Z","iopub.execute_input":"2024-12-07T10:57:02.102286Z","iopub.status.idle":"2024-12-07T10:57:05.245699Z","shell.execute_reply.started":"2024-12-07T10:57:02.102233Z","shell.execute_reply":"2024-12-07T10:57:05.244473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:57:05.247708Z","iopub.execute_input":"2024-12-07T10:57:05.248202Z","iopub.status.idle":"2024-12-07T10:57:05.285634Z","shell.execute_reply.started":"2024-12-07T10:57:05.24815Z","shell.execute_reply":"2024-12-07T10:57:05.284324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T10:57:05.933144Z","iopub.execute_input":"2024-12-07T10:57:05.933655Z","iopub.status.idle":"2024-12-07T10:57:05.959626Z","shell.execute_reply.started":"2024-12-07T10:57:05.933613Z","shell.execute_reply":"2024-12-07T10:57:05.958309Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def date_separator(x):\n    return pd.Series([x.day, x.month, x.year])\n\ntrain_df[['day', 'month', 'year']] = train_df['Policy Start Date'].apply(date_separator)\ntest_df[['day', 'month', 'year']] = test_df['Policy Start Date'].apply(date_separator)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:03:21.686504Z","iopub.execute_input":"2024-12-07T11:03:21.686953Z","iopub.status.idle":"2024-12-07T11:07:00.378857Z","shell.execute_reply.started":"2024-12-07T11:03:21.686916Z","shell.execute_reply":"2024-12-07T11:07:00.377355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:07:00.381273Z","iopub.execute_input":"2024-12-07T11:07:00.381795Z","iopub.status.idle":"2024-12-07T11:07:01.031069Z","shell.execute_reply.started":"2024-12-07T11:07:00.381752Z","shell.execute_reply":"2024-12-07T11:07:01.02957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:07:01.032912Z","iopub.execute_input":"2024-12-07T11:07:01.033611Z","iopub.status.idle":"2024-12-07T11:07:01.631344Z","shell.execute_reply.started":"2024-12-07T11:07:01.03355Z","shell.execute_reply":"2024-12-07T11:07:01.629854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target = 'Premium Amount'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:07:01.634822Z","iopub.execute_input":"2024-12-07T11:07:01.635431Z","iopub.status.idle":"2024-12-07T11:07:01.641351Z","shell.execute_reply.started":"2024-12-07T11:07:01.635368Z","shell.execute_reply":"2024-12-07T11:07:01.639857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_features = train_df.drop(target, axis=1).select_dtypes(include=np.number).columns.values\nnumerical_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:07:01.64326Z","iopub.execute_input":"2024-12-07T11:07:01.643812Z","iopub.status.idle":"2024-12-07T11:07:01.90942Z","shell.execute_reply.started":"2024-12-07T11:07:01.643746Z","shell.execute_reply":"2024-12-07T11:07:01.908062Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_features = train_df.drop(target, axis=1).select_dtypes(include='object').columns.values\ncategorical_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:07:01.911056Z","iopub.execute_input":"2024-12-07T11:07:01.911589Z","iopub.status.idle":"2024-12-07T11:07:02.580198Z","shell.execute_reply.started":"2024-12-07T11:07:01.911535Z","shell.execute_reply":"2024-12-07T11:07:02.57895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:10:42.886404Z","iopub.execute_input":"2024-12-07T11:10:42.886835Z","iopub.status.idle":"2024-12-07T11:10:44.44377Z","shell.execute_reply.started":"2024-12-07T11:10:42.886797Z","shell.execute_reply":"2024-12-07T11:10:44.442335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[numerical_features].astype(np.float_).describe().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:11:31.78565Z","iopub.execute_input":"2024-12-07T11:11:31.786093Z","iopub.status.idle":"2024-12-07T11:11:32.735429Z","shell.execute_reply.started":"2024-12-07T11:11:31.786055Z","shell.execute_reply":"2024-12-07T11:11:32.73412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.describe(include='O').T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:11:34.158995Z","iopub.execute_input":"2024-12-07T11:11:34.159433Z","iopub.status.idle":"2024-12-07T11:11:35.681755Z","shell.execute_reply.started":"2024-12-07T11:11:34.159394Z","shell.execute_reply":"2024-12-07T11:11:35.680293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, FunctionTransformer, LabelEncoder, OneHotEncoder\nfrom sklearn.pipeline import make_pipeline, Pipeline\nfrom sklearn.compose import ColumnTransformer, make_column_selector, make_column_transformer\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import SimpleImputer, IterativeImputer\nimport category_encoders as ce\n\npreprocessing = ColumnTransformer([\n    ('num', make_pipeline(SimpleImputer(strategy='mean'), FunctionTransformer(), StandardScaler()), numerical_features),\n    ('cat', make_pipeline(SimpleImputer(strategy='most_frequent'), ce.cat_boost.CatBoostEncoder()), categorical_features)\n], remainder='drop')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:12:16.8244Z","iopub.execute_input":"2024-12-07T11:12:16.825107Z","iopub.status.idle":"2024-12-07T11:12:16.832358Z","shell.execute_reply.started":"2024-12-07T11:12:16.825064Z","shell.execute_reply":"2024-12-07T11:12:16.83098Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train_df.copy()\ny = X.pop(target)\ny = np.log1p(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:22:02.395091Z","iopub.execute_input":"2024-12-07T11:22:02.396904Z","iopub.status.idle":"2024-12-07T11:22:03.28133Z","shell.execute_reply.started":"2024-12-07T11:22:02.396834Z","shell.execute_reply":"2024-12-07T11:22:03.280007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nX = preprocessing.fit_transform(X, y)\ntestProcessed = preprocessing.transform(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:27:16.47776Z","iopub.execute_input":"2024-12-07T11:27:16.479096Z","iopub.status.idle":"2024-12-07T11:27:28.153246Z","shell.execute_reply.started":"2024-12-07T11:27:16.479041Z","shell.execute_reply":"2024-12-07T11:27:28.152081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lightgbm import LGBMRegressor\nlgb_model = LGBMRegressor(\n    boosting_type='gbdt', \n    num_leaves=31, \n    max_depth=-1, \n    learning_rate=0.1, \n    n_estimators=1000, \n    random_state=42\n)\n\nlgb_model.fit(X,y) \n\n# Predict on the test set\ny_pred = lgb_model.predict(testProcessed)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:53:47.619165Z","iopub.execute_input":"2024-12-07T11:53:47.619632Z","iopub.status.idle":"2024-12-07T11:55:37.316431Z","shell.execute_reply.started":"2024-12-07T11:53:47.619593Z","shell.execute_reply":"2024-12-07T11:55:37.315294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBRegressor\nxgb_model = XGBRegressor(n_estimators=1000, learning_rate=0.05, max_depth=6, random_state=42, n_jobs=-1)\n\nxgb_model.fit(X,y)\ny_pred = xgb_model.predict(testProcessed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:05:42.736928Z","iopub.execute_input":"2024-12-07T12:05:42.73743Z","iopub.status.idle":"2024-12-07T12:07:46.269958Z","shell.execute_reply.started":"2024-12-07T12:05:42.73739Z","shell.execute_reply":"2024-12-07T12:07:46.268786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lightgbm import LGBMRegressor\n\n# Initialize the LightGBM Regressor with optimized parameters\nlgb_model = LGBMRegressor(\n    boosting_type='gbdt',       # Gradient Boosted Decision Trees\n    num_leaves=31,              # Number of leaves, balanced for regression tasks\n    max_depth=-1,               # Unlimited depth (can be tuned based on dataset size)\n    learning_rate=0.05,         # Lower learning rate for gradual optimization\n    n_estimators=2000,          # Increase the number of estimators for better convergence\n    min_child_samples=30,       # Minimum data points in a leaf (to prevent overfitting)\n    min_child_weight=1e-3,      # Minimum sum of hessian (second-order derivative)\n    subsample=0.8,              # Use 80% of data randomly for training to prevent overfitting\n    colsample_bytree=0.8,       # Use 80% of features randomly for training\n    reg_alpha=0.1,              # L1 regularization to reduce overfitting\n    reg_lambda=0.1,             # L2 regularization to reduce overfitting\n    max_bin=255,                # Maximum number of bins for discretizing continuous features\n    random_state=42,            # Ensure reproducibility\n    n_jobs=-1                   # Use all CPU cores\n)\n\n# Train the model\nlgb_model.fit(\n    X, y               # Print logs every 50 iterations\n)\n\n# Predict on the test set\ny_pred = lgb_model.predict(testProcessed)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:24:35.066485Z","iopub.execute_input":"2024-12-07T12:24:35.066884Z","iopub.status.idle":"2024-12-07T12:28:23.28489Z","shell.execute_reply.started":"2024-12-07T12:24:35.066848Z","shell.execute_reply":"2024-12-07T12:28:23.28383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")\nsub[target] = np.expm1(y_pred)\nsub.to_csv(\"submission_lgb_new.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:28:23.287412Z","iopub.execute_input":"2024-12-07T12:28:23.287932Z","iopub.status.idle":"2024-12-07T12:28:25.156058Z","shell.execute_reply.started":"2024-12-07T12:28:23.287878Z","shell.execute_reply":"2024-12-07T12:28:25.155117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}