{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import accuracy_score, mean_squared_error\nimport warnings\n\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:11:47.249791Z","iopub.execute_input":"2024-12-07T12:11:47.250279Z","iopub.status.idle":"2024-12-07T12:11:49.995862Z","shell.execute_reply.started":"2024-12-07T12:11:47.250229Z","shell.execute_reply":"2024-12-07T12:11:49.994769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv', index_col='id', engine='pyarrow')\ntest_df = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv', index_col='id', engine='pyarrow')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:11:49.997811Z","iopub.execute_input":"2024-12-07T12:11:49.998404Z","iopub.status.idle":"2024-12-07T12:11:54.467764Z","shell.execute_reply.started":"2024-12-07T12:11:49.998346Z","shell.execute_reply":"2024-12-07T12:11:54.466642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:11:54.469248Z","iopub.execute_input":"2024-12-07T12:11:54.470126Z","iopub.status.idle":"2024-12-07T12:11:54.501344Z","shell.execute_reply.started":"2024-12-07T12:11:54.47007Z","shell.execute_reply":"2024-12-07T12:11:54.500314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:11:54.504018Z","iopub.execute_input":"2024-12-07T12:11:54.504481Z","iopub.status.idle":"2024-12-07T12:11:54.528209Z","shell.execute_reply.started":"2024-12-07T12:11:54.504429Z","shell.execute_reply":"2024-12-07T12:11:54.52718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def date_separator(x):\n    return pd.Series([x.day, x.month, x.year])\n\ntrain_df[['day', 'month', 'year']] = train_df['Policy Start Date'].apply(date_separator)\ntest_df[['day', 'month', 'year']] = test_df['Policy Start Date'].apply(date_separator)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:11:54.529646Z","iopub.execute_input":"2024-12-07T12:11:54.530125Z","iopub.status.idle":"2024-12-07T12:15:27.543991Z","shell.execute_reply.started":"2024-12-07T12:11:54.530062Z","shell.execute_reply":"2024-12-07T12:15:27.542911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:15:27.545384Z","iopub.execute_input":"2024-12-07T12:15:27.545733Z","iopub.status.idle":"2024-12-07T12:15:28.138005Z","shell.execute_reply.started":"2024-12-07T12:15:27.545698Z","shell.execute_reply":"2024-12-07T12:15:28.136843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:15:28.139395Z","iopub.execute_input":"2024-12-07T12:15:28.139727Z","iopub.status.idle":"2024-12-07T12:15:28.760966Z","shell.execute_reply.started":"2024-12-07T12:15:28.139696Z","shell.execute_reply":"2024-12-07T12:15:28.759655Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target = 'Premium Amount'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:15:28.762463Z","iopub.execute_input":"2024-12-07T12:15:28.762918Z","iopub.status.idle":"2024-12-07T12:15:28.768318Z","shell.execute_reply.started":"2024-12-07T12:15:28.762872Z","shell.execute_reply":"2024-12-07T12:15:28.767106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_features = train_df.drop(target, axis=1).select_dtypes(include=np.number).columns.values\nnumerical_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:15:28.769731Z","iopub.execute_input":"2024-12-07T12:15:28.770124Z","iopub.status.idle":"2024-12-07T12:15:29.032521Z","shell.execute_reply.started":"2024-12-07T12:15:28.770081Z","shell.execute_reply":"2024-12-07T12:15:29.031276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_features = train_df.drop(target, axis=1).select_dtypes(include='object').columns.values\ncategorical_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:15:29.037479Z","iopub.execute_input":"2024-12-07T12:15:29.038308Z","iopub.status.idle":"2024-12-07T12:15:29.666487Z","shell.execute_reply.started":"2024-12-07T12:15:29.038254Z","shell.execute_reply":"2024-12-07T12:15:29.66548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:15:29.668069Z","iopub.execute_input":"2024-12-07T12:15:29.668517Z","iopub.status.idle":"2024-12-07T12:15:31.064191Z","shell.execute_reply.started":"2024-12-07T12:15:29.668469Z","shell.execute_reply":"2024-12-07T12:15:31.063172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[numerical_features].astype(np.float_).describe().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:15:31.065485Z","iopub.execute_input":"2024-12-07T12:15:31.065825Z","iopub.status.idle":"2024-12-07T12:15:31.960908Z","shell.execute_reply.started":"2024-12-07T12:15:31.065793Z","shell.execute_reply":"2024-12-07T12:15:31.959811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.describe(include='O').T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:15:31.961997Z","iopub.execute_input":"2024-12-07T12:15:31.96228Z","iopub.status.idle":"2024-12-07T12:15:33.554866Z","shell.execute_reply.started":"2024-12-07T12:15:31.962251Z","shell.execute_reply":"2024-12-07T12:15:33.553771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, FunctionTransformer, LabelEncoder, OneHotEncoder\nfrom sklearn.pipeline import make_pipeline, Pipeline\nfrom sklearn.compose import ColumnTransformer, make_column_selector, make_column_transformer\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import SimpleImputer, IterativeImputer\nimport category_encoders as ce\n\npreprocessing = ColumnTransformer([\n    ('num', make_pipeline(SimpleImputer(strategy='mean'), FunctionTransformer(), StandardScaler()), numerical_features),\n    ('cat', make_pipeline(SimpleImputer(strategy='most_frequent'), ce.cat_boost.CatBoostEncoder()), categorical_features)\n], remainder='drop')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:15:33.556342Z","iopub.execute_input":"2024-12-07T12:15:33.556704Z","iopub.status.idle":"2024-12-07T12:15:34.07673Z","shell.execute_reply.started":"2024-12-07T12:15:33.556671Z","shell.execute_reply":"2024-12-07T12:15:34.075524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train_df.copy()\ny = X.pop(target)\ny = np.log1p(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:15:34.078016Z","iopub.execute_input":"2024-12-07T12:15:34.078494Z","iopub.status.idle":"2024-12-07T12:15:34.637947Z","shell.execute_reply.started":"2024-12-07T12:15:34.07846Z","shell.execute_reply":"2024-12-07T12:15:34.636747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = preprocessing.fit_transform(X, y)\ntestProcessed = preprocessing.transform(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:15:34.639192Z","iopub.execute_input":"2024-12-07T12:15:34.63954Z","iopub.status.idle":"2024-12-07T12:15:45.822999Z","shell.execute_reply.started":"2024-12-07T12:15:34.639506Z","shell.execute_reply":"2024-12-07T12:15:45.821925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lightgbm import LGBMRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:15:45.82423Z","iopub.execute_input":"2024-12-07T12:15:45.8247Z","iopub.status.idle":"2024-12-07T12:15:46.788241Z","shell.execute_reply.started":"2024-12-07T12:15:45.824652Z","shell.execute_reply":"2024-12-07T12:15:46.78719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_model = LGBMRegressor(\n    boosting_type='gbdt', \n    num_leaves=31, \n    max_depth=-1, \n    learning_rate=0.1, \n    n_estimators=1000, \n    random_state=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:15:46.789451Z","iopub.execute_input":"2024-12-07T12:15:46.789942Z","iopub.status.idle":"2024-12-07T12:15:46.795222Z","shell.execute_reply.started":"2024-12-07T12:15:46.789907Z","shell.execute_reply":"2024-12-07T12:15:46.794201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:15:46.796527Z","iopub.execute_input":"2024-12-07T12:15:46.796892Z","iopub.status.idle":"2024-12-07T12:15:46.988874Z","shell.execute_reply.started":"2024-12-07T12:15:46.796862Z","shell.execute_reply":"2024-12-07T12:15:46.987931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_model = XGBRegressor(\n    n_estimators=1000,\n    learning_rate=0.05,\n    max_depth=6,\n    random_state=42,\n    n_jobs=-1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:15:46.990225Z","iopub.execute_input":"2024-12-07T12:15:46.990548Z","iopub.status.idle":"2024-12-07T12:15:46.995776Z","shell.execute_reply.started":"2024-12-07T12:15:46.990515Z","shell.execute_reply":"2024-12-07T12:15:46.99461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import VotingRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:15:46.997151Z","iopub.execute_input":"2024-12-07T12:15:46.997462Z","iopub.status.idle":"2024-12-07T12:15:47.10317Z","shell.execute_reply.started":"2024-12-07T12:15:46.997431Z","shell.execute_reply":"2024-12-07T12:15:47.101732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = VotingRegressor([('LGB', lgb_model), ('XGB', xgb_model)])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:16:44.813675Z","iopub.execute_input":"2024-12-07T12:16:44.814648Z","iopub.status.idle":"2024-12-07T12:16:44.820378Z","shell.execute_reply.started":"2024-12-07T12:16:44.814588Z","shell.execute_reply":"2024-12-07T12:16:44.819085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.fit(X,y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:17:17.875506Z","iopub.execute_input":"2024-12-07T12:17:17.875999Z","iopub.status.idle":"2024-12-07T12:19:45.004421Z","shell.execute_reply.started":"2024-12-07T12:17:17.87596Z","shell.execute_reply":"2024-12-07T12:19:45.003349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(testProcessed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:19:45.006276Z","iopub.execute_input":"2024-12-07T12:19:45.006646Z","iopub.status.idle":"2024-12-07T12:20:46.425889Z","shell.execute_reply.started":"2024-12-07T12:19:45.006612Z","shell.execute_reply":"2024-12-07T12:20:46.424641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")\nsub[target] = np.expm1(y_pred)\nsub.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:20:46.427246Z","iopub.execute_input":"2024-12-07T12:20:46.427573Z","iopub.status.idle":"2024-12-07T12:20:48.422187Z","shell.execute_reply.started":"2024-12-07T12:20:46.427524Z","shell.execute_reply":"2024-12-07T12:20:48.421072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}