{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30840,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error\nfrom xgboost import XGBRegressor","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-02-03T09:51:23.368453Z","iopub.execute_input":"2025-02-03T09:51:23.368736Z","iopub.status.idle":"2025-02-03T09:51:24.902936Z","shell.execute_reply.started":"2025-02-03T09:51:23.368716Z","shell.execute_reply":"2025-02-03T09:51:24.902071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Enable GPU acceleration\nimport os\nos.environ['CUDA_VISIBLE_DEVICES'] = '0'  # Use first GPU","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T09:51:24.90406Z","iopub.execute_input":"2025-02-03T09:51:24.904535Z","iopub.status.idle":"2025-02-03T09:51:24.908137Z","shell.execute_reply.started":"2025-02-03T09:51:24.904503Z","shell.execute_reply":"2025-02-03T09:51:24.907381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load datasets\ntrain_path = '/kaggle/input/playground-series-s4e12/train.csv'\ntest_path = '/kaggle/input/playground-series-s4e12/test.csv'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T09:51:24.909732Z","iopub.execute_input":"2025-02-03T09:51:24.909923Z","iopub.status.idle":"2025-02-03T09:51:24.925239Z","shell.execute_reply.started":"2025-02-03T09:51:24.909907Z","shell.execute_reply":"2025-02-03T09:51:24.92465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv(train_path)\ntest_df = pd.read_csv(test_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T09:51:24.92618Z","iopub.execute_input":"2025-02-03T09:51:24.926456Z","iopub.status.idle":"2025-02-03T09:51:32.86706Z","shell.execute_reply.started":"2025-02-03T09:51:24.926437Z","shell.execute_reply":"2025-02-03T09:51:32.866348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocessing and Modeling\nX = train_df.drop(columns=['id', 'Premium Amount', 'Policy Start Date'])\ny = train_df['Premium Amount']\ntest_ids = test_df['id']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T09:51:32.868055Z","iopub.execute_input":"2025-02-03T09:51:32.868322Z","iopub.status.idle":"2025-02-03T09:51:33.00833Z","shell.execute_reply.started":"2025-02-03T09:51:32.868299Z","shell.execute_reply":"2025-02-03T09:51:33.007689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define preprocessing\nnumeric_features = X.select_dtypes(include=np.number).columns\ncategorical_features = X.select_dtypes(include='object').columns\n\npreprocessor = ColumnTransformer([\n    ('num', Pipeline([\n        ('imputer', SimpleImputer(strategy='median')),\n        ('scaler', StandardScaler())\n    ]), numeric_features),\n    ('cat', Pipeline([\n        ('imputer', SimpleImputer(strategy='most_frequent')),\n        ('encoder', OneHotEncoder(handle_unknown='ignore', sparse_output=False))\n    ]), categorical_features)\n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T09:51:39.969108Z","iopub.execute_input":"2025-02-03T09:51:39.969433Z","iopub.status.idle":"2025-02-03T09:51:40.11481Z","shell.execute_reply.started":"2025-02-03T09:51:39.969398Z","shell.execute_reply":"2025-02-03T09:51:40.114134Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# GPU-optimized XGBoost model\nxgb_params = {\n    'tree_method': 'gpu_hist',\n    'predictor': 'gpu_predictor',\n    'n_estimators': 2000,\n    'learning_rate': 0.05,\n    'max_depth': 8,\n    'subsample': 0.8,\n    'colsample_bytree': 0.9,\n    'random_state': 42,\n    'verbosity': 1\n}\n\nmodel = Pipeline([\n    ('preprocessor', preprocessor),\n    ('regressor', XGBRegressor(**xgb_params))\n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T09:51:50.39174Z","iopub.execute_input":"2025-02-03T09:51:50.39203Z","iopub.status.idle":"2025-02-03T09:51:50.396255Z","shell.execute_reply.started":"2025-02-03T09:51:50.392009Z","shell.execute_reply":"2025-02-03T09:51:50.39543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Train-validation split\nX_train, X_val, y_train, y_val = train_test_split(\n    X, y, test_size=0.2, random_state=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T09:51:58.974852Z","iopub.execute_input":"2025-02-03T09:51:58.975144Z","iopub.status.idle":"2025-02-03T09:51:59.514102Z","shell.execute_reply.started":"2025-02-03T09:51:58.975121Z","shell.execute_reply":"2025-02-03T09:51:59.51344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train model with early stopping\nmodel.fit(X_train, y_train,)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T09:52:41.014218Z","iopub.execute_input":"2025-02-03T09:52:41.014536Z","iopub.status.idle":"2025-02-03T09:53:21.15434Z","shell.execute_reply.started":"2025-02-03T09:52:41.014509Z","shell.execute_reply":"2025-02-03T09:53:21.153584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Validate model\nval_preds = model.predict(X_val)\nrmse = np.sqrt(mean_squared_error(y_val, val_preds))\nprint(f\"\\nValidation RMSE: {rmse:.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T09:53:21.155312Z","iopub.execute_input":"2025-02-03T09:53:21.155597Z","iopub.status.idle":"2025-02-03T09:53:22.235527Z","shell.execute_reply.started":"2025-02-03T09:53:21.155575Z","shell.execute_reply":"2025-02-03T09:53:22.234646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generate predictions\ntest_preds = model.predict(test_df.drop(columns=['id', 'Policy Start Date']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T09:53:22.236944Z","iopub.execute_input":"2025-02-03T09:53:22.237188Z","iopub.status.idle":"2025-02-03T09:53:25.937481Z","shell.execute_reply.started":"2025-02-03T09:53:22.237168Z","shell.execute_reply":"2025-02-03T09:53:25.936785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create submission file\nsubmission = pd.DataFrame({\n    'id': test_ids,\n    'Premium Amount': test_preds\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T09:53:25.938277Z","iopub.execute_input":"2025-02-03T09:53:25.938521Z","iopub.status.idle":"2025-02-03T09:53:25.943884Z","shell.execute_reply.started":"2025-02-03T09:53:25.9385Z","shell.execute_reply":"2025-02-03T09:53:25.943073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)\nprint(\"Submission file created!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T09:53:29.205444Z","iopub.execute_input":"2025-02-03T09:53:29.205753Z","iopub.status.idle":"2025-02-03T09:53:30.121693Z","shell.execute_reply.started":"2025-02-03T09:53:29.205728Z","shell.execute_reply":"2025-02-03T09:53:30.12081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}