{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":10070859,"sourceType":"datasetVersion","datasetId":6207291}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import Libraries","metadata":{}},{"cell_type":"code","source":"# Import necessary libraries\nimport pandas as pd\nimport numpy as np\n\n# Machine Learning Libraries\nfrom sklearn.model_selection import train_test_split, KFold, cross_val_score, GridSearchCV\nfrom sklearn.metrics import mean_absolute_error\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\n\n# Suppress warnings for cleaner output\nimport warnings\nwarnings.filterwarnings('ignore')\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:22:39.929057Z","iopub.execute_input":"2024-12-01T22:22:39.929941Z","iopub.status.idle":"2024-12-01T22:22:40.808132Z","shell.execute_reply.started":"2024-12-01T22:22:39.929888Z","shell.execute_reply":"2024-12-01T22:22:40.807481Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data","metadata":{}},{"cell_type":"code","source":"# Define file paths\ntrain_path = '/kaggle/input/rwaid-cleaned-dataset-dropaddconvert/train_cleaned.csv'\ntest_path = '/kaggle/input/rwaid-cleaned-dataset-dropaddconvert/test_cleaned.csv'  # Update if different\n\n# Load the datasets\ntrain = pd.read_csv(train_path)\ntest = pd.read_csv(test_path)\n\n# Display the first few rows of the training data\ntrain.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:22:40.80963Z","iopub.execute_input":"2024-12-01T22:22:40.80997Z","iopub.status.idle":"2024-12-01T22:22:43.545529Z","shell.execute_reply.started":"2024-12-01T22:22:40.809943Z","shell.execute_reply":"2024-12-01T22:22:43.54461Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Exploration","metadata":{}},{"cell_type":"code","source":"# Check the shape of the datasets\nprint(f'Training Data Shape: {train.shape}')\nprint(f'Test Data Shape: {test.shape}')\n\n# Summary statistics of the training data\ntrain.describe()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:22:43.546627Z","iopub.execute_input":"2024-12-01T22:22:43.546882Z","iopub.status.idle":"2024-12-01T22:22:44.273053Z","shell.execute_reply.started":"2024-12-01T22:22:43.546857Z","shell.execute_reply":"2024-12-01T22:22:44.272181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define target and features\ntarget = 'Premium Amount'\nid_column = 'id'\n\n# Features are all columns except 'id' and 'Premium Amount'\nfeature_cols = [col for col in train.columns if col not in [id_column, target]]\n\n# Separate features and target in training data\nX = train[feature_cols]\ny = train[target]\n\n# Features in test data\nX_test = test[feature_cols]\n\n# Display feature names\nprint(f'Number of Features: {len(feature_cols)}')\nprint(f'Feature Columns: {feature_cols}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:22:44.274953Z","iopub.execute_input":"2024-12-01T22:22:44.275222Z","iopub.status.idle":"2024-12-01T22:22:44.368234Z","shell.execute_reply.started":"2024-12-01T22:22:44.275197Z","shell.execute_reply":"2024-12-01T22:22:44.367348Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Cross-Validation","metadata":{}},{"cell_type":"code","source":"# Define cross-validation strategy\nkf = KFold(n_splits=5, shuffle=True, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:22:44.369173Z","iopub.execute_input":"2024-12-01T22:22:44.369443Z","iopub.status.idle":"2024-12-01T22:22:44.373507Z","shell.execute_reply.started":"2024-12-01T22:22:44.369418Z","shell.execute_reply":"2024-12-01T22:22:44.372687Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Hyperparameter (Previously Found) Tuning for XGBoost","metadata":{}},{"cell_type":"code","source":"# ------------------------------------\n# Hyperparameter Tuning for XGBoost Using Optuna\n# ------------------------------------\n\n# Example of best parameters found by Optuna\nbest_xgb_params = {\n    'n_estimators': 1339,\n    'max_depth': 12,\n    'learning_rate': 0.021092552123872205,\n    'colsample_bytree': 0.9597977223447891,\n    'gamma': 1.468060636571578,\n    'min_child_weight': 8,\n    'subsample': 1.0\n}\n\nprint(\"Best XGBoost Parameters from Optuna:\")\nfor param, value in best_xgb_params.items():\n    print(f\"{param}: {value}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:22:44.374704Z","iopub.execute_input":"2024-12-01T22:22:44.37547Z","iopub.status.idle":"2024-12-01T22:22:44.388008Z","shell.execute_reply.started":"2024-12-01T22:22:44.37543Z","shell.execute_reply":"2024-12-01T22:22:44.387134Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training the CatBoost Regressor","metadata":{}},{"cell_type":"code","source":"# ------------------------------------\n# Training the CatBoost Regressor\n# ------------------------------------\n\n# Initialize CatBoost Regressor on CPU\ncat_params = {\n    'iterations': 1000,\n    'learning_rate': 0.05,\n    'depth': 10,\n    'loss_function': 'MAE',\n    'eval_metric': 'MAE',\n    'random_seed': 42,\n    'verbose': False,\n    # 'task_type': 'CPU'  # Optional: Explicitly specify CPU usage (default is CPU)\n}\n\ncat_model = CatBoostRegressor(**cat_params)\n\n# Perform cross-validation on CPU\ncat_cv_scores = cross_val_score(\n    cat_model, X, y,\n    scoring='neg_mean_absolute_error',\n    cv=kf,\n    n_jobs=-1  # Utilize all available CPU cores\n)\n\n# Print cross-validation MAE\nprint(f'CatBoost Cross-Validation MAE: {-cat_cv_scores.mean():.4f} ± {cat_cv_scores.std():.4f}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:22:44.389028Z","iopub.execute_input":"2024-12-01T22:22:44.38927Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Ensemble of XGBoost and CatBoost","metadata":{}},{"cell_type":"code","source":"# ------------------------------------\n# Ensemble of XGBoost and CatBoost\n# ------------------------------------\n\n# Initialize lists to store out-of-fold predictions\nxgb_oof = np.zeros(X.shape[0])\ncat_oof = np.zeros(X.shape[0])\n\n# Initialize arrays to store test predictions\nxgb_test_pred = np.zeros(X_test.shape[0])\ncat_test_pred = np.zeros(X_test.shape[0])\n\n# Perform cross-validation and ensemble\nfor fold, (train_idx, valid_idx) in enumerate(kf.split(X)):\n    print(f'Fold {fold + 1}')\n    \n    # Split data\n    X_train_fold, X_valid_fold = X.iloc[train_idx], X.iloc[valid_idx]\n    y_train_fold, y_valid_fold = y.iloc[train_idx], y.iloc[valid_idx]\n    \n    # ----------------------------\n    # Train XGBoost on GPU\n    # ----------------------------\n    xgb_best = XGBRegressor(\n        objective='reg:squarederror',\n        eval_metric='mae',\n        random_state=42,\n        verbosity=0,\n        tree_method='gpu_hist',  # Enables GPU acceleration\n        gpu_id=0,                # Specifies the GPU device (default is 0)\n        **best_xgb_params       # Incorporate the best parameters from Optuna\n    )\n    \n    xgb_best.fit(\n        X_train_fold, y_train_fold,\n        eval_set=[(X_valid_fold, y_valid_fold)],\n        early_stopping_rounds=50,\n        verbose=False\n    )\n    xgb_oof[valid_idx] = xgb_best.predict(X_valid_fold)\n    xgb_test_pred += xgb_best.predict(X_test) / kf.n_splits\n    \n    # ----------------------------\n    # Train CatBoost on CPU\n    # ----------------------------\n    # Re-initialize CatBoost model for each fold to ensure clean state\n    cat_model_fold = CatBoostRegressor(**cat_params)\n    \n    cat_model_fold.fit(\n        X_train_fold, y_train_fold,\n        eval_set=(X_valid_fold, y_valid_fold),\n        early_stopping_rounds=50,\n        verbose=False\n    )\n    cat_oof[valid_idx] = cat_model_fold.predict(X_valid_fold)\n    cat_test_pred += cat_model_fold.predict(X_test) / kf.n_splits\n\n# Calculate MAE for individual models\nxgb_final_mae = mean_absolute_error(y, xgb_oof)\ncat_final_mae = mean_absolute_error(y, cat_oof)\nprint(f'Final XGBoost MAE: {xgb_final_mae:.4f}')\nprint(f'Final CatBoost MAE: {cat_final_mae:.4f}')\n\n# Ensemble predictions by averaging\nensemble_oof = (xgb_oof + cat_oof) / 2\nensemble_mae = mean_absolute_error(y, ensemble_oof)\nprint(f'Ensemble Cross-Validation MAE: {ensemble_mae:.4f}')\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Retrain Models on the Full Training Data","metadata":{}},{"cell_type":"code","source":"# ------------------------------------\n# Retrain Models on the Full Training Data\n# ------------------------------------\n\n# ----------------------------\n# Retrain XGBoost on Full Data with GPU Support\n# ----------------------------\n\n# Initialize XGBoost Regressor with GPU support using the best found parameters\nxgb_best_full = XGBRegressor(\n    objective='reg:squarederror',\n    eval_metric='mae',\n    random_state=42,\n    verbosity=0,\n    tree_method='gpu_hist',  # Enables GPU acceleration\n    gpu_id=0,                # Specifies the GPU device (default is 0)\n    **best_xgb_params       # Incorporate the best parameters from Optuna\n)\n\n# Fit the XGBoost model on the full training data\nxgb_best_full.fit(\n    X, y,\n    eval_set=[(X, y)],             # Optionally include evaluation set for monitoring\n    early_stopping_rounds=50,      # Optional: Prevent overfitting\n    verbose=False                   # Suppress training logs for cleaner output\n)\n\n# ----------------------------\n# Retrain CatBoost on Full Data (CPU)\n# ----------------------------\n\n# Initialize CatBoost Regressor on CPU using predefined parameters\ncat_model_full = CatBoostRegressor(\n    iterations=1000,\n    learning_rate=0.05,\n    depth=10,\n    loss_function='MAE',\n    eval_metric='MAE',\n    random_seed=42,\n    verbose=False                  # Suppress training logs for cleaner output\n    # 'task_type': 'CPU'           # Optional: Explicitly specify CPU usage (default is CPU)\n)\n\n# Fit the CatBoost model on the full training data\ncat_model_full.fit(\n    X, y,\n    verbose=False                   # Suppress training logs for cleaner output\n)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Predictions on the Test Set","metadata":{}},{"cell_type":"code","source":"# ------------------------------------\n# Predictions on the Test Set\n# ------------------------------------\n\n# Predict with XGBoost\nxgb_test_final_pred = xgb_best_full.predict(X_test)\n\n# Predict with CatBoost\ncat_test_final_pred = cat_model_full.predict(X_test)\n\n# Ensemble predictions by averaging\nensemble_test_pred = (xgb_test_final_pred + cat_test_final_pred) / 2\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Prepare and Save the Submission File","metadata":{}},{"cell_type":"code","source":"# ------------------------------------\n# Prepare and Save the Submission File\n# ------------------------------------\n\n# Prepare submission DataFrame\nsubmission = pd.DataFrame({\n    'id': test['id'],\n    'Premium Amount': ensemble_test_pred\n})\n\n# Round the Premium Amount to three decimal places\nsubmission['Premium Amount'] = submission['Premium Amount'].round(3)\n\n# Display the first few rows of the submission file\nsubmission.head()\n\n# Save to CSV\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission file 'submission.csv' created successfully!\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}