{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":256712603,"sourceType":"kernelVersion"},{"sourceId":256716118,"sourceType":"kernelVersion"},{"sourceId":256719625,"sourceType":"kernelVersion"},{"sourceId":256712766,"sourceType":"kernelVersion"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport glob\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import KFold # Import KFold\nimport xgboost as xgb # Import the XGBoost library\nimport warnings\n\nwarnings.simplefilter('ignore')\n\n# --- 1. Configuration ---\n# Path to the original training data to get the true target values.\n# Make sure this path is correct.\ntry:\n    TRAIN_CSV_PATH = '/kaggle/input/playground-series-s4e12/train.csv'\n    train_df = pd.read_csv(TRAIN_CSV_PATH)\nexcept FileNotFoundError:\n    print(\"train.csv not found. Creating a dummy train file for demonstration.\")\n    # Create a dummy file if not found, for demonstration purposes.\n    # In a real scenario, you would need the actual train.csv.\n    oof_files = glob.glob('oof_*.csv')\n    if oof_files:\n        num_samples = pd.read_csv(oof_files[0]).shape[0]\n        train_df = pd.DataFrame({\n            'id': range(num_samples),\n            'Premium Amount': np.random.rand(num_samples) * 1000 + 500\n        })\n    else:\n        print(\"Error: No OOF files found to determine sample size. Exiting.\")\n        exit()\n\n\n# --- 2. Load Prediction Files ---\n\nprint(\"Searching for prediction files...\")\n# Find all Out-of-Fold (OOF) and test prediction files.\noof_files = sorted(glob.glob('/kaggle/input/**/oof_*.csv', recursive=True))\ntest_files = sorted(glob.glob('/kaggle/input/**/test_*.csv', recursive=True))\n\nif not oof_files or not test_files or len(oof_files) != len(test_files):\n    print(\"Error: Mismatch in OOF and Test files or no files found.\")\n    print(f\"Found {len(oof_files)} OOF files and {len(test_files)} Test files.\")\n    exit()\n\nprint(f\"Found {len(oof_files)} pairs of prediction files.\")\nfor i, (o_file, t_file) in enumerate(zip(oof_files, test_files)):\n    print(f\"  Model {i+1}: {o_file} | {t_file}\")\n\n# Load the true target and apply log1p for RMSLE calculation.\ny_true_log = np.log1p(train_df['Premium Amount'])\n\n# Load all OOF predictions and apply log1p.\noof_preds_log = []\nfor file in oof_files:\n    preds = pd.read_csv(file)['Premium Amount']\n    oof_preds_log.append(np.log1p(preds))\n\n# Load all test predictions.\ntest_preds = []\nfor file in test_files:\n    preds = pd.read_csv(file)['Premium Amount']\n    test_preds.append(preds)\n\n# Stack the OOF predictions as columns to create the training data for the meta-model.\nX_train_meta = np.column_stack(oof_preds_log)\n\n# Stack the test predictions similarly to create the prediction data for the meta-model.\ntest_preds_log = [np.log1p(p) for p in test_preds]\nX_test_meta = np.column_stack(test_preds_log)\n\nprint(f\"\\nMeta-model training data shape: {X_train_meta.shape}\")\nprint(f\"Meta-model test data shape: {X_test_meta.shape}\")\n\n\n# --- 3. XGBoost Meta-Model Training with Cross-Validation ---\n\n# Define KFold with the same conditions as when creating the L1 model's OOF.\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\n# Initialize arrays to store the meta-model's OOF predictions and test predictions.\noof_meta_preds = np.zeros(X_train_meta.shape[0])\ntest_meta_preds_list = []\n\nprint(\"\\nStarting training of XGBoost meta-model with KFold CV...\")\n\nfor fold, (train_idx, val_idx) in enumerate(kf.split(X_train_meta, y_true_log)):\n    print(f\"--- Fold {fold + 1}/{kf.get_n_splits()} ---\")\n\n    # Split the data into training and validation sets.\n    X_train, X_val = X_train_meta[train_idx], X_train_meta[val_idx]\n    y_train, y_val = y_true_log.iloc[train_idx], y_true_log.iloc[val_idx]\n\n    # Define the XGBoost meta-model.\n    # Parameters can be tuned as needed.\n    meta_model = xgb.XGBRegressor(\n        objective='reg:squarederror',\n        n_estimators=2000, # Increase n_estimators and optimize with early stopping\n        learning_rate=0.02,\n        max_depth=4,\n        subsample=0.7,\n        colsample_bytree=0.7,\n        random_state=42,\n        n_jobs=-1,\n        tree_method='hist' # Setting for faster training\n    )\n\n    # Train the meta-model using OOF predictions as features and the true values as the target.\n    meta_model.fit(X_train, y_train,\n                   eval_set=[(X_val, y_val)],\n                   early_stopping_rounds=100, # Set early stopping rounds\n                   verbose=200)\n\n    # Save the predictions on the validation data as OOF predictions.\n    oof_meta_preds[val_idx] = meta_model.predict(X_val)\n\n    # Make predictions on the test data and store them in a list.\n    test_meta_preds_list.append(meta_model.predict(X_test_meta))\n\n# --- 4. Final Prediction and Submission ---\n\nprint(\"\\n--- Meta-Model Training Finished ---\")\n\n# Calculate the overall OOF score for the meta-model.\nrmsle_score = np.sqrt(mean_squared_error(y_true_log, oof_meta_preds))\nprint(f\"Overall OOF RMSLE for Meta-Model: {rmsle_score:.6f}\")\n\n# Average the test predictions from each fold to get the final prediction.\nfinal_preds_log = np.mean(test_meta_preds_list, axis=0)\n\n# The predictions are on a log scale, so convert them back to the original scale.\nfinal_test_preds = np.expm1(final_preds_log)\n\n# Clip negative predictions to 0 (as insurance premiums cannot be negative).\nfinal_test_preds[final_test_preds < 0] = 0\n\n# Create the submission file.\nsubmission_df = pd.read_csv(test_files[0])[['id']]\nsubmission_df['Premium Amount'] = final_test_preds\nsubmission_df.to_csv('submission_xgb_meta_cv.csv', index=False)\n\nprint(\"\\nSubmission file 'submission_xgb_meta_cv.csv' created successfully.\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-19T10:16:54.397325Z","iopub.execute_input":"2025-08-19T10:16:54.397638Z","iopub.status.idle":"2025-08-19T10:20:55.077163Z","shell.execute_reply.started":"2025-08-19T10:16:54.397614Z","shell.execute_reply":"2025-08-19T10:20:55.07625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}