{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":84896,"databundleVersionId":10305135}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-18T20:25:59.931724Z","iopub.execute_input":"2026-04-18T20:25:59.932157Z","iopub.status.idle":"2026-04-18T20:25:59.938997Z","shell.execute_reply.started":"2026-04-18T20:25:59.932123Z","shell.execute_reply":"2026-04-18T20:25:59.93832Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Insurance Premium Prediction Notebook\n## Imports, Environment Check, and Kaggle Paths","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport time\nimport warnings\nimport subprocess\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nwarnings.filterwarnings(\"ignore\")\npd.set_option(\"display.max_columns\", 200)\n\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_error\nfrom xgboost import XGBRegressor\nimport lightgbm as lgb\n\ntry:\n    import cudf\n    HAS_CUDF = True\nexcept Exception:\n    cudf = None\n    HAS_CUDF = False\n\ndef has_nvidia_gpu():\n    try:\n        subprocess.run(\n            [\"nvidia-smi\"],\n            stdout=subprocess.PIPE,\n            stderr=subprocess.PIPE,\n            check=True\n        )\n        return True\n    except Exception:\n        return False\n\nHAS_GPU = has_nvidia_gpu()\n\nTRAIN_PATH = \"/kaggle/input/competitions/playground-series-s4e12/train.csv\"\nTEST_PATH  = \"/kaggle/input/competitions/playground-series-s4e12/test.csv\"\n\nprint(f\"HAS_CUDF = {HAS_CUDF}\")\nprint(f\"HAS_GPU  = {HAS_GPU}\")\nprint(f\"TRAIN_PATH exists: {os.path.exists(TRAIN_PATH)}\")\nprint(f\"TEST_PATH exists : {os.path.exists(TEST_PATH)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T20:26:03.169585Z","iopub.execute_input":"2026-04-18T20:26:03.170294Z","iopub.status.idle":"2026-04-18T20:26:11.15868Z","shell.execute_reply.started":"2026-04-18T20:26:03.170261Z","shell.execute_reply":"2026-04-18T20:26:11.157823Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load Training and Test Data","metadata":{}},{"cell_type":"code","source":"train_raw = pd.read_csv(TRAIN_PATH)\ntest_raw = pd.read_csv(TEST_PATH)\n\nprint(\"Train shape:\", train_raw.shape)\nprint(\"Test shape :\", test_raw.shape)\nprint(\"\\nTrain columns:\\n\", train_raw.columns.tolist())\nprint(\"\\nFirst 5 rows of train:\")\ndisplay(train_raw.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T20:26:14.913069Z","iopub.execute_input":"2026-04-18T20:26:14.914223Z","iopub.status.idle":"2026-04-18T20:26:22.247157Z","shell.execute_reply.started":"2026-04-18T20:26:14.914189Z","shell.execute_reply":"2026-04-18T20:26:22.246491Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 2, figsize=(16, 10))\n\ntrain_raw[\"Premium Amount\"].hist(bins=50, ax=axes[0, 0])\naxes[0, 0].set_title(\"Target Distribution: Premium Amount\")\n\nnp.log1p(train_raw[\"Premium Amount\"]).hist(bins=50, ax=axes[0, 1])\naxes[0, 1].set_title(\"Log Target Distribution: log1p(Premium Amount)\")\n\nmissing_train = train_raw.isnull().sum().sort_values(ascending=False)\nmissing_train = missing_train[missing_train > 0]\nif len(missing_train) > 0:\n    missing_train.head(15).plot(kind=\"bar\", ax=axes[1, 0])\n    axes[1, 0].set_title(\"Top Missing Columns in Train\")\nelse:\n    axes[1, 0].text(0.2, 0.5, \"No missing values in train\", fontsize=12)\n    axes[1, 0].set_title(\"Top Missing Columns in Train\")\n    axes[1, 0].axis(\"off\")\n\npolicy_dates = pd.to_datetime(train_raw[\"Policy Start Date\"])\npolicy_dates.dt.to_period(\"M\").astype(str).value_counts().sort_index().plot(ax=axes[1, 1])\naxes[1, 1].set_title(\"Policy Start Date Count by Month\")\naxes[1, 1].tick_params(axis=\"x\", rotation=90)\n\nplt.tight_layout()\nplt.show()\n\ncat_cols_preview = train_raw.select_dtypes(include=[\"object\"]).columns.tolist()\nfor col in cat_cols_preview[:4]:\n    plt.figure(figsize=(10, 4))\n    train_raw[col].astype(str).value_counts().head(10).plot(kind=\"bar\")\n    plt.title(f\"Top Categories in {col}\")\n    plt.xticks(rotation=45)\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T20:26:34.023603Z","iopub.execute_input":"2026-04-18T20:26:34.023909Z","iopub.status.idle":"2026-04-18T20:26:37.552881Z","shell.execute_reply.started":"2026-04-18T20:26:34.023885Z","shell.execute_reply":"2026-04-18T20:26:37.552203Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Date Feature Engineering and Encoding Preparation\nWe extract useful time-based features from the policy start date and prepare the dataset for modeling.","metadata":{}},{"cell_type":"code","source":"train = train_raw.copy()\ntest = test_raw.copy()\n\nfor df in [train, test]:\n    df[\"Policy Start Date\"] = pd.to_datetime(df[\"Policy Start Date\"])\n    df[\"year\"] = df[\"Policy Start Date\"].dt.year.astype(\"float32\")\n    df[\"month\"] = df[\"Policy Start Date\"].dt.month.astype(\"float32\")\n    df[\"day\"] = df[\"Policy Start Date\"].dt.day.astype(\"float32\")\n    df[\"dow\"] = df[\"Policy Start Date\"].dt.dayofweek.astype(\"float32\")\n    df[\"woy\"] = df[\"Policy Start Date\"].dt.isocalendar().week.astype(\"int32\").astype(\"float32\")\n    df[\"seconds\"] = (df[\"Policy Start Date\"].astype(\"int64\") // 10**9).astype(\"float32\")\n\ntrain[\"y\"] = np.log1p(train[\"Premium Amount\"]).astype(\"float32\")\n\nRMV = [\"id\", \"Policy Start Date\", \"Premium Amount\", \"y\"]\nFEATURES = [c for c in train.columns if c not in RMV]\n\ncombined = pd.concat(\n    [train.drop(columns=[\"y\"]), test],\n    axis=0,\n    ignore_index=True\n)\n\nCATS = []\nHIGH_CARDINALITY = []\n\nfor c in FEATURES:\n    dtype_str = str(combined[c].dtype)\n\n    if dtype_str in (\"object\", \"string\", \"str\", \"string[python]\", \"string[pyarrow]\"):\n        CATS.append(c)\n        combined[c] = combined[c].fillna(\"NAN\").astype(str)\n        codes, _ = pd.factorize(combined[c], sort=True)\n        combined[c] = codes.astype(\"int32\")\n\n    if combined[c].dtype == \"int64\":\n        combined[c] = combined[c].astype(\"int32\")\n    elif combined[c].dtype == \"float64\":\n        combined[c] = combined[c].astype(\"float32\")\n\n    if combined[c].nunique(dropna=False) >= 9:\n        HIGH_CARDINALITY.append(c)\n\nn_train = len(train)\ntrain_features = combined.iloc[:n_train].copy().reset_index(drop=True)\ntest_features = combined.iloc[n_train:].copy().reset_index(drop=True)\n\ntrain = pd.concat([train[[\"id\", \"Premium Amount\", \"y\"]].reset_index(drop=True), train_features], axis=1)\ntest = pd.concat([test[[\"id\"]].reset_index(drop=True), test_features], axis=1)\n\nprint(f\"Total FEATURES: {len(FEATURES)}\")\nprint(f\"Categorical FEATURES: {len(CATS)}\")\nprint(f\"High-cardinality FEATURES: {len(HIGH_CARDINALITY)}\")\nprint(\"Sample encoded train shape:\", train.shape)\nprint(\"Sample encoded test shape :\", test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T20:27:07.634075Z","iopub.execute_input":"2026-04-18T20:27:07.634826Z","iopub.status.idle":"2026-04-18T20:27:12.720081Z","shell.execute_reply.started":"2026-04-18T20:27:07.634796Z","shell.execute_reply":"2026-04-18T20:27:12.719352Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Target Encoding and Count Encoding Functions\nThis section defines the helper functions for GPU/pandas groupby aggregation, target encoding, and count encoding.","metadata":{}},{"cell_type":"code","source":"COMBOS = [\n    [\"Annual Income\", \"Health Score\"],\n    [\"Credit Score\", \"Health Score\"],\n    [\"Customer Feedback\", \"Gender\", \"Marital Status\", \"Occupation\", \"Smoking Status\", \"year\"],\n    [\"Exercise Frequency\", \"Health Score\"],\n    [\"Health Score\", \"Marital Status\"],\n    [\"Education Level\", \"Gender\", \"Health Score\"],\n    [\"Health Score\", \"Occupation\"],\n    [\"Age\", \"Health Score\"],\n    [\"Health Score\", \"dow\"],\n    [\"Age\", \"Exercise Frequency\", \"Location\"],\n    [\"Health Score\", \"Smoking Status\", \"month\"],\n    [\"Health Score\", \"Location\", \"Policy Type\"],\n    [\"Health Score\", \"Insurance Duration\"],\n    [\"Health Score\", \"Number of Dependents\"],\n    [\"Customer Feedback\", \"Exercise Frequency\", \"Previous Claims\", \"Property Type\", \"dow\"],\n    [\"Customer Feedback\", \"Health Score\"],\n    [\"Health Score\", \"Property Type\"],\n    [\"Health Score\", \"day\", \"seconds\"],\n    [\"Health Score\", \"year\"],\n    [\"Age\", \"Gender\", \"Insurance Duration\", \"year\"],\n    [\"Health Score\", \"Vehicle Age\"],\n    [\"Credit Score\", \"Annual Income\"],\n    [\"Age\", \"Smoking Status\", \"Policy Type\"],\n    [\"Health Score\", \"woy\"],\n    [\"Gender\", \"Location\", \"Policy Type\", \"Smoking Status\"],\n    [\"Age\", \"Credit Score\"],\n    [\"Health Score\", \"Exercise Frequency\", \"Occupation\"],\n    [\"Annual Income\", \"Insurance Duration\"],\n    [\"Vehicle Age\", \"Credit Score\", \"Policy Type\"],\n    [\"Education Level\", \"Marital Status\", \"Property Type\", \"Location\"],\n]\n\nprint(f\"Number of combo groups: {len(COMBOS)}\")\n\ndef rmse(y_true, y_pred):\n    return np.sqrt(mean_squared_error(y_true, y_pred))\n\ndef gpu_groupby_agg(pdf, cols, target, agg):\n    use_cols = cols + [target]\n\n    if HAS_CUDF:\n        try:\n            gdf = cudf.from_pandas(pdf[use_cols].copy())\n\n            if agg == \"nunique\":\n                stats = gdf.groupby(cols)[target].agg([\"nunique\", \"count\"]).reset_index()\n                stats.columns = cols + [agg, \"count\"]\n            elif agg == \"std\":\n                stats = gdf.groupby(cols)[target].agg([\"std\", \"count\"]).reset_index()\n                stats.columns = cols + [agg, \"count\"]\n            else:\n                stats = gdf.groupby(cols)[target].agg([agg, \"count\"]).reset_index()\n                stats.columns = cols + [agg, \"count\"]\n\n            return stats.to_pandas()\n\n        except Exception:\n            pass\n\n    if agg == \"nunique\":\n        stats = pdf.groupby(cols, dropna=False)[target].agg([\"nunique\", \"count\"]).reset_index()\n    elif agg == \"std\":\n        stats = pdf.groupby(cols, dropna=False)[target].agg([\"std\", \"count\"]).reset_index()\n    else:\n        stats = pdf.groupby(cols, dropna=False)[target].agg([agg, \"count\"]).reset_index()\n\n    stats.columns = cols + [agg, \"count\"]\n    return stats\n\ndef target_encode(train_df, valid_df, test_df, cols, target=\"y\", kfold=5, smooth=30, agg=\"mean\"):\n    feat_name = \"TE_\" + agg.upper() + \"_\" + \"_\".join(cols)\n\n    train_df = train_df.copy()\n    valid_df = valid_df.copy()\n    test_df = test_df.copy()\n\n    train_df[\"kfold_te\"] = np.arange(len(train_df)) % kfold\n    train_df[feat_name] = 0.0\n\n    if agg == \"mean\":\n        global_stat = train_df[target].mean()\n    elif agg == \"median\":\n        global_stat = train_df[target].median()\n    elif agg == \"min\":\n        global_stat = train_df[target].min()\n    elif agg == \"max\":\n        global_stat = train_df[target].max()\n    elif agg == \"std\":\n        global_stat = train_df[target].std()\n    elif agg == \"nunique\":\n        global_stat = 0.0\n    else:\n        raise ValueError(f\"Unsupported agg: {agg}\")\n\n    for fold_id in range(kfold):\n        fit_part = train_df[train_df[\"kfold_te\"] != fold_id].copy()\n        val_mask = train_df[\"kfold_te\"] == fold_id\n\n        stats = gpu_groupby_agg(fit_part, cols, target, agg)\n\n        if agg in [\"nunique\", \"std\"]:\n            stats[\"enc_value\"] = stats[agg] / (stats[\"count\"] + 1.0)\n        else:\n            stats[\"enc_value\"] = (\n                (stats[agg] * stats[\"count\"]) + (global_stat * smooth)\n            ) / (stats[\"count\"] + smooth)\n\n        fold_encoded = train_df.loc[val_mask, cols].merge(\n            stats[cols + [\"enc_value\"]],\n            on=cols,\n            how=\"left\"\n        )[\"enc_value\"].fillna(global_stat).values\n\n        train_df.loc[val_mask, feat_name] = fold_encoded\n\n    full_stats = gpu_groupby_agg(train_df, cols, target, agg)\n\n    if agg in [\"nunique\", \"std\"]:\n        full_stats[\"enc_value\"] = full_stats[agg] / (full_stats[\"count\"] + 1.0)\n    else:\n        full_stats[\"enc_value\"] = (\n            (full_stats[agg] * full_stats[\"count\"]) + (global_stat * smooth)\n        ) / (full_stats[\"count\"] + smooth)\n\n    for df_name, df_obj in [(\"valid\", valid_df), (\"test\", test_df)]:\n        encoded = df_obj[cols].merge(\n            full_stats[cols + [\"enc_value\"]],\n            on=cols,\n            how=\"left\"\n        )[\"enc_value\"].fillna(global_stat).astype(\"float32\").values\n\n        if df_name == \"valid\":\n            valid_df[feat_name] = encoded\n        else:\n            test_df[feat_name] = encoded\n\n    train_df[feat_name] = train_df[feat_name].astype(\"float32\")\n    train_df.drop(columns=[\"kfold_te\"], inplace=True)\n\n    return train_df, valid_df, test_df\n\ndef count_encode(train_df, valid_df, test_df, cols):\n    feat_name = \"CE_\" + \"_\".join(cols)\n\n    counts = (\n        train_df.groupby(cols, dropna=False)\n        .size()\n        .reset_index(name=feat_name)\n    )\n\n    train_df = train_df.merge(counts, on=cols, how=\"left\")\n    valid_df = valid_df.merge(counts, on=cols, how=\"left\")\n    test_df = test_df.merge(counts, on=cols, how=\"left\")\n\n    for df in [train_df, valid_df, test_df]:\n        df[feat_name] = df[feat_name].fillna(0).astype(\"int32\")\n\n    return train_df, valid_df, test_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T20:27:22.128404Z","iopub.execute_input":"2026-04-18T20:27:22.129175Z","iopub.status.idle":"2026-04-18T20:27:22.14799Z","shell.execute_reply.started":"2026-04-18T20:27:22.129144Z","shell.execute_reply":"2026-04-18T20:27:22.147256Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Cross-Validation Configuration","metadata":{}},{"cell_type":"code","source":"FOLDS = 2\nRANDOM_STATE = 42\n\nprint(f\"FOLDS = {FOLDS}\")\n\nkf = KFold(n_splits=FOLDS, shuffle=True, random_state=RANDOM_STATE)\n\noof_xgb = np.zeros(len(train), dtype=np.float32)\noof_lgb = np.zeros(len(train), dtype=np.float32)\npred_xgb = np.zeros(len(test), dtype=np.float32)\npred_lgb = np.zeros(len(test), dtype=np.float32)\n\nfold_scores_xgb = []\nfold_scores_lgb = []\nfeature_importance_xgb = None\nfeature_importance_lgb = None\nfinal_feature_names = None\n\ntotal_start = time.time()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T20:27:28.818794Z","iopub.execute_input":"2026-04-18T20:27:28.819196Z","iopub.status.idle":"2026-04-18T20:27:28.827225Z","shell.execute_reply.started":"2026-04-18T20:27:28.819167Z","shell.execute_reply":"2026-04-18T20:27:28.826513Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model Training with Feature Engineering","metadata":{}},{"cell_type":"code","source":"for fold_id, (train_idx, valid_idx) in enumerate(kf.split(train)):\n    print(\"\\n\" + \"=\" * 60)\n    print(f\"Fold {fold_id + 1}/{FOLDS}\")\n    print(\"=\" * 60)\n\n    x_train = train.loc[train_idx, FEATURES + [\"y\"]].copy().reset_index(drop=True)\n    x_valid = train.loc[valid_idx, FEATURES].copy().reset_index(drop=True)\n    x_test = test[FEATURES].copy().reset_index(drop=True)\n\n    y_train = train.loc[train_idx, \"y\"].values\n    y_valid = train.loc[valid_idx, \"y\"].values\n\n    fe_start = time.time()\n\n    all_groups = [[f] for f in FEATURES] + COMBOS\n\n    for cols in all_groups:\n        x_train, x_valid, x_test = target_encode(\n            x_train, x_valid, x_test,\n            cols=cols,\n            target=\"y\",\n            kfold=7,\n            smooth=30,\n            agg=\"mean\"\n        )\n\n        x_train, x_valid, x_test = target_encode(\n            x_train, x_valid, x_test,\n            cols=cols,\n            target=\"y\",\n            kfold=7,\n            smooth=0,\n            agg=\"median\"\n        )\n\n        if (len(cols) > 1) or (cols[0] in HIGH_CARDINALITY):\n            x_train, x_valid, x_test = target_encode(\n                x_train, x_valid, x_test,\n                cols=cols, target=\"y\", kfold=7, smooth=0, agg=\"min\"\n            )\n            x_train, x_valid, x_test = target_encode(\n                x_train, x_valid, x_test,\n                cols=cols, target=\"y\", kfold=7, smooth=0, agg=\"max\"\n            )\n            x_train, x_valid, x_test = target_encode(\n                x_train, x_valid, x_test,\n                cols=cols, target=\"y\", kfold=7, smooth=0, agg=\"std\"\n            )\n            x_train, x_valid, x_test = target_encode(\n                x_train, x_valid, x_test,\n                cols=cols, target=\"y\", kfold=7, smooth=0, agg=\"nunique\"\n            )\n            x_train, x_valid, x_test = count_encode(x_train, x_valid, x_test, cols)\n\n    fe_time = time.time() - fe_start\n\n    x_train.drop(columns=[\"y\"], inplace=True)\n\n    final_feature_names = x_train.columns.tolist()\n\n    print(f\"Feature engineering time: {fe_time/60:.2f} min\")\n    print(f\"Number of final features: {len(final_feature_names)}\")\n\n    X_tr = x_train.values.astype(np.float32)\n    X_va = x_valid.values.astype(np.float32)\n    X_te = x_test.values.astype(np.float32)\n\n    xgb_params = dict(\n        max_depth=9,\n        colsample_bytree=0.85,\n        subsample=0.85,\n        n_estimators=3000,\n        learning_rate=0.008,\n        early_stopping_rounds=30,\n        eval_metric=\"rmse\",\n        random_state=RANDOM_STATE,\n        tree_method=\"hist\"\n    )\n\n    if HAS_GPU:\n        xgb_params[\"device\"] = \"cuda\"\n    else:\n        xgb_params[\"device\"] = \"cpu\"\n\n    xgb_model = XGBRegressor(**xgb_params)\n    xgb_model.fit(\n        X_tr, y_train,\n        eval_set=[(X_va, y_valid)],\n        verbose=200\n    )\n\n    pred_valid_xgb = xgb_model.predict(X_va)\n    pred_test_xgb = xgb_model.predict(X_te)\n\n    oof_xgb[valid_idx] = pred_valid_xgb\n    pred_xgb += pred_test_xgb / FOLDS\n\n    fold_rmse_xgb = rmse(y_valid, pred_valid_xgb)\n    fold_scores_xgb.append(fold_rmse_xgb)\n\n    lgb_model = lgb.LGBMRegressor(\n        objective=\"regression\",\n        metric=\"rmse\",\n        verbosity=-1,\n        n_estimators=3000,\n        learning_rate=0.008,\n        num_leaves=200,\n        max_depth=-1,\n        min_child_samples=25,\n        subsample=0.85,\n        subsample_freq=1,\n        colsample_bytree=0.85,\n        reg_alpha=0.05,\n        reg_lambda=0.5,\n        random_state=RANDOM_STATE,\n        n_jobs=-1\n    )\n\n    lgb_model.fit(\n        X_tr, y_train,\n        eval_set=[(X_va, y_valid)],\n        callbacks=[lgb.early_stopping(30, verbose=False)]\n    )\n\n    pred_valid_lgb = lgb_model.predict(X_va)\n    pred_test_lgb = lgb_model.predict(X_te)\n\n    oof_lgb[valid_idx] = pred_valid_lgb\n    pred_lgb += pred_test_lgb / FOLDS\n\n    fold_rmse_lgb = rmse(y_valid, pred_valid_lgb)\n    fold_scores_lgb.append(fold_rmse_lgb)\n\n    print(f\"Fold RMSE -> XGB: {fold_rmse_xgb:.5f} | LGB: {fold_rmse_lgb:.5f}\")\n\n    if feature_importance_xgb is None:\n        feature_importance_xgb = np.array(xgb_model.feature_importances_, dtype=np.float64)\n    else:\n        feature_importance_xgb += np.array(xgb_model.feature_importances_, dtype=np.float64)\n\n    if feature_importance_lgb is None:\n        feature_importance_lgb = np.array(lgb_model.feature_importances_, dtype=np.float64)\n    else:\n        feature_importance_lgb += np.array(lgb_model.feature_importances_, dtype=np.float64)\n\n    del x_train, x_valid, x_test, X_tr, X_va, X_te, xgb_model, lgb_model\n    gc.collect()\n\nprint(\"\\nTraining completed.\")\nprint(f\"Total runtime: {(time.time() - total_start)/60:.2f} min\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T20:27:32.189617Z","iopub.execute_input":"2026-04-18T20:27:32.189917Z","iopub.status.idle":"2026-04-18T21:22:17.55594Z","shell.execute_reply.started":"2026-04-18T20:27:32.189893Z","shell.execute_reply":"2026-04-18T21:22:17.55524Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Blending Predictions and Submission File","metadata":{}},{"cell_type":"code","source":"test_ids = np.asarray(test[\"id\"]).ravel()\npred_xgb_1d = np.asarray(pred_xgb).ravel()\npred_lgb_1d = np.asarray(pred_lgb).ravel()\n\nprint(\"len(test_ids)   =\", len(test_ids))\nprint(\"len(pred_xgb)   =\", len(pred_xgb_1d))\nprint(\"len(pred_lgb)   =\", len(pred_lgb_1d))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T21:22:27.781787Z","iopub.execute_input":"2026-04-18T21:22:27.782612Z","iopub.status.idle":"2026-04-18T21:22:27.800352Z","shell.execute_reply.started":"2026-04-18T21:22:27.78258Z","shell.execute_reply":"2026-04-18T21:22:27.799786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"test shape:\", test.shape)\n\ntry:\n    print(\"X_test shape:\", X_test.shape)\nexcept:\n    print(\"X_test not defined\")\n\ntry:\n    print(\"X_test_xgb shape:\", X_test_xgb.shape)\nexcept:\n    print(\"X_test_xgb not defined\")\n\ntry:\n    print(\"X_test_lgb shape:\", X_test_lgb.shape)\nexcept:\n    print(\"X_test_lgb not defined\")\n\nprint(\"pred_xgb shape:\", np.asarray(pred_xgb).shape)\nprint(\"pred_lgb shape:\", np.asarray(pred_lgb).shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T21:22:31.016421Z","iopub.execute_input":"2026-04-18T21:22:31.017204Z","iopub.status.idle":"2026-04-18T21:22:31.022362Z","shell.execute_reply.started":"2026-04-18T21:22:31.017176Z","shell.execute_reply":"2026-04-18T21:22:31.021588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test.columns.tolist())\nprint(\"Duplicate columns:\", test.columns[test.columns.duplicated()].tolist())\n\nid_block = test.loc[:, test.columns == \"id\"]\nprint(\"id_block shape:\", id_block.shape)\ndisplay(id_block.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T21:22:34.615225Z","iopub.execute_input":"2026-04-18T21:22:34.615931Z","iopub.status.idle":"2026-04-18T21:22:34.629329Z","shell.execute_reply.started":"2026-04-18T21:22:34.615899Z","shell.execute_reply":"2026-04-18T21:22:34.628557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_score = 1e18\nbest_weight = 0.5\n\ny_true = np.asarray(train[\"y\"]).ravel()\noof_xgb_1d = np.asarray(oof_xgb).ravel()\noof_lgb_1d = np.asarray(oof_lgb).ravel()\n\nfor w in np.arange(0.0, 1.01, 0.01):\n    blend_oof = w * oof_xgb_1d + (1 - w) * oof_lgb_1d\n    score = rmse(y_true, blend_oof)\n    if score < best_score:\n        best_score = score\n        best_weight = w\n\nprint(f\"OOF RMSE XGB   : {rmse(y_true, oof_xgb_1d):.5f}\")\nprint(f\"OOF RMSE LGB   : {rmse(y_true, oof_lgb_1d):.5f}\")\nprint(f\"Best blend w   : {best_weight:.2f}\")\nprint(f\"Best blend RMSE: {best_score:.5f}\")\n\npred_xgb_1d = np.asarray(pred_xgb).ravel()\npred_lgb_1d = np.asarray(pred_lgb).ravel()\n\nfinal_pred_log = best_weight * pred_xgb_1d + (1 - best_weight) * pred_lgb_1d\nfinal_pred = np.expm1(final_pred_log).ravel()\n\ntest_ids = test.loc[:, test.columns == \"id\"].iloc[:, 0].to_numpy().ravel()\n\nsubmission = pd.DataFrame({\n    \"id\": test_ids,\n    \"Premium Amount\": final_pred\n})\n\nsubmission.to_csv(\"/kaggle/working/submission.csv\", index=False)\nprint(\"\\nsubmission.csv saved to /kaggle/working/submission.csv\")\ndisplay(submission.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T21:22:37.586584Z","iopub.execute_input":"2026-04-18T21:22:37.587353Z","iopub.status.idle":"2026-04-18T21:22:39.918876Z","shell.execute_reply.started":"2026-04-18T21:22:37.58732Z","shell.execute_reply":"2026-04-18T21:22:39.918202Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Validation Performance Visualization","metadata":{}},{"cell_type":"code","source":"results_df = pd.DataFrame({\n    \"Fold\": np.arange(1, FOLDS + 1),\n    \"XGB_RMSE\": fold_scores_xgb,\n    \"LGB_RMSE\": fold_scores_lgb\n})\nresults_df[\"Best_Model_Per_Fold\"] = np.where(\n    results_df[\"XGB_RMSE\"] < results_df[\"LGB_RMSE\"], \"XGB\", \"LGB\"\n)\n\ndisplay(results_df)\n\nplt.figure(figsize=(10, 5))\nplt.plot(results_df[\"Fold\"], results_df[\"XGB_RMSE\"], marker=\"o\", label=\"XGB\")\nplt.plot(results_df[\"Fold\"], results_df[\"LGB_RMSE\"], marker=\"o\", label=\"LGB\")\nplt.xlabel(\"Fold\")\nplt.ylabel(\"RMSE\")\nplt.title(\"Fold-wise Validation RMSE\")\nplt.legend()\nplt.grid(True)\nplt.show()\n\nplt.figure(figsize=(10, 5))\nresults_df[[\"XGB_RMSE\", \"LGB_RMSE\"]].mean().plot(kind=\"bar\")\nplt.ylabel(\"Average RMSE\")\nplt.title(\"Average CV RMSE by Model\")\nplt.xticks(rotation=0)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T21:22:43.934356Z","iopub.execute_input":"2026-04-18T21:22:43.934941Z","iopub.status.idle":"2026-04-18T21:22:44.203013Z","shell.execute_reply.started":"2026-04-18T21:22:43.934911Z","shell.execute_reply":"2026-04-18T21:22:44.202174Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Importance Analysis","metadata":{}},{"cell_type":"code","source":"feature_importance_xgb = feature_importance_xgb / FOLDS\nfeature_importance_lgb = feature_importance_lgb / FOLDS\n\nfi_xgb = pd.DataFrame({\n    \"feature\": final_feature_names,\n    \"importance\": feature_importance_xgb\n}).sort_values(\"importance\", ascending=False).head(25)\n\nfi_lgb = pd.DataFrame({\n    \"feature\": final_feature_names,\n    \"importance\": feature_importance_lgb\n}).sort_values(\"importance\", ascending=False).head(25)\n\nplt.figure(figsize=(12, 8))\nsns.barplot(data=fi_xgb, x=\"importance\", y=\"feature\")\nplt.title(\"Top 25 XGBoost Feature Importances\")\nplt.tight_layout()\nplt.show()\n\nplt.figure(figsize=(12, 8))\nsns.barplot(data=fi_lgb, x=\"importance\", y=\"feature\")\nplt.title(\"Top 25 LightGBM Feature Importances\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T21:22:52.260269Z","iopub.execute_input":"2026-04-18T21:22:52.260713Z","iopub.status.idle":"2026-04-18T21:22:53.025014Z","shell.execute_reply.started":"2026-04-18T21:22:52.26068Z","shell.execute_reply":"2026-04-18T21:22:53.024233Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Prediction Diagnostics","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nsns.histplot(train[\"Premium Amount\"], bins=50, kde=True)\nplt.title(\"Train Target Distribution\")\nplt.show()\n\nplt.figure(figsize=(10, 5))\nsns.histplot(final_pred, bins=50, kde=True)\nplt.title(\"Test Prediction Distribution\")\nplt.show()\n\noof_blend = best_weight * oof_xgb + (1 - best_weight) * oof_lgb\nplt.figure(figsize=(10, 5))\nplt.scatter(train[\"y\"].values, oof_blend, alpha=0.25)\nplt.xlabel(\"True log target\")\nplt.ylabel(\"OOF blended prediction\")\nplt.title(\"OOF Prediction vs True Target\")\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T21:23:01.592131Z","iopub.execute_input":"2026-04-18T21:23:01.592575Z","iopub.status.idle":"2026-04-18T21:23:34.989652Z","shell.execute_reply.started":"2026-04-18T21:23:01.592545Z","shell.execute_reply":"2026-04-18T21:23:34.988929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# ---------------------------\n# Style settings\n# ---------------------------\nsns.set_theme(style=\"whitegrid\", context=\"talk\")\nplt.rcParams[\"figure.figsize\"] = (12, 6)\nplt.rcParams[\"axes.titlesize\"] = 18\nplt.rcParams[\"axes.labelsize\"] = 13\nplt.rcParams[\"xtick.labelsize\"] = 11\nplt.rcParams[\"ytick.labelsize\"] = 11\nplt.rcParams[\"legend.fontsize\"] = 11\nplt.rcParams[\"figure.dpi\"] = 120\n\n# ---------------------------\n# Prepare variables safely\n# ---------------------------\nif \"results_df\" not in globals():\n    results_df = pd.DataFrame({\n        \"Fold\": np.arange(1, FOLDS + 1),\n        \"XGB_RMSE\": np.asarray(fold_scores_xgb).ravel(),\n        \"LGB_RMSE\": np.asarray(fold_scores_lgb).ravel()\n    })\n\nif \"oof_blend\" not in globals():\n    oof_blend = best_weight * np.asarray(oof_xgb).ravel() + (1 - best_weight) * np.asarray(oof_lgb).ravel()\n\ntrain_target = np.asarray(train[\"y\"]).ravel() if \"y\" in train.columns else np.asarray(train[\"Premium Amount\"]).ravel()\nfinal_pred_1d = np.asarray(final_pred).ravel()\n\n# ---------------------------\n# 1) Fold-wise RMSE comparison\n# ---------------------------\nplt.figure(figsize=(12, 6))\nplt.plot(results_df[\"Fold\"], results_df[\"XGB_RMSE\"], marker=\"o\", linewidth=2.5, label=\"XGBoost\")\nplt.plot(results_df[\"Fold\"], results_df[\"LGB_RMSE\"], marker=\"s\", linewidth=2.5, label=\"LightGBM\")\n\nfor x, y in zip(results_df[\"Fold\"], results_df[\"XGB_RMSE\"]):\n    plt.text(x, y, f\"{y:.4f}\", ha=\"center\", va=\"bottom\", fontsize=10)\nfor x, y in zip(results_df[\"Fold\"], results_df[\"LGB_RMSE\"]):\n    plt.text(x, y, f\"{y:.4f}\", ha=\"center\", va=\"top\", fontsize=10)\n\nplt.title(\"Fold-wise RMSE Comparison\", pad=14, weight=\"bold\")\nplt.xlabel(\"Fold\")\nplt.ylabel(\"RMSE\")\nplt.legend(frameon=True)\nplt.grid(True, linestyle=\"--\", alpha=0.35)\nplt.tight_layout()\nplt.show()\n\n# ---------------------------\n# 2) Average RMSE bar chart\n# ---------------------------\navg_scores = pd.DataFrame({\n    \"Model\": [\"XGBoost\", \"LightGBM\"],\n    \"Average_RMSE\": [\n        results_df[\"XGB_RMSE\"].mean(),\n        results_df[\"LGB_RMSE\"].mean()\n    ]\n}).sort_values(\"Average_RMSE\")\n\nplt.figure(figsize=(10, 6))\nax = sns.barplot(data=avg_scores, x=\"Model\", y=\"Average_RMSE\")\nfor p in ax.patches:\n    ax.annotate(f\"{p.get_height():.5f}\",\n                (p.get_x() + p.get_width() / 2, p.get_height()),\n                ha=\"center\", va=\"bottom\", fontsize=11, xytext=(0, 6),\n                textcoords=\"offset points\")\nplt.title(\"Average CV RMSE by Model\", pad=14, weight=\"bold\")\nplt.xlabel(\"\")\nplt.ylabel(\"Average RMSE\")\nplt.grid(True, axis=\"y\", linestyle=\"--\", alpha=0.30)\nplt.tight_layout()\nplt.show()\n\n# ---------------------------\n# 3) Train target vs test prediction distribution\n# ---------------------------\nplt.figure(figsize=(12, 6))\nsns.histplot(train_target, bins=60, kde=True, stat=\"density\", label=\"Train Target\", alpha=0.45)\nsns.histplot(final_pred_1d, bins=60, kde=True, stat=\"density\", label=\"Test Prediction\", alpha=0.45)\nplt.title(\"Train Target vs Test Prediction Distribution\", pad=14, weight=\"bold\")\nplt.xlabel(\"Target / Prediction Value\")\nplt.ylabel(\"Density\")\nplt.legend()\nplt.grid(True, linestyle=\"--\", alpha=0.25)\nplt.tight_layout()\nplt.show()\n\n# ---------------------------\n# 4) OOF prediction vs true target\n# ---------------------------\nplt.figure(figsize=(10, 8))\nplt.scatter(train_target, oof_blend, alpha=0.22, s=18)\n\nmin_v = min(train_target.min(), oof_blend.min())\nmax_v = max(train_target.max(), oof_blend.max())\nplt.plot([min_v, max_v], [min_v, max_v], linestyle=\"--\", linewidth=2)\n\ncorr = np.corrcoef(train_target, oof_blend)[0, 1]\n\nplt.title(f\"OOF Prediction vs True Target  |  Corr = {corr:.4f}\", pad=14, weight=\"bold\")\nplt.xlabel(\"True Target\")\nplt.ylabel(\"OOF Blended Prediction\")\nplt.grid(True, linestyle=\"--\", alpha=0.30)\nplt.tight_layout()\nplt.show()\n\n# ---------------------------\n# 5) Feature importance plots, if available\n# ---------------------------\nif \"fi_xgb\" in globals() and isinstance(fi_xgb, pd.DataFrame):\n    top_xgb = fi_xgb.sort_values(\"importance\", ascending=False).head(20)\n    plt.figure(figsize=(12, 8))\n    sns.barplot(data=top_xgb, x=\"importance\", y=\"feature\")\n    plt.title(\"Top 20 XGBoost Feature Importances\", pad=14, weight=\"bold\")\n    plt.xlabel(\"Importance\")\n    plt.ylabel(\"\")\n    plt.grid(True, axis=\"x\", linestyle=\"--\", alpha=0.25)\n    plt.tight_layout()\n    plt.show()\n\nif \"fi_lgb\" in globals() and isinstance(fi_lgb, pd.DataFrame):\n    top_lgb = fi_lgb.sort_values(\"importance\", ascending=False).head(20)\n    plt.figure(figsize=(12, 8))\n    sns.barplot(data=top_lgb, x=\"importance\", y=\"feature\")\n    plt.title(\"Top 20 LightGBM Feature Importances\", pad=14, weight=\"bold\")\n    plt.xlabel(\"Importance\")\n    plt.ylabel(\"\")\n    plt.grid(True, axis=\"x\", linestyle=\"--\", alpha=0.25)\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T21:23:43.593038Z","iopub.execute_input":"2026-04-18T21:23:43.593426Z","iopub.status.idle":"2026-04-18T21:23:55.627054Z","shell.execute_reply.started":"2026-04-18T21:23:43.593398Z","shell.execute_reply":"2026-04-18T21:23:55.626243Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#Save the files","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nsave_dir = \"/kaggle/working/beautiful_plots\"\nos.makedirs(save_dir, exist_ok=True)\n\nsns.set_theme(style=\"whitegrid\", context=\"talk\")\nplt.rcParams[\"figure.dpi\"] = 140\n\nif \"results_df\" not in globals():\n    results_df = pd.DataFrame({\n        \"Fold\": np.arange(1, FOLDS + 1),\n        \"XGB_RMSE\": np.asarray(fold_scores_xgb).ravel(),\n        \"LGB_RMSE\": np.asarray(fold_scores_lgb).ravel()\n    })\n\nif \"oof_blend\" not in globals():\n    oof_blend = best_weight * np.asarray(oof_xgb).ravel() + (1 - best_weight) * np.asarray(oof_lgb).ravel()\n\ntrain_target = np.asarray(train[\"y\"]).ravel() if \"y\" in train.columns else np.asarray(train[\"Premium Amount\"]).ravel()\nfinal_pred_1d = np.asarray(final_pred).ravel()\n\ndef save_current_fig(filename):\n    plt.tight_layout()\n    plt.savefig(os.path.join(save_dir, filename), dpi=300, bbox_inches=\"tight\")\n    plt.show()\n    plt.close()\n\n# 1) Fold-wise RMSE\nplt.figure(figsize=(12, 6))\nplt.plot(results_df[\"Fold\"], results_df[\"XGB_RMSE\"], marker=\"o\", linewidth=2.5, label=\"XGBoost\")\nplt.plot(results_df[\"Fold\"], results_df[\"LGB_RMSE\"], marker=\"s\", linewidth=2.5, label=\"LightGBM\")\nfor x, y in zip(results_df[\"Fold\"], results_df[\"XGB_RMSE\"]):\n    plt.text(x, y, f\"{y:.4f}\", ha=\"center\", va=\"bottom\", fontsize=10)\nfor x, y in zip(results_df[\"Fold\"], results_df[\"LGB_RMSE\"]):\n    plt.text(x, y, f\"{y:.4f}\", ha=\"center\", va=\"top\", fontsize=10)\nplt.title(\"Fold-wise RMSE Comparison\", pad=14, weight=\"bold\")\nplt.xlabel(\"Fold\")\nplt.ylabel(\"RMSE\")\nplt.legend(frameon=True)\nplt.grid(True, linestyle=\"--\", alpha=0.35)\nsave_current_fig(\"01_foldwise_rmse_comparison.png\")\n\n# 2) Average RMSE\navg_scores = pd.DataFrame({\n    \"Model\": [\"XGBoost\", \"LightGBM\"],\n    \"Average_RMSE\": [\n        results_df[\"XGB_RMSE\"].mean(),\n        results_df[\"LGB_RMSE\"].mean()\n    ]\n}).sort_values(\"Average_RMSE\")\n\nplt.figure(figsize=(10, 6))\nax = sns.barplot(data=avg_scores, x=\"Model\", y=\"Average_RMSE\")\nfor p in ax.patches:\n    ax.annotate(f\"{p.get_height():.5f}\",\n                (p.get_x() + p.get_width() / 2, p.get_height()),\n                ha=\"center\", va=\"bottom\", fontsize=11, xytext=(0, 6),\n                textcoords=\"offset points\")\nplt.title(\"Average CV RMSE by Model\", pad=14, weight=\"bold\")\nplt.xlabel(\"\")\nplt.ylabel(\"Average RMSE\")\nplt.grid(True, axis=\"y\", linestyle=\"--\", alpha=0.30)\nsave_current_fig(\"02_average_cv_rmse.png\")\n\n# 3) Distribution comparison\nplt.figure(figsize=(12, 6))\nsns.histplot(train_target, bins=60, kde=True, stat=\"density\", label=\"Train Target\", alpha=0.45)\nsns.histplot(final_pred_1d, bins=60, kde=True, stat=\"density\", label=\"Test Prediction\", alpha=0.45)\nplt.title(\"Train Target vs Test Prediction Distribution\", pad=14, weight=\"bold\")\nplt.xlabel(\"Target / Prediction Value\")\nplt.ylabel(\"Density\")\nplt.legend()\nplt.grid(True, linestyle=\"--\", alpha=0.25)\nsave_current_fig(\"03_target_vs_prediction_distribution.png\")\n\n# 4) OOF scatter\nplt.figure(figsize=(10, 8))\nplt.scatter(train_target, oof_blend, alpha=0.22, s=18)\nmin_v = min(train_target.min(), oof_blend.min())\nmax_v = max(train_target.max(), oof_blend.max())\nplt.plot([min_v, max_v], [min_v, max_v], linestyle=\"--\", linewidth=2)\ncorr = np.corrcoef(train_target, oof_blend)[0, 1]\nplt.title(f\"OOF Prediction vs True Target  |  Corr = {corr:.4f}\", pad=14, weight=\"bold\")\nplt.xlabel(\"True Target\")\nplt.ylabel(\"OOF Blended Prediction\")\nplt.grid(True, linestyle=\"--\", alpha=0.30)\nsave_current_fig(\"04_oof_vs_true_target.png\")\n\n# 5) XGB feature importance\nif \"fi_xgb\" in globals() and isinstance(fi_xgb, pd.DataFrame):\n    top_xgb = fi_xgb.sort_values(\"importance\", ascending=False).head(20)\n    plt.figure(figsize=(12, 8))\n    sns.barplot(data=top_xgb, x=\"importance\", y=\"feature\")\n    plt.title(\"Top 20 XGBoost Feature Importances\", pad=14, weight=\"bold\")\n    plt.xlabel(\"Importance\")\n    plt.ylabel(\"\")\n    plt.grid(True, axis=\"x\", linestyle=\"--\", alpha=0.25)\n    save_current_fig(\"05_top20_xgb_feature_importance.png\")\n\n# 6) LGB feature importance\nif \"fi_lgb\" in globals() and isinstance(fi_lgb, pd.DataFrame):\n    top_lgb = fi_lgb.sort_values(\"importance\", ascending=False).head(20)\n    plt.figure(figsize=(12, 8))\n    sns.barplot(data=top_lgb, x=\"importance\", y=\"feature\")\n    plt.title(\"Top 20 LightGBM Feature Importances\", pad=14, weight=\"bold\")\n    plt.xlabel(\"Importance\")\n    plt.ylabel(\"\")\n    plt.grid(True, axis=\"x\", linestyle=\"--\", alpha=0.25)\n    save_current_fig(\"06_top20_lgb_feature_importance.png\")\n\n# Save results table too\nresults_df.to_csv(os.path.join(save_dir, \"cv_results_table.csv\"), index=False)\n\nprint(f\"All plots saved in: {save_dir}\")\nprint(\"Files:\", sorted(os.listdir(save_dir)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T21:24:19.328027Z","iopub.execute_input":"2026-04-18T21:24:19.328486Z","iopub.status.idle":"2026-04-18T21:24:49.192318Z","shell.execute_reply.started":"2026-04-18T21:24:19.328454Z","shell.execute_reply":"2026-04-18T21:24:49.191662Z"}},"outputs":[],"execution_count":null}]}