{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":9178166,"sourceType":"datasetVersion","datasetId":5547076}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing libraries and Files","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom hillclimbers import climb_hill, partial\nimport warnings\n\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:25:17.957903Z","iopub.execute_input":"2024-12-03T10:25:17.95864Z","iopub.status.idle":"2024-12-03T10:25:19.40424Z","shell.execute_reply.started":"2024-12-03T10:25:17.958602Z","shell.execute_reply":"2024-12-03T10:25:19.403409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest_df  = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\nsub_df   = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")\norg   = pd.read_csv(\"/kaggle/input/insurance-premium-prediction/Insurance Premium Prediction Dataset.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:25:23.657115Z","iopub.execute_input":"2024-12-03T10:25:23.657643Z","iopub.status.idle":"2024-12-03T10:25:33.336842Z","shell.execute_reply.started":"2024-12-03T10:25:23.657607Z","shell.execute_reply":"2024-12-03T10:25:33.33586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:25:37.546047Z","iopub.execute_input":"2024-12-03T10:25:37.546482Z","iopub.status.idle":"2024-12-03T10:25:37.553327Z","shell.execute_reply.started":"2024-12-03T10:25:37.546442Z","shell.execute_reply":"2024-12-03T10:25:37.552379Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Imputing columns Individually","metadata":{}},{"cell_type":"code","source":"train_df.drop(columns=['id'],inplace=True)\ntest_df.drop(columns=['id'],inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:25:40.909489Z","iopub.execute_input":"2024-12-03T10:25:40.909921Z","iopub.status.idle":"2024-12-03T10:25:41.247323Z","shell.execute_reply.started":"2024-12-03T10:25:40.909881Z","shell.execute_reply":"2024-12-03T10:25:41.246357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print dataset shapes\nprint(train_df.shape)\nprint(test_df.shape)\n\n# Create a figure with two subplots in a single row\nfig, axes = plt.subplots(1, 2, figsize=(12, 5), sharey=True)  # 1 row, 2 columns\n\n# Plot null counts for train_df in the first subplot\ntrain_df.isnull().sum().plot(kind=\"bar\", ax=axes[0], color='skyblue')\naxes[0].set_title(\"Train Data - Null Counts\")\naxes[0].set_ylabel(\"Count\")\n\n# Plot null counts for test_df in the second subplot\ntest_df.isnull().sum().plot(kind=\"bar\", ax=axes[1], color='orange')\naxes[1].set_title(\"Test Data - Null Counts\")\n\n# Display the plots\nplt.tight_layout()  # Adjust spacing between plots\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:25:44.445727Z","iopub.execute_input":"2024-12-03T10:25:44.446074Z","iopub.status.idle":"2024-12-03T10:25:46.016762Z","shell.execute_reply.started":"2024-12-03T10:25:44.446042Z","shell.execute_reply":"2024-12-03T10:25:46.015909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols = test_df.select_dtypes(include=['int','float']).columns.to_list()\ncat_cols = test_df.select_dtypes(include=['O']).columns.to_list()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:25:51.664988Z","iopub.execute_input":"2024-12-03T10:25:51.665585Z","iopub.status.idle":"2024-12-03T10:25:51.78477Z","shell.execute_reply.started":"2024-12-03T10:25:51.66555Z","shell.execute_reply":"2024-12-03T10:25:51.783758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in num_cols:\n    train_df[col].fillna(train_df[col].mean,inplace=True)\n    test_df[col].fillna(test_df[col].mean, inplace=True)\n\nfor col in cat_cols:\n    c_train = train_df[col].value_counts().head(1).index.values[0]\n    c_test  = test_df[col].value_counts().head(1).index.values[0]\n    train_df[col].fillna(c_train,inplace=True)\n    test_df[col].fillna(c_test,inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:25:53.959501Z","iopub.execute_input":"2024-12-03T10:25:53.960174Z","iopub.status.idle":"2024-12-03T10:25:57.440739Z","shell.execute_reply.started":"2024-12-03T10:25:53.960136Z","shell.execute_reply":"2024-12-03T10:25:57.439974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print dataset shapes\nprint(train_df.shape)\nprint(test_df.shape)\n\n# Create a figure with two subplots in a single row\nfig, axes = plt.subplots(1, 2, figsize=(12, 5), sharey=True)  # 1 row, 2 columns\n\n# Plot null counts for train_df in the first subplot\ntrain_df.isnull().sum().plot(kind=\"bar\", ax=axes[0], color='skyblue')\naxes[0].set_title(\"Train Data - Null Counts\")\naxes[0].set_ylabel(\"Count\")\n\n# Plot null counts for test_df in the second subplot\ntest_df.isnull().sum().plot(kind=\"bar\", ax=axes[1], color='orange')\naxes[1].set_title(\"Test Data - Null Counts\")\n\n# Display the plots\nplt.tight_layout()  # Adjust spacing between plots\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:26:00.475398Z","iopub.execute_input":"2024-12-03T10:26:00.475752Z","iopub.status.idle":"2024-12-03T10:26:02.4013Z","shell.execute_reply.started":"2024-12-03T10:26:00.475721Z","shell.execute_reply":"2024-12-03T10:26:02.400496Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Successfully imputed the values**","metadata":{}},{"cell_type":"markdown","source":"# Encoding Categorical Columns","metadata":{}},{"cell_type":"code","source":"cat_cols = ['Gender',\n 'Marital Status',\n 'Education Level',\n 'Occupation',\n 'Location',\n 'Policy Type',\n 'Customer Feedback',\n 'Smoking Status',\n 'Exercise Frequency',\n 'Property Type']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:26:08.799471Z","iopub.execute_input":"2024-12-03T10:26:08.800306Z","iopub.status.idle":"2024-12-03T10:26:08.804235Z","shell.execute_reply.started":"2024-12-03T10:26:08.800239Z","shell.execute_reply":"2024-12-03T10:26:08.803374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert 'Policy Start Date' column to datetime type\ntrain_df['Policy Start Date'] = pd.to_datetime(train_df['Policy Start Date'], errors='coerce')\ntest_df['Policy Start Date'] = pd.to_datetime(test_df['Policy Start Date'], errors='coerce')\n\n# Extract features from the 'Policy Start Date' column\nfor col in ['Policy Start Date']:\n    # Extracting Year, Month, Day\n    train_df[f'{col}_year'] = train_df[col].dt.year\n    train_df[f'{col}_month'] = train_df[col].dt.month\n    train_df[f'{col}_day'] = train_df[col].dt.day\n    train_df[f'{col}_month_name'] = train_df[col].dt.month_name()\n    train_df[f'{col}_day_of_week'] = train_df[col].dt.day_name()\n    train_df[f'{col}_week'] = train_df[col].dt.isocalendar().week\n\n    test_df[f'{col}_year'] = test_df[col].dt.year\n    test_df[f'{col}_month'] = test_df[col].dt.month\n    test_df[f'{col}_day'] = test_df[col].dt.day\n    test_df[f'{col}_month_name'] = test_df[col].dt.month_name()\n    test_df[f'{col}_day_of_week'] = test_df[col].dt.day_name()\n    test_df[f'{col}_week'] = test_df[col].dt.isocalendar().week\n\n    # Add sinusoidal and cosine transformations for Year, Month, and Day\n    train_df[f'{col}_year_sin'] = np.sin(2 * np.pi * train_df[f'{col}_year'])\n    train_df[f'{col}_year_cos'] = np.cos(2 * np.pi * train_df[f'{col}_year'])\n    train_df[f'{col}_month_sin'] = np.sin(2 * np.pi * train_df[f'{col}_month'] / 12)\n    train_df[f'{col}_month_cos'] = np.cos(2 * np.pi * train_df[f'{col}_month'] / 12)\n    train_df[f'{col}_day_sin'] = np.sin(2 * np.pi * train_df[f'{col}_day'] / 31)\n    train_df[f'{col}_day_cos'] = np.cos(2 * np.pi * train_df[f'{col}_day'] / 31)\n\n    test_df[f'{col}_year_sin'] = np.sin(2 * np.pi * test_df[f'{col}_year'])\n    test_df[f'{col}_year_cos'] = np.cos(2 * np.pi * test_df[f'{col}_year'])\n    test_df[f'{col}_month_sin'] = np.sin(2 * np.pi * test_df[f'{col}_month'] / 12)\n    test_df[f'{col}_month_cos'] = np.cos(2 * np.pi * test_df[f'{col}_month'] / 12)\n    test_df[f'{col}_day_sin'] = np.sin(2 * np.pi * test_df[f'{col}_day'] / 31)\n    test_df[f'{col}_day_cos'] = np.cos(2 * np.pi * test_df[f'{col}_day'] / 31)\n\n    # Add Group feature\n    train_df['Group'] = (train_df[f'{col}_year'] - 2020) * 48 + train_df[f'{col}_month'] * 4 + train_df[f'{col}_day'] // 7\n    test_df['Group'] = (test_df[f'{col}_year'] - 2020) * 48 + test_df[f'{col}_month'] * 4 + test_df[f'{col}_day'] // 7\n\n# Now drop the original 'Policy Start Date' column\ntrain_df = train_df.drop(columns=['Policy Start Date'])\ntest_df = test_df.drop(columns=['Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:26:19.181036Z","iopub.execute_input":"2024-12-03T10:26:19.18203Z","iopub.status.idle":"2024-12-03T10:26:22.006934Z","shell.execute_reply.started":"2024-12-03T10:26:19.181982Z","shell.execute_reply":"2024-12-03T10:26:22.006203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:02:57.400114Z","iopub.execute_input":"2024-12-03T10:02:57.400899Z","iopub.status.idle":"2024-12-03T10:02:57.425675Z","shell.execute_reply.started":"2024-12-03T10:02:57.400867Z","shell.execute_reply":"2024-12-03T10:02:57.424918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import KFold\nfrom category_encoders import TargetEncoder\n\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\ntarget_encoder = TargetEncoder()\n\nfor train_index, val_index in kf.split(train_df, train_df['Premium Amount']):\n    train_fold = train_df.iloc[train_index]\n    val_fold = train_df.iloc[val_index]\n    train_fold_encoded = target_encoder.fit_transform(train_fold[cat_cols], train_fold['Premium Amount'])\n    val_fold_encoded = target_encoder.transform(val_fold[cat_cols])\n    train_df.loc[val_index, cat_cols] = val_fold_encoded\n\ntrain_df[cat_cols] = train_df[cat_cols].apply(lambda col: pd.to_numeric(col, errors='coerce'))\ntest_encoded = target_encoder.transform(test_df[cat_cols])\ntest_df[cat_cols] = test_encoded\n\ntest_df[cat_cols] = test_df[cat_cols].apply(lambda col: pd.to_numeric(col, errors='coerce'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:26:30.881965Z","iopub.execute_input":"2024-12-03T10:26:30.882806Z","iopub.status.idle":"2024-12-03T10:27:18.960711Z","shell.execute_reply.started":"2024-12-03T10:26:30.88277Z","shell.execute_reply":"2024-12-03T10:27:18.959693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:04:01.981946Z","iopub.execute_input":"2024-12-03T10:04:01.982988Z","iopub.status.idle":"2024-12-03T10:04:02.005792Z","shell.execute_reply.started":"2024-12-03T10:04:01.982955Z","shell.execute_reply":"2024-12-03T10:04:02.004856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_cols = test_df.select_dtypes(include=['O']).columns.to_list()\ncat_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:27:26.131868Z","iopub.execute_input":"2024-12-03T10:27:26.132727Z","iopub.status.idle":"2024-12-03T10:27:26.593394Z","shell.execute_reply.started":"2024-12-03T10:27:26.13269Z","shell.execute_reply":"2024-12-03T10:27:26.592531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in cat_cols:\n    train_df[col] = pd.to_numeric(train_df[col], errors='coerce')\n    test_df[col]  = pd.to_numeric(test_df[col], errors='coerce')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:27:29.595974Z","iopub.execute_input":"2024-12-03T10:27:29.596968Z","iopub.status.idle":"2024-12-03T10:27:34.132097Z","shell.execute_reply.started":"2024-12-03T10:27:29.59691Z","shell.execute_reply":"2024-12-03T10:27:34.131409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:04:28.672718Z","iopub.execute_input":"2024-12-03T10:04:28.673084Z","iopub.status.idle":"2024-12-03T10:04:28.726469Z","shell.execute_reply.started":"2024-12-03T10:04:28.673049Z","shell.execute_reply":"2024-12-03T10:04:28.72551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:04:31.618073Z","iopub.execute_input":"2024-12-03T10:04:31.618425Z","iopub.status.idle":"2024-12-03T10:04:31.685633Z","shell.execute_reply.started":"2024-12-03T10:04:31.618397Z","shell.execute_reply":"2024-12-03T10:04:31.684837Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Training","metadata":{}},{"cell_type":"markdown","source":"**Loading Dependencies**","metadata":{}},{"cell_type":"code","source":"from catboost import CatBoostRegressor, Pool\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error\nimport xgboost as xgb\nfrom sklearn.base import clone","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:27:39.98519Z","iopub.execute_input":"2024-12-03T10:27:39.985564Z","iopub.status.idle":"2024-12-03T10:27:40.318094Z","shell.execute_reply.started":"2024-12-03T10:27:39.985532Z","shell.execute_reply":"2024-12-03T10:27:40.317208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_valid, Y_train, Y_valid = train_test_split(train_df.drop(columns=['Premium Amount']), train_df['Premium Amount'], test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T09:41:58.700588Z","iopub.execute_input":"2024-12-03T09:41:58.70125Z","iopub.status.idle":"2024-12-03T09:41:59.292849Z","shell.execute_reply.started":"2024-12-03T09:41:58.701219Z","shell.execute_reply":"2024-12-03T09:41:59.291922Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Catboost Model","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_log_error\nfrom catboost import CatBoostRegressor, Pool\nfrom sklearn.model_selection import KFold\nfrom sklearn.base import clone\nimport numpy as np\nimport pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:27:44.546896Z","iopub.execute_input":"2024-12-03T10:27:44.547759Z","iopub.status.idle":"2024-12-03T10:27:44.551895Z","shell.execute_reply.started":"2024-12-03T10:27:44.547723Z","shell.execute_reply":"2024-12-03T10:27:44.55113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# RMSLE calculation function\ndef rmsle(y_true, y_pred):\n    \"\"\"Compute Root Mean Squared Log Error (RMSLE).\"\"\"\n    # Add a small value to avoid log(0)\n    y_true = np.maximum(y_true, 1e-15)\n    y_pred = np.maximum(y_pred, 1e-15)\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:27:47.27276Z","iopub.execute_input":"2024-12-03T10:27:47.273101Z","iopub.status.idle":"2024-12-03T10:27:47.278039Z","shell.execute_reply.started":"2024-12-03T10:27:47.27307Z","shell.execute_reply":"2024-12-03T10:27:47.277023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define CatBoost Regressor\ncatboost_model = CatBoostRegressor(\n    iterations=3000,\n    learning_rate=0.05,\n    depth=6,\n    eval_metric=\"RMSE\",\n    random_seed=42,\n    verbose=200,\n    task_type='GPU',\n    l2_leaf_reg=0.7,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:27:50.113102Z","iopub.execute_input":"2024-12-03T10:27:50.113463Z","iopub.status.idle":"2024-12-03T10:27:50.120392Z","shell.execute_reply.started":"2024-12-03T10:27:50.113431Z","shell.execute_reply":"2024-12-03T10:27:50.119511Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_model_oof_predictions(X, y, model, n_splits=5, random_state=2):\n    \"\"\"\n    Perform KFold cross-validation and generate out-of-fold (OOF) predictions.\n\n    Parameters:\n    - X: DataFrame or array-like, features\n    - y: Series or array-like, target\n    - model: CatBoost model instance\n    - n_splits: int, number of KFold splits\n    - random_state: int, random seed for reproducibility\n\n    Returns:\n    - results: dict containing OOF predictions, fold scores, and trained models\n    \"\"\"\n    # Initialize KFold\n    kf = KFold(n_splits=n_splits, shuffle=True, random_state=random_state)\n\n    # Convert to NumPy arrays for compatibility with CatBoost Pool\n    X_array = X.values if hasattr(X, 'values') else np.array(X)\n    y_array = y.values if hasattr(y, 'values') else np.array(y)\n\n    # Initialize results dictionary\n    results = {\n        'oof_predictions': np.zeros(len(X)),\n        'fold_scores': [],\n        'mean_rmse': 0.0,\n        'oof_models': []\n    }\n\n    # Loop through folds\n    for fold, (train_idx, val_idx) in enumerate(kf.split(X_array), 1):\n        X_train, X_val = X_array[train_idx], X_array[val_idx]\n        y_train, y_val = y_array[train_idx], y_array[val_idx]\n\n        print(f\"\\nTraining - Fold {fold}\")\n\n        # Clone the model for each fold\n        fold_model = clone(model)\n\n        # Create CatBoost Pool objects\n        train_pool = Pool(X_train, y_train)\n        val_pool = Pool(X_val, y_val)\n\n        # Train the model\n        fold_model.fit(\n            X=train_pool,\n            eval_set=val_pool,\n            verbose=500,\n            early_stopping_rounds=300\n        )\n\n        # Get predictions\n        val_pred = fold_model.predict(X_val)\n        results['oof_predictions'][val_idx] = val_pred\n\n        # Calculate RMSLE for this fold\n        fold_rmse = rmsle(np.expm1(y_val), np.expm1(val_pred))\n        results['fold_scores'].append(fold_rmse)\n        print(f\"Fold {fold} RMSLE: {fold_rmse:.4f}\")\n\n        # Save the trained model\n        results['oof_models'].append(fold_model)\n\n    # Calculate overall RMSLE\n    overall_rmse = rmsle(np.expm1(y_array), np.expm1(results['oof_predictions']))\n    results['mean_rmse'] = overall_rmse\n\n    # Display results\n    fold_scores = results['fold_scores']\n    print(\"\\nResults:\")\n    print(f\"Fold RMSLE scores: {', '.join([f'{score:.4f}' for score in fold_scores])}\")\n    print(f\"Mean RMSLE: {overall_rmse:.4f} (±{np.std(fold_scores):.4f})\")\n\n    return results\n\n# Example usage with a DataFrame\n# Make sure `train_df` is preprocessed and features are ready\n# Assuming 'Premium Amount' is the target column\nresults_cat = get_model_oof_predictions(\n    X=train_df.drop(columns=['Premium Amount']),\n    y=np.log1p(train_df['Premium Amount']),  # Apply log-transform to stabilize large values\n    model=catboost_model\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:27:53.768201Z","iopub.execute_input":"2024-12-03T10:27:53.769098Z","iopub.status.idle":"2024-12-03T10:29:42.674044Z","shell.execute_reply.started":"2024-12-03T10:27:53.769062Z","shell.execute_reply":"2024-12-03T10:29:42.673161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions_cat = pd.DataFrame()\nfor fold_id, model in enumerate(results_cat['oof_models']):\n    predictions_cat[fold_id] = model.predict(test_df)\n\npredictions_cat['cat_mean'] = predictions_cat.mean(axis=1)\npredictions_cat['cat_mean_original'] = np.expm1(predictions_cat['cat_mean'])\npredictions_cat.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:29:45.840023Z","iopub.execute_input":"2024-12-03T10:29:45.840739Z","iopub.status.idle":"2024-12-03T10:29:48.467955Z","shell.execute_reply.started":"2024-12-03T10:29:45.840703Z","shell.execute_reply":"2024-12-03T10:29:48.466909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df['Premium Amount'] = predictions_cat['cat_mean_original']\nsub_df.to_csv(\"ps4e12catv2.csv\",index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:11:42.376046Z","iopub.execute_input":"2024-12-03T10:11:42.376905Z","iopub.status.idle":"2024-12-03T10:11:43.707498Z","shell.execute_reply.started":"2024-12-03T10:11:42.376869Z","shell.execute_reply":"2024-12-03T10:11:43.706798Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# XGBoost Model","metadata":{}},{"cell_type":"code","source":"def objective(trial):\n    param = {\n        'objective': 'reg:squarederror',  # Regression task with squared error\n        'eval_metric': 'rmse',            # Use RMSE as evaluation metric\n        'max_depth': trial.suggest_int('max_depth', 3, 10),  # Maximum depth of a tree\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3, log=True),  # Learning rate\n        'n_estimators': trial.suggest_int('n_estimators', 100, 1000, step=100),  # Number of boosting rounds\n        'subsample': trial.suggest_float('subsample', 0.6, 1.0),  # Fraction of samples to train each tree\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.6, 1.0),  # Fraction of features to train each tree\n        'gamma': trial.suggest_float('gamma', 0, 10),  # Minimum loss reduction required to make a further partition\n        'reg_alpha': trial.suggest_float('reg_alpha', 0, 1),  # L1 regularization\n        'reg_lambda': trial.suggest_float('reg_lambda', 0, 1)  # L2 regularization\n    }\n\n    # Train the model\n    model = xgb.XGBRegressor(**param)\n    model.fit(X_train, Y_train)\n\n    # Make predictions and calculate MSE\n    y_pred = model.predict(X_valid)\n    mse = mean_squared_error(Y_valid, y_pred)\n\n    return mse","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:21:09.334972Z","iopub.execute_input":"2024-12-03T06:21:09.335398Z","iopub.status.idle":"2024-12-03T06:21:09.34292Z","shell.execute_reply.started":"2024-12-03T06:21:09.335348Z","shell.execute_reply":"2024-12-03T06:21:09.341839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import optuna \n\nstudy = optuna.create_study(direction='minimize')\nstudy.optimize(objective, n_trials=10)\n\nprint(f\"Best MSE: {study.best_value}\")\nprint(f\"Best hyperparameters: {study.best_params}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nfrom sklearn.base import clone\nimport xgboost as xgb\nimport numpy as np\n\n# Define RMSLE calculation function\ndef rmsle_xgb(y_true, y_pred):\n    y_true = np.maximum(y_true, 1e-15)  # Avoid log(0)\n    y_pred = np.maximum(y_pred, 1e-15)\n    return np.sqrt(np.mean((np.log1p(y_pred) - np.log1p(y_true))**2))\n\n# Updated function for out-of-fold predictions\ndef get_oof_predictions_xgb(X_xgb, y_xgb, model_params_xgb, random_state_xgb=42):\n    skf_xgb = StratifiedKFold(n_splits=5, shuffle=True, random_state=random_state_xgb)\n    \n    # Convert X and y to NumPy arrays\n    X_array_xgb = X_xgb.values if hasattr(X_xgb, 'values') else np.array(X_xgb)\n    y_array_xgb = y_xgb.values if hasattr(y_xgb, 'values') else np.array(y_xgb)\n    \n    results_xgb = {\n        'oof_prediction_xgb': np.zeros(len(X_xgb)),\n        'fold_score_xgb': [],\n        'mean_rmsle_xgb': 0.0,\n        'oof_models_xgb': [],\n    }\n    \n    for fold_xgb, (train_idx_xgb, val_idx_xgb) in enumerate(skf_xgb.split(X_array_xgb, y_array_xgb), 1):\n        X_train_xgb, X_val_xgb = X_array_xgb[train_idx_xgb], X_array_xgb[val_idx_xgb]\n        y_train_xgb, y_val_xgb = y_array_xgb[train_idx_xgb], y_array_xgb[val_idx_xgb]\n        \n        print(f\"\\nTraining - Fold {fold_xgb}\")\n        \n        # Set GPU-specific parameters\n        model_params_xgb['tree_method'] = 'gpu_hist'\n        model_params_xgb['predictor'] = 'gpu_predictor'\n        \n        # Create DMatrix for XGBoost\n        train_dmatrix_xgb = xgb.DMatrix(X_train_xgb, label=np.log1p(y_train_xgb))\n        val_dmatrix_xgb = xgb.DMatrix(X_val_xgb, label=np.log1p(y_val_xgb))\n        \n        # Train the model\n        fold_model_xgb = xgb.train(\n            params=model_params_xgb,\n            dtrain=train_dmatrix_xgb,\n            evals=[(train_dmatrix_xgb, 'train'), (val_dmatrix_xgb, 'valid')],\n            num_boost_round=model_params_xgb.get('n_estimators', 800),\n            early_stopping_rounds=100,\n            verbose_eval=200\n        )\n        \n        # Get predictions and transform back to original scale\n        val_pred_log_xgb = fold_model_xgb.predict(val_dmatrix_xgb)\n        val_pred_xgb = np.expm1(val_pred_log_xgb)  # Inverse log1p\n        results_xgb['oof_prediction_xgb'][val_idx_xgb] = val_pred_xgb\n        \n        # Calculate RMSLE for this fold\n        fold_rmsle_xgb = rmsle_xgb(y_val_xgb, val_pred_xgb)\n        results_xgb['fold_score_xgb'].append(fold_rmsle_xgb)\n        print(f\"Fold {fold_xgb} RMSLE: {fold_rmsle_xgb:.4f}\")\n        \n        # Save the model for this fold\n        results_xgb['oof_models_xgb'].append(fold_model_xgb)\n    \n    # Calculate mean RMSLE across all folds\n    overall_rmsle_xgb = rmsle_xgb(y_array_xgb, results_xgb['oof_prediction_xgb'])\n    results_xgb['mean_rmsle_xgb'] = overall_rmsle_xgb\n    print(f\"\\nMean RMSLE: {overall_rmsle_xgb:.4f} (±{np.std(results_xgb['fold_score_xgb']):.4f})\")\n    \n    return results_xgb\n\n# Define best XGBoost parameters\nbest_xgb_params_xgb = {\n    'max_depth': 9,\n    'learning_rate': 0.017865130858123492,\n    'n_estimators': 800,\n    'subsample': 0.9826024297378023,\n    'colsample_bytree': 0.888218779155892,\n    'gamma': 0.8057155345797573,\n    'reg_alpha': 0.7069793976674651,\n    'reg_lambda': 0.7780604026518645,\n    'objective': 'reg:squarederror',  # Standard regression objective\n    'eval_metric': 'rmse',\n    'seed': 42,\n}\n\n# Example usage\nresults_xgb_final = get_oof_predictions_xgb(\n    X_xgb=train_df.drop(columns=['Premium Amount']),\n    y_xgb=train_df['Premium Amount'],\n    model_params_xgb=best_xgb_params_xgb\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:30:00.991993Z","iopub.execute_input":"2024-12-03T10:30:00.992749Z","iopub.status.idle":"2024-12-03T10:31:04.235778Z","shell.execute_reply.started":"2024-12-03T10:30:00.992713Z","shell.execute_reply":"2024-12-03T10:31:04.23509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ensure the test data is converted into a DMatrix\ntest_dmatrix_xgb = xgb.DMatrix(test_df)\n\n# Create a DataFrame to store fold predictions\npredictions_xgb = pd.DataFrame()\n\n# Iterate through each fold model and generate predictions\nfor fold_id, model_xgb in enumerate(results_xgb_final['oof_models_xgb']):\n    fold_pred_log_xgb = model_xgb.predict(test_dmatrix_xgb)  # Predict log-transformed values\n    predictions_xgb[fold_id] = np.expm1(fold_pred_log_xgb)  # Transform back to original scale\n\n# Calculate the mean prediction across folds\npredictions_xgb['xgb_mean'] = predictions_xgb.mean(axis=1)\n\n# Display the first few rows of predictions\npredictions_xgb.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:33:06.958477Z","iopub.execute_input":"2024-12-03T10:33:06.958829Z","iopub.status.idle":"2024-12-03T10:33:08.077409Z","shell.execute_reply.started":"2024-12-03T10:33:06.9588Z","shell.execute_reply":"2024-12-03T10:33:08.07651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df['Premium Amount'] = predictions_xgb['xgb_mean']\nsub_df.to_csv(\"ps4e12xgbv2.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:33:16.917537Z","iopub.execute_input":"2024-12-03T10:33:16.918443Z","iopub.status.idle":"2024-12-03T10:33:17.865599Z","shell.execute_reply.started":"2024-12-03T10:33:16.918406Z","shell.execute_reply":"2024-12-03T10:33:17.864642Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LGB Model","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:45:59.933239Z","iopub.execute_input":"2024-12-03T06:45:59.933638Z","iopub.status.idle":"2024-12-03T06:46:01.042936Z","shell.execute_reply.started":"2024-12-03T06:45:59.933602Z","shell.execute_reply":"2024-12-03T06:46:01.042057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def objective(trial):\n    param = {\n        'objective': 'regression',  # Regression task\n        'metric': 'rmse',           # Use RMSE as evaluation metric\n        'boosting_type': 'gbdt',    # Gradient Boosting Decision Tree\n        'max_depth': trial.suggest_int('max_depth', -1, 10),  # Maximum depth of a tree (-1 for no limit)\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3, log=True),  # Learning rate\n        'n_estimators': trial.suggest_int('n_estimators', 100, 1000, step=100),  # Number of boosting rounds\n        'subsample': trial.suggest_float('subsample', 0.6, 1.0),  # Fraction of data to train each tree\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.6, 1.0),  # Fraction of features to train each tree\n        'reg_alpha': trial.suggest_float('reg_alpha', 0, 1),  # L1 regularization\n        'reg_lambda': trial.suggest_float('reg_lambda', 0, 1),  # L2 regularization\n        'min_child_weight': trial.suggest_float('min_child_weight', 1e-3, 10, log=True),  # Minimum sum of instance weight in a child\n        'min_split_gain': trial.suggest_float('min_split_gain', 0, 1),  # Minimum gain for splitting\n    }\n\n    model = lgb.LGBMRegressor(**param)\n    model.fit(X_train, Y_train, eval_set=[(X_valid, Y_valid)])\n\n    y_pred = model.predict(X_valid)\n    rmse = mean_squared_error(Y_valid, y_pred, squared=False)  \n\n    return rmse","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T15:50:17.916008Z","iopub.execute_input":"2024-12-02T15:50:17.916418Z","iopub.status.idle":"2024-12-02T15:50:17.924565Z","shell.execute_reply.started":"2024-12-02T15:50:17.916384Z","shell.execute_reply":"2024-12-02T15:50:17.923402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"study_lgb = optuna.create_study(direction='minimize')\nstudy_lgb.optimize(objective, n_trials=10)\n\nprint(f\"Best MSE: {study_lgb.best_value}\")\nprint(f\"Best hyperparameters: {study_lgb.best_params}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T15:50:20.877653Z","iopub.execute_input":"2024-12-02T15:50:20.878097Z","iopub.status.idle":"2024-12-02T15:54:31.162237Z","shell.execute_reply.started":"2024-12-02T15:50:20.878047Z","shell.execute_reply":"2024-12-02T15:54:31.160798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params_lgb = {'learning_rate': 0.08692991511139551, 'num_leaves': 85, 'max_depth': 15, 'min_data_in_leaf': 95,\n          'feature_fraction': 0.7567559292276751, 'bagging_fraction': 0.9472874885021447, 'bagging_freq': 1,\n          'max_bin': 305, 'min_child_weight': 1, 'scale_pos_weight': 4,'n_estimators':200}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T06:46:05.901317Z","iopub.execute_input":"2024-12-03T06:46:05.901961Z","iopub.status.idle":"2024-12-03T06:46:05.907565Z","shell.execute_reply.started":"2024-12-03T06:46:05.901922Z","shell.execute_reply":"2024-12-03T06:46:05.906449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nimport lightgbm as lgb\nimport numpy as np\n\n# Define RMSLE calculation function\ndef rmsle_lgb(y_true, y_pred):\n    y_true = np.maximum(y_true, 1e-15)  # Avoid log(0)\n    y_pred = np.maximum(y_pred, 1e-15)\n    return np.sqrt(np.mean((np.log1p(y_pred) - np.log1p(y_true))**2))\n\n# Updated function for out-of-fold predictions with LightGBM\ndef get_oof_predictions_lgb(X_lgb, y_lgb, model_params_lgb, random_state_lgb=42):\n    skf_lgb = StratifiedKFold(n_splits=5, shuffle=True, random_state=random_state_lgb)\n    \n    # Convert X and y to NumPy arrays\n    X_array_lgb = X_lgb.values if hasattr(X_lgb, 'values') else np.array(X_lgb)\n    y_array_lgb = y_lgb.values if hasattr(y_lgb, 'values') else np.array(y_lgb)\n    \n    results_lgb = {\n        'oof_prediction_lgb': np.zeros(len(X_lgb)),\n        'fold_score_lgb': [],\n        'mean_rmsle_lgb': 0.0,\n        'oof_models_lgb': [],\n    }\n    \n    for fold_lgb, (train_idx_lgb, val_idx_lgb) in enumerate(skf_lgb.split(X_array_lgb, y_array_lgb), 1):\n        X_train_lgb, X_val_lgb = X_array_lgb[train_idx_lgb], X_array_lgb[val_idx_lgb]\n        y_train_lgb, y_val_lgb = y_array_lgb[train_idx_lgb], y_array_lgb[val_idx_lgb]\n        \n        print(f\"\\nTraining - Fold {fold_lgb}\")\n        \n        \n        # Create LightGBM datasets\n        train_data_lgb = lgb.Dataset(X_train_lgb, label=np.log1p(y_train_lgb))\n        val_data_lgb = lgb.Dataset(X_val_lgb, label=np.log1p(y_val_lgb), reference=train_data_lgb)\n        \n        # Train the model\n        fold_model_lgb = lgb.train(\n            params=model_params_lgb,\n            train_set=train_data_lgb,\n            valid_sets=[train_data_lgb, val_data_lgb],\n            num_boost_round=model_params_lgb.get('n_estimators', 200),\n            #early_stopping=100,\n            #verbose_eval=200\n        )\n        \n        # Get predictions and transform back to original scale\n        val_pred_log_lgb = fold_model_lgb.predict(X_val_lgb, num_iteration=fold_model_lgb.best_iteration)\n        val_pred_lgb = np.expm1(val_pred_log_lgb)  # Inverse log1p\n        results_lgb['oof_prediction_lgb'][val_idx_lgb] = val_pred_lgb\n        \n        # Calculate RMSLE for this fold\n        fold_rmsle_lgb = rmsle_lgb(y_val_lgb, val_pred_lgb)\n        results_lgb['fold_score_lgb'].append(fold_rmsle_lgb)\n        print(f\"Fold {fold_lgb} RMSLE: {fold_rmsle_lgb:.4f}\")\n        \n        # Save the model for this fold\n        results_lgb['oof_models_lgb'].append(fold_model_lgb)\n    \n    # Calculate mean RMSLE across all folds\n    overall_rmsle_lgb = rmsle_lgb(y_array_lgb, results_lgb['oof_prediction_lgb'])\n    results_lgb['mean_rmsle_lgb'] = overall_rmsle_lgb\n    print(f\"\\nMean RMSLE: {overall_rmsle_lgb:.4f} (±{np.std(results_lgb['fold_score_lgb']):.4f})\")\n    \n    return results_lgb\n\n# Define best LightGBM parameters\nparams_lgb = {\n    'learning_rate': 0.08692991511139551,\n    'num_leaves': 85,\n    'max_depth': 15,\n    'min_data_in_leaf': 95,\n    'feature_fraction': 0.7567559292276751,\n    'bagging_fraction': 0.9472874885021447,\n    'bagging_freq': 1,\n    'max_bin': 305,\n    'min_child_weight': 1,\n    'scale_pos_weight': 4,\n    'n_estimators': 200,\n    'objective': 'regression',\n    'metric': 'rmse',\n    'seed': 42\n}\n\n# Example usage\nresults_lgb_final = get_oof_predictions_lgb(\n    X_lgb=train_df.drop(columns=['Premium Amount']),\n    y_lgb=train_df['Premium Amount'],\n    model_params_lgb=params_lgb\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:37:22.636593Z","iopub.execute_input":"2024-12-03T10:37:22.636941Z","iopub.status.idle":"2024-12-03T10:38:45.640515Z","shell.execute_reply.started":"2024-12-03T10:37:22.636909Z","shell.execute_reply":"2024-12-03T10:38:45.639583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a DataFrame to store predictions from each fold\npredictions_lgb = pd.DataFrame()\n\nfor fold_id, model in enumerate(results_lgb_final['oof_models_lgb']):\n    # Predict for the test dataset using the best iteration\n    predictions_lgb[f'fold_{fold_id}'] = model.predict(test_df, num_iteration=model.best_iteration)\n\n# Compute the mean prediction across all folds\npredictions_lgb['lgb_mean'] = predictions_lgb.mean(axis=1)\n\n# Optionally, if the target was log-transformed during training, apply inverse transformation\n# (remove this line if the target wasn't log-transformed)\npredictions_lgb['lgb_mean_original'] = np.expm1(predictions_lgb['lgb_mean'])\n\n# Display the final predictions\npredictions_lgb.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:39:34.895742Z","iopub.execute_input":"2024-12-03T10:39:34.89643Z","iopub.status.idle":"2024-12-03T10:39:49.029066Z","shell.execute_reply.started":"2024-12-03T10:39:34.896397Z","shell.execute_reply":"2024-12-03T10:39:49.028265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df['Premium Amount'] = predictions_lgb['lgb_mean_original']\nsub_df.to_csv('p4e12lgbv2.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:40:05.919751Z","iopub.execute_input":"2024-12-03T10:40:05.920099Z","iopub.status.idle":"2024-12-03T10:40:07.259141Z","shell.execute_reply.started":"2024-12-03T10:40:05.920069Z","shell.execute_reply":"2024-12-03T10:40:07.258148Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# HILL Climber","metadata":{}},{"cell_type":"code","source":"results_lgb_final","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:43:26.215675Z","iopub.execute_input":"2024-12-03T10:43:26.216539Z","iopub.status.idle":"2024-12-03T10:43:26.222626Z","shell.execute_reply.started":"2024-12-03T10:43:26.216504Z","shell.execute_reply":"2024-12-03T10:43:26.221559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create the out-of-fold predictions DataFrame\noof_train_df = pd.DataFrame()\n\n# Assuming 'xgb_res', 'lgb_res', and 'results_cat' contain the out-of-fold predictions\noof_train_df['xgb_mean'] = results_xgb_final['oof_prediction_xgb']  # XGBoost predictions (inverse log-transformed)\noof_train_df['lgb_mean_original'] = results_lgb_final['oof_prediction_lgb']  # LightGBM predictions (inverse log-transformed)\noof_train_df['cat_mean_original'] = np.expm1(results_cat['oof_predictions'])  # Assuming 'results_cat' contains CatBoost out-of-fold predictions\n\n# Display the first few rows of the oof_train_df\noof_train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:51:03.72499Z","iopub.execute_input":"2024-12-03T10:51:03.725406Z","iopub.status.idle":"2024-12-03T10:51:03.765178Z","shell.execute_reply.started":"2024-12-03T10:51:03.725371Z","shell.execute_reply":"2024-12-03T10:51:03.764312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_pred = pd.concat([predictions_xgb['xgb_mean'],predictions_lgb['lgb_mean_original'],predictions_cat['cat_mean_original']],axis=1)\ntest_pred.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:51:06.41996Z","iopub.execute_input":"2024-12-03T10:51:06.420301Z","iopub.status.idle":"2024-12-03T10:51:06.433758Z","shell.execute_reply.started":"2024-12-03T10:51:06.420274Z","shell.execute_reply":"2024-12-03T10:51:06.43281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_log_error\n\n# Define RMSLE calculation function with handling of zero and negative values\ndef root_mean_square_log_error(y_true, y_pred):\n    # Ensure there are no zero or negative values\n    y_true = np.maximum(y_true, 1e-15)\n    y_pred = np.maximum(y_pred, 1e-15)\n    \n    return np.sqrt(mean_squared_log_error(y_true, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:54:04.402659Z","iopub.execute_input":"2024-12-03T10:54:04.403066Z","iopub.status.idle":"2024-12-03T10:54:04.408709Z","shell.execute_reply.started":"2024-12-03T10:54:04.403033Z","shell.execute_reply":"2024-12-03T10:54:04.407591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Call climb_hill function with RMSLE as the evaluation metric\nhc_test_pred_probs, hc_oof_pred_probs = climb_hill(\n    train=train_df, \n    oof_pred_df=oof_train_df, \n    test_pred_df=test_pred,\n    target='Premium Amount',\n    objective='minimize', \n    eval_metric=partial(root_mean_square_log_error),  # Use RMSLE as the evaluation metric\n    negative_weights=True, \n    precision=0.001, \n    plot_hill=True, \n    plot_hist=False,\n    return_oof_preds=True\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:54:07.231914Z","iopub.execute_input":"2024-12-03T10:54:07.232281Z","iopub.status.idle":"2024-12-03T10:55:21.11972Z","shell.execute_reply.started":"2024-12-03T10:54:07.232236Z","shell.execute_reply":"2024-12-03T10:55:21.118802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df['Premium Amount'] = hc_test_pred_probs\nsub_df.to_csv('ps4e12ensemblev3.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:55:48.149983Z","iopub.execute_input":"2024-12-03T10:55:48.150348Z","iopub.status.idle":"2024-12-03T10:55:49.517389Z","shell.execute_reply.started":"2024-12-03T10:55:48.150316Z","shell.execute_reply":"2024-12-03T10:55:49.51667Z"}},"outputs":[],"execution_count":null}]}