{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_log_error\nimport xgboost as xgb\nimport matplotlib.pyplot as plt\n\n# 1. Load the Training Data\ntrain_data = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\n\n# Define categorical columns\ncategorical_columns = [\n    'Education Level', 'Exercise Frequency', \n    'Occupation', 'Location', 'Policy Type', 'Smoking Status',\n    'Property Type', 'Customer Feedback', 'Gender',\n    'Marital Status'\n]\n\n# Preprocessing Function\ndef preprocess_data(df, label_encoders=None, is_train=True, latest_date=None):\n    # Convert 'Policy Start Date' to datetime\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'], errors='coerce')\n    \n    if is_train:\n        latest_date = df['Policy Start Date'].max()\n    \n    # Assign quarter intervals\n    def assign_quarter_interval(policy_date, latest):\n        if pd.isnull(policy_date):\n            return -1  # Assign -1 for missing dates\n        diff_years = latest.year - policy_date.year\n        diff_months = latest.month - policy_date.month\n        total_diff_months = diff_years * 12 + diff_months\n        interval = total_diff_months // 3  # Integer division for quarterly intervals\n        return interval\n    \n    df['Policy_Start_Interval'] = df['Policy Start Date'].apply(lambda x: assign_quarter_interval(x, latest_date))\n    df['Policy_Start_Interval'] = df['Policy_Start_Interval'].astype(int)\n    \n    # Drop the original 'Policy Start Date' column\n    df = df.drop('Policy Start Date', axis=1)\n    \n    # Encode categorical columns\n    if is_train:\n        label_encoders = {}\n        for column in categorical_columns:\n            # Handle missing values by filling with 'Missing'\n            if df[column].isnull().any():\n                df[column] = df[column].fillna('Missing')\n            \n            encoder = LabelEncoder()\n            df[column] = encoder.fit_transform(df[column])\n            label_encoders[column] = encoder\n        return df, label_encoders, latest_date\n    else:\n        for column in categorical_columns:\n            # Handle missing values by filling with 'Missing'\n            if df[column].isnull().any():\n                df[column] = df[column].fillna('Missing')\n            \n            # Use the existing encoder\n            if label_encoders and column in label_encoders:\n                encoder = label_encoders[column]\n                # Handle unseen labels by assigning a new value\n                df[column] = df[column].map(lambda s: '<unknown>' if s not in encoder.classes_ else s)\n                # Update classes_ to include '<unknown>' if needed\n                if '<unknown>' not in encoder.classes_:\n                    encoder.classes_ = np.append(encoder.classes_, '<unknown>')\n                df[column] = encoder.transform(df[column])\n            else:\n                raise ValueError(f'Label encoder for {column} not provided.')\n        return df\n\n# 2. Preprocess Training Data\ntrain_data, label_encoders, latest_date = preprocess_data(train_data, is_train=True)\n\n# Separate Features and Target\nX_train_full = train_data.drop(['id', 'Premium Amount'], axis=1)\nY_train_full = train_data['Premium Amount']\n\n# 3. Load and Preprocess Test Data\ntest_data = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\n\n# **Preserve the 'id' column before preprocessing**\ntest_ids = test_data['id'].copy()\n\n# Preprocess Test Data using the same label encoders and latest_date from training\ntest_data_processed = preprocess_data(test_data, label_encoders=label_encoders, is_train=False, latest_date=latest_date)\n\n# **Ensure 'id' is not included in the features for prediction**\n# Drop 'id' from test_data_processed to create the feature set\nX_test = test_data_processed.drop(['id'], axis=1, errors='ignore')  # Use 'errors=\"ignore\"' in case 'id' was already dropped\n\n# 4. Train the Model on the Full Training Data\n# Log-transform the target variable\nY_train_full_log = np.log1p(Y_train_full)\n\n# Initialize the XGBoost Regressor with early_stopping_rounds set in the constructor\nmodel = xgb.XGBRegressor(\n    objective='reg:squarederror',\n    random_state=42,\n    n_estimators=1000,\n    learning_rate=0.05,\n    max_depth=8,\n    subsample=0.8,\n    colsample_bytree=0.8,\n    early_stopping_rounds=50,  # Moved here as per warning\n    eval_metric='rmse',\n    verbosity=1\n)\n\n# Since you want to train on the entire dataset, reserve a small portion for early stopping\n# Here, we'll use a simple split for early stopping\nX_train_part, X_val_part, y_train_part, y_val_part = train_test_split(\n    X_train_full, Y_train_full_log, test_size=0.2, random_state=42\n)\n\n# Fit the model with early stopping\nmodel.fit(\n    X_train_part, y_train_part,\n    eval_set=[(X_val_part, y_val_part)],\n    verbose=100\n)\n\n# Optional: Retrain on the full dataset without early stopping using the best iteration\n# Uncomment the following lines if you prefer to train on all data after finding optimal rounds\noptimal_n_estimators = model.best_iteration\nmodel = xgb.XGBRegressor(\n    objective='reg:squarederror',\n    random_state=42,\n    n_estimators=optimal_n_estimators,\n    learning_rate=0.05,\n    max_depth=8,\n    subsample=0.8,\n    colsample_bytree=0.8,\n    eval_metric='rmse',\n    verbosity=1\n)\nmodel.fit(\n    X_train_full, Y_train_full_log,\n    verbose=100\n)\n\n# 5. Make Predictions on the Test Data\npreds_log_test = model.predict(X_test)\npreds_test = np.expm1(preds_log_test)  # Reverse the log transformation\n\n# 6. Prepare the Submission File\n# **Use the preserved 'id's stored in test_ids**\nsubmission = pd.DataFrame({\n    'id': test_ids,  # Use the preserved 'id's\n    'Premium Amount': preds_test\n})\n\n# Verify the submission format\nprint(submission.head())\n\n# Save the submission to a CSV file\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission file 'submission.csv' created successfully.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T17:40:29.647684Z","iopub.execute_input":"2025-01-08T17:40:29.648084Z","iopub.status.idle":"2025-01-08T17:42:16.594532Z","shell.execute_reply.started":"2025-01-08T17:40:29.648058Z","shell.execute_reply":"2025-01-08T17:42:16.593522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}