{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h2 style=\"font-family:Comic Sans MS, cursive, sans-serif; font-size:32px; font-weight:bold; text-align:center; background-color:#FFFFE0; padding:15px; border-radius:10px; box-shadow: 2px 2px 10px rgba(0, 0, 0, 0.3);\">\n    <span style=\"color:#FF6347;\">PS</span> - \n    <span style=\"color:#1E90FF;\">S4E12</span> | \n    <span style=\"color:#FF1493;\">Insurance Dataset</span> |\n    <span style=\"color:#32CD32;\">Regression </span> \n    \n</h2>","metadata":{}},{"cell_type":"code","source":"print('Begin')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:01:41.681618Z","iopub.execute_input":"2024-12-02T07:01:41.682848Z","iopub.status.idle":"2024-12-02T07:01:41.717973Z","shell.execute_reply.started":"2024-12-02T07:01:41.682794Z","shell.execute_reply":"2024-12-02T07:01:41.716817Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <span style=\"color:#2E86C1; font-family:Arial, sans-serif; font-weight:600;\">Import the Libraries</span>\n\n---","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom datetime import datetime\nimport lightgbm as lgb\nfrom sklearn.metrics import mean_squared_log_error\nfrom sklearn.model_selection import KFold","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:01:41.720418Z","iopub.execute_input":"2024-12-02T07:01:41.720877Z","iopub.status.idle":"2024-12-02T07:01:44.353154Z","shell.execute_reply.started":"2024-12-02T07:01:41.72083Z","shell.execute_reply":"2024-12-02T07:01:44.352092Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n## <span style=\"color:#2E86C1; font-family:Arial, sans-serif; font-weight:600;\">Import the Datasets</span>\n\n---","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest_df = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\nsubmission_df = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:01:44.354317Z","iopub.execute_input":"2024-12-02T07:01:44.354833Z","iopub.status.idle":"2024-12-02T07:01:55.691792Z","shell.execute_reply.started":"2024-12-02T07:01:44.354799Z","shell.execute_reply":"2024-12-02T07:01:55.690545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:01:55.694622Z","iopub.execute_input":"2024-12-02T07:01:55.695129Z","iopub.status.idle":"2024-12-02T07:01:55.738731Z","shell.execute_reply.started":"2024-12-02T07:01:55.695057Z","shell.execute_reply":"2024-12-02T07:01:55.737585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:01:55.740258Z","iopub.execute_input":"2024-12-02T07:01:55.740717Z","iopub.status.idle":"2024-12-02T07:01:55.766976Z","shell.execute_reply.started":"2024-12-02T07:01:55.740664Z","shell.execute_reply":"2024-12-02T07:01:55.765509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:01:55.769017Z","iopub.execute_input":"2024-12-02T07:01:55.769465Z","iopub.status.idle":"2024-12-02T07:01:55.785342Z","shell.execute_reply.started":"2024-12-02T07:01:55.769402Z","shell.execute_reply":"2024-12-02T07:01:55.784165Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <span style=\"color:#2E86C1; font-family:Arial, sans-serif; font-weight:600;\">EDA and Data Preprocessing</span>\n\n---","metadata":{}},{"cell_type":"code","source":"print(f'Shape of the train set: {train_df.shape}')\nprint(f'Shape of the test set: {test_df.shape}')\nprint(f'Shape of the sample submission: {submission_df.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:01:55.786661Z","iopub.execute_input":"2024-12-02T07:01:55.787026Z","iopub.status.idle":"2024-12-02T07:01:55.801523Z","shell.execute_reply.started":"2024-12-02T07:01:55.786991Z","shell.execute_reply":"2024-12-02T07:01:55.800188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:01:55.80293Z","iopub.execute_input":"2024-12-02T07:01:55.803433Z","iopub.status.idle":"2024-12-02T07:01:56.45067Z","shell.execute_reply.started":"2024-12-02T07:01:55.803383Z","shell.execute_reply":"2024-12-02T07:01:56.449384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:01:56.451932Z","iopub.execute_input":"2024-12-02T07:01:56.452303Z","iopub.status.idle":"2024-12-02T07:01:56.876768Z","shell.execute_reply.started":"2024-12-02T07:01:56.452269Z","shell.execute_reply":"2024-12-02T07:01:56.87563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the string column names\nstring_columns = train_df.select_dtypes(include=['object', 'category']).columns.tolist()\n\n# Display the string column names\nprint(\"String column names:\")\nprint(string_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:01:56.87992Z","iopub.execute_input":"2024-12-02T07:01:56.880283Z","iopub.status.idle":"2024-12-02T07:01:57.051511Z","shell.execute_reply.started":"2024-12-02T07:01:56.880249Z","shell.execute_reply":"2024-12-02T07:01:57.050265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Identify categorical variables based on data types\ncategorical_columns = train_df.select_dtypes(include=['object', 'category']).columns.tolist()\n\n# Change all the values to lowercase.\ntrain_df[categorical_columns] = train_df[categorical_columns].map(lambda x: x.lower() if isinstance(x, str) else x)\ntest_df[categorical_columns] = test_df[categorical_columns].map(lambda x: x.lower() if isinstance(x, str) else x)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:01:57.052774Z","iopub.execute_input":"2024-12-02T07:01:57.053369Z","iopub.status.idle":"2024-12-02T07:02:05.17572Z","shell.execute_reply.started":"2024-12-02T07:01:57.053315Z","shell.execute_reply":"2024-12-02T07:02:05.174639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count unique categories in each categorical column\nunique_categories_counts = {col: train_df[col].nunique() for col in categorical_columns}\n\n# Convert the counts to a pandas Series for easy plotting\nunique_categories_series = pd.Series(unique_categories_counts)\n\n# Plotting the unique category counts as a horizontal bar plot\nplt.figure(figsize=(12, 8))\nax = sns.barplot(x=unique_categories_series.values, y=unique_categories_series.index, palette='viridis')\n\n# Adding labels on the side of each bar\nfor i in ax.containers:\n    ax.bar_label(i, label_type='edge')\n\n# Displaying grid lines\nplt.grid(True, which='both', axis='x', linestyle='--', linewidth=0.7)\n\nplt.title('Number of Unique Categories in Categorical Columns')\nplt.xlabel('Number of Unique Categories')\nplt.ylabel('Categorical Columns')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:02:05.177251Z","iopub.execute_input":"2024-12-02T07:02:05.177621Z","iopub.status.idle":"2024-12-02T07:02:06.736509Z","shell.execute_reply.started":"2024-12-02T07:02:05.177583Z","shell.execute_reply":"2024-12-02T07:02:06.735344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['Policy Start Date'] = pd.to_datetime(train_df['Policy Start Date'])\ntest_df['Policy Start Date'] = pd.to_datetime(test_df['Policy Start Date'])\n\ntrain_df['Policy Month'] = train_df['Policy Start Date'].dt.month.astype(int)\ntest_df['Policy Month'] = test_df['Policy Start Date'].dt.month.astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:03:05.430991Z","iopub.execute_input":"2024-12-02T07:03:05.431628Z","iopub.status.idle":"2024-12-02T07:03:06.2786Z","shell.execute_reply.started":"2024-12-02T07:03:05.431568Z","shell.execute_reply":"2024-12-02T07:03:06.277473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns_to_remove = ['id', 'Policy Start Date', 'Premium Amount']\nX = train_df.drop(columns_to_remove, axis = 1)\ny = train_df['Premium Amount']\n\nX_test = test_df.drop(columns_to_remove[:-1], axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:03:06.280962Z","iopub.execute_input":"2024-12-02T07:03:06.281467Z","iopub.status.idle":"2024-12-02T07:03:06.662544Z","shell.execute_reply.started":"2024-12-02T07:03:06.281405Z","shell.execute_reply":"2024-12-02T07:03:06.661301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert all object type columns to category in train set\nX[X.select_dtypes(['object']).columns] = X.select_dtypes(['object']).apply(lambda x: x.astype('category'))\n\n# Convert all object type columns to category in test set\nX_test[X_test.select_dtypes(['object']).columns] = X_test.select_dtypes(['object']).apply(lambda x: x.astype('category'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:03:06.663948Z","iopub.execute_input":"2024-12-02T07:03:06.664297Z","iopub.status.idle":"2024-12-02T07:03:11.380509Z","shell.execute_reply.started":"2024-12-02T07:03:06.664264Z","shell.execute_reply":"2024-12-02T07:03:11.37926Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <span style=\"color:#2E86C1; font-family:Arial, sans-serif; font-weight:600;\">Model Training and Evaluation</span>\n\n---","metadata":{}},{"cell_type":"code","source":"# RMSLE Metric Function\ndef rmsle_metric(preds, train_data):\n    \"\"\"\n    Custom RMSLE evaluation metric for LightGBM regression\n    \"\"\"\n    y_true = train_data.get_label()\n    preds = np.clip(preds, 0, None)\n    \n    # Calculate RMSLE\n    rmsle = np.sqrt(mean_squared_log_error(y_true, preds))\n    return 'rmsle', rmsle, False\n\n# Define parameters\nparams = {\n    'n_estimators': 2500,\n    'objective': 'regression', \n    'metric': 'None',\n    'boosting_type': 'gbdt',\n    'learning_rate': 0.05,\n    'num_leaves': 31,\n    'random_state': 42,\n    'verbose': -1\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:03:11.382749Z","iopub.execute_input":"2024-12-02T07:03:11.383158Z","iopub.status.idle":"2024-12-02T07:03:11.390733Z","shell.execute_reply.started":"2024-12-02T07:03:11.383121Z","shell.execute_reply":"2024-12-02T07:03:11.38924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def run_lightgbm_cv(X, y, test_df=None, n_splits=5, random_state=42):\n    \"\"\"\n    Run cross-validation for LightGBM regression\n    \"\"\"\n    # Initialize KFold\n    kf = KFold(n_splits=n_splits, shuffle=True, random_state=random_state)\n    \n    # Initialize lists to hold scores\n    rmsle_scores = []\n    \n    # Initialize test predictions if test set is provided\n    if test_df is not None:\n        test_preds = np.zeros(len(test_df))\n    \n    # Run cross-validation\n    for fold, (train_index, val_index) in enumerate(kf.split(X), start=1):\n        print(f\"\\nFold {fold}/{n_splits}\")\n        print(\"=\" * 50)\n        \n        # Split data for this fold\n        X_train_fold = X.iloc[train_index]\n        X_val_fold = X.iloc[val_index]\n        y_train_fold = y.iloc[train_index]\n        y_val_fold = y.iloc[val_index]\n        \n        # Create LightGBM datasets\n        train_data = lgb.Dataset(X_train_fold, label=y_train_fold)\n        val_data = lgb.Dataset(X_val_fold, label=y_val_fold, reference=train_data)\n        \n        # Train model\n        model = lgb.train(\n            params,\n            train_data,\n            valid_sets=[train_data, val_data],\n            valid_names=['Train', 'Valid'],\n            feval=rmsle_metric,\n            callbacks=[\n                lgb.log_evaluation(100),\n                lgb.early_stopping(500)\n            ]\n        )\n        \n        # Predict on validation set\n        val_preds = model.predict(X_val_fold)\n        val_preds = np.clip(val_preds, 0, None)  # Ensure no negative predictions\n        \n        # Calculate RMSLE for this fold\n        fold_rmsle = np.sqrt(mean_squared_log_error(y_val_fold, val_preds))\n        rmsle_scores.append(fold_rmsle)\n        \n        print(f\"Fold {fold} RMSLE: {fold_rmsle:.4f}\")\n        \n        # Predict on test set if provided\n        if test_df is not None:\n            test_preds += model.predict(test_df) / n_splits\n    \n    # Print overall results\n    print(\"\\n\" + \"=\" * 50)\n    print(f\"Average RMSLE across folds: {np.mean(rmsle_scores):.4f}\")\n    print(f\"Std of RMSLE across folds: {np.std(rmsle_scores):.4f}\")\n    \n    # Return results\n    results = {\n        'fold_rmsle': rmsle_scores,\n        'mean_rmsle': np.mean(rmsle_scores),\n        'std_rmsle': np.std(rmsle_scores)\n    }\n    \n    if test_df is not None:\n        results['test_predictions'] = test_preds\n    \n    return results\n\n# Run cross-validation\nresults = run_lightgbm_cv(X, y, test_df=X_test)\n\n# Access the mean RMSLE\nprint(f\"Mean RMSLE: {results['mean_rmsle']:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:03:11.392783Z","iopub.execute_input":"2024-12-02T07:03:11.393275Z","iopub.status.idle":"2024-12-02T07:03:28.014458Z","shell.execute_reply.started":"2024-12-02T07:03:11.393232Z","shell.execute_reply":"2024-12-02T07:03:28.013225Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <span style=\"color:#2E86C1; font-family:Arial, sans-serif; font-weight:600;\">Submission</span>\n\n---","metadata":{}},{"cell_type":"code","source":"submission_df['Premium Amount'] = results['test_predictions']\nsubmission_df.to_csv('submission.csv', index = False)\nsubmission_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:03:28.015725Z","iopub.execute_input":"2024-12-02T07:03:28.016064Z","iopub.status.idle":"2024-12-02T07:03:29.723889Z","shell.execute_reply.started":"2024-12-02T07:03:28.016032Z","shell.execute_reply":"2024-12-02T07:03:29.722584Z"}},"outputs":[],"execution_count":null}]}