{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":9178166,"sourceType":"datasetVersion","datasetId":5547076}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"border: 5px solid; border-color: rgb(146, 109, 255); border-radius: 30px; position: relative; width: 95%; height: 300px; display: flex; justify-content: center; align-items: center; background-color: #f0f0f0;\">\n    <img src=\"https://www.kaggle.com/competitions/84896/images/header\" style=\"position: absolute; top: 0; left: 0; width: 100%; height: 100%; z-index: 0; opacity: 1.0; border-radius: 30px\">\n    <div style=\"position: relative; z-index: 1; text-align: center; background-color: rgba(36, 219, 255, 0); color: orange; display: flex; flex-direction: column; align-items: center; text-align: center; justify-content: center; width: 100%; margin: 10%; padding: 5px; border-radius: 20px\">\n        <h1 style=\"text-align: center; width: 100%; font-size: 75px; color:rgb(146, 109, 255)\" ><b>Insurance Premium 💲💸</b></h1>\n    </div>\n</div>","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport os\nbase_path = \"/kaggle/input/playground-series-s4e12\"\ndf = pd.read_csv(os.path.join(base_path, 'train.csv'))\ndf_test = pd.read_csv(os.path.join(base_path, 'test.csv'))\ndf_sample = pd.read_csv(os.path.join(base_path, 'sample_submission.csv'))\ndf_origin = pd.read_csv(r\"/kaggle/input/insurance-premium-prediction/Insurance Premium Prediction Dataset.csv\")\nTARGET = 'Premium Amount'\n\nimport warnings\nwarnings.filterwarnings('ignore')\n%load_ext autoreload\n%autoreload 2\n%matplotlib inline\n\nsns.set()\nsns.set_palette('cool')\nSNS_CMAP = 'cool'\nplt.style.use(\"dark_background\")\nplt.rcParams['grid.color'] = '#444444'\ncolors = sns.palettes.color_palette(SNS_CMAP)\npd.options.mode.chained_assignment = None","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-31T17:06:39.767454Z","iopub.execute_input":"2024-12-31T17:06:39.767766Z","iopub.status.idle":"2024-12-31T17:06:50.573387Z","shell.execute_reply.started":"2024-12-31T17:06:39.76774Z","shell.execute_reply":"2024-12-31T17:06:50.572732Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <div style=\"width: 100%\"><h1 style=\"text-align: center; font-family: 'Roboto', sans-serif; color: rgb(219, 36, 255); background-color: rgba(73, 182, 255, 0.6); padding: 30px; border: 5px solid rgb(219, 36, 255); border-style: solid; border-radius: 10px;\"> EDA </h1></div>","metadata":{}},{"cell_type":"code","source":"# df = df.drop('id', axis=1)\n# df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n# df['year'] = df['Policy Start Date'].dt.year\n# df['days_elapsed'] = (df['Policy Start Date'] -  df['Policy Start Date'].min()).dt.days\n\n# num_cols = [col for col in df.columns if len(df[col].value_counts())>6]\n# cat_cols = [col for col in df.columns if len(df[col].value_counts())<=6]\n\ndf.head().style.background_gradient(cmap=SNS_CMAP)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:14:51.895846Z","iopub.execute_input":"2024-12-01T08:14:51.896247Z","iopub.status.idle":"2024-12-01T08:14:52.034919Z","shell.execute_reply.started":"2024-12-01T08:14:51.896209Z","shell.execute_reply":"2024-12-01T08:14:52.03378Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.describe().iloc[1:].style.background_gradient(cmap=SNS_CMAP)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:15:04.104737Z","iopub.execute_input":"2024-12-01T08:15:04.10534Z","iopub.status.idle":"2024-12-01T08:15:04.861852Z","shell.execute_reply.started":"2024-12-01T08:15:04.105298Z","shell.execute_reply":"2024-12-01T08:15:04.860743Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"agg_df = df.agg([\"nunique\", \"unique\", lambda x:x.isna().sum(), \"dtypes\"]).T\nagg_df['unique'] = agg_df['unique'].apply(lambda x: x if len(x)<10 else x[:10])\nagg_df.style.apply(lambda s: [f'background-color: rgba({colors[0][0]*255}, {colors[0][1]*255}, {colors[0][2]*255}, 0.5)' if i % 2 == 0 else f'background-color: rgba({colors[5][0]*255}, {colors[5][1]*255}, {colors[5][2]*255}, 0.5)' for i in range(len(s))])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:15:25.37505Z","iopub.execute_input":"2024-12-01T08:15:25.375431Z","iopub.status.idle":"2024-12-01T08:15:29.079237Z","shell.execute_reply.started":"2024-12-01T08:15:25.375396Z","shell.execute_reply":"2024-12-01T08:15:29.077954Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_histograms_continous(df: pd.DataFrame, target: str, cols:list[str]=None, ncols: int = 3)->None:\n    df = df.copy()\n    if cols is None:\n        cols = df.columns\n    nrows = (len(cols)+ncols-1)//ncols\n    fig, axes = plt.subplots(nrows, ncols, figsize=(4*ncols, 4*nrows))\n    df['TARGET_BINNED'] =  pd.cut(df[TARGET], bins = 5)\n    \n    for i, col in enumerate(cols):\n        if len(df[col].unique())>10:\n            sns.histplot(data=df, x=col, bins=20, hue='TARGET_BINNED', multiple='stack', ax=axes.flatten()[i], legend=False, palette=SNS_CMAP)\n        else:\n            sns.histplot(data=df, x=col, hue='TARGET_BINNED', discrete=True, shrink=0.8, multiple='stack', ax=axes.flatten()[i], legend=False, palette=SNS_CMAP)\n    plt.show()\n\nplot_histograms_continous(df.drop(['Policy Start Date'], axis=1), TARGET)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T16:38:17.797235Z","iopub.execute_input":"2024-12-02T16:38:17.798433Z","iopub.status.idle":"2024-12-02T16:38:41.852571Z","shell.execute_reply.started":"2024-12-02T16:38:17.798362Z","shell.execute_reply":"2024-12-02T16:38:41.851319Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_datetime_info(df:pd.DataFrame)->pd.DataFrame:\n    df = df.copy()\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    df['days_elapsed'] = (df['Policy Start Date'] -  df['Policy Start Date'].min()).dt.days\n    df['year'] = df['Policy Start Date'].dt.year\n    df['month'] = df['Policy Start Date'].dt.month\n    return df\n\ndf = extract_datetime_info(df)\nfig, ax = plt.subplots(1, 2, figsize=(16, 8))\nsns.histplot(data=df, x='year', bins=20, hue='TARGET_BINNED', multiple='stack', legend=False, palette=SNS_CMAP, ax=ax[0])\nsns.violinplot(data=df, y='Credit Score', x='year', ax=ax[1], palette=SNS_CMAP)\nfig.suptitle('Is Date/Year correlated with the data ?')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T05:06:21.501025Z","iopub.execute_input":"2024-12-03T05:06:21.501483Z","iopub.status.idle":"2024-12-03T05:06:24.854438Z","shell.execute_reply.started":"2024-12-03T05:06:21.501443Z","shell.execute_reply":"2024-12-03T05:06:24.853042Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\" margin: 0% 20%; text-align: center; font-family: 'Roboto', sans-serif; color: rgba(146, 146, 230, 1); background-color: rgba(36, 219, 215, 0.15); padding: 20px; border-style: solid; border-radius: 10px;\"> <p> The Immediate observation is how balanced all features are. \n\nYear/Time elapsed is the only categorical var presenting a slight Normal behaviour but the distributions across features is again balanced </p></div>","metadata":{}},{"cell_type":"code","source":"def plot_nans(df:pd.DataFrame):\n    nan_cols = [col for col in df.columns if df[col].isna().sum() > 100]\n    num_cols = [TARGET, 'Health Score', 'Credit Score']\n    \n    fig, ax = plt.subplots(len(nan_cols), 3, width_ratios=[3, 3, 1], figsize=(16, 8*len(nan_cols)))\n    for i, col in enumerate(nan_cols):\n        targ_num_cols = [numcol for numcol in num_cols if numcol != col]\n        df[f\"{col}_is_nan\"] = df[col].isna()\n        sns.violinplot(data=df, y=targ_num_cols[0], x=f\"{col}_is_nan\", ax=ax.flatten()[3*i], palette=SNS_CMAP)\n        sns.violinplot(data=df, y=targ_num_cols[1], x=f\"{col}_is_nan\", ax=ax.flatten()[3*i+1], palette=SNS_CMAP)\n        b = df[col].isna().sum()*100/df.shape[0]  # Complement percentage for category B\n        a = 100-b \n        data = pd.DataFrame({\n            \"NaN\": [\"A\", \"A\"],  \n            \"Type\": [\"Non-NaN\", \"NaN\"],  \n            \"Percentage\": [a, b]  \n        })\n        \n        sns.barplot(\n            data=data,\n            x=\"NaN\",\n            y=\"Percentage\",\n            hue=\"Type\",\n            dodge=False,  \n            palette=[colors[0], colors[-1]],\n            ax=ax.flatten()[3*i+2]\n        )\n        \n        ax.flatten()[3*i+2].legend().set_visible(False)\n        \n    fig.suptitle(\"Significance of nan values ?\")\n    fig.subplots_adjust(top=0.95, hspace=0.4) \n    plt.show()\n\nplot_nans(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:06:03.226016Z","iopub.execute_input":"2024-12-03T12:06:03.22643Z","iopub.status.idle":"2024-12-03T12:06:53.758493Z","shell.execute_reply.started":"2024-12-03T12:06:03.226388Z","shell.execute_reply":"2024-12-03T12:06:53.757275Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\" margin: 0% 20%; text-align: center; font-family: 'Roboto', sans-serif; color: rgba(146, 146, 230, 1); background-color: rgba(36, 219, 215, 0.15); padding: 20px; border-style: solid; border-radius: 10px;\"> <p> That wasn't exactly what I was hoping for :(\n\nI will still encode Occupation, Feedback and Previous Claims however due to the number of missing values</p></div>","metadata":{}},{"cell_type":"code","source":"from matplotlib.colors import LinearSegmentedColormap\n\ndef enocde_categoricals(df: pd.DataFrame)->pd.DataFrame:\n    df = df.copy()  \n    initial = df.isna().sum()\n    df['Gender'] = df['Gender'].map({\n        'Male': 0,\n        'Female': 1\n    }) \n    df['Smoking Status'] = df['Smoking Status'].map({\n        'No': 0,\n        'Yes': 1\n    }) \n    df['Marital Status'] = df['Marital Status'].map({\n        'Single': 0,\n        'Married': 1,\n        'Divorced': 2,\n    })\n    df['Education Level'] = df['Education Level'].map({\n        \"High School\": 0,\n        \"Bachelor's\": 1,\n        \"Master's\": 2,\n        \"PhD\": 3\n    }) \n    df['Occupation'] = df['Occupation'].map({\n        'Self-Employed': -1,\n        'Unemployed': 0,\n        'Employed': 1,\n    })\n    df['Location'] = df['Location'].map({\n        'Rural': 0,\n        'Suburban': 1,\n        'Urban': 2,\n    })\n    df['Policy Type'] = df['Policy Type'].map({\n        \"Premium\": 1,\n        \"Comprehensive\": 2,\n        \"Basic\": 0,\n    }) \n    df['Customer Feedback'] = df['Customer Feedback'].map({\n        'Poor': -1,\n        'Average': 0,\n        'Good': 1,\n    })\n    df['Exercise Frequency'] = df['Exercise Frequency'].map({\n        \"Weekly\": 2, \"Monthly\": 1, \"Daily\": 3, \"Rarely\": 0,\n    })\n    df['Property Type'] = df['Property Type'].map({\n        'House': 2,\n        'Apartment': 0,\n        'Condo': 1,\n    })\n    \n    final = df.isna().sum()\n    assert initial.equals(final)\n    return df\n\ndef plot_heatmap(df, cmap=SNS_CMAP):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    plt.figure(figsize=(15, 15))\n    sns.heatmap(df.select_dtypes(include=numerics).corr(),\n                cmap=cmap, annot=True, annot_kws={'fontsize':7}, fmt='.1g', vmin=-1, vmax=1, center= 0)\n    plt.title(\"Feature Correlation\")\n    plt.show()\n\nnum_cols = [col for col in df.columns.drop(TARGET) if len(df[col].value_counts())>6]\nord_cols = [col for col in df.columns.drop(TARGET) if len(df[col].value_counts())<=6]\ndf = df[num_cols + ord_cols + [TARGET]]\nplot_heatmap(enocde_categoricals(df[list(df.columns.drop(TARGET))+[TARGET]]),\n            LinearSegmentedColormap.from_list(\"hmap_cmap\", [colors[1], (0, 0, 0), colors[-2]]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T14:23:49.808306Z","iopub.execute_input":"2024-12-03T14:23:49.808925Z","iopub.status.idle":"2024-12-03T14:23:59.113762Z","shell.execute_reply.started":"2024-12-03T14:23:49.808869Z","shell.execute_reply":"2024-12-03T14:23:59.112591Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_numerical_distributions(df: pd.DataFrame, num_cols: list[str], target: str):\n    df = df.copy()\n    \n    fig, ax = plt.subplots(len(num_cols), 1, figsize=(15, 6*len(num_cols)))\n\n    for i, col in enumerate(num_cols):\n        df[col] = pd.cut(df[col], bins=30).apply(lambda x: np.ceil(x.right) if pd.notnull(x) else np.nan)\n        sns.boxplot(data=df, x=col, y=target, palette=SNS_CMAP, ax=ax[i])\n        num_ticks = max(1, len(df[col].unique())//10)\n        x_ticks = ax[i].get_xticks()\n        ax[i].set_xticks(x_ticks[::num_ticks])  # Keep every 5th tick\n        ax[i].set_xticklabels([int(tick) for tick in x_ticks[::num_ticks]])  # Convert tick labels to integers\n\n    plt.tight_layout()\n    fig.suptitle('Distribution of Target vs Numerical Features')\n    fig.subplots_adjust(top=0.97) \n\n\n    plt.show()\n\nnum_cols = [col for col in df.columns.drop([TARGET, 'Policy Start Date']) if len(df[col].value_counts())>6]\nplot_numerical_distributions(df, num_cols, TARGET)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T01:56:16.686785Z","iopub.execute_input":"2024-12-08T01:56:16.687793Z","iopub.status.idle":"2024-12-08T01:56:24.629836Z","shell.execute_reply.started":"2024-12-08T01:56:16.687735Z","shell.execute_reply":"2024-12-08T01:56:24.628561Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\" margin: 0% 20%; text-align: center; font-family: 'Roboto', sans-serif; color: rgba(146, 146, 230, 1); background-color: rgba(36, 219, 215, 0.15); padding: 20px; border-style: solid; border-radius: 10px;\"> <p> Previous Claims seems to be the feature with the strongest relation with the target. <br>\nThe Premium Amount increases with more number of Previous Claims</p></div>","metadata":{"execution":{"iopub.status.busy":"2024-12-08T02:29:17.387889Z","iopub.execute_input":"2024-12-08T02:29:17.388828Z","iopub.status.idle":"2024-12-08T02:29:17.41552Z","shell.execute_reply.started":"2024-12-08T02:29:17.388782Z","shell.execute_reply":"2024-12-08T02:29:17.413885Z"}}},{"cell_type":"code","source":"sns.pairplot(df[num_cols+[TARGET]].sample(5000), diag_kind='kde', hue=TARGET, plot_kws={'alpha': 0.5}, palette=SNS_CMAP)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T02:27:08.921127Z","iopub.execute_input":"2024-12-08T02:27:08.921539Z","iopub.status.idle":"2024-12-08T02:28:44.206745Z","shell.execute_reply.started":"2024-12-08T02:27:08.921499Z","shell.execute_reply":"2024-12-08T02:28:44.205239Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### <div style=\"text-align: center; font-family: 'Roboto', sans-serif; color: rgba(109, 146, 255, 1.0); background-color: rgba(219, 36, 255, 0.3); padding: 10px; border-style: solid; border-radius: 10px;\"> Categorical Columns </div>","metadata":{}},{"cell_type":"code","source":" def plot_categoricals_against_numeric_targets(df:pd.DataFrame, cat_cols = None, targs = None):\n    if cat_cols is None:\n        cat_cols = [col for col in df.columns if len(df[col].unique())<10]\n    fig,ax = plt.subplots(len(cat_cols), 2,figsize=(15, 5*len(cat_cols)))\n\n    for i, col in enumerate(cat_cols):\n        sns.kdeplot(data=df, x=targs[0], hue=df[col], fill=True, ax=ax[i][0], palette=SNS_CMAP)\n        ax[i][0].set_title(f\"Distribution of {targs[0]} for {col}\")\n        sns.boxplot(df, y=targs[1], x=col, ax=ax[i][1], palette=SNS_CMAP)\n        ax[i][1].set_title(f\"Boxplot of {targs[1]} for {col}\")\n\n    fig.tight_layout()\n    plt.show()\n\nplot_categoricals_against_numeric_targets(df, cat_cols, [TARGET, TARGET])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T04:43:47.692913Z","iopub.execute_input":"2024-12-05T04:43:47.693296Z","iopub.status.idle":"2024-12-05T04:45:12.553035Z","shell.execute_reply.started":"2024-12-05T04:43:47.693263Z","shell.execute_reply":"2024-12-05T04:45:12.551907Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### <div style=\"text-align: center; font-family: 'Roboto', sans-serif; color: rgba(109, 146, 255, 1.0); background-color: rgba(219, 36, 255, 0.3); padding: 10px; border-style: solid; border-radius: 10px;\"> Original Data </div>","metadata":{}},{"cell_type":"code","source":"def plot_histograms_continous(df: pd.DataFrame, target: str, cols:list[str]=None, ncols: int = 3)->None:\n    df = df.copy()\n    if cols is None:\n        cols = df.columns\n    nrows = (len(cols)+ncols-1)//ncols\n    fig, axes = plt.subplots(nrows, ncols, figsize=(4*ncols, 4*nrows))\n    df['TARGET_BINNED'] =  pd.cut(df[TARGET], bins = 5)\n    \n    for i, col in enumerate(cols):\n        if len(df[col].unique())>10:\n            sns.histplot(data=df, x=col, bins=20, hue='TARGET_BINNED', multiple='stack', ax=axes.flatten()[i], legend=False, palette='vlag')\n        else:\n            sns.histplot(data=df, x=col, hue='TARGET_BINNED', discrete=True, shrink=0.8, multiple='stack', ax=axes.flatten()[i], legend=False, palette='vlag')\n    plt.show()\n\nplot_histograms_continous(df_origin.drop(['Policy Start Date'], axis=1), TARGET)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T03:34:55.563786Z","iopub.execute_input":"2024-12-14T03:34:55.56419Z","iopub.status.idle":"2024-12-14T03:35:06.393313Z","shell.execute_reply.started":"2024-12-14T03:34:55.564156Z","shell.execute_reply":"2024-12-14T03:35:06.392163Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import scipy.stats as stats\n\ndef make_probaplots(df, cols, ncols=3):\n    nrows = (len(cols)+ncols-1)//ncols\n    fig,ax = plt.subplots(nrows, ncols ,figsize=(15, 6*nrows))\n    for i, col in enumerate(cols):\n        stats.probplot(df[col], dist=\"norm\", plot=ax.flatten()[i])\n        ax.flatten()[i].set_title(col)\n    plt.show()\n\nmake_probaplots(df.dropna(), num_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T03:45:49.497663Z","iopub.execute_input":"2024-12-08T03:45:49.498641Z","iopub.status.idle":"2024-12-08T03:45:59.138575Z","shell.execute_reply.started":"2024-12-08T03:45:49.498595Z","shell.execute_reply":"2024-12-08T03:45:59.137452Z"},"jupyter":{"source_hidden":true},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <div style=\"width: 100%\"><h1 style=\"text-align: center; font-family: 'Roboto', sans-serif; color: rgb(219, 36, 255); background-color: rgba(73, 182, 255, 0.6); padding: 30px; border: 5px solid rgb(219, 36, 255); border-style: solid; border-radius: 10px;\"> Preprocessing </h1></div>","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(os.path.join(base_path, 'train.csv'))\ndf_test = pd.read_csv(os.path.join(base_path, 'test.csv')).drop(['id'], axis=1)\ndf_origin = pd.read_csv(r\"/kaggle/input/insurance-premium-prediction/Insurance Premium Prediction Dataset.csv\")\n# df = pd.concat([df, df_origin]).drop(['id'], axis=1)\n\ndf = df[~df[TARGET].isna()]\ndf[TARGET] = np.log1p(df[TARGET])\ndf.shape, df_origin.shape, df_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T17:06:50.574444Z","iopub.execute_input":"2024-12-31T17:06:50.574649Z","iopub.status.idle":"2024-12-31T17:06:57.473007Z","shell.execute_reply.started":"2024-12-31T17:06:50.574631Z","shell.execute_reply":"2024-12-31T17:06:57.472296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def encode_nans(df: pd.DataFrame, df_test: pd.DataFrame)->pd.DataFrame:\n    for col in df.columns:\n        if df[col].isna().sum() > 100000:\n            df[f\"{col}_is_na\"] = df[col].isna().astype(int)\n            df_test[f\"{col}_is_na\"] = df_test[col].isna().astype(int)\n    return df, df_test\n    \ndef enocde_categoricals(df: pd.DataFrame, cat_cols: list[str])->pd.DataFrame:\n    initial_cat_count = {}\n    for col in df[cat_cols]:\n        cats = [x for x in df[col].unique() if not pd.isna(x)]\n        initial_cat_count[col] = len(cats)\n    df['Gender'] = df['Gender'].map({\n        'Male': 0,\n        'Female': 1\n    }) \n    df['Smoking Status'] = df['Smoking Status'].map({\n        'No': 0,\n        'Yes': 1\n    }) \n    df['Marital Status'] = df['Marital Status'].map({\n        'Single': 0,\n        'Married': 1,\n        'Divorced': 2,\n    })\n    df['Education Level'] = df['Education Level'].map({\n        \"High School\": 0,\n        \"Bachelor's\": 1,\n        \"Master's\": 2,\n        \"PhD\": 3\n    }) \n    df['Occupation'] = df['Occupation'].map({\n        'Self-Employed': -1,\n        'Unemployed': 0,\n        'Employed': 1,\n    })\n    df['Location'] = df['Location'].map({\n        'Rural': 0,\n        'Suburban': 1,\n        'Urban': 2,\n    })\n    df['Policy Type'] = df['Policy Type'].map({\n        \"Premium\": 1,\n        \"Comprehensive\": 2,\n        \"Basic\": 0,\n    }) \n    df['Customer Feedback'] = df['Customer Feedback'].map({\n        'Poor': -1,\n        'Average': 0,\n        'Good': 1,\n    })\n    df['Exercise Frequency'] = df['Exercise Frequency'].map({\n        \"Weekly\": 2, \"Monthly\": 1, \"Daily\": 3, \"Rarely\": 0,\n    })\n    df['Property Type'] = df['Property Type'].map({\n        'House': 2,\n        'Apartment': 0,\n        'Condo': 1,\n    })\n    final_cat_count = {}\n    for col in df[cat_cols]:\n        cats = [x for x in df[col].unique() if not pd.isna(x)]\n        final_cat_count[col] = len(cats)\n\n    if (initial_cat_count != final_cat_count):\n        print(initial_cat_count)\n        print(final_cat_count)\n        print(df.head())\n        raise ValueError(f\"Categories lost for shape {df.shape}\")\n    return df\n\ndef clrd(text: str, color: str = None, con: bool = None, c1:str = 'ok', c2:str = 'error')->str:\n    text = str(text)\n    color_codes = {\n        'ok': '\\033[1;92m',\n        'error': '\\033[91m',\n        'warning': '\\033[93m',\n        'success': '\\033[92m',\n        'status': '\\033[95m',\n        'special': '\\033[94m',\n        'log': '\\033[96m',\n        'reset': '\\033[0m',\n    }\n    if con is not None:\n        color = c1 if con else c2\n    color_code = color_codes.get(color, color_codes['reset'])\n    return f\"{color_code}{text}{color_codes['reset']}\"\n\ndef _feat_eng_date(df:pd.DataFrame)->None:\n    \"\"\"\n    Attempt to encode cyclic nature of variables\n    \"\"\"\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    df['year'] = df['Policy Start Date'].dt.year\n    df['Day'] = df['Policy Start Date'].dt.day\n    df['Month'] = df['Policy Start Date'].dt.month\n    df['Day_of_week'] = df['Policy Start Date'].dt.dayofweek\n    df['Week'] = df['Policy Start Date'].dt.isocalendar().week\n    df['Year_sin'] = np.sin(2 * np.pi * df['year'])\n    df['Year_cos'] = np.cos(2 * np.pi * df['year'])\n    df['Month_sin'] = np.sin(2 * np.pi * df['Month'] / 12) \n    df['Month_cos'] = np.cos(2 * np.pi * df['Month'] / 12)\n    df['Day_sin'] = np.sin(2 * np.pi * df['Day'] / 31)  \n    df['Day_cos'] = np.cos(2 * np.pi * df['Day'] / 31)\n    df['Group']=(df['year']-2020)*48+df['Month']*4+df['Day']//7\n\ndef encode_dates(df:pd.DataFrame, df_test: pd.DataFrame):\n    _feat_eng_date(df)\n    _feat_eng_date(df_test)\n    # df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    # df['year'] = df['Policy Start Date'].dt.year\n    df_test['Policy Start Date'] = pd.to_datetime(df_test['Policy Start Date'])\n    minday = df['Policy Start Date'].min()\n    df['days_elapsed'] = (df['Policy Start Date'] -  minday).dt.days\n    df_test['days_elapsed'] = (df_test['Policy Start Date'] -  minday).dt.days\n\n    return df, df_test","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-31T17:06:59.404054Z","iopub.execute_input":"2024-12-31T17:06:59.404332Z","iopub.status.idle":"2024-12-31T17:06:59.435714Z","shell.execute_reply.started":"2024-12-31T17:06:59.404311Z","shell.execute_reply":"2024-12-31T17:06:59.434882Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feat_preproc(df:pd.DataFrame, df_test:pd.DataFrame, cat_cols)->pd.DataFrame:\n    try:\n        df, df_test = df.copy(), df_test.copy()\n        df, df_test = encode_dates(df, df_test)\n        df, df_test = encode_nans(df, df_test)\n        df = enocde_categoricals(df, cat_cols)\n        df_test = enocde_categoricals(df_test, cat_cols)\n        return df, df_test\n    except Exception as e:\n        print(f\"Error with base preprocessing on df of shape {df.shape} and {df_test.shape}\")\n        print(df.head())\n        print(df.columns)\n        print(\"=========================================================\")\n        print(df_test.head())\n        print(df_test.columns)\n        print(\"=========================================================\")\n        raise e\n\nnum_cols = ['Age', 'Annual Income', 'Health Score', 'Previous Claims', 'Vehicle Age', 'Credit Score',\n            'Insurance Duration', 'days_elapsed']\ncat_cols = ['Gender', 'Marital Status', 'Number of Dependents', 'Education Level', 'Occupation',\n            'Location','Policy Type', 'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type', 'year']\nnan_cols = ['Number of Dependents_is_na', 'Occupation_is_na', 'Previous Claims_is_na', \n            'Credit Score_is_na']\nohe_cols = ['Marital Status', 'Occupation', 'Property Type']\nord_cols = ['Gender', 'Number of Dependents', 'Education Level', 'Location', 'Policy Type',\n            'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'year']\n\ndf, df_test = feat_preproc(df, df_test,  cat_cols)\nprint(df.shape, df_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T17:07:04.337103Z","iopub.execute_input":"2024-12-31T17:07:04.337408Z","iopub.status.idle":"2024-12-31T17:07:09.192357Z","shell.execute_reply.started":"2024-12-31T17:07:04.337383Z","shell.execute_reply":"2024-12-31T17:07:09.191532Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### <div style=\"text-align: center; font-family: 'Roboto', sans-serif; color: rgba(109, 146, 255, 1.0); background-color: rgba(219, 36, 255, 0.3); padding: 10px; border-style: solid; border-radius: 10px;\"> Preproc Pipelines </div>","metadata":{}},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import RobustScaler, StandardScaler, OneHotEncoder, OrdinalEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import FunctionTransformer\n\ntime_vars = ['Day', 'Month', 'Day_of_week', 'Week', 'Year_sin', 'Year_cos',\n             'Month_sin', 'Month_cos', 'Day_sin', 'Day_cos', 'Group']\n\nbase_preproc = Pipeline([\n    ('select_feat', ColumnTransformer([\n        ('select', 'passthrough', num_cols+cat_cols)\n    ], verbose_feature_names_out=False).set_output(transform='pandas')),\n    ('imp', SimpleImputer(strategy='constant', fill_value=-1))\n])\n\ncat_preproc = Pipeline([\n    ('make_object', ColumnTransformer([\n        ('select', 'passthrough', num_cols),\n        ('make_object', FunctionTransformer(\n            lambda x: x.astype(str)).set_output(transform='pandas'), cat_cols),\n    ], verbose_feature_names_out=False).set_output(transform='pandas')),\n    ('imp', SimpleImputer(strategy='constant', fill_value=-1).set_output(transform='pandas'))\n])\n\npreproc1 = Pipeline([\n    ('base', base_preproc),\n    ('scaler', StandardScaler()),\n])\n\npreproc2 = Pipeline([\n    ('ohe-all', ColumnTransformer([\n        ('imp', OneHotEncoder(), cat_cols),\n        ('select', 'passthrough', num_cols),\n    ])),\n    ('imp', SimpleImputer(strategy='constant', fill_value=-1)),\n    ('scaler', StandardScaler()),\n])\n\npreproc3 = Pipeline([\n    ('num-only', ColumnTransformer([\n        ('select', 'passthrough', num_cols),\n    ])),\n    ('imp', SimpleImputer(strategy='mean')),\n    ('scaler', RobustScaler()),\n])\n\npreproc4 = Pipeline([\n    ('select', ColumnTransformer([\n        ('select', 'passthrough', num_cols+cat_cols+nan_cols),\n    ])),\n    ('imp', SimpleImputer(strategy='mean'))\n])\n\npreproc5 = Pipeline([\n    ('select', ColumnTransformer([\n        ('select', 'passthrough', num_cols+cat_cols+nan_cols+time_vars),\n    ])),\n    ('imp', SimpleImputer(strategy='mean'))\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T17:07:14.63241Z","iopub.execute_input":"2024-12-31T17:07:14.632713Z","iopub.status.idle":"2024-12-31T17:07:14.965441Z","shell.execute_reply.started":"2024-12-31T17:07:14.632689Z","shell.execute_reply":"2024-12-31T17:07:14.964572Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <div style=\"width: 100%\"><h1 style=\"text-align: center; font-family: 'Roboto', sans-serif; color: rgb(219, 36, 255); background-color: rgba(73, 182, 255, 0.6); padding: 30px; border: 5px solid rgb(219, 36, 255); border-style: solid; border-radius: 10px;\"> Baseline </h1></div>","metadata":{"execution":{"iopub.status.busy":"2024-12-08T04:50:49.966534Z","iopub.execute_input":"2024-12-08T04:50:49.966953Z","iopub.status.idle":"2024-12-08T04:50:50.008757Z","shell.execute_reply.started":"2024-12-08T04:50:49.966919Z","shell.execute_reply":"2024-12-08T04:50:50.007662Z"}}},{"cell_type":"code","source":"import optuna\nfrom sklearn.linear_model import LinearRegression, Ridge\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import BaggingRegressor, StackingRegressor, VotingRegressor\nfrom sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.pipeline import make_pipeline\n\nfrom sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error\nimport torch\nif torch.cuda.is_available():\n    from cuml.svm import SVR                \nelse:\n    from sklearn.svm import SVR\n\nfrom tqdm import tqdm\nfrom sklearn.model_selection import KFold, cross_val_score, StratifiedKFold\nimport copy\nimport os","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T17:08:58.266249Z","iopub.execute_input":"2024-12-31T17:08:58.26656Z","iopub.status.idle":"2024-12-31T17:08:58.313443Z","shell.execute_reply.started":"2024-12-31T17:08:58.266539Z","shell.execute_reply":"2024-12-31T17:08:58.312586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_and_evaluate_regression_model(model, X, y, X_test=None, cv=5, name=None, model_name=None):\n    \"\"\"\n    Train and evaluate a regression model using cross-validation.\n\n    Args:\n        model: The model to train and evaluate.\n        X: Features for training and evaluation.\n        y: Target labels.\n        X_test: Optional test set for predictions after cross-validation.\n        cv: Number of cross-validation folds.\n        name: Optional name of the model for display.\n\n    Returns:\n        metrics: Dictionary containing metrics for all folds.\n        oof_preds: Out-of-fold predictions for training data.\n        test_preds: Predictions on X_test if provided.\n    \"\"\"\n    folds = KFold(n_splits=cv, shuffle=True, random_state=42).split(X, y)\n\n    metrics = {\n        'rmse': [],\n        'mae': [],\n        'r2': [],\n    }\n    \n    if model_name is None:\n        if type(model) is Pipeline:\n            model_name = model[-1].__class__.__name__\n        else:\n            model_name = model.__class__.__name__\n    if name is None:\n        name = model_name\n\n    print(f\"{name} ( {clrd(model_name, 'status')} )\")\n    for train_index, test_index in tqdm(folds, desc=f\"Fitting {cv} folds on {model_name}\"):\n        if isinstance(X, pd.DataFrame):\n            X_train, X_valid = X.iloc[train_index], X.iloc[test_index]\n        else:  \n            X_train, X_valid = X[train_index], X[test_index]\n        if isinstance(y, pd.Series):\n            y_train, y_valid = y.iloc[train_index], y.iloc[test_index]\n        else:  \n            y_train, y_valid = y[train_index], y[test_index]\n\n        model_fold = copy.deepcopy(model)\n        model_fold.fit(X_train, y_train)\n\n        y_pred = model_fold.predict(X_valid)\n\n        metrics['rmse'].append(np.sqrt(mean_squared_error(y_valid, y_pred))  )\n        metrics['mae'].append(mean_absolute_error(y_valid, y_pred)  )\n        metrics['r2'].append(r2_score(y_valid, y_pred)  )\n\n    for k, v in metrics.items():\n        metrics[k] = np.mean(v)\n\n    model.fit(X, y)\n    print(f\"{clrd('rmse ', 'ok')} : {  metrics['rmse']:.4f}   {clrd('mae', 'log')}: {metrics['mae']:.4f}   {clrd('r2', 'log')}: {metrics['r2']:.4f}\")\n    print('-'*50)\n\ndef fit_and_evaluate_models(models, df, df_test=None, DEBUG=False)->None:\n    for name, model_pipe in models.items():\n        try:\n            train_and_evaluate_regression_model(model_pipe, df, df[TARGET], name=name)\n        except Exception as e:\n            print(clrd(f\"Error evaluating model\", 'error'))\n            if DEBUG:\n                raise e\n\nmodels = {}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T17:09:11.744389Z","iopub.execute_input":"2024-12-31T17:09:11.744716Z","iopub.status.idle":"2024-12-31T17:09:11.78835Z","shell.execute_reply.started":"2024-12-31T17:09:11.744687Z","shell.execute_reply":"2024-12-31T17:09:11.787589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"models.update({\n    'gbr1': make_pipeline(preproc4, GradientBoostingRegressor()),\n    'xgbr1': make_pipeline(preproc4, XGBRegressor(verbose=0)),\n    'lgbm1': make_pipeline(preproc4, LGBMRegressor(verbosity=-1)),\n    'cb3': make_pipeline(preproc4, CatBoostRegressor(verbose=False)),\n})\n\nfor name, model_pipe in models.items():\n    try:\n        train_and_evaluate_regression_model(model_pipe, df, df[TARGET], name=name)\n    except Exception as e:\n        print(clrd(f\"Error evaluating model\", 'error'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T05:50:17.949165Z","iopub.execute_input":"2024-12-10T05:50:17.949577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred1 = models['lgbm1'].predict(df_test)\npred2 = models['cb3'].predict(df_test)\npred3 = models['xgbr1'].predict(df_test)\npred = 0.72*pred1 + 0.16*pred1 + 0.12*pred3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T08:35:38.177981Z","iopub.execute_input":"2024-12-10T08:35:38.178376Z","iopub.status.idle":"2024-12-10T08:35:38.223652Z","shell.execute_reply.started":"2024-12-10T08:35:38.178343Z","shell.execute_reply":"2024-12-10T08:35:38.222622Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### <div style=\"text-align: center; font-family: 'Roboto', sans-serif; color: rgba(109, 146, 255, 1.0); background-color: rgba(219, 36, 255, 0.3); padding: 10px; border-style: solid; border-radius: 10px;\"> LGBM </div>","metadata":{}},{"cell_type":"code","source":"def objective(trial):\n    params = {\n        \"learning_rate\": trial.suggest_loguniform(\"learning_rate\", 0.005, 0.3),\n        \"n_estimators\": trial.suggest_int(\"n_estimators\", 50, 500),\n        \"feature_fraction\": trial.suggest_float(\"feature_fraction\", 0.4, 1.0),\n        \"bagging_freq\": trial.suggest_int(\"bagging_freq\", 0, 7),\n        \"min_child_samples\": trial.suggest_int(\"min_child_samples\", 5, 100),\n        \"subsample_for_bin\": trial.suggest_int(\"subsample_for_bin\", 50000, 1000000),\n        \"min_split_gain\": trial.suggest_float(\"min_split_gain\", 0.0, 0.3),\n        \"min_child_weight\": trial.suggest_loguniform(\"min_child_weight\", 0.0001, 0.1),\n        \"subsample\": trial.suggest_float(\"subsample\", 0.6, 1.0),\n        \"subsample_freq\": trial.suggest_int(\"subsample_freq\", 0, 7),\n        \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.5, 1.0),\n    }\n\n    _use_custom_depth = trial.suggest_categorical(\"_use_custom_depth\", [False, True])\n    _use_regularization = trial.suggest_categorical(\"_use_regularization\", [False, True])\n    \n    if _use_custom_depth:\n        params[\"max_depth\"] = trial.suggest_int(\"max_depth\", 1, 20)\n        params[\"num_leaves\"] = trial.suggest_int(\"num_leaves\", 2, min(256, 2**params[\"max_depth\"]))\n    if _use_regularization:\n        params[\"lambda_l1\"] = trial.suggest_float(\"lambda_l1\", 1e-9, 10.0, log=True)\n        params[\"lambda_l2\"] = trial.suggest_float(\"lambda_l2\", 1e-9, 10.0, log=True)\n\n    if params[\"bagging_freq\"] > 0:\n        params[\"bagging_fraction\"] = trial.suggest_float(\"bagging_fraction\", 0.4, 1.0)\n\n    try:\n        metrics = {\"rmse\": []}\n        X, y = df, df[TARGET]\n        folds = KFold(n_splits=5, shuffle=True, random_state=42).split(X, y)\n        for train_index, test_index in folds:        \n            X_train, X_valid = X.iloc[train_index], X.iloc[test_index]\n            y_train, y_valid = y.iloc[train_index], y.iloc[test_index]\n            model_trial = LGBMRegressor(\n                eval_metric= \"rmse\",\n                verbosity=-1,\n                **params,\n            )\n            trial_pipe = make_pipeline(preproc4, model_trial)\n            trial_pipe.fit(X_train, y_train)\n            y_pred =  trial_pipe.predict(X_valid)\n            metrics[\"rmse\"].append(np.sqrt(mean_squared_error(y_valid, y_pred)))\n\n        return np.mean(metrics[\"rmse\"])\n    except Exception as e:\n        return 0.001","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params1 = {'learning_rate': 0.027786857053511582, 'n_estimators': 485, 'feature_fraction': 0.9982956311629986, 'bagging_freq': 0, 'min_child_samples': 41, 'subsample_for_bin': 61890, 'min_split_gain': 0.1839535418898852, 'min_child_weight': 0.004640531400929176, 'subsample': 0.7271295816046559, 'subsample_freq': 3, 'colsample_bytree': 0.7055809400505509, 'max_depth': 14, 'num_leaves': 92}\nparams2 = {'learning_rate': 0.021540191995084293, 'n_estimators': 452, 'feature_fraction': 0.905724211458762, 'bagging_freq': 3, 'min_child_samples': 18, 'subsample_for_bin': 252505, 'min_split_gain': 0.0005567535443960353, 'min_child_weight': 0.018844650323861564, 'subsample': 0.6638313473000301, 'subsample_freq': 7, 'colsample_bytree': 0.7347493248482618, 'max_depth': 16, 'num_leaves': 140, 'lambda_l1': 4.6586705828191044e-09, 'lambda_l2': 0.5698678568963467, 'bagging_fraction': 0.7201403676948238}\n\nmodels.update({\n    \"lgbm3\":  make_pipeline(preproc4, LGBMRegressor(eval_metric= \"rmse\",\n                                                    verbosity=-1, **params1)),\n    \"lgbm4\":  make_pipeline(preproc4, LGBMRegressor(eval_metric= \"rmse\",\n                                                    verbosity=-1, **params2)),\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T17:18:42.863433Z","iopub.execute_input":"2024-12-31T17:18:42.863704Z","iopub.status.idle":"2024-12-31T17:18:42.901226Z","shell.execute_reply.started":"2024-12-31T17:18:42.863683Z","shell.execute_reply":"2024-12-31T17:18:42.900335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fit_and_evaluate_models(models, df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T02:50:50.313768Z","iopub.execute_input":"2024-12-23T02:50:50.314245Z","iopub.status.idle":"2024-12-23T03:05:15.05385Z","shell.execute_reply.started":"2024-12-23T02:50:50.314204Z","shell.execute_reply":"2024-12-23T03:05:15.052308Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### <div style=\"text-align: center; font-family: 'Roboto', sans-serif; color: rgba(109, 146, 255, 1.0); background-color: rgba(219, 36, 255, 0.3); padding: 10px; border-style: solid; border-radius: 10px;\"> XGBoost </div>","metadata":{}},{"cell_type":"code","source":"def objective(trial):\n    params = {\n        \"eta\": trial.suggest_loguniform(\"eta\", 1e-3, 0.35), #learning_rate\n        \"gamma\": trial.suggest_loguniform(\"gamma\", 1e-6, 1.0),\n        \"max_depth\": trial.suggest_int(\"max_depth\", 3, 20),\n        \"min_child_weight\": trial.suggest_int(\"min_child_weight\", 1, 300),\n        \"max_delta_step\": trial.suggest_int(\"max_delta_step\", 0, 10),\n        \"subsample\": trial.suggest_float(\"subsample\", 0.5, 1.0),\n        \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.3, 1.0),\n        \"colsample_bylevel\": trial.suggest_float(\"colsample_bylevel\", 0.3, 1.0),\n        'lambda': trial.suggest_loguniform('lambda', 1e-3, 10.0),\n        'alpha': trial.suggest_loguniform('alpha', 1e-3, 10.0),\n        'n_estimators': trial.suggest_int('n_estimators', 500, 5000),\n    }\n    \n    try:\n        metrics = {\"rmse\": []}\n        X, y = df, df[TARGET]\n        folds = KFold(n_splits=5, shuffle=True, random_state=42).split(X, y)\n        for train_index, test_index in folds:        \n            X_train, X_valid = X.iloc[train_index], X.iloc[test_index]\n            y_train, y_valid = y.iloc[train_index], y.iloc[test_index]\n            model_trial = XGBRegressor(\n                eval_metric= \"rmse\",\n                verbose=0,\n                **params,\n            )\n            trial_pipe = make_pipeline(preproc4, model_trial)\n            trial_pipe.fit(X_train, y_train)\n            y_pred =  trial_pipe.predict(X_valid)\n            metrics[\"rmse\"].append(np.sqrt(mean_squared_error(y_valid, y_pred)))\n\n        return np.mean(metrics[\"rmse\"])\n    except Exception as e:\n        return 0.001","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params1 = {'eta': 0.0017744497628967722, 'gamma': 0.0015001738084495292, 'max_depth': 17, 'min_child_weight': 108, 'max_delta_step': 5, 'subsample': 0.5359765640004988, 'colsample_bytree': 0.9273668369613144, 'colsample_bylevel': 0.4950897889801614, 'lambda': 0.00900121577635916, 'alpha': 0.0009970166537485116, 'n_estimators': 2405}\nparams2 = {'booster': 'gbtree', 'lambda': 0.00265211599854828, 'alpha': 2.434889653648006e-05, 'subsample': 0.8178477253822083, 'colsample_bytree': 0.9084817815609364, 'max_depth': 7, 'min_child_weight': 5, 'eta': 0.1258495736818396, 'gamma': 0.395087653517694, 'grow_policy': 'depthwise'}\nmodels.update({\n    'xgb2': make_pipeline(preproc4, XGBRegressor(eval_metric= \"rmse\",\n                                                    verbose=0, **params1)),\n    'xgb3': make_pipeline(preproc4, XGBRegressor(eval_metric= \"rmse\",\n                                                    verbose=0, **params2)),\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T17:18:48.120273Z","iopub.execute_input":"2024-12-31T17:18:48.120604Z","iopub.status.idle":"2024-12-31T17:18:48.159464Z","shell.execute_reply.started":"2024-12-31T17:18:48.120575Z","shell.execute_reply":"2024-12-31T17:18:48.158588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fit_and_evaluate_models(models, df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T04:48:03.721868Z","iopub.execute_input":"2024-12-27T04:48:03.722317Z","iopub.status.idle":"2024-12-27T05:30:13.725592Z","shell.execute_reply.started":"2024-12-27T04:48:03.722276Z","shell.execute_reply":"2024-12-27T05:30:13.724024Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### <div style=\"text-align: center; font-family: 'Roboto', sans-serif; color: rgba(109, 146, 255, 1.0); background-color: rgba(219, 36, 255, 0.3); padding: 10px; border-style: solid; border-radius: 10px;\"> CatBoost </div>","metadata":{}},{"cell_type":"code","source":"def objective(trial):\n    params = {\n        \"depth\": trial.suggest_int(\"depth\", 1, 12),\n        \"boosting_type\": trial.suggest_categorical(\"boosting_type\", [\"Ordered\", \"Plain\"]),\n        \"bootstrap_type\": trial.suggest_categorical(\n            \"bootstrap_type\", [\"Bayesian\", \"Bernoulli\", \"MVS\"]\n        ),\n\n        \"iterations\":trial.suggest_int(\"iterations\", 100, 1000),\n        'l2_leaf_reg': trial.suggest_loguniform('l2_leaf_reg', 1e-3, 10.0),\n        'max_bin': trial.suggest_int('max_bin', 200, 400),\n        'learning_rate': trial.suggest_float(\"learning_rate\", 1e-3, 1e-1, log=True),\n        'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 1, 300),\n        'random_strength': trial.suggest_float(\"random_strength\", 1e-8, 10.0, log=True),\n    }\n\n    if params[\"bootstrap_type\"] == \"Bayesian\":\n        params[\"bagging_temperature\"] = trial.suggest_float(\"bagging_temperature\", 0, 10)\n    elif params[\"bootstrap_type\"] == \"Bernoulli\":\n        params[\"subsample\"] = trial.suggest_float(\"subsample\", 0.1, 1)\n\n    try:\n        metrics = {\"rmse\": []}\n        X, y = df, df[TARGET]\n        folds = KFold(n_splits=3, shuffle=True, random_state=42).split(X, y)\n        for train_index, test_index in folds:        \n            X_train, X_valid = X.iloc[train_index], X.iloc[test_index]\n            y_train, y_valid = y.iloc[train_index], y.iloc[test_index]\n            model_trial = CatBoostRegressor(\n                loss_function= \"RMSE\",\n                task_type='GPU',\n                verbose=0,\n                **params,\n            )\n            trial_pipe = make_pipeline(preproc4, model_trial)\n            trial_pipe.fit(X_train, y_train)\n            y_pred =  trial_pipe.predict(X_valid)\n            metrics[\"rmse\"].append(np.sqrt(mean_squared_error(y_valid, y_pred)))\n\n        return np.mean(metrics[\"rmse\"])\n    except Exception as e:\n        return 0.001","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params1 = {'depth': 12, 'boosting_type': 'Ordered', 'bootstrap_type': 'Bernoulli', 'iterations': 954, 'l2_leaf_reg': 0.008487675094469389, 'max_bin': 281, 'learning_rate': 0.057599842487784304, 'min_data_in_leaf': 257, 'random_strength': 0.0066110992258185735, 'subsample': 0.5180732137983772}\nparams2 = {'depth': 10, 'boosting_type': 'Ordered', 'bootstrap_type': 'Bernoulli', 'iterations': 522, 'l2_leaf_reg': 2.174993484363667, 'max_bin': 264, 'learning_rate': 0.08328400775397035, 'min_data_in_leaf': 156, 'random_strength': 0.000947515685813026, 'subsample': 0.7579693803966996}\n\nmodels.update({\n    'cb2': make_pipeline(preproc4, CatBoostRegressor(loss_function= \"RMSE\", task_type='GPU', verbose=0, **params1)),\n    'cb3': make_pipeline(preproc4, CatBoostRegressor(loss_function= \"RMSE\", task_type='GPU', verbose=0, **params2)),\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T17:19:10.636456Z","iopub.execute_input":"2024-12-31T17:19:10.636728Z","iopub.status.idle":"2024-12-31T17:19:10.674805Z","shell.execute_reply.started":"2024-12-31T17:19:10.636707Z","shell.execute_reply":"2024-12-31T17:19:10.674119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fit_and_evaluate_models(models, df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T09:04:46.553988Z","iopub.execute_input":"2024-12-31T09:04:46.554376Z","iopub.status.idle":"2024-12-31T13:15:53.49191Z","shell.execute_reply.started":"2024-12-31T09:04:46.554345Z","shell.execute_reply":"2024-12-31T13:15:53.489102Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### <div style=\"text-align: center; font-family: 'Roboto', sans-serif; color: rgba(109, 146, 255, 1.0); background-color: rgba(219, 36, 255, 0.3); padding: 10px; border-style: solid; border-radius: 10px;\"> Autogluon </div>","metadata":{}},{"cell_type":"code","source":"!pip install -q autogluon.tabular[all]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from autogluon.tabular import TabularDataset, TabularPredictor\n\ndf.to_csv(os.path.join(r\"/kaggle/working/\", \"tabular_data.csv\"))\ntrain_data = TabularDataset(os.path.join(r\"/kaggle/working/\", \"tabular_data.csv\"))\ndf_test.to_csv(os.path.join(r\"/kaggle/working/\", \"tabular_data.csv\"))\ntest_data = TabularDataset(os.path.join(r\"/kaggle/working/\", \"tabular_data.csv\"))\n\ntrain_data.shape, test_data.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T16:38:55.743807Z","iopub.execute_input":"2024-12-23T16:38:55.744326Z","iopub.status.idle":"2024-12-23T16:40:02.558439Z","shell.execute_reply.started":"2024-12-23T16:38:55.744284Z","shell.execute_reply":"2024-12-23T16:40:02.556933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictor = TabularPredictor(\n    label=TARGET,\n    eval_metric='rmse'  \n).fit(\n    train_data=train_data,\n    time_limit=7 * 3600,       #7 hrs\n    presets='best_quality',     \n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <div style=\"width: 100%\"><h1 style=\"text-align: center; font-family: 'Roboto', sans-serif; color: rgb(219, 36, 255); background-color: rgba(73, 182, 255, 0.6); padding: 30px; border: 5px solid rgb(219, 36, 255); border-style: solid; border-radius: 10px;\"> Ensembles </h1></div>","metadata":{}},{"cell_type":"code","source":"stk1 = StackingRegressor(estimators = [\n    ('lgbm', models['lgbm4']),\n    ('xgb', models['xgb3']),\n    ('cb', models['cb3']),\n])\nstk1.fit(df, df[TARGET])\n\nstk2 = StackingRegressor(estimators = [\n    ('lgbm', models['lgbm4']),\n    ('xgb', models['xgb2']),\n    ('cb', models['lgbm3']),\n])\n\nstk2.fit(df, df[TARGET])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T17:20:09.037845Z","iopub.execute_input":"2024-12-31T17:20:09.038192Z","iopub.status.idle":"2024-12-31T17:28:18.78116Z","shell.execute_reply.started":"2024-12-31T17:20:09.038164Z","shell.execute_reply":"2024-12-31T17:28:18.78026Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <div style=\"width: 100%\"><h1 style=\"text-align: center; font-family: 'Roboto', sans-serif; color: rgb(219, 36, 255); background-color: rgba(73, 182, 255, 0.6); padding: 30px; border: 5px solid rgb(219, 36, 255); border-style: solid; border-radius: 10px;\"> Blended Preds </h1></div>","metadata":{}},{"cell_type":"code","source":"def blend_preds(pred_paths, weights, target)->np.ndarray:\n    assert len(pred_paths)==len(weights), \":/\"\n    blended_preds = None\n    for path, weight in zip(pred_paths, weights):\n        df = pd.read_csv(path)        \n        preds = df[target].to_numpy()\n        if blended_preds is None:\n            blended_preds = preds * weight\n        else:\n            blended_preds += preds * weight\n    \n    return blended_preds/np.sum(weights)\n\nblend3 = blend_preds([ 'autogluon.csv', 'ensemble1.csv', 'ensemble2.csv', 'ensemble3.csv'],\n           [0.72, 0.10, 0.05, 0.13], TARGET)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <div style=\"width: 100%\"><h1 style=\"text-align: center; font-family: 'Roboto', sans-serif; color: rgb(219, 36, 255); background-color: rgba(73, 182, 255, 0.6); padding: 30px; border: 5px solid rgb(219, 36, 255); border-style: solid; border-radius: 10px;\"> Final Prediction </h1></div>","metadata":{}},{"cell_type":"code","source":"df_sub = df_sample.copy()\ndf_sub[TARGET] = blend3\ndf_sub[TARGET] = np.expm1(df_sub[TARGET])\ndf_sub.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T18:24:41.007616Z","iopub.execute_input":"2024-12-31T18:24:41.007937Z","iopub.status.idle":"2024-12-31T18:24:42.336225Z","shell.execute_reply.started":"2024-12-31T18:24:41.007909Z","shell.execute_reply":"2024-12-31T18:24:42.335301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(15, 6))\nsns.histplot(pred2, palette=SNS_CMAP, ax=ax[0])\nsns.histplot(pred3, palette=SNS_CMAP, ax=ax[1])\nplt.title(\"Prediction Distribution\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T18:19:16.504119Z","iopub.execute_input":"2024-12-31T18:19:16.504425Z","iopub.status.idle":"2024-12-31T18:19:20.874985Z","shell.execute_reply.started":"2024-12-31T18:19:16.5044Z","shell.execute_reply":"2024-12-31T18:19:20.874101Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null}]}