{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30805,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:30:10.956782Z","iopub.execute_input":"2024-12-09T07:30:10.957094Z","iopub.status.idle":"2024-12-09T07:30:11.933376Z","shell.execute_reply.started":"2024-12-09T07:30:10.957064Z","shell.execute_reply":"2024-12-09T07:30:11.932535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q scikit-learn==1.5.2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:30:11.935041Z","iopub.execute_input":"2024-12-09T07:30:11.935842Z","iopub.status.idle":"2024-12-09T07:30:25.086144Z","shell.execute_reply.started":"2024-12-09T07:30:11.935797Z","shell.execute_reply":"2024-12-09T07:30:25.085273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sklearn\nsklearn.__version__","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:30:25.087773Z","iopub.execute_input":"2024-12-09T07:30:25.088171Z","iopub.status.idle":"2024-12-09T07:30:25.493759Z","shell.execute_reply.started":"2024-12-09T07:30:25.088127Z","shell.execute_reply":"2024-12-09T07:30:25.492884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.gridspec as gridspec\nfrom sklearn.cluster import KMeans\nfrom sklearn.model_selection import cross_val_score\nfrom xgboost import XGBRegressor\nimport lightgbm as lgb\nfrom sklearn.impute import SimpleImputer\nimport seaborn as sns\nimport numpy as np\nfrom scipy.stats import norm\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom scipy import stats\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import root_mean_squared_log_error,mean_squared_error, mean_absolute_error, r2_score\nimport optuna\nfrom optuna.samplers import TPESampler\nfrom optuna.samplers import RandomSampler\nimport shap\nimport warnings\nwarnings.filterwarnings('ignore')\n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:30:25.496435Z","iopub.execute_input":"2024-12-09T07:30:25.497134Z","iopub.status.idle":"2024-12-09T07:30:32.651126Z","shell.execute_reply.started":"2024-12-09T07:30:25.497104Z","shell.execute_reply":"2024-12-09T07:30:32.65041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndf_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\nids = df_test['id'].values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:30:32.652209Z","iopub.execute_input":"2024-12-09T07:30:32.652926Z","iopub.status.idle":"2024-12-09T07:30:40.846078Z","shell.execute_reply.started":"2024-12-09T07:30:32.652873Z","shell.execute_reply":"2024-12-09T07:30:40.845321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option('display.float_format', lambda x: '%.2f' % x)\ndf_train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:30:40.847112Z","iopub.execute_input":"2024-12-09T07:30:40.847388Z","iopub.status.idle":"2024-12-09T07:30:41.446312Z","shell.execute_reply.started":"2024-12-09T07:30:40.847362Z","shell.execute_reply":"2024-12-09T07:30:41.44541Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:30:41.447393Z","iopub.execute_input":"2024-12-09T07:30:41.447691Z","iopub.status.idle":"2024-12-09T07:30:41.453224Z","shell.execute_reply.started":"2024-12-09T07:30:41.447663Z","shell.execute_reply":"2024-12-09T07:30:41.45238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['Premium Amount'].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:30:41.454442Z","iopub.execute_input":"2024-12-09T07:30:41.454801Z","iopub.status.idle":"2024-12-09T07:30:41.513504Z","shell.execute_reply.started":"2024-12-09T07:30:41.454764Z","shell.execute_reply":"2024-12-09T07:30:41.512571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.distplot(df_train['Premium Amount'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:30:41.514576Z","iopub.execute_input":"2024-12-09T07:30:41.514851Z","iopub.status.idle":"2024-12-09T07:30:46.000288Z","shell.execute_reply.started":"2024-12-09T07:30:41.514824Z","shell.execute_reply":"2024-12-09T07:30:45.999356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:30:46.002872Z","iopub.execute_input":"2024-12-09T07:30:46.003165Z","iopub.status.idle":"2024-12-09T07:30:46.557734Z","shell.execute_reply.started":"2024-12-09T07:30:46.00313Z","shell.execute_reply":"2024-12-09T07:30:46.556835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:30:46.558944Z","iopub.execute_input":"2024-12-09T07:30:46.559315Z","iopub.status.idle":"2024-12-09T07:30:46.577176Z","shell.execute_reply.started":"2024-12-09T07:30:46.559276Z","shell.execute_reply":"2024-12-09T07:30:46.576332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical = [var for var in df_train.columns if df_train[var].dtype=='object']\nnumerical = [var for var in df_train.columns if df_train[var].dtype != 'object' and (var != 'Premium Amount')]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:30:46.578197Z","iopub.execute_input":"2024-12-09T07:30:46.578446Z","iopub.status.idle":"2024-12-09T07:30:46.588077Z","shell.execute_reply.started":"2024-12-09T07:30:46.578421Z","shell.execute_reply":"2024-12-09T07:30:46.58741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_column = 'Premium Amount'\nprint(\"Target Column:\", target_column)\nprint(\"\\nCategorical Columns:\", categorical)\nprint(\"\\nNumerical Columns:\", numerical)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:30:46.588961Z","iopub.execute_input":"2024-12-09T07:30:46.589248Z","iopub.status.idle":"2024-12-09T07:30:46.598968Z","shell.execute_reply.started":"2024-12-09T07:30:46.589222Z","shell.execute_reply":"2024-12-09T07:30:46.598166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filtered_columns = [col for col in categorical if col != 'Policy Start Date']\n\nfig, axes = plt.subplots(len(filtered_columns), 2, figsize=(15, 5 * len(filtered_columns)))\n\nfor i, column in enumerate(filtered_columns):\n \n    sns.countplot(data=df_train, x=column, ax=axes[i, 0], palette='tab10')\n    axes[i, 0].set_title(f'Distribution of {column}', fontsize=14)\n    axes[i, 0].set_xlabel(column, fontsize=12)\n    axes[i, 0].set_ylabel('Count', fontsize=12)\n    sns.despine(ax=axes[i, 0])\n\n  \n    sns.boxplot(data=df_train, x=column, y=target_column, ax=axes[i, 1], palette='tab10')\n    axes[i, 1].set_title(f'{column} vs {target_column}', fontsize=14)\n    axes[i, 1].set_xlabel(column, fontsize=12)\n    axes[i, 1].set_ylabel(target_column, fontsize=12)\n    sns.despine(ax=axes[i, 1])\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:30:46.59998Z","iopub.execute_input":"2024-12-09T07:30:46.60023Z","iopub.status.idle":"2024-12-09T07:31:00.380258Z","shell.execute_reply.started":"2024-12-09T07:30:46.600205Z","shell.execute_reply":"2024-12-09T07:31:00.379377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"palette = sns.color_palette('tab10', len(numerical))\ncolor_dict = dict(zip(numerical, palette))\n\n# Create a grid of subplots for histograms, boxplots, and scatterplots/violin plots\nfig = plt.figure(figsize=(30, 10 * len(numerical)))\ngs = gridspec.GridSpec(2 * len(numerical), 2, figure=fig)\n\ndf_binned = df_train.copy()\n\nfor i, column in enumerate(numerical):\n\n    if df_train[column].nunique() > 50: discrete = False\n    else : discrete = True\n    \n    # Plot histogram with a unique color\n    ax_hist = fig.add_subplot(gs[2 * i, 0])\n    sns.histplot(\n        data=df_train, x=column, fill=True, common_norm=False, alpha=0.6,\n        linewidth=0.8, color=color_dict[column], ax=ax_hist,  discrete = discrete\n    )\n    \n    # Plot boxplot with the same unique color\n    ax_box = fig.add_subplot(gs[2 * i + 1, 0])\n    sns.boxplot(data=df_train, x=column, ax=ax_box, color=color_dict[column])\n    ax_box.set_title(f'{column} vs Target (Boxplot)', fontsize=14)\n    sns.despine(ax=ax_box)\n\n    # Conditional plot: violin plot or barplot based on unique values, fallback to scatterplot\n    ax_conditional = fig.add_subplot(gs[2 * i:2 * i + 2, 1])  # Merges 2 rows\n    if df_train[column].nunique() <= 10:\n        # If the column has 10 or fewer unique values, use a violin plot\n        sns.violinplot(data=df_train, x=column, y=target_column, ax=ax_conditional, color=color_dict[column], alpha=0.6)\n        ax_conditional.set_title(f'{column} vs {target_column} (Violin Plot)', fontsize=14)\n    else:\n        # Bin the column into 10 intervals, but keep original target column values\n        df_binned['Binned Column'] = pd.cut(df_train[column], bins=10)\n        sns.violinplot(data=df_binned, x='Binned Column', y=target_column, ax=ax_conditional, color=color_dict[column], alpha=0.6)\n        ax_conditional.set_title(f'{column} (Binned) vs {target_column} (Violin Plot)', fontsize=14)\n        ax_conditional.set_xlabel(f'{column} (Binned)', fontsize=12)\n\nplt.tight_layout()  # Adjust subplots to fit into the figure area\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:31:00.381304Z","iopub.execute_input":"2024-12-09T07:31:00.381593Z","iopub.status.idle":"2024-12-09T07:31:33.77865Z","shell.execute_reply.started":"2024-12-09T07:31:00.381561Z","shell.execute_reply":"2024-12-09T07:31:33.777291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(15,9))\nplt.title(\"Visualizing Missing Values\")\nsns.heatmap(df_train.isnull(), cbar=False, cmap=sns.color_palette('magma'), yticklabels=False);\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:31:33.779899Z","iopub.execute_input":"2024-12-09T07:31:33.780171Z","iopub.status.idle":"2024-12-09T07:31:53.738415Z","shell.execute_reply.started":"2024-12-09T07:31:33.780146Z","shell.execute_reply":"2024-12-09T07:31:53.737566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the correlation matrix\ncorrelation_matrix = df_train[numerical].corr()\n\n# Plot the heatmap\nplt.figure(figsize=(12, 8))\nsns.heatmap(correlation_matrix, annot=True, fmt=\".2f\", cmap=\"coolwarm\", cbar=True, linewidths=0.5)\nplt.title(\"Correlation Heatmap of Numerical Variables\", fontsize=16)\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:31:53.739568Z","iopub.execute_input":"2024-12-09T07:31:53.739849Z","iopub.status.idle":"2024-12-09T07:31:54.432043Z","shell.execute_reply.started":"2024-12-09T07:31:53.739821Z","shell.execute_reply":"2024-12-09T07:31:54.431155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def date(df):\n\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    df['Year'] = df['Policy Start Date'].dt.year\n    df['Day'] = df['Policy Start Date'].dt.day\n    df['Month'] = df['Policy Start Date'].dt.month\n    df['Month_name'] = df['Policy Start Date'].dt.month_name()\n    df['Day_of_week'] = df['Policy Start Date'].dt.day_name()\n    df['Week'] = df['Policy Start Date'].dt.isocalendar().week\n    min_year = df['Year'].min()\n    max_year = df['Year'].max()\n    df.drop('Policy Start Date', axis=1, inplace=True)\n\n    return df\n\n# Apply the date function to both datasets\ndf_train = date(df_train)\ndf_test = date(df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:31:54.433146Z","iopub.execute_input":"2024-12-09T07:31:54.433426Z","iopub.status.idle":"2024-12-09T07:31:56.727936Z","shell.execute_reply.started":"2024-12-09T07:31:54.433398Z","shell.execute_reply":"2024-12-09T07:31:56.72719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical.remove('Policy Start Date')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:31:56.729024Z","iopub.execute_input":"2024-12-09T07:31:56.729305Z","iopub.status.idle":"2024-12-09T07:31:56.733372Z","shell.execute_reply.started":"2024-12-09T07:31:56.729278Z","shell.execute_reply":"2024-12-09T07:31:56.73258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Shape of training data (num_rows, num_columns)\nprint(df_train[numerical[1:]].shape)\n\n# Number of missing values in each column of training data\nmissing_val_count_by_column = (df_train[numerical[1:]].isnull().sum())\nprint(missing_val_count_by_column[missing_val_count_by_column > 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:31:56.734636Z","iopub.execute_input":"2024-12-09T07:31:56.734872Z","iopub.status.idle":"2024-12-09T07:31:56.819407Z","shell.execute_reply.started":"2024-12-09T07:31:56.734848Z","shell.execute_reply":"2024-12-09T07:31:56.818284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_imputer = SimpleImputer( strategy='median')\ndf_train[numerical[1:]]=num_imputer.fit_transform(df_train[numerical[1:]])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:31:56.820879Z","iopub.execute_input":"2024-12-09T07:31:56.82134Z","iopub.status.idle":"2024-12-09T07:31:57.972924Z","shell.execute_reply.started":"2024-12-09T07:31:56.821292Z","shell.execute_reply":"2024-12-09T07:31:57.972226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Shape of training data (cat_rows, cat_columns)\nprint(df_train[categorical].shape)\n\n# Number of missing values in each column of training data\nmissing_val_count_by_column = (df_train[categorical].isnull().sum())\nprint(missing_val_count_by_column[missing_val_count_by_column > 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:31:57.974033Z","iopub.execute_input":"2024-12-09T07:31:57.974306Z","iopub.status.idle":"2024-12-09T07:31:58.695428Z","shell.execute_reply.started":"2024-12-09T07:31:57.97428Z","shell.execute_reply":"2024-12-09T07:31:58.694568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"obj_imputer = SimpleImputer( strategy='constant', fill_value='Unknown')\ndf_train[categorical] = obj_imputer.fit_transform(df_train[categorical])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:31:58.696598Z","iopub.execute_input":"2024-12-09T07:31:58.696864Z","iopub.status.idle":"2024-12-09T07:32:00.26322Z","shell.execute_reply.started":"2024-12-09T07:31:58.696837Z","shell.execute_reply":"2024-12-09T07:32:00.262466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.drop(columns=['id'], inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:00.264195Z","iopub.execute_input":"2024-12-09T07:32:00.264444Z","iopub.status.idle":"2024-12-09T07:32:00.391199Z","shell.execute_reply.started":"2024-12-09T07:32:00.26442Z","shell.execute_reply":"2024-12-09T07:32:00.390479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test[numerical[1:]]=num_imputer.fit_transform(df_test[numerical[1:]])\ndf_test[categorical]=obj_imputer.fit_transform(df_test[categorical])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:00.392332Z","iopub.execute_input":"2024-12-09T07:32:00.39271Z","iopub.status.idle":"2024-12-09T07:32:02.208365Z","shell.execute_reply.started":"2024-12-09T07:32:00.392671Z","shell.execute_reply":"2024-12-09T07:32:02.2075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:02.209472Z","iopub.execute_input":"2024-12-09T07:32:02.20985Z","iopub.status.idle":"2024-12-09T07:32:02.606177Z","shell.execute_reply.started":"2024-12-09T07:32:02.209812Z","shell.execute_reply":"2024-12-09T07:32:02.605348Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Scaling Features","metadata":{}},{"cell_type":"code","source":"scaler = StandardScaler()\nscaler.fit(df_train[numerical[1:]])\ndf_train[numerical[1:]] = scaler.transform(df_train[numerical[1:]])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:02.607299Z","iopub.execute_input":"2024-12-09T07:32:02.607678Z","iopub.status.idle":"2024-12-09T07:32:02.87727Z","shell.execute_reply.started":"2024-12-09T07:32:02.607638Z","shell.execute_reply":"2024-12-09T07:32:02.876315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler.fit(df_test[numerical[1:]])\ndf_test[numerical[1:]] = scaler.transform(df_test[numerical[1:]])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:02.882759Z","iopub.execute_input":"2024-12-09T07:32:02.883052Z","iopub.status.idle":"2024-12-09T07:32:03.046553Z","shell.execute_reply.started":"2024-12-09T07:32:02.883024Z","shell.execute_reply":"2024-12-09T07:32:03.04556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical = [var for var in df_train.columns if df_train[var].dtype=='object']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:03.047689Z","iopub.execute_input":"2024-12-09T07:32:03.048083Z","iopub.status.idle":"2024-12-09T07:32:03.053469Z","shell.execute_reply.started":"2024-12-09T07:32:03.048043Z","shell.execute_reply":"2024-12-09T07:32:03.052608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"enc = OrdinalEncoder()\ndf_train[categorical] = enc.fit_transform(df_train[categorical])\ndf_test[categorical] = enc.fit_transform(df_test[categorical])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:03.054687Z","iopub.execute_input":"2024-12-09T07:32:03.055067Z","iopub.status.idle":"2024-12-09T07:32:08.329547Z","shell.execute_reply.started":"2024-12-09T07:32:03.055035Z","shell.execute_reply":"2024-12-09T07:32:08.328494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10,10))\n\ncorr_matrix = df_train.corr()\nlower = corr_matrix.where(np.tril(np.ones(corr_matrix.shape), k=-1).astype(bool))\n\nsns.heatmap(lower, annot=True, fmt='.2f', cbar=False, center=0);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:08.331147Z","iopub.execute_input":"2024-12-09T07:32:08.331557Z","iopub.status.idle":"2024-12-09T07:32:11.617625Z","shell.execute_reply.started":"2024-12-09T07:32:08.331493Z","shell.execute_reply":"2024-12-09T07:32:11.61679Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"high_corr = [\n    column for column in lower.columns if any((lower[column] > 0.6)|(lower[column] < -0.6))\n]\nhigh_corr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:11.618784Z","iopub.execute_input":"2024-12-09T07:32:11.619142Z","iopub.status.idle":"2024-12-09T07:32:11.634146Z","shell.execute_reply.started":"2024-12-09T07:32:11.619102Z","shell.execute_reply":"2024-12-09T07:32:11.633133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.drop(columns=['Month'], inplace=True)\ndf_test.drop(columns=['Month'], inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:11.6351Z","iopub.execute_input":"2024-12-09T07:32:11.635348Z","iopub.status.idle":"2024-12-09T07:32:11.794966Z","shell.execute_reply.started":"2024-12-09T07:32:11.635324Z","shell.execute_reply":"2024-12-09T07:32:11.794238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_features = df_train.drop(columns=['Premium Amount'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:11.796051Z","iopub.execute_input":"2024-12-09T07:32:11.796412Z","iopub.status.idle":"2024-12-09T07:32:11.887004Z","shell.execute_reply.started":"2024-12-09T07:32:11.796376Z","shell.execute_reply":"2024-12-09T07:32:11.886058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def outlier_std(data, col, threshold=3):\n    mean = data[col].mean()\n    std = data[col].std()\n    up_bound = mean + threshold * std\n    low_bound = mean - threshold * std\n    anomalies = pd.concat([data[col]>up_bound, data[col]<low_bound], axis=1).any(axis=1)\n    return anomalies, up_bound, low_bound","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:11.888199Z","iopub.execute_input":"2024-12-09T07:32:11.888497Z","iopub.status.idle":"2024-12-09T07:32:11.893872Z","shell.execute_reply.started":"2024-12-09T07:32:11.888468Z","shell.execute_reply":"2024-12-09T07:32:11.893033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_column_outliers(data, columns=None, function=outlier_std, threshold=3):\n    if columns:\n        columns_to_check = columns\n    else:\n        columns_to_check = data.columns\n\n    outliers = pd.Series(data=[False]*len(data), index=data_features.index, name='is_outlier')\n    comparison_table = {}\n    for column in columns_to_check:\n        anomalies, upper_bound, lower_bound = function(data, column, threshold=threshold)\n        comparison_table[column] = [upper_bound, lower_bound, sum(anomalies), 100*sum(anomalies)/len(anomalies)]\n        outliers[anomalies[anomalies].index] = True\n\n    comparison_table = pd.DataFrame(comparison_table).T\n    comparison_table.columns=['upper_bound', 'lower_bound', 'anomalies_count', 'anomalies_percentage']\n    comparison_table = comparison_table.sort_values(by='anomalies_percentage', ascending=False)\n\n    return comparison_table, outliers\n\ndef anomalies_report(outliers):\n    print(\"Total number of outliers: {}\\nPercentage of outliers:   {:.2f}%\".format(\n            sum(outliers), 100*sum(outliers)/len(outliers)))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:11.89496Z","iopub.execute_input":"2024-12-09T07:32:11.895241Z","iopub.status.idle":"2024-12-09T07:32:11.906712Z","shell.execute_reply.started":"2024-12-09T07:32:11.895215Z","shell.execute_reply":"2024-12-09T07:32:11.905979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"comparison_table, std_outliers = get_column_outliers(data_features)\nanomalies_report(std_outliers)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:11.907758Z","iopub.execute_input":"2024-12-09T07:32:11.908015Z","iopub.status.idle":"2024-12-09T07:32:16.637145Z","shell.execute_reply.started":"2024-12-09T07:32:11.907991Z","shell.execute_reply":"2024-12-09T07:32:16.636264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"comparison_table","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:16.638431Z","iopub.execute_input":"2024-12-09T07:32:16.63871Z","iopub.status.idle":"2024-12-09T07:32:16.648632Z","shell.execute_reply.started":"2024-12-09T07:32:16.638683Z","shell.execute_reply":"2024-12-09T07:32:16.647717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['is_outlier'] = std_outliers\ndf_train = df_train.drop(df_train[df_train.is_outlier == True].index).reset_index(drop=True)\ndf_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:16.64979Z","iopub.execute_input":"2024-12-09T07:32:16.650152Z","iopub.status.idle":"2024-12-09T07:32:17.097576Z","shell.execute_reply.started":"2024-12-09T07:32:16.650113Z","shell.execute_reply":"2024-12-09T07:32:17.096693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split train data into features and target\nX = df_train.drop(columns=[target_column, 'id', 'is_outlier'])\ny = df_train[target_column]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:36:51.98004Z","iopub.execute_input":"2024-12-09T07:36:51.980427Z","iopub.status.idle":"2024-12-09T07:36:52.049923Z","shell.execute_reply.started":"2024-12-09T07:36:51.980393Z","shell.execute_reply":"2024-12-09T07:36:52.048922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:36:56.205071Z","iopub.execute_input":"2024-12-09T07:36:56.205429Z","iopub.status.idle":"2024-12-09T07:36:56.451348Z","shell.execute_reply.started":"2024-12-09T07:36:56.205395Z","shell.execute_reply":"2024-12-09T07:36:56.450303Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def objective(trial):\n    # Define parameter search space\n    param = {\n        \"objective\": \"regression\",\n        \"metric\": \"rmse\",\n        \"boosting_type\": trial.suggest_categorical(\"boosting_type\", [\"gbdt\", \"dart\"]),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 200, 512),\n        'min_child_samples': trial.suggest_int('min_child_samples', 5, 1000),\n        \"learning_rate\": trial.suggest_loguniform(\"learning_rate\", 1e-4, 1e-1),\n        'reg_alpha': trial.suggest_loguniform('reg_alpha', 1e-8, 10),\n        'reg_lambda': trial.suggest_loguniform('reg_lambda', 1e-8, 10),\n        \"feature_fraction\": trial.suggest_uniform(\"feature_fraction\", 0.6, 1.0),\n        \"bagging_fraction\": trial.suggest_uniform(\"bagging_fraction\", 0.6, 1.0),\n        \"bagging_freq\": trial.suggest_int(\"bagging_freq\", 5, 12),\n        \"min_data_in_leaf\": trial.suggest_int(\"min_data_in_leaf\", 20, 100),\n        \"max_depth\": trial.suggest_int(\"max_depth\", -1, 16),  # -1 means no limit\n        \"lambda_l1\": trial.suggest_loguniform(\"lambda_l1\", 1e-4, 10.0),\n        \"lambda_l2\": trial.suggest_loguniform(\"lambda_l2\", 1e-4, 10.0),\n        'colsample_bytree': trial.suggest_uniform('colsample_bytree', 0.5, 1),\n        'subsample': trial.suggest_uniform('subsample', 0.5, 1),\n        'subsample_freq': trial.suggest_int('subsample_freq', 0, 10),\n        \"device_type\": \"gpu\",  # Enable GPU support\n        \"seed\" : 42\n\n    }\n\n    # Create a LightGBM dataset\n    dtrain = lgb.Dataset(X_train, label=y_train)\n    dval = lgb.Dataset(X_val, label=y_val, reference=dtrain)\n\n    # Train LightGBM model\n    model = lgb.train(\n        param,\n        dtrain,\n        valid_sets=[dval],\n    )\n\n    # Predict on validation set\n    y_val_pred = model.predict(X_val)\n    \n    # Compute RMSLE using sklearn's root_mean_squared_log_error\n    rmsle = root_mean_squared_log_error(y_val, np.maximum(y_val_pred, 0))\n    return rmsle\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:17.394852Z","iopub.execute_input":"2024-12-09T07:32:17.395112Z","iopub.status.idle":"2024-12-09T07:32:17.40287Z","shell.execute_reply.started":"2024-12-09T07:32:17.395088Z","shell.execute_reply":"2024-12-09T07:32:17.401962Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"study_name = \"LGBM\"\nstudy = optuna.create_study(study_name=study_name,sampler=TPESampler(), pruner=optuna.pruners.HyperbandPruner( min_resource=1, max_resource=100, reduction_factor=3), direction=\"minimize\", load_if_exists=True)\nstudy.optimize(objective, n_trials=100)\nprint(f\"best optimized rmsle: {study.best_value:0.5f}\")\nprint(f\"best hyperparameters: {study.best_params}\")\nlgb_params = study.best_params","metadata":{"execution":{"iopub.status.busy":"2024-12-05T10:03:49.146651Z","iopub.execute_input":"2024-12-05T10:03:49.14735Z","iopub.status.idle":"2024-12-05T10:51:29.437925Z","shell.execute_reply.started":"2024-12-05T10:03:49.147318Z","shell.execute_reply":"2024-12-05T10:51:29.437015Z"},"_kg_hide-input":true}},{"cell_type":"code","source":"best_params = {'boosting_type': 'dart', \n                       'num_leaves': 506,\n                       'min_child_samples': 88, \n                       'learning_rate': 0.023941956540168206, \n                       'reg_alpha': 5.544302771418543e-07,\n                       'reg_lambda': 7.792319838879878e-06,\n                       'feature_fraction': 0.9785381529571701,\n                       'bagging_fraction': 0.6377113808460491,\n                       'bagging_freq': 10,\n                       'min_data_in_leaf': 45,\n                       'max_depth': 14,\n                       'lambda_l1': 0.07576182964120466,\n                       'lambda_l2': 9.86798320013589,\n                       'colsample_bytree': 0.8908685026291246,\n                       'subsample': 0.5636997605277937,\n                       'subsample_freq': 0,\n                        'seed': 42}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:32:17.404079Z","iopub.execute_input":"2024-12-09T07:32:17.404505Z","iopub.status.idle":"2024-12-09T07:32:17.417354Z","shell.execute_reply.started":"2024-12-09T07:32:17.404467Z","shell.execute_reply":"2024-12-09T07:32:17.416666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_model = lgb.train(\n    best_params,\n    lgb.Dataset(X, label=y),\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:37:03.010785Z","iopub.execute_input":"2024-12-09T07:37:03.011134Z","iopub.status.idle":"2024-12-09T07:37:50.397746Z","shell.execute_reply.started":"2024-12-09T07:37:03.011107Z","shell.execute_reply":"2024-12-09T07:37:50.396763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Performance Metrics\ny_pred = final_model.predict(X)\n\n# Calcul des métriques\nrmsle = root_mean_squared_log_error(y, y_pred)\nrmse = np.sqrt(mean_squared_error(y, y_pred))\nmae = mean_absolute_error(y, y_pred)\nr2 = r2_score(y, y_pred)\nmape = np.mean(np.abs((y - y_pred) / y)) * 100\n\n# Display performance metrics\nprint(f\"\\nPerformance Metrics:\\n{'-'*30}\")\nprint(f\"RMSLE: {rmsle:.4f}\")\nprint(f\"RMSE: {rmse:.4f}\")\nprint(f\"MAE: {mae:.4f}\")\nprint(f\"R²: {r2:.4f}\")\nprint(f\"MAPE: {mape:.2f}%\")\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:38:31.779889Z","iopub.execute_input":"2024-12-09T07:38:31.780624Z","iopub.status.idle":"2024-12-09T07:38:38.042481Z","shell.execute_reply.started":"2024-12-09T07:38:31.780591Z","shell.execute_reply":"2024-12-09T07:38:38.041517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. Feature Importance\ndf_feature_importance = (\n    pd.DataFrame({\n        'feature': final_model.feature_name(),\n        'importance': final_model.feature_importance(),\n    })\n    .sort_values('importance', ascending=False)\n)\n\ndf_feature_importance","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:38:42.975043Z","iopub.execute_input":"2024-12-09T07:38:42.975392Z","iopub.status.idle":"2024-12-09T07:38:42.987475Z","shell.execute_reply.started":"2024-12-09T07:38:42.975364Z","shell.execute_reply":"2024-12-09T07:38:42.986684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3. Residual Analysis\nresiduals = y - y_pred\n\n# Residuals vs Predicted Values\nplt.figure(figsize=(12, 6))\nsns.scatterplot(x=y_pred, y=residuals, alpha=0.6, color=\"#007acc\")\nplt.axhline(y=0, color='red', linestyle='--', linewidth=1.5)\nplt.title(\"Residuals vs Predicted Values\", fontsize=16, fontweight='bold')\nplt.xlabel(\"Predicted Values\", fontsize=12)\nplt.ylabel(\"Residuals\", fontsize=12)\nplt.tight_layout()\nplt.show()\n\n# Residual Distribution\nplt.figure(figsize=(10, 6))\nsns.histplot(residuals, bins=30, kde=True, color=\"#55a630\")\nplt.axvline(x=0, color='red', linestyle='--', linewidth=1.5)\nplt.title(\"Distribution of Residuals\", fontsize=16, fontweight='bold')\nplt.xlabel(\"Residuals\", fontsize=12)\nplt.ylabel(\"Frequency\", fontsize=12)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:38:53.104551Z","iopub.execute_input":"2024-12-09T07:38:53.105375Z","iopub.status.idle":"2024-12-09T07:39:00.561701Z","shell.execute_reply.started":"2024-12-09T07:38:53.105327Z","shell.execute_reply":"2024-12-09T07:39:00.560863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make predictions on the test set\ntest_predictions = final_model.predict(df_test, num_iteration=final_model.best_iteration)\n\n# Prepare submission file\nsubmission = pd.DataFrame({'id': ids, 'Premium Amount': test_predictions})\nsubmission.to_csv(\"submission.csv\", index=False)\n\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T07:41:39.929854Z","iopub.execute_input":"2024-12-09T07:41:39.930706Z","iopub.status.idle":"2024-12-09T07:41:45.626327Z","shell.execute_reply.started":"2024-12-09T07:41:39.930671Z","shell.execute_reply":"2024-12-09T07:41:45.625431Z"}},"outputs":[],"execution_count":null}]}