{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:54.89486Z","iopub.execute_input":"2024-12-06T03:19:54.895294Z","iopub.status.idle":"2024-12-06T03:19:55.228727Z","shell.execute_reply.started":"2024-12-06T03:19:54.895247Z","shell.execute_reply":"2024-12-06T03:19:55.227815Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q scikit-learn==1.5.2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:55.230609Z","iopub.execute_input":"2024-12-06T03:19:55.231394Z","iopub.status.idle":"2024-12-06T03:20:08.017229Z","shell.execute_reply.started":"2024-12-06T03:19:55.231351Z","shell.execute_reply":"2024-12-06T03:20:08.016183Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sklearn\nsklearn.__version__","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:20:08.018799Z","iopub.execute_input":"2024-12-06T03:20:08.019183Z","iopub.status.idle":"2024-12-06T03:20:08.435484Z","shell.execute_reply.started":"2024-12-06T03:20:08.01914Z","shell.execute_reply":"2024-12-06T03:20:08.434692Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport matplotlib.gridspec as gridspec\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=UserWarning, module=\"seaborn\")\nwarnings.filterwarnings(\"ignore\", category=FutureWarning, module=\"seaborn\")\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import root_mean_squared_log_error,mean_squared_error, mean_absolute_error, r2_score\n\nimport optuna\nimport lightgbm as lgb\n\nimport torch\nfrom sklearn.pipeline import Pipeline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:20:08.436823Z","iopub.execute_input":"2024-12-06T03:20:08.437443Z","iopub.status.idle":"2024-12-06T03:20:25.537389Z","shell.execute_reply.started":"2024-12-06T03:20:08.437401Z","shell.execute_reply":"2024-12-06T03:20:25.536359Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df=pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest_df=pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:20:25.540495Z","iopub.execute_input":"2024-12-06T03:20:25.541431Z","iopub.status.idle":"2024-12-06T03:20:34.100269Z","shell.execute_reply.started":"2024-12-06T03:20:25.541401Z","shell.execute_reply":"2024-12-06T03:20:34.099491Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check dataset shape and first rows\nprint(f\"Dataset contains {train_df.shape[0]} rows and {train_df.shape[1]} columns.\")\ntrain_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:20:34.101294Z","iopub.execute_input":"2024-12-06T03:20:34.101561Z","iopub.status.idle":"2024-12-06T03:20:34.135463Z","shell.execute_reply.started":"2024-12-06T03:20:34.101535Z","shell.execute_reply":"2024-12-06T03:20:34.134623Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:20:34.13664Z","iopub.execute_input":"2024-12-06T03:20:34.136917Z","iopub.status.idle":"2024-12-06T03:20:34.691102Z","shell.execute_reply.started":"2024-12-06T03:20:34.13689Z","shell.execute_reply":"2024-12-06T03:20:34.690173Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save 'id' column for submission\ntest_ids = test_df['id']\n\n# Define the target column\ntarget_column = 'Premium Amount'\n\n# Select categorical and numerical columns (initial)\ncategorical_columns = train_df.select_dtypes(include=['object']).columns\nnumerical_columns = train_df.select_dtypes(exclude=['object']).columns\n\n# Print out column information\nprint(\"Target Column:\", target_column)\nprint(\"\\nCategorical Columns:\", categorical_columns.tolist())\nprint(\"\\nNumerical Columns:\", numerical_columns.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:20:34.692154Z","iopub.execute_input":"2024-12-06T03:20:34.692454Z","iopub.status.idle":"2024-12-06T03:20:34.800076Z","shell.execute_reply.started":"2024-12-06T03:20:34.692427Z","shell.execute_reply":"2024-12-06T03:20:34.799197Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.describe().round(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:20:34.801144Z","iopub.execute_input":"2024-12-06T03:20:34.801424Z","iopub.status.idle":"2024-12-06T03:20:35.373044Z","shell.execute_reply.started":"2024-12-06T03:20:34.801398Z","shell.execute_reply":"2024-12-06T03:20:35.372248Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for column in categorical_columns:\n    num_unique = train_df[column].nunique()\n    print(f\"'{column}' has {num_unique} unique categories.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:20:35.374381Z","iopub.execute_input":"2024-12-06T03:20:35.375135Z","iopub.status.idle":"2024-12-06T03:20:36.117463Z","shell.execute_reply.started":"2024-12-06T03:20:35.375091Z","shell.execute_reply":"2024-12-06T03:20:36.116487Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print top 10 unique value counts for each categorical column\nfor column in categorical_columns:\n    print(f\"\\nTop value counts in '{column}':\\n{train_df[column].value_counts().head(10)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:20:36.118515Z","iopub.execute_input":"2024-12-06T03:20:36.118788Z","iopub.status.idle":"2024-12-06T03:20:37.200826Z","shell.execute_reply.started":"2024-12-06T03:20:36.118762Z","shell.execute_reply":"2024-12-06T03:20:37.199966Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"The mean of columns:\")\nprint(train_df[numerical_columns].mean())\n\nprint(\"\\nThe std dev of columns:\")\nprint(train_df[numerical_columns].std())\n\nprint(\"\\nThe skewness of columns:\")\nprint(train_df[numerical_columns].skew())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:20:37.202Z","iopub.execute_input":"2024-12-06T03:20:37.202304Z","iopub.status.idle":"2024-12-06T03:20:37.640424Z","shell.execute_reply.started":"2024-12-06T03:20:37.202261Z","shell.execute_reply":"2024-12-06T03:20:37.639446Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(15,9))\nplt.title(\"Visualizing Missing Values\")\nsns.heatmap(train_df.isnull(), cbar=False, cmap=sns.color_palette('magma'), yticklabels=False);\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:20:37.641534Z","iopub.execute_input":"2024-12-06T03:20:37.641768Z","iopub.status.idle":"2024-12-06T03:20:57.810047Z","shell.execute_reply.started":"2024-12-06T03:20:37.641744Z","shell.execute_reply":"2024-12-06T03:20:57.809141Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a color palette for the columns\npalette = sns.color_palette('tab10', len(numerical_columns))\ncolor_dict = dict(zip(numerical_columns, palette))\n\n# Create a grid of subplots for histograms, boxplots, and scatterplots/violin plots\nfig = plt.figure(figsize=(30, 10 * len(numerical_columns)))\ngs = gridspec.GridSpec(2 * len(numerical_columns), 2, figure=fig)\n\ndf_binned = train_df.copy()\n\nfor i, column in enumerate(numerical_columns):\n\n    if train_df[column].nunique() > 50: discrete = False\n    else : discrete = True\n    \n    # Plot histogram with a unique color\n    ax_hist = fig.add_subplot(gs[2 * i, 0])\n    sns.histplot(\n        data=train_df, x=column, fill=True, common_norm=False, alpha=0.6,\n        linewidth=0.8, color=color_dict[column], ax=ax_hist,  discrete = discrete\n    )\n    \n    # Plot boxplot with the same unique color\n    ax_box = fig.add_subplot(gs[2 * i + 1, 0])\n    sns.boxplot(data=train_df, x=column, ax=ax_box, color=color_dict[column])\n    ax_box.set_title(f'{column} vs Target (Boxplot)', fontsize=14)\n    sns.despine(ax=ax_box)\n\n    # Conditional plot: violin plot or barplot based on unique values, fallback to scatterplot\n    ax_conditional = fig.add_subplot(gs[2 * i:2 * i + 2, 1])  # Merges 2 rows\n    if train_df[column].nunique() <= 10:\n        # If the column has 10 or fewer unique values, use a violin plot\n        sns.violinplot(data=train_df, x=column, y=target_column, ax=ax_conditional, color=color_dict[column], alpha=0.6)\n        ax_conditional.set_title(f'{column} vs {target_column} (Violin Plot)', fontsize=14)\n    else:\n        # Bin the column into 10 intervals, but keep original target column values\n        df_binned['Binned Column'] = pd.cut(train_df[column], bins=10)\n        sns.violinplot(data=df_binned, x='Binned Column', y=target_column, ax=ax_conditional, color=color_dict[column], alpha=0.6)\n        ax_conditional.set_title(f'{column} (Binned) vs {target_column} (Violin Plot)', fontsize=14)\n        ax_conditional.set_xlabel(f'{column} (Binned)', fontsize=12)\n\nplt.tight_layout()  # Adjust subplots to fit into the figure area\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:20:57.814572Z","iopub.execute_input":"2024-12-06T03:20:57.814945Z","iopub.status.idle":"2024-12-06T03:21:35.189221Z","shell.execute_reply.started":"2024-12-06T03:20:57.814914Z","shell.execute_reply":"2024-12-06T03:21:35.187829Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filtrer les colonnes catégorielles et exclure 'Policy Start Date'\nfiltered_columns = [col for col in categorical_columns if col != 'Policy Start Date']\n\n# Créer des sous-graphiques pour barplots et boxplots\nfig, axes = plt.subplots(len(filtered_columns), 2, figsize=(15, 5 * len(filtered_columns)))\n\nfor i, column in enumerate(filtered_columns):\n    # Barplot à gauche\n    sns.countplot(data=train_df, x=column, ax=axes[i, 0], palette='tab10')\n    axes[i, 0].set_title(f'Distribution of {column}', fontsize=14)\n    axes[i, 0].set_xlabel(column, fontsize=12)\n    axes[i, 0].set_ylabel('Count', fontsize=12)\n    sns.despine(ax=axes[i, 0])\n\n    # Boxplot à droite\n    sns.boxplot(data=train_df, x=column, y=target_column, ax=axes[i, 1], palette='tab10')\n    axes[i, 1].set_title(f'{column} vs {target_column}', fontsize=14)\n    axes[i, 1].set_xlabel(column, fontsize=12)\n    axes[i, 1].set_ylabel(target_column, fontsize=12)\n    sns.despine(ax=axes[i, 1])\n\nplt.tight_layout()  # Ajustement global des sous-graphiques\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:21:35.190797Z","iopub.execute_input":"2024-12-06T03:21:35.191063Z","iopub.status.idle":"2024-12-06T03:21:48.628155Z","shell.execute_reply.started":"2024-12-06T03:21:35.191037Z","shell.execute_reply":"2024-12-06T03:21:48.627332Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the correlation matrix\ncorrelation_matrix = train_df[numerical_columns].corr()\n\n# Plot the heatmap\nplt.figure(figsize=(12, 8))\nsns.heatmap(correlation_matrix, annot=True, fmt=\".2f\", cmap=\"coolwarm\", cbar=True, linewidths=0.5)\nplt.title(\"Correlation Heatmap of Numerical Variables\", fontsize=16)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:21:48.629239Z","iopub.execute_input":"2024-12-06T03:21:48.629556Z","iopub.status.idle":"2024-12-06T03:21:49.508568Z","shell.execute_reply.started":"2024-12-06T03:21:48.629529Z","shell.execute_reply":"2024-12-06T03:21:49.507538Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def date(df):\n\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    df['Year'] = df['Policy Start Date'].dt.year\n    df['Day'] = df['Policy Start Date'].dt.day\n    df['Month'] = df['Policy Start Date'].dt.month\n    df['Month_name'] = df['Policy Start Date'].dt.month_name()\n    df['Day_of_week'] = df['Policy Start Date'].dt.day_name()\n    df['Week'] = df['Policy Start Date'].dt.isocalendar().week\n    df['Year_sin'] = np.sin(2 * np.pi * df['Year'])\n    df['Year_cos'] = np.cos(2 * np.pi * df['Year'])\n    min_year = df['Year'].min()\n    max_year = df['Year'].max()\n    df['Year_sin'] = np.sin(2 * np.pi * (df['Year'] - min_year) / (max_year - min_year))\n    df['Year_cos'] = np.cos(2 * np.pi * (df['Year'] - min_year) / (max_year - min_year))\n    df['Month_sin'] = np.sin(2 * np.pi * df['Month'] / 12) \n    df['Month_cos'] = np.cos(2 * np.pi * df['Month'] / 12)\n    df['Day_sin'] = np.sin(2 * np.pi * df['Day'] / 31)  \n    df['Day_cos'] = np.cos(2 * np.pi * df['Day'] / 31)\n    df['Group']=(df['Year']-2020)*48+df['Month']*4+df['Day']//7\n    \n    df.drop('Policy Start Date', axis=1, inplace=True)\n\n    return df\n\n# Apply the date function to both datasets\ntrain_df = date(train_df)\ntest_df = date(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:21:49.509558Z","iopub.execute_input":"2024-12-06T03:21:49.509798Z","iopub.status.idle":"2024-12-06T03:21:52.175622Z","shell.execute_reply.started":"2024-12-06T03:21:49.509774Z","shell.execute_reply":"2024-12-06T03:21:52.174841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define features and target\nnumerical_features = [\n    'Age', 'Annual Income', 'Number of Dependents', 'Health Score', \n    'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration', \n    'Year_sin', 'Year_cos', 'Month_sin', 'Month_cos', 'Day_sin', 'Day_cos'\n]\ncategorical_features = [\n    'Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location',\n    'Policy Type', 'Customer Feedback', 'Smoking Status', 'Exercise Frequency', \n    'Property Type', 'Month_name', 'Day_of_week'\n]\ntarget_column = 'Premium Amount'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:21:52.176715Z","iopub.execute_input":"2024-12-06T03:21:52.176995Z","iopub.status.idle":"2024-12-06T03:21:52.182118Z","shell.execute_reply.started":"2024-12-06T03:21:52.176968Z","shell.execute_reply":"2024-12-06T03:21:52.181094Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split train data into features and target\nX = train_df.drop(columns=[target_column, 'id', 'Group', 'Year', 'Month', 'Day', 'Week'])\ny = train_df[target_column]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:21:52.183112Z","iopub.execute_input":"2024-12-06T03:21:52.183398Z","iopub.status.idle":"2024-12-06T03:21:52.376831Z","shell.execute_reply.started":"2024-12-06T03:21:52.183373Z","shell.execute_reply":"2024-12-06T03:21:52.375574Z"},"_kg_hide-input":false},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocessing pipeline for numerical features\nnum_pipeline = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='median')),\n    #('scaler', StandardScaler())                       # Scale numerical features\n])\n\n# Preprocessing pipeline for categorical features\ncat_pipeline = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='constant', fill_value='Unknown')),  # Handle missing values\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))                      # Encode categorical features\n])\n\n# Combine pipelines into a ColumnTransformer\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', num_pipeline, numerical_features),\n        ('cat', cat_pipeline, categorical_features)\n    ]\n)\n\n# Preprocess train and test data\nX_processed = preprocessor.fit_transform(X)\ntest_processed = preprocessor.transform(test_df.drop(columns=['id', 'Group', 'Year', 'Month', 'Day', 'Week']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:21:52.378124Z","iopub.execute_input":"2024-12-06T03:21:52.378839Z","iopub.status.idle":"2024-12-06T03:22:03.807038Z","shell.execute_reply.started":"2024-12-06T03:21:52.378794Z","shell.execute_reply":"2024-12-06T03:22:03.806287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split the data\nX_train, X_val, y_train, y_val = train_test_split(X_processed, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:22:03.807999Z","iopub.execute_input":"2024-12-06T03:22:03.808243Z","iopub.status.idle":"2024-12-06T03:22:04.111004Z","shell.execute_reply.started":"2024-12-06T03:22:03.80822Z","shell.execute_reply":"2024-12-06T03:22:04.110233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define Optuna optimization function\ndef objective(trial):\n    # Define parameter search space\n    param = {\n        \"objective\": \"regression\",\n        \"metric\": \"rmse\",\n        \"boosting_type\": trial.suggest_categorical(\"boosting_type\", [\"gbdt\", \"dart\"]),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 200, 512),\n        \"learning_rate\": trial.suggest_loguniform(\"learning_rate\", 1e-4, 1e-1),\n        \"feature_fraction\": trial.suggest_uniform(\"feature_fraction\", 0.6, 1.0),\n        \"bagging_fraction\": trial.suggest_uniform(\"bagging_fraction\", 0.6, 1.0),\n        \"bagging_freq\": trial.suggest_int(\"bagging_freq\", 5, 12),\n        \"min_data_in_leaf\": trial.suggest_int(\"min_data_in_leaf\", 20, 100),\n        \"max_depth\": trial.suggest_int(\"max_depth\", -1, 16),  # -1 means no limit\n        \"lambda_l1\": trial.suggest_loguniform(\"lambda_l1\", 1e-4, 10.0),\n        \"lambda_l2\": trial.suggest_loguniform(\"lambda_l2\", 1e-4, 10.0),\n        \"device_type\": \"gpu\",  # Enable GPU support\n        \"seed\" : 42\n\n    }\n\n    # Create a LightGBM dataset\n    dtrain = lgb.Dataset(X_train, label=y_train)\n    dval = lgb.Dataset(X_val, label=y_val, reference=dtrain)\n\n    # Train LightGBM model\n    model = lgb.train(\n        param,\n        dtrain,\n        valid_sets=[dval],\n    )\n\n    # Predict on validation set\n    y_val_pred = model.predict(X_val)\n    \n    # Compute RMSLE using sklearn's root_mean_squared_log_error\n    rmsle = root_mean_squared_log_error(y_val, np.maximum(y_val_pred, 0))\n    return rmsle\n\n# Run Optuna study\nstudy = optuna.create_study(direction=\"minimize\")\nstudy.optimize(objective, n_trials=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:22:04.112081Z","iopub.execute_input":"2024-12-06T03:22:04.112454Z","iopub.status.idle":"2024-12-06T03:22:53.473786Z","shell.execute_reply.started":"2024-12-06T03:22:04.112417Z","shell.execute_reply":"2024-12-06T03:22:53.472838Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize or update the best_params dictionary\nbest_params = {\n    'boosting_type': 'dart',\n    'num_leaves': 384,\n    'learning_rate': 0.024680120465142227,\n    'feature_fraction': 0.9883068358315126,\n    'bagging_fraction': 0.7201712704805496,\n    'bagging_freq': 7,\n    'min_data_in_leaf': 50,\n    'max_depth': 15,\n    'lambda_l1': 0.0011290211269753322,\n    'lambda_l2': 3.056310541294088,\n    'seed': 42\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:22:53.474973Z","iopub.execute_input":"2024-12-06T03:22:53.475289Z","iopub.status.idle":"2024-12-06T03:22:53.479933Z","shell.execute_reply.started":"2024-12-06T03:22:53.475242Z","shell.execute_reply":"2024-12-06T03:22:53.47917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train final model with best parameters\n#best_params = study.best_params\n\nfinal_model = lgb.train(\n    best_params,\n    lgb.Dataset(X_processed, label=y),\n)","metadata":{"trusted":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-12-06T03:22:53.480914Z","iopub.execute_input":"2024-12-06T03:22:53.48122Z","iopub.status.idle":"2024-12-06T03:23:38.224873Z","shell.execute_reply.started":"2024-12-06T03:22:53.481193Z","shell.execute_reply":"2024-12-06T03:23:38.223929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Performance Metrics\ny_pred = final_model.predict(X_processed)\n\n# Calcul des métriques\nrmsle = root_mean_squared_log_error(y, y_pred)\nrmse = np.sqrt(mean_squared_error(y, y_pred))\nmae = mean_absolute_error(y, y_pred)\nr2 = r2_score(y, y_pred)\nmape = np.mean(np.abs((y - y_pred) / y)) * 100\n\n# Display performance metrics\nprint(f\"\\nPerformance Metrics:\\n{'-'*30}\")\nprint(f\"RMSLE: {rmsle:.4f}\")\nprint(f\"RMSE: {rmse:.4f}\")\nprint(f\"MAE: {mae:.4f}\")\nprint(f\"R²: {r2:.4f}\")\nprint(f\"MAPE: {mape:.2f}%\")\n\n# 2. Feature Importance\nimportances = final_model.feature_importance(importance_type='split')  # or 'gain'\nfeatures = preprocessor.get_feature_names_out()\nsorted_indices = importances.argsort()[::-1]\n\n# Create a DataFrame for feature importances\nimportance_df = pd.DataFrame({\n    'Feature': [features[i] for i in sorted_indices],\n    'Importance': importances[sorted_indices]\n})\n\n# Plot top 10 feature importances\nplt.figure(figsize=(12, 6))\nsns.barplot(data=importance_df.head(10), x='Importance', y='Feature', palette=\"coolwarm\")\nplt.title(\"Top 10 Feature Importances\", fontsize=16, fontweight='bold')\nplt.xlabel(\"Feature Importance\", fontsize=12)\nplt.ylabel(\"Feature\", fontsize=12)\nplt.tight_layout()\nplt.show()\n\n# 3. Residual Analysis\nresiduals = y - y_pred\n\n# Residuals vs Predicted Values\nplt.figure(figsize=(12, 6))\nsns.scatterplot(x=y_pred, y=residuals, alpha=0.6, color=\"#007acc\")\nplt.axhline(y=0, color='red', linestyle='--', linewidth=1.5)\nplt.title(\"Residuals vs Predicted Values\", fontsize=16, fontweight='bold')\nplt.xlabel(\"Predicted Values\", fontsize=12)\nplt.ylabel(\"Residuals\", fontsize=12)\nplt.tight_layout()\nplt.show()\n\n# Residual Distribution\nplt.figure(figsize=(10, 6))\nsns.histplot(residuals, bins=30, kde=True, color=\"#55a630\")\nplt.axvline(x=0, color='red', linestyle='--', linewidth=1.5)\nplt.title(\"Distribution of Residuals\", fontsize=16, fontweight='bold')\nplt.xlabel(\"Residuals\", fontsize=12)\nplt.ylabel(\"Frequency\", fontsize=12)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:23:38.226213Z","iopub.execute_input":"2024-12-06T03:23:38.22657Z","iopub.status.idle":"2024-12-06T03:23:51.534582Z","shell.execute_reply.started":"2024-12-06T03:23:38.226541Z","shell.execute_reply":"2024-12-06T03:23:51.533736Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"before KNN imputer :\n\n___\n\nPerformance Metrics:\n- RMSLE: 1.0634\n- RMSE: 906.9466\n- MAE: 625.4777\n- R²: -0.0993\n- MAPE: 204.97%","metadata":{}},{"cell_type":"code","source":"# Make predictions on the test set\ntest_predictions = final_model.predict(test_processed, num_iteration=final_model.best_iteration)\n\n# Prepare submission file\nsubmission = pd.DataFrame({'id': test_df['id'], 'Premium Amount': test_predictions})\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-12-06T03:23:51.535784Z","iopub.execute_input":"2024-12-06T03:23:51.536166Z","iopub.status.idle":"2024-12-06T03:23:56.737409Z","shell.execute_reply.started":"2024-12-06T03:23:51.536127Z","shell.execute_reply":"2024-12-06T03:23:56.73666Z"},"_kg_hide-output":true,"trusted":true},"outputs":[],"execution_count":null}]}