{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:03:45.946154Z","iopub.execute_input":"2024-12-27T09:03:45.946725Z","iopub.status.idle":"2024-12-27T09:03:45.960728Z","shell.execute_reply.started":"2024-12-27T09:03:45.946652Z","shell.execute_reply":"2024-12-27T09:03:45.959367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:03:46.444984Z","iopub.execute_input":"2024-12-27T09:03:46.445499Z","iopub.status.idle":"2024-12-27T09:03:51.219928Z","shell.execute_reply.started":"2024-12-27T09:03:46.445451Z","shell.execute_reply":"2024-12-27T09:03:51.21878Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Reduce memory usage","metadata":{}},{"cell_type":"code","source":"def reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2    \n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)    \n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: print('Mem. usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:03:51.221786Z","iopub.execute_input":"2024-12-27T09:03:51.222143Z","iopub.status.idle":"2024-12-27T09:03:51.236152Z","shell.execute_reply.started":"2024-12-27T09:03:51.222109Z","shell.execute_reply":"2024-12-27T09:03:51.234875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:03:51.237954Z","iopub.execute_input":"2024-12-27T09:03:51.238443Z","iopub.status.idle":"2024-12-27T09:03:51.254149Z","shell.execute_reply.started":"2024-12-27T09:03:51.238394Z","shell.execute_reply":"2024-12-27T09:03:51.253105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:03:51.257008Z","iopub.execute_input":"2024-12-27T09:03:51.257489Z","iopub.status.idle":"2024-12-27T09:03:51.91795Z","shell.execute_reply.started":"2024-12-27T09:03:51.25744Z","shell.execute_reply":"2024-12-27T09:03:51.916899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n\ndf['Policy Start Date'].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:03:51.919397Z","iopub.execute_input":"2024-12-27T09:03:51.919844Z","iopub.status.idle":"2024-12-27T09:03:52.332618Z","shell.execute_reply.started":"2024-12-27T09:03:51.919798Z","shell.execute_reply":"2024-12-27T09:03:52.331486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:03:52.334296Z","iopub.execute_input":"2024-12-27T09:03:52.33486Z","iopub.status.idle":"2024-12-27T09:03:52.913841Z","shell.execute_reply.started":"2024-12-27T09:03:52.334813Z","shell.execute_reply":"2024-12-27T09:03:52.912574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:03:52.915381Z","iopub.execute_input":"2024-12-27T09:03:52.917116Z","iopub.status.idle":"2024-12-27T09:03:53.678915Z","shell.execute_reply.started":"2024-12-27T09:03:52.917063Z","shell.execute_reply":"2024-12-27T09:03:53.677771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_col = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', 'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration', 'Premium Amount']\n\ndef plot_histogram(df, numerical_columns):\n    n_cols = 3  # Number of histograms per row\n    n_rows = (len(numerical_columns) + n_cols - 1) // n_cols  # Calculate the number of rows needed\n    \n    fig, axes = plt.subplots(n_rows, n_cols, figsize=(18, 5 * n_rows))  # Adjust figsize as needed\n    axes = axes.flatten()  # Flatten the 2D array of axes into a 1D array for easier indexing\n    \n    for i, column in enumerate(numerical_columns):\n        sns.histplot(df[column], bins=20, kde=True, color='blue', ax=axes[i])\n        axes[i].set_title(f'Distribution of {column}')\n        axes[i].set_xlabel(column)\n        axes[i].set_ylabel('Frequency')\n    \n    # Hide any unused subplots\n    for j in range(len(numerical_columns), len(axes)):\n        fig.delaxes(axes[j])\n    \n    plt.tight_layout()  # Adjust layout to prevent overlapping\n    plt.show()\n\n# Call the function\nplot_histogram(df, numerical_col)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:03:53.680276Z","iopub.execute_input":"2024-12-27T09:03:53.680732Z","iopub.status.idle":"2024-12-27T09:04:40.485931Z","shell.execute_reply.started":"2024-12-27T09:03:53.680663Z","shell.execute_reply":"2024-12-27T09:04:40.484407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import skew\n\nskewness_value = skew(df['Premium Amount'])\nprint(f\"Skewness: {skewness_value}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:04:40.487658Z","iopub.execute_input":"2024-12-27T09:04:40.488094Z","iopub.status.idle":"2024-12-27T09:04:40.51918Z","shell.execute_reply.started":"2024-12-27T09:04:40.488049Z","shell.execute_reply":"2024-12-27T09:04:40.51795Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The premium amount is positively skewed, let's log transform it.","metadata":{}},{"cell_type":"code","source":"df[\"Log Premium Amount\"] = np.log1p(df[\"Premium Amount\"])\nplot_histogram(df, [\"Log Premium Amount\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:04:40.524319Z","iopub.execute_input":"2024-12-27T09:04:40.524836Z","iopub.status.idle":"2024-12-27T09:04:46.068776Z","shell.execute_reply.started":"2024-12-27T09:04:40.524786Z","shell.execute_reply":"2024-12-27T09:04:46.06774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"skewness_value = skew(df['Log Premium Amount'])\nprint(f\"Skewness: {skewness_value}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:04:46.070397Z","iopub.execute_input":"2024-12-27T09:04:46.070841Z","iopub.status.idle":"2024-12-27T09:04:46.100737Z","shell.execute_reply.started":"2024-12-27T09:04:46.070792Z","shell.execute_reply":"2024-12-27T09:04:46.099421Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now it's negatively skewed :( we will leave this for now since Light GBM is robust to skewed data.","metadata":{}},{"cell_type":"code","source":"import math\ncategorical_col = ['Gender', 'Marital Status', 'Education Level','Occupation','Location', 'Policy Type', 'Smoking Status', 'Exercise Frequency', 'Property Type']\n\ndef plot_pie_charts(df, categorical_columns):\n    # Number of charts and layout\n    n_cols = 3  # Number of columns in the grid\n    n_rows = math.ceil(len(categorical_columns) / n_cols)  # Calculate required rows\n    \n    # Create subplots\n    fig, axes = plt.subplots(n_rows, n_cols, figsize=(15, 5 * n_rows))  # Adjust size as needed\n    axes = axes.flatten()  # Flatten the axes array for easy indexing\n    \n    for i, col in enumerate(categorical_columns):\n        counts = df[col].value_counts(normalize=True) * 100  # Calculate percentages\n        \n        # Plot pie chart\n        counts.plot(\n            kind='pie',\n            autopct='%1.1f%%', \n            startangle=90,\n            labels=counts.index,\n            colors=plt.cm.Paired.colors[:len(counts)],\n            ax=axes[i]  # Plot on the specific subplot\n        )\n        \n        axes[i].set_title(f'Percentage Distribution of {col}')\n        axes[i].set_ylabel('')  # Remove default ylabel for a cleaner look\n    \n    # Hide any unused subplots\n    for j in range(i + 1, len(axes)):\n        fig.delaxes(axes[j])\n    \n    plt.tight_layout()  # Adjust layout for better spacing\n    plt.show()\n\nplot_pie_charts(df, categorical_col)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:04:46.102161Z","iopub.execute_input":"2024-12-27T09:04:46.10256Z","iopub.status.idle":"2024-12-27T09:04:47.85852Z","shell.execute_reply.started":"2024-12-27T09:04:46.102518Z","shell.execute_reply":"2024-12-27T09:04:47.85736Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"All of the values here are quite evenly distributed.","metadata":{}},{"cell_type":"markdown","source":"## Correlation","metadata":{}},{"cell_type":"code","source":"numerical_df = df.select_dtypes(include = ['int64', 'float64'])\n\n# Compute the correlation matrix\ncorrelation_matrix = numerical_df.corr()\n\n# Plot the heatmap\nplt.figure(figsize=(8,8))\nsns.heatmap(correlation_matrix, annot = True, fmt = '.2f', cmap= 'coolwarm', vmin = -1, vmax = 1)\nplt.title('Correlation Heatmap')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:04:47.860053Z","iopub.execute_input":"2024-12-27T09:04:47.860462Z","iopub.status.idle":"2024-12-27T09:04:49.014971Z","shell.execute_reply.started":"2024-12-27T09:04:47.860418Z","shell.execute_reply":"2024-12-27T09:04:49.012644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Scatter plot\nplt.figure(figsize=(10, 6))\nsns.scatterplot(data=df, x='Annual Income', y='Credit Score', alpha=0.6, color='blue')\n\n# Add labels and title\nplt.title('Annual Income vs. Credit Score', fontsize=16)\nplt.xlabel('Annual Income', fontsize=14)\nplt.ylabel('Credit Score', fontsize=14)\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:04:49.016637Z","iopub.execute_input":"2024-12-27T09:04:49.017191Z","iopub.status.idle":"2024-12-27T09:04:52.492369Z","shell.execute_reply.started":"2024-12-27T09:04:49.017141Z","shell.execute_reply":"2024-12-27T09:04:52.491331Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There's no strong correlations between these numerical features.","metadata":{}},{"cell_type":"code","source":"from scipy.stats import chi2_contingency\ncategorical_features = ['Gender', 'Marital Status', 'Education Level','Occupation','Location', 'Policy Type', 'Smoking Status', 'Exercise Frequency', 'Property Type']\n\n# Loop through pairs of categorical features\nfor feature1 in categorical_features:\n    for feature2 in categorical_features:\n        if feature1 != feature2:  # Avoid testing a feature with itself\n            # Create a contingency table\n            contingency_table = pd.crosstab(df[feature1], df[feature2])\n            \n            # Perform Chi-squared test\n            chi2, p, dof, expected = chi2_contingency(contingency_table)\n            \n            # Print the results\n            print(f\"Chi-Squared Test between {feature1} and {feature2}\")\n            print(f\"Chi-squared Statistic: {chi2:.2f}\")\n            print(f\"P-value: {p:.4f}\")\n            print(f\"Degrees of Freedom: {dof}\")\n            print(\"-\" * 50)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:04:52.494049Z","iopub.execute_input":"2024-12-27T09:04:52.494459Z","iopub.status.idle":"2024-12-27T09:05:08.672662Z","shell.execute_reply.started":"2024-12-27T09:04:52.494416Z","shell.execute_reply":"2024-12-27T09:05:08.671378Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As we know that P-values less than 0.05 are likely associated, we have these pairs that have p-values less than 0.05\n- Gender and Marital Status (lowest deviation from independence based on the chi-squared statistics)\n- Policy Type and Education Level\n- Smoking Status and Gender\n- Exercise Frequency and Education Level\n- Property Type and Marital Status\n- Marital Status and Occupation\n- Property Type and Occupation (highest deviation from independence based on chi-squared statistics)\n\n\nLet's do a Cramer V test to see the strength of association.","metadata":{}},{"cell_type":"code","source":"def cramers_v(contingency_table):\n    \"\"\"Calculate Cramér's V for a contingency table.\"\"\"\n    chi2, _, _, _ = chi2_contingency(contingency_table)\n    n = contingency_table.sum().sum()  # Total sample size\n    k = min(contingency_table.shape)  # Smaller dimension of the table (rows or columns)\n    return np.sqrt(chi2 / (n * (k - 1)))\n\n# Selected pairs\npairs = [\n    ('Gender', 'Marital Status'),\n    ('Policy Type', 'Education Level'),\n    ('Smoking Status', 'Gender'),\n    ('Exercise Frequency' , 'Education Level'),\n    ('Property Type' , 'Marital Status'),\n    ('Marital Status', 'Occupation'),\n    ('Policy Type', 'Education Level')\n]\n\n# Compute Cramér's V for each pair\nfor feature1, feature2 in pairs:\n    contingency_table = pd.crosstab(df[feature1], df[feature2])\n    v = cramers_v(contingency_table)\n    print(f\"Cramér's V between {feature1} and {feature2}: {v:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:05:08.673919Z","iopub.execute_input":"2024-12-27T09:05:08.674226Z","iopub.status.idle":"2024-12-27T09:05:10.3014Z","shell.execute_reply.started":"2024-12-27T09:05:08.674197Z","shell.execute_reply":"2024-12-27T09:05:10.300345Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"All less than 0.1, weak assocation.","metadata":{}},{"cell_type":"markdown","source":"## Comments\n\n- Features are not strongly correlated, indicating that each feature may provide unique and independent information to the model.\n- An even distribution suggests the dataset has good variability in its features, avoiding extreme imbalances or concentrated values.\n- The lack of strong correlations between features means the relationships in the data might be complex, rather than linear or directly apparent.\n\nWe will be using tree-based models since they are better at finding complex, non-linear patterns in data where pairwise relationships are weak.\n","metadata":{}},{"cell_type":"code","source":"numerical_features = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', 'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration']\n\n# Create subplots\nn_cols = 3  # Number of plots per row\nn_rows = (len(numerical_features) + n_cols - 1) // n_cols  # Calculate rows needed\n\nfig, axes = plt.subplots(n_rows, n_cols, figsize=(20, 5 * n_rows))  # Adjust size\naxes = axes.flatten()  # Flatten the axes array for easy indexing\n\n# Loop through features and create scatter plots\nfor i, feature in enumerate(numerical_features):\n    sns.scatterplot(data=df, x=feature, y='Premium Amount', ax=axes[i], alpha=0.6, color='blue')\n    axes[i].set_title(f'Premium Amount vs {feature}', fontsize=14)\n    axes[i].set_xlabel(feature, fontsize=12)\n    axes[i].set_ylabel('Premium Amount', fontsize=12)\n\n# Hide unused subplots\nfor j in range(len(numerical_features), len(axes)):\n    fig.delaxes(axes[j])\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:05:10.30272Z","iopub.execute_input":"2024-12-27T09:05:10.303056Z","iopub.status.idle":"2024-12-27T09:05:33.968366Z","shell.execute_reply.started":"2024-12-27T09:05:10.303024Z","shell.execute_reply":"2024-12-27T09:05:33.966798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_features = ['Gender', 'Marital Status', 'Education Level','Occupation','Location', 'Policy Type', 'Smoking Status', 'Exercise Frequency', 'Property Type', 'Customer Feedback']\n\n# Create subplots\nn_cols = 3  # Number of plots per row\nn_rows = (len(categorical_features) + n_cols - 1) // n_cols  # Calculate rows needed\n\nfig, axes = plt.subplots(n_rows, n_cols, figsize=(20, 5 * n_rows))  # Adjust size\naxes = axes.flatten()  # Flatten the axes array for easy indexing\n\n# Loop through features and create box plots\nfor i, feature in enumerate(categorical_features):\n    sns.boxplot(data=df, x=feature, y='Premium Amount', ax=axes[i], palette='coolwarm')\n    axes[i].set_title(f'Premium Amount by {feature}', fontsize=14)\n    axes[i].set_xlabel(feature, fontsize=12)\n    axes[i].set_ylabel('Premium Amount', fontsize=12)\n\n# Hide unused subplots\nfor j in range(len(categorical_features), len(axes)):\n    fig.delaxes(axes[j])\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:05:33.970629Z","iopub.execute_input":"2024-12-27T09:05:33.971066Z","iopub.status.idle":"2024-12-27T09:05:42.126984Z","shell.execute_reply.started":"2024-12-27T09:05:33.971024Z","shell.execute_reply":"2024-12-27T09:05:42.125755Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"df['Insurance Duration'].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:05:42.128595Z","iopub.execute_input":"2024-12-27T09:05:42.129072Z","iopub.status.idle":"2024-12-27T09:05:42.138732Z","shell.execute_reply.started":"2024-12-27T09:05:42.129012Z","shell.execute_reply":"2024-12-27T09:05:42.137666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Days Since Policy Start'] = (pd.Timestamp.now() - df['Policy Start Date']).dt.days\ndf['Policy Start Month'] = pd.to_datetime(df['Policy Start Date']).dt.month\ndf['Policy Start Year'] = pd.to_datetime(df['Policy Start Date']).dt.year\ndf['income_age_ratio'] = df['Annual Income']/df['Age']\ndf['income_dependents_ratio'] = df['Annual Income']/df['Number of Dependents']\ndf['claims_income_ratio'] = df['Previous Claims']/ df['Annual Income']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:05:42.140163Z","iopub.execute_input":"2024-12-27T09:05:42.140583Z","iopub.status.idle":"2024-12-27T09:05:42.348089Z","shell.execute_reply.started":"2024-12-27T09:05:42.140524Z","shell.execute_reply":"2024-12-27T09:05:42.346943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_features = ['Days Since Policy Start', 'Policy Start Month', 'Policy Start Year', 'income_age_ratio', 'income_dependents_ratio', 'claims_income_ratio']\n\n# Create subplots\nn_cols = 3  # Number of plots per row\nn_rows = (len(numerical_features) + n_cols - 1) // n_cols  # Calculate rows needed\n\nfig, axes = plt.subplots(n_rows, n_cols, figsize=(20, 5 * n_rows))  # Adjust size\naxes = axes.flatten()  # Flatten the axes array for easy indexing\n\n# Loop through features and create scatter plots\nfor i, feature in enumerate(numerical_features):\n    sns.scatterplot(data=df, x=feature, y='Premium Amount', ax=axes[i], alpha=0.6, color='blue')\n    axes[i].set_title(f'Premium Amount vs {feature}', fontsize=14)\n    axes[i].set_xlabel(feature, fontsize=12)\n    axes[i].set_ylabel('Premium Amount', fontsize=12)\n\n# Hide unused subplots\nfor j in range(len(numerical_features), len(axes)):\n    fig.delaxes(axes[j])\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:05:42.350183Z","iopub.execute_input":"2024-12-27T09:05:42.350662Z","iopub.status.idle":"2024-12-27T09:05:59.245398Z","shell.execute_reply.started":"2024-12-27T09:05:42.350611Z","shell.execute_reply":"2024-12-27T09:05:59.243769Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Encode categorical features","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# List of columns to encode\ncol_encode = [\"Gender\", \"Marital Status\", \"Education Level\", \"Occupation\", \"Location\", \"Policy Type\", \"Policy Start Date\", \"Customer Feedback\", \"Smoking Status\", \"Exercise Frequency\", \"Property Type\"]\n\n# Initialize LabelEncoder\nlabel = LabelEncoder()\n\n#Loop through each column\nfor col in col_encode:\n    #create a new col\n    df[f'{col}_code'] = label.fit_transform(df[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:05:59.246667Z","iopub.execute_input":"2024-12-27T09:05:59.247091Z","iopub.status.idle":"2024-12-27T09:06:01.909936Z","shell.execute_reply.started":"2024-12-27T09:05:59.247047Z","shell.execute_reply":"2024-12-27T09:06:01.908745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:06:01.911317Z","iopub.execute_input":"2024-12-27T09:06:01.91179Z","iopub.status.idle":"2024-12-27T09:06:01.919738Z","shell.execute_reply.started":"2024-12-27T09:06:01.911745Z","shell.execute_reply":"2024-12-27T09:06:01.918618Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_log_error\nfrom lightgbm import LGBMRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:06:01.921246Z","iopub.execute_input":"2024-12-27T09:06:01.921688Z","iopub.status.idle":"2024-12-27T09:06:01.933287Z","shell.execute_reply.started":"2024-12-27T09:06:01.921623Z","shell.execute_reply":"2024-12-27T09:06:01.931957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"reduced_df = reduce_mem_usage(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:06:01.935014Z","iopub.execute_input":"2024-12-27T09:06:01.93546Z","iopub.status.idle":"2024-12-27T09:06:02.317956Z","shell.execute_reply.started":"2024-12-27T09:06:01.935407Z","shell.execute_reply":"2024-12-27T09:06:02.316583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Separate features and target\ntarget_column = \"Log Premium Amount\"\nX = reduced_df[['Gender_code', 'Marital Status_code', 'Education Level_code', 'Occupation_code', 'Location_code', 'Policy Type_code', 'Policy Start Date_code', 'Customer Feedback_code', 'Smoking Status_code',\n        'Exercise Frequency_code', 'Property Type_code','Days Since Policy Start', 'Policy Start Month', 'Policy Start Year', 'income_age_ratio', 'income_dependents_ratio', 'claims_income_ratio', 'Age', 'Annual Income', 'Number of Dependents', \n        'Health Score', 'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration']]\ny = reduced_df[target_column]\n\n# train test split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.2, random_state = 42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:06:02.319565Z","iopub.execute_input":"2024-12-27T09:06:02.320049Z","iopub.status.idle":"2024-12-27T09:06:02.736173Z","shell.execute_reply.started":"2024-12-27T09:06:02.320001Z","shell.execute_reply":"2024-12-27T09:06:02.734781Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Modelling","metadata":{}},{"cell_type":"code","source":"model = LGBMRegressor(objective=\"regression\", random_state=42, verbose = -1)\nmodel.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:06:02.737505Z","iopub.execute_input":"2024-12-27T09:06:02.737908Z","iopub.status.idle":"2024-12-27T09:06:14.446016Z","shell.execute_reply.started":"2024-12-27T09:06:02.737838Z","shell.execute_reply":"2024-12-27T09:06:14.444773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on the test set\ny_pred_log = model.predict(X_test)  # Predictions are still in the log scale\n\n# Revert predictions and true values to the original scale\ny_pred = np.exp(y_pred_log) - 1  # Reverting np.log1p transformation\ny_true = np.exp(y_test) - 1      # Reverting np.log1p transformation\n\n# Clip negative values to 0 to avoid errors with log metrics\ny_pred = np.maximum(0, y_pred)\ny_true = np.maximum(0, y_true)\n\n# Calculate RMSLE\nrmsle_value = np.sqrt(mean_squared_log_error(y_true, y_pred))\nrmsle_value","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:06:14.453174Z","iopub.execute_input":"2024-12-27T09:06:14.453593Z","iopub.status.idle":"2024-12-27T09:06:15.759731Z","shell.execute_reply.started":"2024-12-27T09:06:14.453546Z","shell.execute_reply":"2024-12-27T09:06:15.758522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def lgbm_objective(trial):\n    # Hyperparameter search space\n    lgbm_params = {\n        \"objective\": \"regression\",\n        \"metric\": \"rmse\",  # LightGBM's internal metric\n        \"n_estimators\": 500,\n        \"bagging_freq\": trial.suggest_int(\"bagging_freq\", 10, 60),\n        \"min_data_in_leaf\": trial.suggest_int(\"min_data_in_leaf\", 30, 75),\n        \"max_depth\": trial.suggest_int(\"max_depth\", 5, 25),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 150, 350),\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.01, 0.03),\n        \"feature_fraction\": trial.suggest_float(\"feature_fraction\", 0.7, 0.88),\n        \"bagging_fraction\": trial.suggest_float(\"bagging_fraction\", 0.5, 0.6),\n        \"lambda_l1\": trial.suggest_float(\"lambda_l1\", 0.07, 0.09),\n        \"lambda_l2\": trial.suggest_float(\"lambda_l2\", 0.03, 0.05),\n    }\n\n    # Train the LightGBM model\n    lgbm_model = LGBMRegressor(**lgbm_params, random_state=1)\n    lgbm_model.fit(X_train, y_train)  # X_train and y_train are log-transformed\n\n    # Predict on the test set\n    y_pred_log = lgbm_model.predict(X_test)  # Predictions are still in the log scale\n\n    # Revert predictions and true values to the original scale\n    y_pred = np.exp(y_pred_log) - 1  # Reverting np.log1p transformation\n    y_true = np.exp(y_test) - 1      # Reverting np.log1p transformation\n\n    # Clip negative values to 0 to avoid errors with log metrics\n    y_pred = np.maximum(0, y_pred)\n    y_true = np.maximum(0, y_true)\n\n    # Calculate RMSLE\n    rmsle_value = np.sqrt(mean_squared_log_error(y_true, y_pred))\n    return rmsle_value\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:06:15.761306Z","iopub.execute_input":"2024-12-27T09:06:15.761815Z","iopub.status.idle":"2024-12-27T09:06:15.771633Z","shell.execute_reply.started":"2024-12-27T09:06:15.761762Z","shell.execute_reply":"2024-12-27T09:06:15.770441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import optuna\n\n# # Define the study\n# study_LGBM = optuna.create_study(study_name=\"LGBM_Insurance\", direction=\"minimize\")\n\n# # Optimize the objective function\n# study_LGBM.optimize(lgbm_objective, n_trials=30, show_progress_bar=True)\n\n# # Print the best results\n# print(f\"Best RMSLE: {study_LGBM.best_value:.4f}\")\n# print(\"Best Parameters:\", study_LGBM.best_params)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:06:15.773013Z","iopub.execute_input":"2024-12-27T09:06:15.773338Z","iopub.status.idle":"2024-12-27T09:06:15.790094Z","shell.execute_reply.started":"2024-12-27T09:06:15.773307Z","shell.execute_reply":"2024-12-27T09:06:15.788958Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##### Best Parameters: {'bagging_freq': 59, 'min_data_in_leaf': 30, 'max_depth': 5, 'num_leaves': 233, 'learning_rate': 0.010188884186242364, 'feature_fraction': 0.7011249533969797, 'bagging_fraction': 0.5705588691222849, 'lambda_l1': 0.07026381775905029, 'lambda_l2': 0.03728916594092147}","metadata":{}},{"cell_type":"code","source":"submit_df_ori = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\nsubmit_df = reduce_mem_usage(submit_df_ori)\nfor col in col_encode:\n    #create a new col\n    submit_df[f'{col}_code'] = label.fit_transform(submit_df[col])\n\n\nsubmit_df.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:06:15.791623Z","iopub.execute_input":"2024-12-27T09:06:15.792022Z","iopub.status.idle":"2024-12-27T09:06:21.110951Z","shell.execute_reply.started":"2024-12-27T09:06:15.791987Z","shell.execute_reply":"2024-12-27T09:06:21.109781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit_df['Policy Start Date'] = pd.to_datetime(submit_df['Policy Start Date'])\nsubmit_df['Days Since Policy Start'] = (pd.Timestamp.now() - submit_df['Policy Start Date']).dt.days\nsubmit_df['Policy Start Month'] = pd.to_datetime(submit_df['Policy Start Date']).dt.month\nsubmit_df['Policy Start Year'] = pd.to_datetime(submit_df['Policy Start Date']).dt.year\nsubmit_df['income_age_ratio'] = submit_df['Annual Income']/submit_df['Age']\nsubmit_df['income_dependents_ratio'] = submit_df['Annual Income']/submit_df['Number of Dependents']\nsubmit_df['claims_income_ratio'] = submit_df['Previous Claims']/ submit_df['Annual Income']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:06:21.112134Z","iopub.execute_input":"2024-12-27T09:06:21.112459Z","iopub.status.idle":"2024-12-27T09:06:21.524202Z","shell.execute_reply.started":"2024-12-27T09:06:21.112428Z","shell.execute_reply":"2024-12-27T09:06:21.523036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Best parameters from Optuna\nbest_params = {\n    \"objective\": \"regression\",\n    \"metric\": \"rmse\",\n    \"n_estimators\": 500,  # This was set manually during Optuna trials\n    \"bagging_freq\": 59,\n    \"min_data_in_leaf\": 30,\n    \"max_depth\": 5,\n    \"num_leaves\": 233,\n    \"learning_rate\": 0.010188884186242364,\n    \"feature_fraction\": 0.7011249533969797,\n    \"bagging_fraction\": 0.5705588691222849,\n    \"lambda_l1\": 0.07026381775905029,\n    \"lambda_l2\": 0.03728916594092147,\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:06:21.525724Z","iopub.execute_input":"2024-12-27T09:06:21.526081Z","iopub.status.idle":"2024-12-27T09:06:21.532605Z","shell.execute_reply.started":"2024-12-27T09:06:21.526049Z","shell.execute_reply":"2024-12-27T09:06:21.53118Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# K Fold","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_log_error\n\n\n# Initialize KFold\nkf = KFold(n_splits=5, shuffle=True, random_state=42)  # 5-fold cross-validation\n\n# Store RMSLE for each fold\nrmsle_scores = []\n\n# Loop through each fold\nfor fold, (train_idx, val_idx) in enumerate(kf.split(X_train, y_train)):\n    print(f\"Fold {fold + 1}\")\n    \n    # Split the data\n    X_tr, X_val = X_train.iloc[train_idx], X_train.iloc[val_idx]\n    y_tr, y_val = y_train.iloc[train_idx], y_train.iloc[val_idx]\n    \n    # Train LightGBM model\n    final_model = LGBMRegressor(**best_params, random_state=42)\n    final_model.fit(X_tr, y_tr, eval_set=[(X_val, y_val)], eval_metric='rmse')\n    \n    # Predict on validation set\n    y_val_pred_log = final_model.predict(X_val)  # Predictions are still in log scale\n    \n    # Revert log1p transformation\n    y_val_pred = np.expm1(y_val_pred_log)\n    y_val_true = np.expm1(y_val)\n    \n    # Clip negative predictions to 0\n    y_val_pred = np.maximum(0, y_val_pred)\n    y_val_true = np.maximum(0, y_val_true)\n    \n    # Calculate RMSLE\n    rmsle = np.sqrt(mean_squared_log_error(y_val_true, y_val_pred))\n    rmsle_scores.append(rmsle)\n    print(f\"RMSLE for Fold {fold + 1}: {rmsle:.4f}\")\n\n# Calculate average and standard deviation of RMSLE\nmean_rmsle = np.mean(rmsle_scores)\nstd_rmsle = np.std(rmsle_scores)\nprint(f\"\\nMean RMSLE: {mean_rmsle:.4f}\")\nprint(f\"Standard Deviation of RMSLE: {std_rmsle:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:07:46.719982Z","iopub.execute_input":"2024-12-27T09:07:46.720377Z","iopub.status.idle":"2024-12-27T09:12:53.099272Z","shell.execute_reply.started":"2024-12-27T09:07:46.720343Z","shell.execute_reply":"2024-12-27T09:12:53.097781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Best parameters from Optuna\n# best_params = {\n#     \"objective\": \"regression\",\n#     \"metric\": \"rmse\",\n#     \"n_estimators\": 500,  # This was set manually during Optuna trials\n#     \"bagging_freq\": 59,\n#     \"min_data_in_leaf\": 30,\n#     \"max_depth\": 5,\n#     \"num_leaves\": 233,\n#     \"learning_rate\": 0.010188884186242364,\n#     \"feature_fraction\": 0.7011249533969797,\n#     \"bagging_fraction\": 0.5705588691222849,\n#     \"lambda_l1\": 0.07026381775905029,\n#     \"lambda_l2\": 0.03728916594092147,\n# }\n\n# # Recreate the trained model\n# final_model = LGBMRegressor(**best_params, random_state=42)\n\n# # Fit the model to your original training data\n# final_model.fit(X_train, y_train)  # X_train and y_train are log-transformed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:06:21.739933Z","iopub.status.idle":"2024-12-27T09:06:21.740274Z","shell.execute_reply.started":"2024-12-27T09:06:21.740111Z","shell.execute_reply":"2024-12-27T09:06:21.740128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Predict on the test set\n# y_pred_log = final_model.predict(X_test)  # Predictions are still in the log scale\n\n# # Revert predictions and true values to the original scale\n# y_pred = np.exp(y_pred_log) - 1  # Reverting np.log1p transformation\n# y_true = np.exp(y_test) - 1      # Reverting np.log1p transformation\n\n# # Clip negative values to 0 to avoid errors with log metrics\n# y_pred = np.maximum(0, y_pred)\n# y_true = np.maximum(0, y_true)\n\n# rmsle_value = np.sqrt(mean_squared_log_error(y_true, y_pred))\n# rmsle_value","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:06:21.74273Z","iopub.status.idle":"2024-12-27T09:06:21.743132Z","shell.execute_reply.started":"2024-12-27T09:06:21.742954Z","shell.execute_reply":"2024-12-27T09:06:21.742973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transformed_col = ['Gender_code', 'Marital Status_code', 'Education Level_code', 'Occupation_code', 'Location_code', 'Policy Type_code', 'Policy Start Date_code', 'Customer Feedback_code', 'Smoking Status_code',\n        'Exercise Frequency_code', 'Property Type_code','Days Since Policy Start', 'Policy Start Month', 'Policy Start Year', 'income_age_ratio', 'income_dependents_ratio', 'claims_income_ratio', 'Age', 'Annual Income', 'Number of Dependents', \n        'Health Score', 'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration']\n\n\npredictions = final_model.predict(submit_df[transformed_col])\n\noutput = pd.DataFrame({'id': submit_df.id, 'Premium Amount': predictions})\noutput.to_csv('submission.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:12:53.101826Z","iopub.execute_input":"2024-12-27T09:12:53.102851Z","iopub.status.idle":"2024-12-27T09:13:17.491449Z","shell.execute_reply.started":"2024-12-27T09:12:53.102797Z","shell.execute_reply":"2024-12-27T09:13:17.490374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}