{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Relevant Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, TensorDataset, random_split, Subset\nimport torch.optim as optim\n\nimport matplotlib.pyplot as plt\nimport matplotlib.gridspec as gridspec\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T13:56:28.923081Z","iopub.execute_input":"2024-12-17T13:56:28.923539Z","iopub.status.idle":"2024-12-17T13:56:28.929946Z","shell.execute_reply.started":"2024-12-17T13:56:28.923498Z","shell.execute_reply":"2024-12-17T13:56:28.928686Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load in data","metadata":{}},{"cell_type":"markdown","source":"Some of the code has been taken/modified from https://www.kaggle.com/code/adrienmorel97/eda-lightgbm-optuna-1-0644-v1.\nI had similar ideas for the data exploration, and didn't think it was necessary to rewrite it.","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\ndf_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndf_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\ndf_submission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-17T13:28:03.765929Z","iopub.execute_input":"2024-12-17T13:28:03.766313Z","iopub.status.idle":"2024-12-17T13:28:11.735884Z","shell.execute_reply.started":"2024-12-17T13:28:03.766281Z","shell.execute_reply":"2024-12-17T13:28:11.73475Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data exploration","metadata":{}},{"cell_type":"code","source":"print(f\"shape of data {df_train.shape} \\n\")\n\n# Select categorical and numerical columns\ncategorical_columns = df_train.select_dtypes(include=['object']).columns\nnumerical_columns = df_train.select_dtypes(exclude=['object']).columns\n\nprint(\"\\nCategorical Columns:\", categorical_columns.tolist())\nprint(\"\\nNumerical Columns:\", numerical_columns.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T13:28:11.738055Z","iopub.execute_input":"2024-12-17T13:28:11.738506Z","iopub.status.idle":"2024-12-17T13:28:11.933608Z","shell.execute_reply.started":"2024-12-17T13:28:11.738457Z","shell.execute_reply":"2024-12-17T13:28:11.932364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for column in categorical_columns:\n    num_unique = df_train[column].nunique()\n    print(f\"'{column}' has {num_unique} unique categories.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T13:28:11.935141Z","iopub.execute_input":"2024-12-17T13:28:11.935588Z","iopub.status.idle":"2024-12-17T13:28:12.895966Z","shell.execute_reply.started":"2024-12-17T13:28:11.935534Z","shell.execute_reply":"2024-12-17T13:28:12.89466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_column = 'Premium Amount'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T13:28:12.898017Z","iopub.execute_input":"2024-12-17T13:28:12.898334Z","iopub.status.idle":"2024-12-17T13:28:12.903628Z","shell.execute_reply.started":"2024-12-17T13:28:12.898302Z","shell.execute_reply":"2024-12-17T13:28:12.902325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in categorical_columns:\n    if col == 'Policy Start Date': \n        continue\n        \n    unique_categories = df_train[col].nunique()\n    palette = sns.color_palette('tab10', unique_categories)  \n    \n    plt.figure(figsize=(8, 4))\n    category_counts = df_train[col].value_counts()\n    category_colors = [palette[i] for i in range(len(category_counts))]\n    \n    plt.bar(category_counts.index, category_counts.values, color=category_colors)\n    plt.title(f'Distribution of {col}')\n    plt.xlabel(col)\n    plt.ylabel('Count')\n    plt.xticks(rotation=45)\n    plt.show()\n    \n    plt.figure(figsize=(8, 4))\n    sns.boxplot(\n        data=df_train, \n        x=col, \n        y=target_column, \n        palette=palette \n    )\n    plt.title(f'{target_column} vs {col} (Boxplot)')\n    plt.xlabel(col)\n    plt.ylabel(target_column)\n    plt.xticks(rotation=45)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:08:27.494287Z","iopub.execute_input":"2024-12-17T14:08:27.494727Z","iopub.status.idle":"2024-12-17T14:08:39.178462Z","shell.execute_reply.started":"2024-12-17T14:08:27.494685Z","shell.execute_reply":"2024-12-17T14:08:39.177319Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Each categorical column is evenly distributed","metadata":{}},{"cell_type":"code","source":"# Calculate the correlation matrix\ncorrelation_matrix = df_train[numerical_columns].corr()\n\n# Plot the heatmap\nplt.figure(figsize=(12, 8))\nsns.heatmap(correlation_matrix, annot=True, fmt=\".2f\", cmap=\"coolwarm\", cbar=True, linewidths=0.5)\nplt.title(\"Correlation Heatmap of Numerical Variables\", fontsize=16)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T13:43:49.187977Z","iopub.execute_input":"2024-12-17T13:43:49.188383Z","iopub.status.idle":"2024-12-17T13:43:50.370564Z","shell.execute_reply.started":"2024-12-17T13:43:49.188347Z","shell.execute_reply":"2024-12-17T13:43:50.369295Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There is nothing really that significant here. No numerial features seems to have a linear relationship with Premium Amount. Elsewhere, credit score and annual income have a slight negative correlation - doesn't seem intuitive. \n\nSo far, looks like non linear models must be considered.","metadata":{}},{"cell_type":"code","source":"# Copy of the DataFrame and target column\ndf_binned = df_train.copy()\ntarget_column = 'Premium Amount'\n\n# Define numerical columns to exclude 'id' and 'Premium Amount'\nfiltered_numerical_columns = [col for col in numerical_columns if col not in ['id', target_column]]\n\n# Color palette for the columns\npalette = sns.color_palette('tab10', len(filtered_numerical_columns))\ncolor_dict = dict(zip(filtered_numerical_columns, palette))\n\n# Grid for subplots\nfig = plt.figure(figsize=(30, 10 * len(filtered_numerical_columns)))\ngs = gridspec.GridSpec(2 * len(filtered_numerical_columns), 2, figure=fig)\n\nfor i, column in enumerate(filtered_numerical_columns):\n\n    # Determine if column is discrete\n    discrete = df_train[column].nunique() <= 50\n\n    # Plot histogram\n    ax_hist = fig.add_subplot(gs[2 * i, 0])\n    sns.histplot(\n        data=df_train, x=column, fill=True, common_norm=False, alpha=0.6,\n        linewidth=0.8, color=color_dict[column], ax=ax_hist, discrete=discrete\n    )\n    ax_hist.set_title(f'Histogram of {column}', fontsize=14)\n\n    # Plot boxplot\n    ax_box = fig.add_subplot(gs[2 * i + 1, 0])\n    sns.boxplot(data=df_train, x=column, ax=ax_box, color=color_dict[column])\n    ax_box.set_title(f'Boxplot of {column}', fontsize=14)\n    sns.despine(ax=ax_box)\n\n    # Violin plot or binned violin plot\n    ax_conditional = fig.add_subplot(gs[2 * i:2 * i + 2, 1])\n    if df_train[column].nunique() <= 10:\n        sns.violinplot(data=df_train, x=column, y=target_column, ax=ax_conditional, \n                       color=color_dict[column], alpha=0.6)\n        ax_conditional.set_title(f'{column} vs {target_column} (Violin Plot)', fontsize=14)\n    else:\n        df_binned['Binned Column'] = pd.cut(df_train[column], bins=10)\n        sns.violinplot(data=df_binned, x='Binned Column', y=target_column, ax=ax_conditional, \n                       color=color_dict[column], alpha=0.6)\n        ax_conditional.set_title(f'{column} (Binned) vs {target_column} (Violin Plot)', fontsize=14)\n        ax_conditional.set_xlabel(f'{column} (Binned)', fontsize=12)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T13:57:10.611463Z","iopub.execute_input":"2024-12-17T13:57:10.61202Z","iopub.status.idle":"2024-12-17T13:57:45.166488Z","shell.execute_reply.started":"2024-12-17T13:57:10.611959Z","shell.execute_reply":"2024-12-17T13:57:45.164981Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"So far, data looks like it will require a non linear model. I will attempt with a neural network.","metadata":{}},{"cell_type":"markdown","source":"# Preprocessing\nReplace missing values with mean where possible.\n\nHandle policy start data by converting dates into cylical features.\n\nOne hot encode categorical features.\n\nNormalise data.","metadata":{}}]}