{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-23T12:12:45.264929Z","iopub.execute_input":"2024-12-23T12:12:45.265216Z","iopub.status.idle":"2024-12-23T12:12:45.2687Z","shell.execute_reply.started":"2024-12-23T12:12:45.265195Z","shell.execute_reply":"2024-12-23T12:12:45.267913Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nprint(\"Num GPUs Available: \", len(tf.config.list_physical_devices('GPU')))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:44:18.877868Z","iopub.execute_input":"2024-12-27T13:44:18.878179Z","iopub.status.idle":"2024-12-27T13:44:28.563752Z","shell.execute_reply.started":"2024-12-27T13:44:18.87816Z","shell.execute_reply":"2024-12-27T13:44:28.562799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nprint(\"Is GPU available:\", torch.cuda.is_available())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:45:46.287876Z","iopub.execute_input":"2024-12-27T13:45:46.288231Z","iopub.status.idle":"2024-12-27T13:45:49.755933Z","shell.execute_reply.started":"2024-12-27T13:45:46.288204Z","shell.execute_reply":"2024-12-27T13:45:49.755125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!nvcc --version\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:46:40.958429Z","iopub.execute_input":"2024-12-27T13:46:40.958742Z","iopub.status.idle":"2024-12-27T13:46:41.124581Z","shell.execute_reply.started":"2024-12-27T13:46:40.958721Z","shell.execute_reply":"2024-12-27T13:46:41.123628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!nvidia-smi\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:44:57.754594Z","iopub.execute_input":"2024-12-27T13:44:57.755195Z","iopub.status.idle":"2024-12-27T13:44:57.994513Z","shell.execute_reply.started":"2024-12-27T13:44:57.755168Z","shell.execute_reply":"2024-12-27T13:44:57.993633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib as mp\nimport matplotlib.pyplot as plt\n\nimport seaborn as sns\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:30:44.044647Z","iopub.execute_input":"2024-12-27T13:30:44.044842Z","iopub.status.idle":"2024-12-27T13:30:45.043673Z","shell.execute_reply.started":"2024-12-27T13:30:44.044824Z","shell.execute_reply":"2024-12-27T13:30:45.042771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import warnings\n\nwarnings.filterwarnings(action=\"ignore\", message=\"^internal gelsd\")\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:31:53.706864Z","iopub.execute_input":"2024-12-27T13:31:53.707181Z","iopub.status.idle":"2024-12-27T13:31:53.711183Z","shell.execute_reply.started":"2024-12-27T13:31:53.70716Z","shell.execute_reply":"2024-12-27T13:31:53.710303Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_path = \"/kaggle/input/playground-series-s4e12/\"\n\n\ntrain_db = pd.read_csv(dataset_path +'train.csv')\ntest_db = pd.read_csv(dataset_path +'test.csv')\n\n\nsample_db = pd.read_csv(dataset_path + \"sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:31:54.277407Z","iopub.execute_input":"2024-12-27T13:31:54.277693Z","iopub.status.idle":"2024-12-27T13:32:02.716149Z","shell.execute_reply.started":"2024-12-27T13:31:54.277672Z","shell.execute_reply":"2024-12-27T13:32:02.71511Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Train Keys: \", train_db.keys())\n\nprint(\"Test Keys: \", test_db.keys())\n\nprint(\"Sample Keys: \", sample_db.keys())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:32:02.717257Z","iopub.execute_input":"2024-12-27T13:32:02.717502Z","iopub.status.idle":"2024-12-27T13:32:02.725768Z","shell.execute_reply.started":"2024-12-27T13:32:02.717482Z","shell.execute_reply":"2024-12-27T13:32:02.72476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Train Data Shape: \", train_db.shape)\n\nprint(\"Test Data Shape: \", test_db.shape)\n\nprint(\"Sample Data Shape: \", sample_db.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:35:33.803467Z","iopub.execute_input":"2024-12-27T13:35:33.803811Z","iopub.status.idle":"2024-12-27T13:35:33.8096Z","shell.execute_reply.started":"2024-12-27T13:35:33.803789Z","shell.execute_reply":"2024-12-27T13:35:33.808715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_db.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:35:36.003367Z","iopub.execute_input":"2024-12-27T13:35:36.003668Z","iopub.status.idle":"2024-12-27T13:35:36.044385Z","shell.execute_reply.started":"2024-12-27T13:35:36.003647Z","shell.execute_reply":"2024-12-27T13:35:36.04364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_db.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:35:48.645058Z","iopub.execute_input":"2024-12-27T13:35:48.645401Z","iopub.status.idle":"2024-12-27T13:35:49.187679Z","shell.execute_reply.started":"2024-12-27T13:35:48.645373Z","shell.execute_reply":"2024-12-27T13:35:49.186894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_db.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:35:51.022422Z","iopub.execute_input":"2024-12-27T13:35:51.022726Z","iopub.status.idle":"2024-12-27T13:35:51.593595Z","shell.execute_reply.started":"2024-12-27T13:35:51.022706Z","shell.execute_reply":"2024-12-27T13:35:51.592868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_db.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:35:54.18985Z","iopub.execute_input":"2024-12-27T13:35:54.190141Z","iopub.status.idle":"2024-12-27T13:35:54.548868Z","shell.execute_reply.started":"2024-12-27T13:35:54.190122Z","shell.execute_reply":"2024-12-27T13:35:54.548124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_db.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:35:56.405417Z","iopub.execute_input":"2024-12-27T13:35:56.405717Z","iopub.status.idle":"2024-12-27T13:35:56.724924Z","shell.execute_reply.started":"2024-12-27T13:35:56.405695Z","shell.execute_reply":"2024-12-27T13:35:56.724089Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*Understand the structure and relationship between the datasets:\n\ntrain_db: Typically contains features and a target variable for training.\n\ntest_db: Features without the target variable (for predictions).\n\nsample_db: Shows the expected format for predictions.\n\n--------------------\n\n*Key observations from the dataset:\n\n--> Both datasets have missing values, especially in critical columns like Age, Annual Income, Number of Dependents, and Occupation.\n\n--> Many categorical features (Gender, Marital Status, Policy Type, etc.) and a mix of numerical features (Health Score, Credit Score, Vehicle Age, etc.).","metadata":{}},{"cell_type":"code","source":"train_db.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:36:02.500812Z","iopub.execute_input":"2024-12-27T13:36:02.501101Z","iopub.status.idle":"2024-12-27T13:36:03.023562Z","shell.execute_reply.started":"2024-12-27T13:36:02.501081Z","shell.execute_reply":"2024-12-27T13:36:03.022798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_db.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:36:04.788791Z","iopub.execute_input":"2024-12-27T13:36:04.789093Z","iopub.status.idle":"2024-12-27T13:36:05.140082Z","shell.execute_reply.started":"2024-12-27T13:36:04.789072Z","shell.execute_reply":"2024-12-27T13:36:05.13937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_db.hist(bins=50, figsize=(20,15))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:36:07.417335Z","iopub.execute_input":"2024-12-27T13:36:07.417643Z","iopub.status.idle":"2024-12-27T13:36:09.813078Z","shell.execute_reply.started":"2024-12-27T13:36:07.417621Z","shell.execute_reply":"2024-12-27T13:36:09.812125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Train Dataset Summary:\")\nprint(train_db.describe(include='all'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:36:28.364832Z","iopub.execute_input":"2024-12-27T13:36:28.365157Z","iopub.status.idle":"2024-12-27T13:36:30.398101Z","shell.execute_reply.started":"2024-12-27T13:36:28.36513Z","shell.execute_reply":"2024-12-27T13:36:30.397105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\nTest Dataset Summary:\")\nprint(test_db.describe(include='all'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:36:33.518734Z","iopub.execute_input":"2024-12-27T13:36:33.519011Z","iopub.status.idle":"2024-12-27T13:36:34.809827Z","shell.execute_reply.started":"2024-12-27T13:36:33.518992Z","shell.execute_reply":"2024-12-27T13:36:34.809013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nplt.figure(figsize=(12, 8))\nsns.heatmap(train_db.isnull(), cbar=False, cmap='viridis')\nplt.title(\"Missing Values in Train Dataset\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:37:00.52213Z","iopub.execute_input":"2024-12-27T13:37:00.522469Z","iopub.status.idle":"2024-12-27T13:37:21.062839Z","shell.execute_reply.started":"2024-12-27T13:37:00.522444Z","shell.execute_reply":"2024-12-27T13:37:21.061777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Numerical feature distribution\nnum_cols = train_db.select_dtypes(include=['float64', 'int64']).columns\n\nfor col in num_cols:\n    sns.histplot(train_db[col], kde=True)\n    plt.title(f\"Distribution of {col}\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:38:19.208924Z","iopub.execute_input":"2024-12-27T13:38:19.209298Z","iopub.status.idle":"2024-12-27T13:39:04.840837Z","shell.execute_reply.started":"2024-12-27T13:38:19.209267Z","shell.execute_reply":"2024-12-27T13:39:04.839888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Categorical feature distribution\ncat_cols = train_db.drop(columns=['Policy Start Date'])\ncat_cols = cat_cols.select_dtypes(include=['object']).columns\n\nfor col in cat_cols:\n    sns.countplot(x=col, data=train_db)\n    plt.title(f\"Distribution of {col}\")\n    plt.xticks(rotation=45)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:39:04.8421Z","iopub.execute_input":"2024-12-27T13:39:04.842437Z","iopub.status.idle":"2024-12-27T13:39:12.111739Z","shell.execute_reply.started":"2024-12-27T13:39:04.842406Z","shell.execute_reply":"2024-12-27T13:39:12.110797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#corelation float / int data\nobj_data_type = train_db.select_dtypes(include=['object']).columns\n\nnum_datas = train_db.drop(columns=obj_data_type)\n\nplt.figure(figsize=(12, 8))\nsns.heatmap(num_datas.corr(), annot=True, cmap='coolwarm')\nplt.title(\"Correlation Matrix\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T13:39:12.113312Z","iopub.execute_input":"2024-12-27T13:39:12.113644Z","iopub.status.idle":"2024-12-27T13:39:13.173328Z","shell.execute_reply.started":"2024-12-27T13:39:12.113621Z","shell.execute_reply":"2024-12-27T13:39:13.172456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# List of categorical columns\ncategorical_cols = ['Gender', 'Marital Status', 'Education Level', 'Occupation',\n                    'Location', 'Policy Type', 'Customer Feedback', 'Smoking Status',\n                    'Exercise Frequency', 'Property Type']\n\n# Apply Label Encoding\nlabel_encoder = LabelEncoder()\nfor col in categorical_cols:\n    train_db[col] = label_encoder.fit_transform(train_db[col]).astype(float)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import datetime\n\n# Convert to datetime\ntrain_db['Policy Start Date'] = pd.to_datetime(train_db['Policy Start Date'])\n\n# Extract year, month, day, etc.\ntrain_db['Policy Start Year'] = train_db['Policy Start Date'].dt.year\ntrain_db['Policy Start Month'] = train_db['Policy Start Date'].dt.month\ntrain_db['Policy Start Day'] = train_db['Policy Start Date'].dt.day\n\n\n# Calculate days since policy start\ncurrent_date = datetime.datetime.now()\ntrain_db['Days Since Policy Start'] = (current_date - train_db['Policy Start Date']).dt.days\n\nprint(train_db[['Policy Start Year', 'Policy Start Month','Policy Start Day', 'Days Since Policy Start']].head())\n\n# Convert datetime to integer (nanoseconds since epoch) in place\ntrain_db['Policy Start Date'] = train_db['Policy Start Date'].view('int64')\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot histograms for each numerical column\nplt.figure(figsize=(15, 10))\nfor i, col in enumerate(train_db, 1):\n    plt.subplot(5,5, i)  # Adjust the grid size based on the number of columns\n    train_db[col].hist(bins=30, grid=False)\n    plt.title(col)\n    plt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate skewness for numerical columns\nskewness = train_db[num_cols].skew()\nprint(\"Skewness of numerical columns:\\n\", skewness)\n\n# Highlight highly skewed columns\nhighly_skewed = skewness[abs(skewness) > 1]\nprint(\"\\nHighly skewed columns:\\n\", highly_skewed)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Bar plot of skewness\nskewness.sort_values().plot(kind='bar', figsize=(10, 5), title=\"Skewness of Numerical Features\")\nplt.ylabel(\"Skewness Value\")\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', 'Previous Claims', 'Credit Score', 'Vehicle Age', 'Insurance Duration']\nfor col in num_cols:\n    train_db[col].fillna(train_db[col].median(), inplace=True)\n\ntrain_db['Annual Income'].fillna(train_db['Annual Income'].mean(), inplace=True)\n\ntrain_db['Previous Claims'].fillna(0, inplace=True)\n\ncat_cols = ['Marital Status', 'Occupation', 'Customer Feedback']\nfor col in cat_cols:\n    train_db[col].fillna(train_db[col].mode()[0], inplace=True)\n\ntrain_db['Occupation'].fillna('Unknown', inplace=True)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Remaining missing values:\\n\", train_db.isnull().sum())\n\ntrain_db","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-23T12:12:27.457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(22, 22))\n#sns.heatmap(train_db.corr(), annot=True, cmap='coolwarm')\n\nsns.heatmap(train_db.corr(), cbar=False, annot=True, square=True, fmt='.2f', annot_kws={'size': 8})\n\nplt.title(\"Correlation Matrix\")\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# correlation with target \ncorrmat = train_db.corr()\nsorted_corrmat = corrmat.iloc[:, 0].sort_values(ascending=False)\nsorted_corrmat","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Now similarly convert the data in Test file and remove the missing values\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# List of categorical columns\ncategorical_cols = ['Gender', 'Marital Status', 'Education Level', 'Occupation',\n                    'Location', 'Policy Type', 'Customer Feedback', 'Smoking Status',\n                    'Exercise Frequency', 'Property Type']\n\n# Apply Label Encoding\nlabel_encoder = LabelEncoder()\nfor col in categorical_cols:\n    test_db[col] = label_encoder.fit_transform(test_db[col]).astype(float)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import datetime\n\n# Convert to datetime\ntest_db['Policy Start Date'] = pd.to_datetime(test_db['Policy Start Date'])\n\n# Extract year, month, day, etc.\ntest_db['Policy Start Year'] = test_db['Policy Start Date'].dt.year\ntest_db['Policy Start Month'] = test_db['Policy Start Date'].dt.month\ntest_db['Policy Start Day'] = test_db['Policy Start Date'].dt.day\n\n\n# Calculate days since policy start\ncurrent_date = datetime.datetime.now()\ntest_db['Days Since Policy Start'] = (current_date - test_db['Policy Start Date']).dt.days\n\nprint(test_db[['Policy Start Year', 'Policy Start Month','Policy Start Day', 'Days Since Policy Start']].head())\n\n\n# Convert datetime to integer (nanoseconds since epoch) in place\ntest_db['Policy Start Date'] = test_db['Policy Start Date'].view('int64')\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_db.dtypes)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', 'Previous Claims', 'Credit Score', 'Vehicle Age', 'Insurance Duration']\nfor col in num_cols:\n    test_db[col].fillna(test_db[col].median(), inplace=True)\n\ntest_db['Annual Income'].fillna(test_db['Annual Income'].mean(), inplace=True)\n\ntest_db['Previous Claims'].fillna(0, inplace=True)\n\ncat_cols = ['Marital Status', 'Occupation', 'Customer Feedback']\nfor col in cat_cols:\n    test_db[col].fillna(test_db[col].mode()[0], inplace=True)\n\ntest_db['Occupation'].fillna('Unknown', inplace=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_db.isnull().sum()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare the correlation data\ncorrelation_data = {\n    'Feature': [\n        'Health Score', 'Credit Score', 'Days Since Policy Start',\n        'Number of Dependents', 'Policy Start Month', 'Smoking Status',\n        'Occupation', 'Customer Feedback', 'Location', 'Age',\n        'Previous Claims', 'Premium Amount', 'Property Type',\n        'Insurance Duration', 'Exercise Frequency', 'Policy Type',\n        'Policy Start Day', 'Policy Start Date', 'Annual Income',\n        'Education Level', 'Policy Start Year', 'Gender',\n        'Vehicle Age', 'Marital Status'\n    ],\n    'Correlation': [\n        0.001340, 0.000909, 0.000829, 0.000717, 0.000621, 0.000611,\n        0.000496, 0.000330, 0.000071, -0.000134, -0.000135, -0.000292,\n        -0.000319, -0.000350, -0.000584, -0.000677, -0.000692,\n        -0.000829, -0.000864, -0.000868, -0.000912, -0.001454,\n        -0.001461, -0.001597\n    ]\n}\n\n\ncorrelation_df = pd.DataFrame(correlation_data)\n\n# Use absolute values for proportions\ncorrelation_df['Absolute Correlation'] = correlation_df['Correlation'].abs()\n\n# Normalize the absolute values for pie chart proportions\ncorrelation_df['Proportion'] = correlation_df['Absolute Correlation'] / correlation_df['Absolute Correlation'].sum()\n\ncorrelation_df.sort_values(by='Proportion', ascending=False, inplace=True)\ncorrelation_df.head()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport matplotlib.cm as cm\n\n# Generate a colormap\ncolors = cm.tab20.colors[:len(correlation_df)]\n\n# Pie chart for all features\nplt.figure(figsize=(10, 10))\nplt.pie(\n    correlation_df['Proportion'], \n    labels=correlation_df['Feature'], \n    autopct='%1.1f%%', \n    startangle=90, \n    colors=colors\n)\nplt.title('Feature Correlation Proportions')\n\nplt.savefig('feature_correlation_pie_chart_all.png', dpi=300, bbox_inches='tight')\n\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Feature Correlation Proportions\nThis chart highlights the contribution of all features to their correlation with the target (Premium Amount).\n\nKey Observations:\nFeatures like Marital Status, Vehicle Age, Gender, and Health Score have the highest absolute correlation values.\n\nFeatures such as Customer Feedback and Policy Start Day have minimal correlation and may contribute less to prediction performance.","metadata":{}},{"cell_type":"code","source":"# Focus on top 10 features\ntop_features = correlation_df.head(10)\n\n# Generate colors for the top features\ncolors = cm.tab10.colors[:len(top_features)]\n\n# Pie chart for top features\nplt.figure(figsize=(10, 10))\nplt.pie(\n    top_features['Proportion'], \n    labels=top_features['Feature'], \n    autopct='%1.1f%%', \n    startangle=90, \n    colors=colors\n)\nplt.title('Top 10 Features Correlation Proportions')\n\nplt.savefig('feature_correlation_pie_chart_top_10.png', dpi=300, bbox_inches='tight')\n\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Top 10 Feature Correlation Proportions\nFocuses on the most significant features.\n\nKey Observations:\nFeatures such as Vehicle Age, Gender, and Health Score dominate the top 10, indicating their strong predictive potential.\n\nThese features should be prioritized during feature selection for model development.","metadata":{}},{"cell_type":"code","source":"# now all the data is ready for training and testing\n\n# prepare the data first for modeling","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Define features (X) and target (y)\nX = train_db.drop(columns=['Premium Amount', 'id'])  # Drop target and identifier\ny = train_db['Premium Amount']  # Target variable\n\n\n# Split the data into 80% training and 20% validation\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Verify the split\nprint(f\"Training data shape: {X_train.shape}, Validation data shape: {X_val.shape}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\n# Initialize scaler\nscaler = StandardScaler()\n\n# Scale the training and validation features\nX_train_scaled = scaler.fit_transform(X_train)\nX_val_scaled = scaler.transform(X_val)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"markdown","source":"# Train a Baseline Model\n","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_squared_log_error\nimport numpy as np\n\n# Initialize and train the model\nbaseline_model = LinearRegression()\nbaseline_model.fit(X_train_scaled, y_train)\n\n# Predict on the validation set\ny_pred_baseline = baseline_model.predict(X_val_scaled)\n\n# Define RMSLE function\ndef rmsle(y_true, y_pred):\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))\n\n# Evaluate the model using RMSLE\nprint(\"Baseline Model RMSLE:\", rmsle(y_val, y_pred_baseline))\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train Advanced Models","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\n# Initialize and train the model\n# rf_model = RandomForestRegressor(random_state=42, n_estimators=100, max_depth=10)\n# rf_model.fit(X_train_scaled, y_train)\n\nrf_model = RandomForestRegressor(random_state=42, n_estimators=100, max_depth=10, n_jobs=-1)\nrf_model.fit(X_train_scaled, y_train)\n\n# Predict on the validation set\ny_pred_rf = rf_model.predict(X_val_scaled)\n\n# Evaluate the model\nprint(\"Random Forest RMSLE:\", rmsle(y_val, y_pred_rf))\n","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-23T12:12:27.455Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# XGBoost Regressor","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBRegressor\n\n# Initialize and train the model\nxgb_model = XGBRegressor(random_state=42, n_estimators=100, learning_rate=0.1, max_depth=6)\nxgb_model.fit(X_train_scaled, y_train)\n\n# Predict on the validation set\ny_pred_xgb = xgb_model.predict(X_val_scaled)\n\n# Evaluate the model\nprint(\"XGBoost RMSLE:\", rmsle(y_val, y_pred_xgb))\n","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-23T12:12:27.455Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Perform Cross-Validation","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\n\n# Perform cross-validation\ncv_scores_rf = cross_val_score(rf_model, X_train_scaled, y_train, scoring='neg_mean_squared_log_error', cv=5)\ncv_rmsle_rf = np.sqrt(-cv_scores_rf.mean())\n\nprint(\"Random Forest Cross-Validation RMSLE:\", cv_rmsle_rf)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-23T12:12:27.455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Perform cross-validation\ncv_scores_xgb = cross_val_score(xgb_model, X_train_scaled, y_train, scoring='neg_mean_squared_log_error', cv=5)\ncv_rmsle_xgb = np.sqrt(-cv_scores_xgb.mean())\n\nprint(\"XGBoost Cross-Validation RMSLE:\", cv_rmsle_xgb)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-23T12:12:27.456Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Hyperparameter Tuning","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\n\n# Define parameter grid\nparam_grid_rf = {\n    'n_estimators': [100, 200],\n    'max_depth': [10, 20],\n    'min_samples_split': [2, 5]\n}\n\n# Perform grid search\ngrid_search_rf = GridSearchCV(estimator=RandomForestRegressor(random_state=42),\n                              param_grid=param_grid_rf,\n                              scoring='neg_mean_squared_log_error',\n                              cv=3)\n\ngrid_search_rf.fit(X_train_scaled, y_train)\n\n# Best model and parameters\nbest_rf_model = grid_search_rf.best_estimator_\nprint(\"Best Random Forest Parameters:\", grid_search_rf.best_params_)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-23T12:12:27.456Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":" # Predict on Test Data","metadata":{}},{"cell_type":"code","source":"# Scale the test features\nX_test_scaled = scaler.transform(test_db.drop(columns=['id', 'Policy Start Date']))\n\n# Predict using the best model\ntest_predictions = best_rf_model.predict(X_test_scaled)\n\n# Prepare the submission file\nsample_submission['Premium Amount'] = test_predictions\nsample_submission.to_csv('submission.csv', index=False)\nprint(\"Predictions saved to submission.csv\")\n","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-23T12:12:27.456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Prepare the Data for CNN","metadata":{}},{"cell_type":"code","source":"# Reshape training and validation data for CNN\nX_train_cnn = np.expand_dims(X_train_scaled, axis=2)  # Add a channel dimension\nX_val_cnn = np.expand_dims(X_val_scaled, axis=2)      # Add a channel dimension\n\nprint(\"Shape of CNN Training Data:\", X_train_cnn.shape)\nprint(\"Shape of CNN Validation Data:\", X_val_cnn.shape)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-23T12:12:27.456Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Build the CNN Model","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv1D, Flatten, Dense, Dropout, MaxPooling1D\n\n# Define the CNN model\ncnn_model = Sequential([\n    Conv1D(filters=32, kernel_size=3, activation='relu', input_shape=(X_train_cnn.shape[1], X_train_cnn.shape[2])),\n    MaxPooling1D(pool_size=2),\n    Dropout(0.2),\n    Conv1D(filters=64, kernel_size=3, activation='relu'),\n    MaxPooling1D(pool_size=2),\n    Flatten(),\n    Dense(64, activation='relu'),\n    Dropout(0.3),\n    Dense(1, activation='linear')  # Regression output\n])\n\n# Compile the model\ncnn_model.compile(optimizer='adam', loss='mean_squared_logarithmic_error', metrics=['mean_squared_error'])\ncnn_model.summary()\n","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-23T12:12:27.456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the CNN\nhistory = cnn_model.fit(\n    X_train_cnn, y_train,\n    validation_data=(X_val_cnn, y_val),\n    epochs=50,\n    batch_size=32,\n    verbose=1\n)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-23T12:12:27.456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Evaluate the CNN\n\n# Predict on the validation set\ny_pred_cnn = cnn_model.predict(X_val_cnn)\n\n# Compute RMSLE\nrmsle = lambda y_true, y_pred: np.sqrt(mean_squared_log_error(y_true, y_pred))\nprint(\"CNN RMSLE:\", rmsle(y_val, y_pred_cnn))\n","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-23T12:12:27.456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize Training Progress\n\nimport matplotlib.pyplot as plt\n\n# Plot training and validation loss\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Training and Validation Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-23T12:12:27.456Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Predict on Test Data","metadata":{}},{"cell_type":"code","source":"# Reshape test data for CNN\nX_test_cnn = np.expand_dims(X_test_scaled, axis=2)\n\n# Predict using the CNN model\ntest_predictions_cnn = cnn_model.predict(X_test_cnn)\n\n# Prepare submission file\nsample_submission['Premium Amount'] = test_predictions_cnn.flatten()\nsample_submission.to_csv('cnn_submission.csv', index=False)\nprint(\"CNN Predictions saved to cnn_submission.csv\")\n","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-23T12:12:27.456Z"}},"outputs":[],"execution_count":null}]}