{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30805,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-07T17:53:15.559423Z","iopub.execute_input":"2024-12-07T17:53:15.559795Z","iopub.status.idle":"2024-12-07T17:53:15.566363Z","shell.execute_reply.started":"2024-12-07T17:53:15.559765Z","shell.execute_reply":"2024-12-07T17:53:15.565582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T17:53:15.761825Z","iopub.execute_input":"2024-12-07T17:53:15.762075Z","iopub.status.idle":"2024-12-07T17:53:19.182344Z","shell.execute_reply.started":"2024-12-07T17:53:15.76205Z","shell.execute_reply":"2024-12-07T17:53:19.181322Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T17:53:19.184096Z","iopub.execute_input":"2024-12-07T17:53:19.184557Z","iopub.status.idle":"2024-12-07T17:53:19.887797Z","shell.execute_reply.started":"2024-12-07T17:53:19.184515Z","shell.execute_reply":"2024-12-07T17:53:19.886931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T17:53:19.888745Z","iopub.execute_input":"2024-12-07T17:53:19.888982Z","iopub.status.idle":"2024-12-07T17:53:19.894585Z","shell.execute_reply.started":"2024-12-07T17:53:19.888959Z","shell.execute_reply":"2024-12-07T17:53:19.893756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T17:53:19.896353Z","iopub.execute_input":"2024-12-07T17:53:19.89664Z","iopub.status.idle":"2024-12-07T17:53:22.056989Z","shell.execute_reply.started":"2024-12-07T17:53:19.896615Z","shell.execute_reply":"2024-12-07T17:53:22.056003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T17:53:22.058379Z","iopub.execute_input":"2024-12-07T17:53:22.058837Z","iopub.status.idle":"2024-12-07T17:53:22.081012Z","shell.execute_reply.started":"2024-12-07T17:53:22.058787Z","shell.execute_reply":"2024-12-07T17:53:22.080071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T17:53:22.081925Z","iopub.execute_input":"2024-12-07T17:53:22.082575Z","iopub.status.idle":"2024-12-07T17:53:22.648696Z","shell.execute_reply.started":"2024-12-07T17:53:22.082549Z","shell.execute_reply":"2024-12-07T17:53:22.647762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T17:53:22.64986Z","iopub.execute_input":"2024-12-07T17:53:22.650224Z","iopub.status.idle":"2024-12-07T17:53:23.195545Z","shell.execute_reply.started":"2024-12-07T17:53:22.650184Z","shell.execute_reply":"2024-12-07T17:53:23.194665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T17:53:23.196922Z","iopub.execute_input":"2024-12-07T17:53:23.19764Z","iopub.status.idle":"2024-12-07T17:53:23.553235Z","shell.execute_reply.started":"2024-12-07T17:53:23.197595Z","shell.execute_reply":"2024-12-07T17:53:23.552448Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = train_data.copy()\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T18:01:57.807698Z","iopub.execute_input":"2024-12-07T18:01:57.808572Z","iopub.status.idle":"2024-12-07T18:01:57.961498Z","shell.execute_reply.started":"2024-12-07T18:01:57.808532Z","shell.execute_reply":"2024-12-07T18:01:57.960528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Count the occurrences of each age\nage_counts = df['Age'].value_counts()\n\n# Plot the bar chart\nplt.bar(age_counts.index, age_counts.values)\nplt.xlabel('Age')\nplt.ylabel('Count')\nplt.title('Age Distribution')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T18:01:58.054916Z","iopub.execute_input":"2024-12-07T18:01:58.055213Z","iopub.status.idle":"2024-12-07T18:01:58.274007Z","shell.execute_reply.started":"2024-12-07T18:01:58.055185Z","shell.execute_reply":"2024-12-07T18:01:58.273133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\n\nsns.kdeplot(df['Annual Income'], shade=True, color='blue')\nplt.xlabel('Annual Income')\nplt.ylabel('Density')\nplt.title('Income Distribution (KDE)')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T18:01:58.275322Z","iopub.execute_input":"2024-12-07T18:01:58.2756Z","iopub.status.idle":"2024-12-07T18:02:02.939741Z","shell.execute_reply.started":"2024-12-07T18:01:58.275573Z","shell.execute_reply":"2024-12-07T18:02:02.938796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.fillna({\n    'Education Level': 'Unknown',\n    'Occupation': 'Unknown',\n    'Customer Feedback': 'Unknown',\n   'Marital Status' : 'Unknown'\n})\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T18:02:02.941201Z","iopub.execute_input":"2024-12-07T18:02:02.941586Z","iopub.status.idle":"2024-12-07T18:02:03.372863Z","shell.execute_reply.started":"2024-12-07T18:02:02.941553Z","shell.execute_reply":"2024-12-07T18:02:03.372152Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_columns = ['Education Level', 'Occupation', 'Policy Type', 'Location', 'Customer Feedback','Marital Status']\nfor col in categorical_columns:\n    df[col].value_counts().plot(kind='bar', figsize=(8, 4), title=f'{col} Distribution')\n    plt.xlabel(col)\n    plt.ylabel('Count')\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T18:02:03.373852Z","iopub.execute_input":"2024-12-07T18:02:03.374136Z","iopub.status.idle":"2024-12-07T18:02:04.725909Z","shell.execute_reply.started":"2024-12-07T18:02:03.374109Z","shell.execute_reply":"2024-12-07T18:02:04.724997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# List of numerical columns\nnumerical_columns = ['Number of Dependents', 'Health Score', 'Credit Score', 'Previous Claims', 'Age', 'Annual Income','Insurance Duration','Vehicle Age']\n\nfor col in numerical_columns:\n    # Histogram\n    plt.figure(figsize=(8, 4))\n    df[col].dropna().plot(kind='hist', bins=30, title=f'{col} Histogram', alpha=0.7, color='blue')\n    plt.xlabel(col)\n    plt.ylabel('Frequency')\n    plt.show()\n\n    # Boxplot\n    plt.figure(figsize=(8, 4))\n    sns.boxplot(data=df, x=col, color='blue')\n    plt.title(f'{col} Boxplot')\n    plt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T18:02:04.727901Z","iopub.execute_input":"2024-12-07T18:02:04.728263Z","iopub.status.idle":"2024-12-07T18:02:08.652297Z","shell.execute_reply.started":"2024-12-07T18:02:04.728221Z","shell.execute_reply":"2024-12-07T18:02:08.651434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Replace missing values with the mean for Health Score, Credit Score, and Age\ndf['Health Score'] = df['Health Score'].fillna(df['Health Score'].mean())\ndf['Credit Score'] = df['Credit Score'].fillna(df['Credit Score'].mean())\ndf['Age'] = df['Age'].fillna(df['Age'].mean())\n\n# Replace missing values with the median for other columns\ndf['Number of Dependents'] = df['Number of Dependents'].fillna(df['Number of Dependents'].median())\ndf['Previous Claims'] = df['Previous Claims'].fillna(df['Previous Claims'].median())\ndf['Annual Income'] = df['Annual Income'].fillna(df['Annual Income'].median())\ndf['Vehicle Age'] = df['Vehicle Age'].fillna(df['Vehicle Age'].median())\ndf['Insurance Duration'] = df['Insurance Duration'].fillna(df['Insurance Duration'].median())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T18:02:08.653583Z","iopub.execute_input":"2024-12-07T18:02:08.653869Z","iopub.status.idle":"2024-12-07T18:02:08.833497Z","shell.execute_reply.started":"2024-12-07T18:02:08.653842Z","shell.execute_reply":"2024-12-07T18:02:08.832835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T18:02:08.834483Z","iopub.execute_input":"2024-12-07T18:02:08.834734Z","iopub.status.idle":"2024-12-07T18:02:08.858275Z","shell.execute_reply.started":"2024-12-07T18:02:08.83471Z","shell.execute_reply":"2024-12-07T18:02:08.857506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().any()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T18:02:08.859517Z","iopub.execute_input":"2024-12-07T18:02:08.860107Z","iopub.status.idle":"2024-12-07T18:02:09.386686Z","shell.execute_reply.started":"2024-12-07T18:02:08.860068Z","shell.execute_reply":"2024-12-07T18:02:09.385809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nimport pandas as pd\n\n# Identify categorical columns\ncategorical_columns = ['Gender', 'Marital Status', 'Education Level', 'Occupation', \n                       'Location', 'Policy Type', 'Customer Feedback', \n                       'Smoking Status', 'Exercise Frequency', 'Property Type']\n\n# Initialize LabelEncoder\nlabel_encoders = {}\n\nfor col in categorical_columns:\n    # Apply Label Encoding\n    le = LabelEncoder()\n    df[col] = le.fit_transform(df[col].astype(str))  # Handle NaN by converting to string\n    label_encoders[col] = le  # Save the encoder for future reference\n\ndf\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T18:02:09.387947Z","iopub.execute_input":"2024-12-07T18:02:09.388335Z","iopub.status.idle":"2024-12-07T18:02:12.086197Z","shell.execute_reply.started":"2024-12-07T18:02:09.388293Z","shell.execute_reply":"2024-12-07T18:02:12.085361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert 'Policy Start Date' to datetime format\ndf['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n\n# Extract 'Days Since Policy Start'\ndf['Days Since Policy Start'] = (pd.to_datetime('today') - df['Policy Start Date']).dt.days\n\n# Drop all other date-related columns\ndf = df.drop(['Policy Start Date'], axis=1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T18:02:57.703795Z","iopub.execute_input":"2024-12-07T18:02:57.704137Z","iopub.status.idle":"2024-12-07T18:02:57.875957Z","shell.execute_reply.started":"2024-12-07T18:02:57.704106Z","shell.execute_reply":"2024-12-07T18:02:57.874935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T18:02:59.861775Z","iopub.execute_input":"2024-12-07T18:02:59.862572Z","iopub.status.idle":"2024-12-07T18:03:00.584876Z","shell.execute_reply.started":"2024-12-07T18:02:59.86251Z","shell.execute_reply":"2024-12-07T18:03:00.583959Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compute the correlation matrix\ncorrelation_matrix = df.corr()\n\n# Plot the correlation matrix\nplt.figure(figsize=(12, 8))  # Set figure size\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm', fmt=\".2f\", linewidths=0.5)\n\n# Add title\nplt.title(\"Correlation Matrix\", fontsize=16)\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T18:04:06.942319Z","iopub.execute_input":"2024-12-07T18:04:06.943143Z","iopub.status.idle":"2024-12-07T18:04:09.285756Z","shell.execute_reply.started":"2024-12-07T18:04:06.943106Z","shell.execute_reply":"2024-12-07T18:04:09.284869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}