{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":84896,"databundleVersionId":10305135}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:57:08.192368Z","iopub.execute_input":"2024-12-29T06:57:08.192692Z","iopub.status.idle":"2024-12-29T06:57:08.595095Z","shell.execute_reply.started":"2024-12-29T06:57:08.192666Z","shell.execute_reply":"2024-12-29T06:57:08.594029Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🔍 Exploratory Data Analysis (EDA)  \n\nEDA is a vital process that helps us uncover insights and patterns within the dataset. By exploring the data, we can identify trends, inconsistencies, and relationships that guide our next steps. Let’s dive into the analysis! 🚀  \n\n---\n\n### 1️⃣ 📜 **Dataset Overview**  \n\nTo start, we examine the structure and key details of the dataset:  \n\n- **Shape:** Understand the number of rows and columns in the dataset.\n\n- **Feature Types:** Classify the features into numerical, categorical, or mixed types.\n\n- **Missing Values:** Highlight any gaps in the data that may require cleaning or imputation.\n\n- **Duplicates:** Identify redundant entries that may skew our analysis.  \n\n---\n\n### 2️⃣ 🧹 **Missing Values Analysis**  \n\nVisualizing missing values allows us to see the extent and distribution of gaps within the dataset.  \n\n- **Key Insight:** Are missing values localized to specific features or widespread?\n  \n- **Observation:** Features with excessive missing data might need to be dropped, while others may require imputation.  \n\n---\n\n### 3️⃣ 📈 **Correlation Analysis**  \n\nUnderstanding the relationships between features is crucial for feature selection and model building.  \n- **Key Insight:** Which features are strongly correlated with the target variable?\n  \n- **Observation:** Highly correlated independent variables may indicate multicollinearity, which needs to be addressed.  \n\n---  ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import accuracy_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:57:08.596462Z","iopub.execute_input":"2024-12-29T06:57:08.597059Z","iopub.status.idle":"2024-12-29T06:57:09.68702Z","shell.execute_reply.started":"2024-12-29T06:57:08.596993Z","shell.execute_reply":"2024-12-29T06:57:09.686163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\nsubmission = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:57:09.688663Z","iopub.execute_input":"2024-12-29T06:57:09.689152Z","iopub.status.idle":"2024-12-29T06:57:29.205048Z","shell.execute_reply.started":"2024-12-29T06:57:09.689111Z","shell.execute_reply":"2024-12-29T06:57:29.203828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:57:29.206768Z","iopub.execute_input":"2024-12-29T06:57:29.207188Z","iopub.status.idle":"2024-12-29T06:57:29.24625Z","shell.execute_reply.started":"2024-12-29T06:57:29.207156Z","shell.execute_reply":"2024-12-29T06:57:29.245205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape\ntrain.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:57:29.247213Z","iopub.execute_input":"2024-12-29T06:57:29.247575Z","iopub.status.idle":"2024-12-29T06:57:29.881752Z","shell.execute_reply.started":"2024-12-29T06:57:29.247546Z","shell.execute_reply":"2024-12-29T06:57:29.880693Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🧹 Missing Values Analysis with Data Visualization","metadata":{}},{"cell_type":"code","source":"target_column = 'Premium Amount'\n\n# Select categorical and numerical columns (initial)\ncategorical_columns = train.select_dtypes(include=['object']).columns\nnumerical_columns = train.select_dtypes(exclude=['object']).columns\n\n# Print out column information\n\nprint(\"\\nCategorical Columns:\", categorical_columns.tolist())\nprint(\"\\nNumerical Columns:\", numerical_columns.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:57:29.883073Z","iopub.execute_input":"2024-12-29T06:57:29.883444Z","iopub.status.idle":"2024-12-29T06:57:30.076492Z","shell.execute_reply.started":"2024-12-29T06:57:29.883412Z","shell.execute_reply":"2024-12-29T06:57:30.075251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.set_theme() \n\n# Filtrer les colonnes catégorielles et exclure 'Policy Start Date'\nfiltered_columns = [col for col in categorical_columns if col != 'Policy Start Date']\n\n# Créer des sous-graphiques pour barplots et boxplots\nfig, axes = plt.subplots(len(filtered_columns), 2, figsize=(15, 5 * len(filtered_columns)))\n\nfor i, column in enumerate(filtered_columns):\n    # Barplot à gauche\n    sns.countplot(data=train, x=column, ax=axes[i, 0], palette='tab10')\n    axes[i, 0].set_title(f'Distribution of {column}', fontsize=14)\n    axes[i, 0].set_xlabel(column, fontsize=12)\n    axes[i, 0].set_ylabel('Count', fontsize=12)\n    sns.despine(ax=axes[i, 0])\n\n    # Boxplot à droite\n    sns.boxplot(data=train, x=column, y=target_column, ax=axes[i, 1], palette='tab10')\n    axes[i, 1].set_title(f'{column} vs {target_column}', fontsize=14)\n    axes[i, 1].set_xlabel(column, fontsize=12)\n    axes[i, 1].set_ylabel(target_column, fontsize=12)\n    sns.despine(ax=axes[i, 1])\n\nplt.tight_layout()  # Ajustement global des sous-graphiques\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:57:30.077713Z","iopub.execute_input":"2024-12-29T06:57:30.078069Z","iopub.status.idle":"2024-12-29T06:57:46.338658Z","shell.execute_reply.started":"2024-12-29T06:57:30.078039Z","shell.execute_reply":"2024-12-29T06:57:46.337558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(15,9))\nplt.title(\"Visualizing Missing Values\")\nsns.heatmap(train.isnull(), cbar=False, cmap=sns.color_palette('coolwarm'), yticklabels=False);\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:57:46.341484Z","iopub.execute_input":"2024-12-29T06:57:46.341776Z","iopub.status.idle":"2024-12-29T06:58:10.184306Z","shell.execute_reply.started":"2024-12-29T06:57:46.34175Z","shell.execute_reply":"2024-12-29T06:58:10.182919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:58:10.186584Z","iopub.execute_input":"2024-12-29T06:58:10.187005Z","iopub.status.idle":"2024-12-29T06:58:10.801508Z","shell.execute_reply.started":"2024-12-29T06:58:10.18697Z","shell.execute_reply":"2024-12-29T06:58:10.800415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def cleaned_data(df):\n    df['Age'] = df['Age'].fillna(df['Age'].mean())\n    df['Gender'] = df['Gender'].map({'Male':1, 'Female':0})\n    df['Annual Income'] = df['Annual Income'].fillna(df['Annual Income'].mean())\n    df['Marital Status'] = df['Marital Status'].map({'Single':1, 'Married':0, 'Divorced':2})\n    df['Marital Status'] = df['Marital Status'].fillna(1)\n    df['Number of Dependents'] = df['Number of Dependents'].fillna(df['Number of Dependents'].mean())\n    df['Education Level'] = df['Education Level'].map({\"Master's\":1, 'PhD':0, 'High School':2, \"Bachelor's\":3})\n    df['Occupation'] = df['Occupation'].map({'Employed':1, 'Self-Employed':0, 'Unemployed':2})\n    df['Occupation'] = df['Occupation'].fillna(1)\n    df['Health Score'] = df['Health Score'].fillna(df['Health Score'].mean())\n    df['Location'] = df['Location'].map({'Suburban':1, 'Rural':0, 'Urban':2})\n    df['Policy Type'] = df['Policy Type'].map({'Premium':1, 'Comprehensive':0, 'Basic':2})\n    df['Previous Claims'] = df['Previous Claims'].fillna(df['Previous Claims'].mean())\n    df['Vehicle Age'] = df['Vehicle Age'].fillna(df['Vehicle Age'].mean())\n    df['Credit Score'] = df['Credit Score'].fillna(df['Credit Score'].mean())\n    df['Insurance Duration'] = df['Insurance Duration'].fillna(df['Insurance Duration'].mean())\n    df.drop(columns=['Policy Start Date', 'id'], inplace=True)\n    df['Customer Feedback'] = df['Customer Feedback'].map({'Average':1, 'Poor':0, 'Good':2})\n    df['Customer Feedback'] = df['Customer Feedback'].fillna(1)\n    df['Smoking Status'] = df['Smoking Status'].map({'Yes':1, 'No':0})\n    df['Exercise Frequency'] = df['Exercise Frequency'].map({'Monthly':3, 'Weekly':2, 'Rarely':0, \"Daily\":1})\n    df['Property Type'] = df['Property Type'].map({'House':1, 'Apartment':0, 'Condo':2})\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:58:10.802542Z","iopub.execute_input":"2024-12-29T06:58:10.802991Z","iopub.status.idle":"2024-12-29T06:58:10.813419Z","shell.execute_reply.started":"2024-12-29T06:58:10.802949Z","shell.execute_reply":"2024-12-29T06:58:10.812273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cleaned_data(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:58:10.814404Z","iopub.execute_input":"2024-12-29T06:58:10.814686Z","iopub.status.idle":"2024-12-29T06:58:11.967037Z","shell.execute_reply.started":"2024-12-29T06:58:10.814656Z","shell.execute_reply":"2024-12-29T06:58:11.965943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"correlation=train.corr()\nplt.figure(figsize=(10,10))\nsns.heatmap(correlation,cbar=True,square=True,annot=True,fmt='.3f',annot_kws={'size':6},cmap='Blues')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:58:11.968089Z","iopub.execute_input":"2024-12-29T06:58:11.968377Z","iopub.status.idle":"2024-12-29T06:58:14.422121Z","shell.execute_reply.started":"2024-12-29T06:58:11.96835Z","shell.execute_reply":"2024-12-29T06:58:14.421058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:58:14.423251Z","iopub.execute_input":"2024-12-29T06:58:14.423567Z","iopub.status.idle":"2024-12-29T06:58:14.460704Z","shell.execute_reply.started":"2024-12-29T06:58:14.42353Z","shell.execute_reply":"2024-12-29T06:58:14.459551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cleaned_data(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:58:14.461723Z","iopub.execute_input":"2024-12-29T06:58:14.462081Z","iopub.status.idle":"2024-12-29T06:58:15.256618Z","shell.execute_reply.started":"2024-12-29T06:58:14.462051Z","shell.execute_reply":"2024-12-29T06:58:15.255525Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🤖 Model Prediction and Evaluation 🚀","metadata":{}},{"cell_type":"code","source":"y= train['Premium Amount']\nx = train.drop(columns=['Premium Amount'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:58:15.257799Z","iopub.execute_input":"2024-12-29T06:58:15.25833Z","iopub.status.idle":"2024-12-29T06:58:15.351011Z","shell.execute_reply.started":"2024-12-29T06:58:15.258289Z","shell.execute_reply":"2024-12-29T06:58:15.350056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model =  LinearRegression()\nmodel.fit(x, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:58:15.352104Z","iopub.execute_input":"2024-12-29T06:58:15.352403Z","iopub.status.idle":"2024-12-29T06:58:16.738996Z","shell.execute_reply.started":"2024-12-29T06:58:15.352376Z","shell.execute_reply":"2024-12-29T06:58:16.736797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:58:16.740107Z","iopub.execute_input":"2024-12-29T06:58:16.740451Z","iopub.status.idle":"2024-12-29T06:58:16.808594Z","shell.execute_reply.started":"2024-12-29T06:58:16.740421Z","shell.execute_reply":"2024-12-29T06:58:16.807495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:58:16.809421Z","iopub.execute_input":"2024-12-29T06:58:16.809758Z","iopub.status.idle":"2024-12-29T06:58:16.819408Z","shell.execute_reply.started":"2024-12-29T06:58:16.809726Z","shell.execute_reply":"2024-12-29T06:58:16.818467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test = submission['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:58:16.820803Z","iopub.execute_input":"2024-12-29T06:58:16.821245Z","iopub.status.idle":"2024-12-29T06:58:16.838699Z","shell.execute_reply.started":"2024-12-29T06:58:16.821202Z","shell.execute_reply":"2024-12-29T06:58:16.837345Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:58:16.839545Z","iopub.execute_input":"2024-12-29T06:58:16.84022Z","iopub.status.idle":"2024-12-29T06:58:16.853074Z","shell.execute_reply.started":"2024-12-29T06:58:16.840172Z","shell.execute_reply":"2024-12-29T06:58:16.85178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn import svm\ny= train['Premium Amount']\nx = train.drop(columns=['Premium Amount'])\nclassifier=svm.SVC(kernel='linear')\nclassifier.fit(x,y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:59:59.555569Z","iopub.execute_input":"2024-12-29T06:59:59.556003Z","iopub.status.idle":"2024-12-29T06:59:59.569784Z","shell.execute_reply.started":"2024-12-29T06:59:59.555965Z","shell.execute_reply":"2024-12-29T06:59:59.568159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_log_error\n\n# Ensure predictions and true values are non-negative\ny_pred = np.maximum(0, y_pred) \ny_test = np.maximum(0, y_test)  \n\n# Calculate RMSLE\nrmsle = np.sqrt(mean_squared_log_error(y_test, y_pred))\nprint(f'Root Mean Squared Logarithmic Error (RMSLE): {rmsle}')","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-29T06:58:31.201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mysub = pd.DataFrame(submission['id'])","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-29T06:58:31.202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mysub['Premium Amount'] = y_pred","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-29T06:58:31.202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mysub","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-29T06:58:31.202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-29T06:58:31.202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.head()","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-29T06:58:31.202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mysub.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-29T06:58:31.202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}