{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30839,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:15:59.3598Z","iopub.execute_input":"2025-02-01T10:15:59.360212Z","iopub.status.idle":"2025-02-01T10:15:59.376863Z","shell.execute_reply.started":"2025-02-01T10:15:59.360183Z","shell.execute_reply":"2025-02-01T10:15:59.375491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.metrics import mean_squared_error, r2_score\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.tree import DecisionTreeRegressor\nimport xgboost as xgb\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:15:59.378357Z","iopub.execute_input":"2025-02-01T10:15:59.378755Z","iopub.status.idle":"2025-02-01T10:15:59.384543Z","shell.execute_reply.started":"2025-02-01T10:15:59.378713Z","shell.execute_reply":"2025-02-01T10:15:59.383293Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Our dataset is already split into training and testing sets. Let's store them in separate dataframes and begin the analysis","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\nX = df_train.iloc[:, :-1]\ny = df_train.iloc[:, -1]\n\n\ndf_train.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:15:59.39439Z","iopub.execute_input":"2025-02-01T10:15:59.394772Z","iopub.status.idle":"2025-02-01T10:16:12.932326Z","shell.execute_reply.started":"2025-02-01T10:15:59.394739Z","shell.execute_reply":"2025-02-01T10:16:12.931408Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:16:12.933449Z","iopub.execute_input":"2025-02-01T10:16:12.933716Z","iopub.status.idle":"2025-02-01T10:16:12.952063Z","shell.execute_reply.started":"2025-02-01T10:16:12.93369Z","shell.execute_reply":"2025-02-01T10:16:12.951181Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's count the number of records in the training and testing datasets","metadata":{}},{"cell_type":"code","source":"df_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:16:12.954Z","iopub.execute_input":"2025-02-01T10:16:12.954276Z","iopub.status.idle":"2025-02-01T10:16:12.972035Z","shell.execute_reply.started":"2025-02-01T10:16:12.954252Z","shell.execute_reply":"2025-02-01T10:16:12.97108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:16:12.97331Z","iopub.execute_input":"2025-02-01T10:16:12.973619Z","iopub.status.idle":"2025-02-01T10:16:12.993398Z","shell.execute_reply.started":"2025-02-01T10:16:12.973593Z","shell.execute_reply":"2025-02-01T10:16:12.992492Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Lets check the null values in both train and test dataset","metadata":{}},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:16:12.994481Z","iopub.execute_input":"2025-02-01T10:16:12.99479Z","iopub.status.idle":"2025-02-01T10:16:13.692922Z","shell.execute_reply.started":"2025-02-01T10:16:12.994759Z","shell.execute_reply":"2025-02-01T10:16:13.692048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:16:13.693826Z","iopub.execute_input":"2025-02-01T10:16:13.694128Z","iopub.status.idle":"2025-02-01T10:16:14.144356Z","shell.execute_reply.started":"2025-02-01T10:16:13.694101Z","shell.execute_reply":"2025-02-01T10:16:14.143333Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Lets now handle the null values before starting the EDA","metadata":{}},{"cell_type":"markdown","source":"Let's separate our dataset into numerical and categorical features","metadata":{}},{"cell_type":"code","source":"numerical_columns = X.select_dtypes(include=[np.number]).columns.tolist()\ncategorical_columns = X.select_dtypes(exclude=[np.number]).columns.tolist()\n\nprint(\"\\nNumerical Columns:\")\nprint(numerical_columns)\nprint(f\"\\nTotal number of numerical columns: {len(numerical_columns)}\")\n\nprint(\"\\nCategorical Columns:\")\nprint(categorical_columns)\nprint(f\"\\nTotal number of categorical columns: {len(categorical_columns)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:16:14.145271Z","iopub.execute_input":"2025-02-01T10:16:14.145548Z","iopub.status.idle":"2025-02-01T10:16:14.327225Z","shell.execute_reply.started":"2025-02-01T10:16:14.145526Z","shell.execute_reply":"2025-02-01T10:16:14.326069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train[numerical_columns] = df_train[numerical_columns].fillna(df_train[numerical_columns].mean())\ndf_test[numerical_columns] = df_test[numerical_columns].fillna(df_test[numerical_columns].mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:16:14.330414Z","iopub.execute_input":"2025-02-01T10:16:14.330684Z","iopub.status.idle":"2025-02-01T10:16:14.917647Z","shell.execute_reply.started":"2025-02-01T10:16:14.330659Z","shell.execute_reply":"2025-02-01T10:16:14.91641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train[categorical_columns] = df_train[categorical_columns].fillna(\"Missing\")\ndf_test[categorical_columns] = df_test[categorical_columns].fillna(\"Missing\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:16:14.919185Z","iopub.execute_input":"2025-02-01T10:16:14.919455Z","iopub.status.idle":"2025-02-01T10:16:17.424687Z","shell.execute_reply.started":"2025-02-01T10:16:14.919433Z","shell.execute_reply":"2025-02-01T10:16:17.423219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:16:17.425676Z","iopub.execute_input":"2025-02-01T10:16:17.426036Z","iopub.status.idle":"2025-02-01T10:16:18.123743Z","shell.execute_reply.started":"2025-02-01T10:16:17.426005Z","shell.execute_reply":"2025-02-01T10:16:18.122701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:16:18.124809Z","iopub.execute_input":"2025-02-01T10:16:18.125224Z","iopub.status.idle":"2025-02-01T10:16:18.57726Z","shell.execute_reply.started":"2025-02-01T10:16:18.125186Z","shell.execute_reply":"2025-02-01T10:16:18.576275Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Let's begin the Exploratory Data Analysis (EDA) of the numerical features.","metadata":{}},{"cell_type":"markdown","source":"Lets start with the basic statistics","metadata":{}},{"cell_type":"code","source":"df_train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:16:18.578186Z","iopub.execute_input":"2025-02-01T10:16:18.578502Z","iopub.status.idle":"2025-02-01T10:16:19.20036Z","shell.execute_reply.started":"2025-02-01T10:16:18.578476Z","shell.execute_reply":"2025-02-01T10:16:19.199308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:16:19.201379Z","iopub.execute_input":"2025-02-01T10:16:19.201697Z","iopub.status.idle":"2025-02-01T10:16:19.20759Z","shell.execute_reply.started":"2025-02-01T10:16:19.201672Z","shell.execute_reply":"2025-02-01T10:16:19.206771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features=['Age','Annual Income','Number of Dependents','Health Score','Previous Claims','Vehicle Age','Credit Score','Insurance Duration']\nplt.figure(figsize=(12, 6))\n\nfor i, feature in enumerate(features, 1):\n    plt.subplot(2, 4, i)  # Adjust grid size based on the number of features\n    sns.boxplot(y=df_train[feature])  # Replace train_dataset with your actual dataset\n    plt.title(feature)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:16:19.208636Z","iopub.execute_input":"2025-02-01T10:16:19.208991Z","iopub.status.idle":"2025-02-01T10:16:21.182356Z","shell.execute_reply.started":"2025-02-01T10:16:19.208961Z","shell.execute_reply":"2025-02-01T10:16:21.181273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def remove_outliers(df, column):\n    Q1 = df[column].quantile(0.25)  # First quartile (25th percentile)\n    Q3 = df[column].quantile(0.75)  # Third quartile (75th percentile)\n    IQR = Q3 - Q1  # Interquartile range\n    \n    # Define lower and upper bounds\n    lower_bound = Q1 - 1.5 * IQR\n    upper_bound = Q3 + 1.5 * IQR\n    \n    # Remove outliers\n    df = df[(df[column] >= lower_bound) & (df[column] <= upper_bound)]\n    return df\n\n# Remove outliers from 'Annual Income' and 'Previous Claims'\ndf_train = remove_outliers(df_train, 'Annual Income')\ndf_train = remove_outliers(df_train, 'Previous Claims')\n\n# Check the dataset after removing outliers\nprint(df_train.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:16:21.183372Z","iopub.execute_input":"2025-02-01T10:16:21.183669Z","iopub.status.idle":"2025-02-01T10:16:21.761087Z","shell.execute_reply.started":"2025-02-01T10:16:21.183643Z","shell.execute_reply":"2025-02-01T10:16:21.760035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features=['Age','Annual Income','Number of Dependents','Health Score','Previous Claims','Vehicle Age','Credit Score','Insurance Duration']\nplt.figure(figsize=(12, 6))\n\nfor i, feature in enumerate(features, 1):\n    plt.subplot(2, 4, i)  # Adjust grid size based on the number of features\n    sns.boxplot(y=df_train[feature])  # Replace train_dataset with your actual dataset\n    plt.title(feature)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:16:21.762123Z","iopub.execute_input":"2025-02-01T10:16:21.762385Z","iopub.status.idle":"2025-02-01T10:16:23.417789Z","shell.execute_reply.started":"2025-02-01T10:16:21.762363Z","shell.execute_reply":"2025-02-01T10:16:23.416838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\n\nfor i, feature in enumerate(features, 1):\n    plt.subplot(2, 4, i)  # Adjust grid size based on the number of features\n    sns.histplot(df_train[feature], bins=30, kde=True)  # KDE=True adds a density curve\n    plt.title(feature)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:16:23.418705Z","iopub.execute_input":"2025-02-01T10:16:23.419007Z","iopub.status.idle":"2025-02-01T10:17:00.530762Z","shell.execute_reply.started":"2025-02-01T10:16:23.418981Z","shell.execute_reply":"2025-02-01T10:17:00.529658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 10))\n\nfor i, feature in enumerate(features, 1):\n    plt.subplot(3, 3, i)  # Adjust grid size based on the number of features\n    sns.scatterplot(x=df_train[feature], y=df_train['Premium Amount'])  # Replace 'Target' with actual target variable\n    plt.title(f\"{feature} vs Target\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:17:00.531725Z","iopub.execute_input":"2025-02-01T10:17:00.532039Z","iopub.status.idle":"2025-02-01T10:17:19.36822Z","shell.execute_reply.started":"2025-02-01T10:17:00.532014Z","shell.execute_reply":"2025-02-01T10:17:19.367104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.pairplot(df_train[features])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:17:19.369409Z","iopub.execute_input":"2025-02-01T10:17:19.369793Z","iopub.status.idle":"2025-02-01T10:19:41.064366Z","shell.execute_reply.started":"2025-02-01T10:17:19.369758Z","shell.execute_reply":"2025-02-01T10:19:41.062918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n# Set up the matplotlib figure\nplt.figure(figsize=(12, 8))\n\ncorr_matrix = df_train[features].corr()\n\n# Set up the matplotlib figure\nplt.figure(figsize=(12, 8))\n\n# Create the heatmap\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', fmt='.2f', linewidths=0.5)\n\n# Display the heatmap\nplt.title('Correlation Heatmap for Numerical Features')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:19:41.065605Z","iopub.execute_input":"2025-02-01T10:19:41.065906Z","iopub.status.idle":"2025-02-01T10:19:41.805873Z","shell.execute_reply.started":"2025-02-01T10:19:41.065875Z","shell.execute_reply":"2025-02-01T10:19:41.804452Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Let's begin the Exploratory Data Analysis (EDA) of the categorical features.","metadata":{}},{"cell_type":"code","source":"featues_categorical = ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location', 'Policy Type', 'Policy Start Date', 'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:19:41.807051Z","iopub.execute_input":"2025-02-01T10:19:41.807402Z","iopub.status.idle":"2025-02-01T10:19:41.811857Z","shell.execute_reply.started":"2025-02-01T10:19:41.80737Z","shell.execute_reply":"2025-02-01T10:19:41.810647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(15, 12))\n\nfor i, feature in enumerate(featues_categorical, 1):\n    plt.subplot(4, 3, i)  # Adjust number of rows and columns based on the number of features\n    sns.countplot(data=df_train, x=feature)\n    plt.title(f'Distribution of {feature}')\n    plt.xticks(rotation=45)  # Rotate x-axis labels if needed\n\nplt.tight_layout() \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T10:19:41.8131Z","iopub.execute_input":"2025-02-01T10:19:41.813484Z","iopub.status.idle":"2025-02-01T11:04:15.526522Z","shell.execute_reply.started":"2025-02-01T10:19:41.813446Z","shell.execute_reply":"2025-02-01T11:04:15.525405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for feature in featues_categorical:\n    plt.figure(figsize=(7, 7))\n    df_train[feature].value_counts().plot.pie(autopct='%1.1f%%', figsize=(7,7))\n    plt.title(f'Proportions of {feature}')\n    plt.ylabel('')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T11:08:52.480678Z","iopub.execute_input":"2025-02-01T11:08:52.481064Z","iopub.status.idle":"2025-02-01T11:43:18.967447Z","shell.execute_reply.started":"2025-02-01T11:08:52.481034Z","shell.execute_reply":"2025-02-01T11:43:18.966277Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Lets start with modelling","metadata":{}},{"cell_type":"markdown","source":"\nSince we have a large dataset, let's use the XGBoost Regressor, as it performs well with large datasets.","metadata":{}},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.impute import SimpleImputer\nfrom xgboost import XGBRegressor\nfrom sklearn.model_selection import train_test_split\n\n\n# Split into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n\n# Identify numerical and categorical columns\nnumerical_features = X.select_dtypes(include=[\"int64\", \"float64\"]).columns\ncategorical_features = X.select_dtypes(include=[\"object\", \"category\"]).columns\n\n# Preprocessing for numerical data\nnumerical_transformer = Pipeline(steps=[\n    \n    (\"scaler\", StandardScaler())  # Standardize numerical features\n])\n\n# Preprocessing for categorical data\ncategorical_transformer = Pipeline(steps=[\n   \n    (\"encoder\", OneHotEncoder(handle_unknown=\"ignore\"))  # One-hot encoding for categorical variables\n])\n\n# Combine transformers into a preprocessor\npreprocessor = ColumnTransformer(transformers=[\n    (\"num\", numerical_transformer, numerical_features),\n    (\"cat\", categorical_transformer, categorical_features)\n])\n\n# Create a pipeline with XGBoost Regressor\npipeline = Pipeline(steps=[\n    (\"preprocessor\", preprocessor),\n    (\"regressor\", XGBRegressor(n_estimators=100, learning_rate=0.1, max_depth=5, random_state=42))\n])\n\n# Train the model\npipeline.fit(X_train, y_train)\n\n# Evaluate the model\nscore = pipeline.score(X_test, y_test)\nprint(f\"Model R² Score: {score:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T11:44:26.033507Z","iopub.execute_input":"2025-02-01T11:44:26.033811Z","iopub.status.idle":"2025-02-01T11:44:59.145235Z","shell.execute_reply.started":"2025-02-01T11:44:26.033785Z","shell.execute_reply":"2025-02-01T11:44:59.144238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test_dataset = preprocessor.transform(df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T12:00:01.577223Z","iopub.execute_input":"2025-02-01T12:00:01.578795Z","iopub.status.idle":"2025-02-01T12:00:06.190988Z","shell.execute_reply.started":"2025-02-01T12:00:01.578738Z","shell.execute_reply":"2025-02-01T12:00:06.189892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = pipeline.predict(X_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T12:00:09.244627Z","iopub.execute_input":"2025-02-01T12:00:09.245013Z","iopub.status.idle":"2025-02-01T12:00:13.292723Z","shell.execute_reply.started":"2025-02-01T12:00:09.244978Z","shell.execute_reply":"2025-02-01T12:00:13.291892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test_pred = pipeline.predict(df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T12:00:52.511605Z","iopub.execute_input":"2025-02-01T12:00:52.512071Z","iopub.status.idle":"2025-02-01T12:01:05.394866Z","shell.execute_reply.started":"2025-02-01T12:00:52.512035Z","shell.execute_reply":"2025-02-01T12:01:05.394074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\nsubmission_df['Premium Amount'] = y_test_pred\nsubmission_df.to_csv('submission.csv', index=False)\nprint(submission_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T12:02:28.756966Z","iopub.execute_input":"2025-02-01T12:02:28.757421Z","iopub.status.idle":"2025-02-01T12:02:30.266995Z","shell.execute_reply.started":"2025-02-01T12:02:28.757388Z","shell.execute_reply":"2025-02-01T12:02:30.266034Z"}},"outputs":[],"execution_count":null}]}