{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-24T09:00:35.456788Z","iopub.execute_input":"2024-12-24T09:00:35.457733Z","iopub.status.idle":"2024-12-24T09:00:35.464328Z","shell.execute_reply.started":"2024-12-24T09:00:35.457694Z","shell.execute_reply":"2024-12-24T09:00:35.463506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ncolumns_name = df.columns\nfeatures_X = df.drop(columns=[\"id\",\"Marital Status\",\"Policy Start Date\",\"Customer Feedback\",\"Property Type\",\"Policy Type\",\"Premium Amount\"])\npred = df[[\"Premium Amount\"]]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T09:00:37.36683Z","iopub.execute_input":"2024-12-24T09:00:37.367766Z","iopub.status.idle":"2024-12-24T09:00:40.882758Z","shell.execute_reply.started":"2024-12-24T09:00:37.36771Z","shell.execute_reply":"2024-12-24T09:00:40.881821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Exploratory Data Analysis -> Fill in / Remove NaN columns \n\n# Remove columns that have excessive NaN values\nfeatures_X=features_X.drop([\"Occupation\",\"Previous Claims\"],axis=1)\n\n# Impute Age, Credit Score, Health Score and Annual Income with the median\nfeatures_X[\"Age\"] = features_X[\"Age\"].fillna(features_X[\"Age\"].median())\nfeatures_X[\"Annual Income\"] = features_X[\"Annual Income\"].fillna(features_X[\"Annual Income\"].median())\nfeatures_X[\"Health Score\"] = features_X[\"Health Score\"].fillna(features_X[\"Health Score\"].median())\nfeatures_X[\"Credit Score\"] = features_X[\"Credit Score\"].fillna(features_X[\"Credit Score\"].median())\n\n# Impute Number of Dependents with mode \nfeatures_X[\"Number of Dependents\"] = features_X[\"Number of Dependents\"].fillna(features_X[\"Number of Dependents\"].mode()[0])\n\n# Drop the rows where that column has very few missing value\nfeatures_X = features_X.dropna(subset=[\"Vehicle Age\",\"Insurance Duration\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T09:00:44.289356Z","iopub.execute_input":"2024-12-24T09:00:44.28971Z","iopub.status.idle":"2024-12-24T09:00:44.610078Z","shell.execute_reply.started":"2024-12-24T09:00:44.289679Z","shell.execute_reply":"2024-12-24T09:00:44.609398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split features_X to categorial_columns and numerial_columns \ncategorial_columns = features_X.select_dtypes(include=[\"object\",\"category\"])\nnumerical_columns = features_X.select_dtypes(include=[\"number\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T09:00:49.230746Z","iopub.execute_input":"2024-12-24T09:00:49.231421Z","iopub.status.idle":"2024-12-24T09:00:49.31172Z","shell.execute_reply.started":"2024-12-24T09:00:49.231375Z","shell.execute_reply":"2024-12-24T09:00:49.310984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize categorial columns data\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nplt.figure(figsize=(30, 20))  # Set the overall figure size\nfor i, col in enumerate(categorial_columns):\n    plt.subplot(1, len(categorial_columns.columns), i + 1)  # Dynamically handle subplot positions\n    \n    x = categorial_columns[col].value_counts()\n    plt.pie(\n        x.values, \n        labels=x.index, \n        autopct=\"%1.1f%%\", \n        textprops={'fontsize': 14}  # Adjust font size for readability\n    )\n    plt.title(col, fontsize=18)  # Add a title for each subplot\n\nplt.tight_layout()  # Ensure subplots don't overlap\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T09:00:51.530842Z","iopub.execute_input":"2024-12-24T09:00:51.531662Z","iopub.status.idle":"2024-12-24T09:00:52.470529Z","shell.execute_reply.started":"2024-12-24T09:00:51.531626Z","shell.execute_reply":"2024-12-24T09:00:52.469648Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize numerical data\n\nnum_cols = len(numerical_columns.columns)\nrows = (num_cols + 2) // 3\nplt.figure(figsize=(20, rows * 5))\n\n# Plot boxplots for initial data\nfor i, col in enumerate(numerical_columns.columns, 1):\n    plt.subplot(rows, 3, i)\n    sns.boxplot(x=numerical_columns[col])\n    plt.xlabel(col)\n    plt.title(f'Boxplot for {col}', fontsize=14)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T09:00:54.959385Z","iopub.execute_input":"2024-12-24T09:00:54.959704Z","iopub.status.idle":"2024-12-24T09:00:56.25858Z","shell.execute_reply.started":"2024-12-24T09:00:54.959678Z","shell.execute_reply":"2024-12-24T09:00:56.257805Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Replace Outliers with mean value\nwhile True:\n    Q1 = numerical_columns[\"Annual Income\"].quantile(0.25)\n    Q3 = numerical_columns[\"Annual Income\"].quantile(0.75)\n    IQR = Q3 - Q1\n    LowerBound = Q1 - 1.5 * IQR\n    UpperBound = Q3 + 1.5 * IQR\n    \n    # Check for outliers and replace\n    numerical_columns.loc[numerical_columns[\"Annual Income\"] >= UpperBound, \"Annual Income\"] = UpperBound\n    numerical_columns.loc[numerical_columns[\"Annual Income\"] <= LowerBound, \"Annual Income\"] = LowerBound\n\n    outliers = ((numerical_columns[\"Annual Income\"] > UpperBound) | (numerical_columns[\"Annual Income\"] < LowerBound))\n    if not outliers.any():\n        break\n\nprint(sns.histplot(numerical_columns[\"Annual Income\"]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T09:00:58.878283Z","iopub.execute_input":"2024-12-24T09:00:58.879187Z","iopub.status.idle":"2024-12-24T09:00:59.96084Z","shell.execute_reply.started":"2024-12-24T09:00:58.879138Z","shell.execute_reply":"2024-12-24T09:00:59.960024Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy import stats\n\n# Apply Box-Cox transformation first\nnumerical_columns[\"Annual Income\"], fitted_lambda = stats.boxcox(numerical_columns[\"Annual Income\"] + 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T09:01:01.997445Z","iopub.execute_input":"2024-12-24T09:01:01.998077Z","iopub.status.idle":"2024-12-24T09:01:07.423273Z","shell.execute_reply.started":"2024-12-24T09:01:01.998043Z","shell.execute_reply":"2024-12-24T09:01:07.422576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Make both columns consistent with the index across both columns\nnumerical_columns = numerical_columns.reset_index(drop=True)\nX = numerical_columns\n\n# Divide data into training and validation data\npred = pred.loc[X.index]\nX = X.reset_index(drop=True)\npred = pred.reset_index(drop=True)\n\nX_train,X_val,y_train,y_val = train_test_split(X,pred,test_size=0.2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T09:01:09.104156Z","iopub.execute_input":"2024-12-24T09:01:09.104485Z","iopub.status.idle":"2024-12-24T09:01:09.376407Z","shell.execute_reply.started":"2024-12-24T09:01:09.104456Z","shell.execute_reply":"2024-12-24T09:01:09.375685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T09:01:11.55209Z","iopub.execute_input":"2024-12-24T09:01:11.552875Z","iopub.status.idle":"2024-12-24T09:01:11.567965Z","shell.execute_reply.started":"2024-12-24T09:01:11.552839Z","shell.execute_reply":"2024-12-24T09:01:11.567009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model Development\n# Import necessary libraries\nfrom sklearn.linear_model import LinearRegression, Lasso\n\n# Import evaluation metrics\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import cross_val_score\n\nlrmodel = LinearRegression()\nlrmodel.fit(X_train,y_train)\ny_pred = lrmodel.predict(X_val)\nmae = mean_absolute_error(y_val, y_pred)\nprint(f'The Mean Absolute Error for Linear Regression is {mae}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T09:01:15.979049Z","iopub.execute_input":"2024-12-24T09:01:15.97966Z","iopub.status.idle":"2024-12-24T09:01:16.44853Z","shell.execute_reply.started":"2024-12-24T09:01:15.979617Z","shell.execute_reply":"2024-12-24T09:01:16.446947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lasmodel = Lasso()\nlasmodel.fit(X_train,y_train)\ny_pred = lasmodel.predict(X_val)\nmae = mean_absolute_error(y_val, y_pred)\nprint(f\"The Mean Absolute Error for Lasso is {mae}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T09:01:22.191213Z","iopub.execute_input":"2024-12-24T09:01:22.192006Z","iopub.status.idle":"2024-12-24T09:01:22.309567Z","shell.execute_reply.started":"2024-12-24T09:01:22.191972Z","shell.execute_reply":"2024-12-24T09:01:22.307752Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\nfeatures_X = test_df.drop(columns=[\"id\",\"Marital Status\",\"Policy Start Date\",\"Customer Feedback\",\"Property Type\",\"Policy Type\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T09:01:26.227026Z","iopub.execute_input":"2024-12-24T09:01:26.227385Z","iopub.status.idle":"2024-12-24T09:01:29.378972Z","shell.execute_reply.started":"2024-12-24T09:01:26.22735Z","shell.execute_reply":"2024-12-24T09:01:29.377987Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Same preprocessing steps for test data","metadata":{}},{"cell_type":"code","source":"# Remove columns that have excessive NaN values\nfeatures_X=features_X.drop([\"Occupation\",\"Previous Claims\"],axis=1)\n\n# Impute Age, Credit Score, Health Score and Annual Income with the median\nfeatures_X[\"Age\"] = features_X[\"Age\"].fillna(features_X[\"Age\"].median())\nfeatures_X[\"Annual Income\"] = features_X[\"Annual Income\"].fillna(features_X[\"Annual Income\"].median())\nfeatures_X[\"Health Score\"] = features_X[\"Health Score\"].fillna(features_X[\"Health Score\"].median())\nfeatures_X[\"Credit Score\"] = features_X[\"Credit Score\"].fillna(features_X[\"Credit Score\"].median())\n\n# Impute Number of Dependents with mode \nfeatures_X[\"Number of Dependents\"] = features_X[\"Number of Dependents\"].fillna(features_X[\"Number of Dependents\"].mode()[0])\n\n# Drop the rows where that column has very few missing value\nfeatures_X[\"Vehicle Age\"] = features_X[\"Vehicle Age\"].fillna(features_X[\"Vehicle Age\"].median())\nfeatures_X[\"Insurance Duration\"] = features_X[\"Insurance Duration\"].fillna(features_X[\"Insurance Duration\"].median())\n\n# Split features_X to categorial_columns and numerial_columns \nnumerical_columns = features_X.select_dtypes(include=[\"number\"])\n\n\nwhile True:\n    Q1 = numerical_columns[\"Annual Income\"].quantile(0.25)\n    Q3 = numerical_columns[\"Annual Income\"].quantile(0.75)\n    IQR = Q3 - Q1\n    LowerBound = Q1 - 1.5 * IQR\n    UpperBound = Q3 + 1.5 * IQR\n    \n    # Check for outliers and replace\n    numerical_columns.loc[numerical_columns[\"Annual Income\"] >= UpperBound, \"Annual Income\"] = UpperBound\n    numerical_columns.loc[numerical_columns[\"Annual Income\"] <= LowerBound, \"Annual Income\"] = LowerBound\n\n    outliers = ((numerical_columns[\"Annual Income\"] > UpperBound) | (numerical_columns[\"Annual Income\"] < LowerBound))\n    if not outliers.any():\n        break\n\n# Apply Box-Cox transformation first\nnumerical_columns[\"Annual Income\"], fitted_lambda = stats.boxcox(numerical_columns[\"Annual Income\"] + 1)\n\nnumerical_columns = numerical_columns.reset_index(drop=True)\n\nX = numerical_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T09:02:10.102206Z","iopub.execute_input":"2024-12-24T09:02:10.103109Z","iopub.status.idle":"2024-12-24T09:02:15.367017Z","shell.execute_reply.started":"2024-12-24T09:02:10.103061Z","shell.execute_reply":"2024-12-24T09:02:15.366304Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_pred = lrmodel.predict(X).flatten()\nsubmission = pd.DataFrame({\n    \"id\": test_df[\"id\"],\n    \"Premium Amount\": test_pred\n})\nsubmission.to_csv(\"submission.csv\",index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T09:02:17.714615Z","iopub.execute_input":"2024-12-24T09:02:17.715317Z","iopub.status.idle":"2024-12-24T09:02:19.198706Z","shell.execute_reply.started":"2024-12-24T09:02:17.715283Z","shell.execute_reply":"2024-12-24T09:02:19.197975Z"}},"outputs":[],"execution_count":null}]}