{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":9178166,"sourceType":"datasetVersion","datasetId":5547076}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Loading Dependencies and files","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:03.342652Z","iopub.execute_input":"2024-12-02T07:56:03.343447Z","iopub.status.idle":"2024-12-02T07:56:06.258283Z","shell.execute_reply.started":"2024-12-02T07:56:03.343393Z","shell.execute_reply":"2024-12-02T07:56:06.257108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\nsub_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")\norg_df = pd.read_csv(\"/kaggle/input/insurance-premium-prediction/Insurance Premium Prediction Dataset.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:06.260406Z","iopub.execute_input":"2024-12-02T07:56:06.260912Z","iopub.status.idle":"2024-12-02T07:56:19.188598Z","shell.execute_reply.started":"2024-12-02T07:56:06.26087Z","shell.execute_reply":"2024-12-02T07:56:19.187599Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Basic Analysis","metadata":{}},{"cell_type":"code","source":"print(train_df.shape)\nprint(\"#####\")\nprint(train_df.describe())\nprint(\"####\")\ntrain_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:19.190671Z","iopub.execute_input":"2024-12-02T07:56:19.191052Z","iopub.status.idle":"2024-12-02T07:56:20.555363Z","shell.execute_reply.started":"2024-12-02T07:56:19.191019Z","shell.execute_reply":"2024-12-02T07:56:20.554389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isnull().sum().plot(kind='bar')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:20.557137Z","iopub.execute_input":"2024-12-02T07:56:20.557431Z","iopub.status.idle":"2024-12-02T07:56:21.579611Z","shell.execute_reply.started":"2024-12-02T07:56:20.557402Z","shell.execute_reply":"2024-12-02T07:56:21.578567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"org_df.isnull().sum().plot(kind='bar')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:21.581122Z","iopub.execute_input":"2024-12-02T07:56:21.581548Z","iopub.status.idle":"2024-12-02T07:56:22.093241Z","shell.execute_reply.started":"2024-12-02T07:56:21.581467Z","shell.execute_reply":"2024-12-02T07:56:22.092126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.concat([train_df.drop(columns=['id']),org_df],axis=0)\ntrain.sample(4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:22.094504Z","iopub.execute_input":"2024-12-02T07:56:22.094834Z","iopub.status.idle":"2024-12-02T07:56:22.612445Z","shell.execute_reply.started":"2024-12-02T07:56:22.094803Z","shell.execute_reply":"2024-12-02T07:56:22.611416Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.sample(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:22.613785Z","iopub.execute_input":"2024-12-02T07:56:22.614057Z","iopub.status.idle":"2024-12-02T07:56:22.676791Z","shell.execute_reply.started":"2024-12-02T07:56:22.61403Z","shell.execute_reply":"2024-12-02T07:56:22.675557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(train.isnull().sum()/len(train)*100).plot(kind='bar')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:22.678335Z","iopub.execute_input":"2024-12-02T07:56:22.678719Z","iopub.status.idle":"2024-12-02T07:56:23.792646Z","shell.execute_reply.started":"2024-12-02T07:56:22.678686Z","shell.execute_reply":"2024-12-02T07:56:23.791552Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:23.794063Z","iopub.execute_input":"2024-12-02T07:56:23.794441Z","iopub.status.idle":"2024-12-02T07:56:24.123019Z","shell.execute_reply.started":"2024-12-02T07:56:23.794399Z","shell.execute_reply":"2024-12-02T07:56:24.121917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols = train.select_dtypes(include=['int','float']).columns.tolist()\ncat_cols = train.select_dtypes(include=['O']).columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:24.126217Z","iopub.execute_input":"2024-12-02T07:56:24.126601Z","iopub.status.idle":"2024-12-02T07:56:25.040283Z","shell.execute_reply.started":"2024-12-02T07:56:24.126547Z","shell.execute_reply":"2024-12-02T07:56:25.039401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_trf = Pipeline(steps=[\n    ('imputer',SimpleImputer(strategy='most_frequent')),\n])\n\nnum_trf = Pipeline(steps=[\n    ('imputer',SimpleImputer(strategy='mean')),\n])\n\nTransformer = ColumnTransformer(transformers=[\n    ('cat_trf',cat_trf,cat_cols),\n    ('num_trf',num_trf,num_cols)\n])\n\nTransformer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:25.041458Z","iopub.execute_input":"2024-12-02T07:56:25.041808Z","iopub.status.idle":"2024-12-02T07:56:25.062797Z","shell.execute_reply.started":"2024-12-02T07:56:25.041776Z","shell.execute_reply":"2024-12-02T07:56:25.061724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_pre = Transformer.fit_transform(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:25.064253Z","iopub.execute_input":"2024-12-02T07:56:25.064696Z","iopub.status.idle":"2024-12-02T07:56:30.691365Z","shell.execute_reply.started":"2024-12-02T07:56:25.064649Z","shell.execute_reply":"2024-12-02T07:56:30.690272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_transformed = Transformer.named_transformers_['num_trf'].transform(train[num_cols])\ncat_transformed = Transformer.named_transformers_['cat_trf'].transform(train[cat_cols])\ntrain_pre_combined = np.hstack([num_transformed, cat_transformed])\ntransformed_columns = num_cols + cat_cols\ntrain_pre_df = pd.DataFrame(train_pre_combined, columns=transformed_columns, index=train.index)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:30.693117Z","iopub.execute_input":"2024-12-02T07:56:30.693568Z","iopub.status.idle":"2024-12-02T07:56:33.189468Z","shell.execute_reply.started":"2024-12-02T07:56:30.693503Z","shell.execute_reply":"2024-12-02T07:56:33.188576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_pre_df.sample(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:33.190768Z","iopub.execute_input":"2024-12-02T07:56:33.191091Z","iopub.status.idle":"2024-12-02T07:56:33.261892Z","shell.execute_reply.started":"2024-12-02T07:56:33.19106Z","shell.execute_reply":"2024-12-02T07:56:33.260622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.sample(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:33.263207Z","iopub.execute_input":"2024-12-02T07:56:33.263622Z","iopub.status.idle":"2024-12-02T07:56:33.332789Z","shell.execute_reply.started":"2024-12-02T07:56:33.263583Z","shell.execute_reply":"2024-12-02T07:56:33.331761Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA - Univariate","metadata":{}},{"cell_type":"markdown","source":"* Checking Distribution of categorical columns","metadata":{}},{"cell_type":"code","source":"for col in cat_cols:\n    train_pre_df[col].value_counts().head(10).plot(kind='bar')\n    plt.xticks(rotation=90)\n    plt.title=f\"Distribution of {col}\"\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:33.334066Z","iopub.execute_input":"2024-12-02T07:56:33.33438Z","iopub.status.idle":"2024-12-02T07:56:37.28859Z","shell.execute_reply.started":"2024-12-02T07:56:33.334349Z","shell.execute_reply":"2024-12-02T07:56:37.287445Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> All the columns have an equal distribution except for the OCCUPATION Column where the number of emplyed is equal to the sum of the rest two categories. We can further improve the analysis by creating buckets on a monthly basis of the Policy start date because the column has a very high cardinality","metadata":{}},{"cell_type":"markdown","source":"# EDA - Bivariate Analysis","metadata":{}},{"cell_type":"code","source":"cat_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:37.290096Z","iopub.execute_input":"2024-12-02T07:56:37.290464Z","iopub.status.idle":"2024-12-02T07:56:37.29721Z","shell.execute_reply.started":"2024-12-02T07:56:37.290417Z","shell.execute_reply":"2024-12-02T07:56:37.296118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:37.298793Z","iopub.execute_input":"2024-12-02T07:56:37.299568Z","iopub.status.idle":"2024-12-02T07:56:37.310597Z","shell.execute_reply.started":"2024-12-02T07:56:37.299491Z","shell.execute_reply":"2024-12-02T07:56:37.309578Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.hexbin(x=train_pre_df['Credit Score'].astype('int'),\n          y=train_pre_df['Premium Amount'].astype('int'),\n          gridsize=30,cmap='Blues')\nplt.colorbar(label='Count in bin')\nplt.xlabel(\"Customer Feedback\")\nplt.ylabel(\"Premium Amount\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T08:18:08.125387Z","iopub.execute_input":"2024-12-02T08:18:08.125804Z","iopub.status.idle":"2024-12-02T08:18:08.7286Z","shell.execute_reply.started":"2024-12-02T08:18:08.12577Z","shell.execute_reply":"2024-12-02T08:18:08.727265Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Most of the Distribution lies in the 0-1000 range**","metadata":{}},{"cell_type":"code","source":"sns.scatterpylabellot(x=train_pre_df['Policy Type'],\n                y=train_pre_df['Premium Amount'].astype('int'),\n                hue = train_pre_df['Property Type'])\n\nplt.legend(title=\"Property Type\", loc=\"upper center\", bbox_to_anchor=(0.5,-0.05), ncol=3)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T08:09:29.439661Z","iopub.execute_input":"2024-12-02T08:09:29.440473Z","iopub.status.idle":"2024-12-02T08:10:01.871383Z","shell.execute_reply.started":"2024-12-02T08:09:29.440435Z","shell.execute_reply":"2024-12-02T08:10:01.870279Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Most of the premium policy types have apartments**","metadata":{}},{"cell_type":"code","source":"sns.scatterplot(x=train_pre_df['Premium Amount'].astype('int'), \n                y=train_pre_df['Marital Status'], \n                hue=train_pre_df['Occupation'])\n\nplt.legend(title='Occupation', loc='upper center', bbox_to_anchor=(0.5, -0.05), ncol=3)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:56:37.311798Z","iopub.execute_input":"2024-12-02T07:56:37.312107Z","iopub.status.idle":"2024-12-02T07:57:07.979943Z","shell.execute_reply.started":"2024-12-02T07:56:37.312064Z","shell.execute_reply":"2024-12-02T07:57:07.978889Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Most of the married man are employed and the distribution of the premium column is same across all the categories**","metadata":{}},{"cell_type":"code","source":"sns.scatterplot(x=train_pre_df['Age'].astype('int'),y = train_pre_df['Annual Income'].astype('int'),hue=train_pre_df['Smoking Status'])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:57:07.981257Z","iopub.execute_input":"2024-12-02T07:57:07.981623Z","iopub.status.idle":"2024-12-02T07:58:23.337961Z","shell.execute_reply.started":"2024-12-02T07:57:07.981588Z","shell.execute_reply":"2024-12-02T07:58:23.336777Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**WOO!! thats a lot of data in the plot.  What we can analysze from this is that there is a kind of equal distrubution of smokers and non smokers along both the axis****","metadata":{}},{"cell_type":"code","source":"plt.hexbin(train_pre_df['Annual Income'].astype('int'), \n           train_pre_df['Premium Amount'].astype('int'), \n           gridsize=30, cmap='Blues')\n\nplt.colorbar(label='Count in bin')\nplt.title = 'Hexbin Plot: Annual Income vs Premium Amount'\nplt.xlabel('Annual Income')\nplt.ylabel('Premium Amount')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:58:23.339301Z","iopub.execute_input":"2024-12-02T07:58:23.339664Z","iopub.status.idle":"2024-12-02T07:58:23.897108Z","shell.execute_reply.started":"2024-12-02T07:58:23.339632Z","shell.execute_reply":"2024-12-02T07:58:23.896094Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Majority of the columns are in the range 0-1000(Premium Amount) and the annual income in the range 0-40000**","metadata":{}},{"cell_type":"code","source":"sns.violinplot(x=train_pre_df['Annual Income'].astype('int'), y=train_pre_df['Policy Type'], hue=train_pre_df['Occupation'])\nplt.title = 'Income Distribution by Policy Type and Occupation'\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:58:23.898447Z","iopub.execute_input":"2024-12-02T07:58:23.898898Z","iopub.status.idle":"2024-12-02T07:58:29.846994Z","shell.execute_reply.started":"2024-12-02T07:58:23.898841Z","shell.execute_reply":"2024-12-02T07:58:29.845849Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**This column also has an even distribution**","metadata":{}},{"cell_type":"code","source":"crosstab = pd.crosstab(train_pre_df['Education Level'], train_pre_df['Gender'])\nsns.heatmap(crosstab, annot=True, cmap=\"Blues\", fmt='g', cbar=False)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:58:29.8482Z","iopub.execute_input":"2024-12-02T07:58:29.848488Z","iopub.status.idle":"2024-12-02T07:58:30.273236Z","shell.execute_reply.started":"2024-12-02T07:58:29.84846Z","shell.execute_reply":"2024-12-02T07:58:30.272262Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Seems like even this field is equall distributed**","metadata":{}},{"cell_type":"code","source":"sns.countplot(data=train_pre_df, x='Gender', hue='Marital Status')\nplt.title ='Gender vs Marital Status'\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T07:58:30.274411Z","iopub.execute_input":"2024-12-02T07:58:30.274841Z","iopub.status.idle":"2024-12-02T07:58:32.142116Z","shell.execute_reply.started":"2024-12-02T07:58:30.274794Z","shell.execute_reply":"2024-12-02T07:58:32.141014Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**The single section is high for both categories**","metadata":{}}]}