{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport warnings\nimport plotly.express as px","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:44.507004Z","iopub.execute_input":"2026-08-05T17:38:44.50738Z","iopub.status.idle":"2026-08-05T17:38:45.626346Z","shell.execute_reply.started":"2026-08-05T17:38:44.507359Z","shell.execute_reply":"2026-08-05T17:38:45.625759Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"warnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:45.627354Z","iopub.execute_input":"2026-08-05T17:38:45.627813Z","iopub.status.idle":"2026-08-05T17:38:45.631807Z","shell.execute_reply.started":"2026-08-05T17:38:45.627793Z","shell.execute_reply":"2026-08-05T17:38:45.63104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:45.632584Z","iopub.execute_input":"2026-08-05T17:38:45.632824Z","iopub.status.idle":"2026-08-05T17:38:51.135718Z","shell.execute_reply.started":"2026-08-05T17:38:45.632804Z","shell.execute_reply":"2026-08-05T17:38:51.135073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:51.13727Z","iopub.execute_input":"2026-08-05T17:38:51.137505Z","iopub.status.idle":"2026-08-05T17:38:51.732751Z","shell.execute_reply.started":"2026-08-05T17:38:51.137487Z","shell.execute_reply":"2026-08-05T17:38:51.73173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# dropna \ndf_copy = df.copy()\ndf_copy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:51.733588Z","iopub.execute_input":"2026-08-05T17:38:51.734518Z","iopub.status.idle":"2026-08-05T17:38:52.523814Z","shell.execute_reply.started":"2026-08-05T17:38:51.734497Z","shell.execute_reply":"2026-08-05T17:38:52.522904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_copy.dropna(inplace=True)\ndf_copy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:52.524608Z","iopub.execute_input":"2026-08-05T17:38:52.524822Z","iopub.status.idle":"2026-08-05T17:38:53.456611Z","shell.execute_reply.started":"2026-08-05T17:38:52.524806Z","shell.execute_reply":"2026-08-05T17:38:53.455984Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **`TASK - Business Edition`**\n## Calculate the percentaage of missing values ? \n## Handle data types \n## Plotting the data ( EDA ) to explore potential handling techniques \n## Detect the outlier \n## Explore How to visulaize the Date and time data in column ( Policy Start Date ) ","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:53.457407Z","iopub.execute_input":"2026-08-05T17:38:53.457722Z","iopub.status.idle":"2026-08-05T17:38:54.101489Z","shell.execute_reply.started":"2026-08-05T17:38:53.457695Z","shell.execute_reply":"2026-08-05T17:38:54.100574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"count_age_na = df['Age'].isnull().sum() # Ammar\ncount_age_na","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:54.102269Z","iopub.execute_input":"2026-08-05T17:38:54.10258Z","iopub.status.idle":"2026-08-05T17:38:54.112298Z","shell.execute_reply.started":"2026-08-05T17:38:54.102536Z","shell.execute_reply":"2026-08-05T17:38:54.111488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"( count_age_na/1200000 ) * 100 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:54.113999Z","iopub.execute_input":"2026-08-05T17:38:54.114266Z","iopub.status.idle":"2026-08-05T17:38:54.137323Z","shell.execute_reply.started":"2026-08-05T17:38:54.114247Z","shell.execute_reply":"2026-08-05T17:38:54.136627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:54.140031Z","iopub.execute_input":"2026-08-05T17:38:54.140344Z","iopub.status.idle":"2026-08-05T17:38:54.148449Z","shell.execute_reply.started":"2026-08-05T17:38:54.140322Z","shell.execute_reply":"2026-08-05T17:38:54.147887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.shape[1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:54.149579Z","iopub.execute_input":"2026-08-05T17:38:54.149778Z","iopub.status.idle":"2026-08-05T17:38:54.158818Z","shell.execute_reply.started":"2026-08-05T17:38:54.149763Z","shell.execute_reply":"2026-08-05T17:38:54.158271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.shape[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:54.159751Z","iopub.execute_input":"2026-08-05T17:38:54.160016Z","iopub.status.idle":"2026-08-05T17:38:54.16956Z","shell.execute_reply.started":"2026-08-05T17:38:54.15997Z","shell.execute_reply":"2026-08-05T17:38:54.168972Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Clean code \n# Resualbility ","metadata":{}},{"cell_type":"code","source":"for i in df.columns :\n    col_nun = df[i].isnull().sum() \n    na_per = ( col_nun / df.shape[0] ) * 100 \n\n    print(f\"missing in {i} is : {na_per.round(2)} % \")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:54.170315Z","iopub.execute_input":"2026-08-05T17:38:54.170654Z","iopub.status.idle":"2026-08-05T17:38:54.770788Z","shell.execute_reply.started":"2026-08-05T17:38:54.170638Z","shell.execute_reply":"2026-08-05T17:38:54.770158Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **`Step 1 - Handling Data types`**","metadata":{}},{"cell_type":"markdown","source":"## change the data type of Policy Start Date to Date&time","metadata":{}},{"cell_type":"code","source":"df['Policy Start Date'] = df['Policy Start Date'].astype('datetime64[ns]')\ndf.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:54.77147Z","iopub.execute_input":"2026-08-05T17:38:54.771786Z","iopub.status.idle":"2026-08-05T17:38:55.830686Z","shell.execute_reply.started":"2026-08-05T17:38:54.771756Z","shell.execute_reply":"2026-08-05T17:38:55.829792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat = df.select_dtypes('object').columns\ncat # Hannan ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:55.831446Z","iopub.execute_input":"2026-08-05T17:38:55.831769Z","iopub.status.idle":"2026-08-05T17:38:56.26867Z","shell.execute_reply.started":"2026-08-05T17:38:55.831749Z","shell.execute_reply":"2026-08-05T17:38:56.268075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in cat:\n    df[i] = df[i].astype(\"category\")\n\ndf.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:56.269354Z","iopub.execute_input":"2026-08-05T17:38:56.269593Z","iopub.status.idle":"2026-08-05T17:38:57.200905Z","shell.execute_reply.started":"2026-08-05T17:38:56.269568Z","shell.execute_reply":"2026-08-05T17:38:57.200026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in df.columns :\n    print(df[i].value_counts() )\n    print(\"___________________\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:57.201831Z","iopub.execute_input":"2026-08-05T17:38:57.202956Z","iopub.status.idle":"2026-08-05T17:38:57.683282Z","shell.execute_reply.started":"2026-08-05T17:38:57.202903Z","shell.execute_reply":"2026-08-05T17:38:57.682364Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **`Remember to Do Ordinal Encoding on these columns because the order is important`**","metadata":{}},{"cell_type":"code","source":"df['Previous Claims'] = df['Previous Claims'].fillna(-1.0) # Unkown ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:57.68435Z","iopub.execute_input":"2026-08-05T17:38:57.685004Z","iopub.status.idle":"2026-08-05T17:38:57.704596Z","shell.execute_reply.started":"2026-08-05T17:38:57.684975Z","shell.execute_reply":"2026-08-05T17:38:57.703988Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#df['Number of Dependents'] = df['Number of Dependents'].astype('category')\ndf['Previous Claims'] = df['Previous Claims'].astype('category')\n\ndf.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:57.705371Z","iopub.execute_input":"2026-08-05T17:38:57.705691Z","iopub.status.idle":"2026-08-05T17:38:57.818409Z","shell.execute_reply.started":"2026-08-05T17:38:57.705673Z","shell.execute_reply":"2026-08-05T17:38:57.817724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.drop('id' , axis=1)\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:57.819217Z","iopub.execute_input":"2026-08-05T17:38:57.819981Z","iopub.status.idle":"2026-08-05T17:38:57.882127Z","shell.execute_reply.started":"2026-08-05T17:38:57.819948Z","shell.execute_reply":"2026-08-05T17:38:57.881322Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **`So for now we handled all the data types in our DataFrame , and ended up with ,minimizing the data size to 96.1 MB `**","metadata":{}},{"cell_type":"code","source":"num = df.select_dtypes(['float64' , 'int64']).columns\nnum","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:57.882853Z","iopub.execute_input":"2026-08-05T17:38:57.883086Z","iopub.status.idle":"2026-08-05T17:38:57.974132Z","shell.execute_reply.started":"2026-08-05T17:38:57.883071Z","shell.execute_reply":"2026-08-05T17:38:57.973321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:57.974845Z","iopub.execute_input":"2026-08-05T17:38:57.975158Z","iopub.status.idle":"2026-08-05T17:38:57.980686Z","shell.execute_reply.started":"2026-08-05T17:38:57.975138Z","shell.execute_reply":"2026-08-05T17:38:57.979692Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"_______________________________________________\n# **`Step 2 - Handling Null Values `**","metadata":{}},{"cell_type":"code","source":"def null_percentage(col) : \n    \"\"\" Function for Finding the percentage of missing value \"\"\"\n    col_nun = df[col].isnull().sum() \n    na_per = ( col_nun / df.shape[0] ) * 100 \n   \n    return na_per\n\nfor i in df.columns :\n    if null_percentage(i)> 0 :        \n        print(f'{i} Null : {null_percentage(i).round(2)} %  | {df[i].dtype}')\n        ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:57.981529Z","iopub.execute_input":"2026-08-05T17:38:57.981853Z","iopub.status.idle":"2026-08-05T17:38:58.078503Z","shell.execute_reply.started":"2026-08-05T17:38:57.981828Z","shell.execute_reply":"2026-08-05T17:38:58.07787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.describe().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:58.079254Z","iopub.execute_input":"2026-08-05T17:38:58.079731Z","iopub.status.idle":"2026-08-05T17:38:58.623404Z","shell.execute_reply.started":"2026-08-05T17:38:58.079707Z","shell.execute_reply":"2026-08-05T17:38:58.62275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:58.624126Z","iopub.execute_input":"2026-08-05T17:38:58.62445Z","iopub.status.idle":"2026-08-05T17:38:59.158373Z","shell.execute_reply.started":"2026-08-05T17:38:58.624432Z","shell.execute_reply":"2026-08-05T17:38:59.157682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"marital_mode = df['Marital Status'].mode()[0]\nmarital_mode","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:59.159094Z","iopub.execute_input":"2026-08-05T17:38:59.159312Z","iopub.status.idle":"2026-08-05T17:38:59.17302Z","shell.execute_reply.started":"2026-08-05T17:38:59.159295Z","shell.execute_reply":"2026-08-05T17:38:59.172171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"edu_mode = df['Education Level'].mode()[0]\nedu_mode","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:59.173844Z","iopub.execute_input":"2026-08-05T17:38:59.17412Z","iopub.status.idle":"2026-08-05T17:38:59.223027Z","shell.execute_reply.started":"2026-08-05T17:38:59.174098Z","shell.execute_reply":"2026-08-05T17:38:59.222351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#df['Age'] = df['Age'].fillna(0) # Zero Filling \n#df['Age'] = df['Age'].fillna(df['Age'].mean()) # Avg Filling\ndf['Age'] = df['Age'].fillna(df['Age'].median()) # Median Filling\ndf['Marital Status'] = df['Marital Status'].fillna(marital_mode) # Categorical \n#df.dropna(subset = ['Vehicle Age'] , inplace= True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:59.227799Z","iopub.execute_input":"2026-08-05T17:38:59.227985Z","iopub.status.idle":"2026-08-05T17:38:59.257338Z","shell.execute_reply.started":"2026-08-05T17:38:59.227972Z","shell.execute_reply":"2026-08-05T17:38:59.256532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Annual Income'] = df['Annual Income'].fillna(df['Annual Income'].mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:59.258208Z","iopub.execute_input":"2026-08-05T17:38:59.258535Z","iopub.status.idle":"2026-08-05T17:38:59.279748Z","shell.execute_reply.started":"2026-08-05T17:38:59.258518Z","shell.execute_reply":"2026-08-05T17:38:59.279174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Number of Dependents'] = df['Number of Dependents'].fillna(-1.0) # Unkown ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:59.280442Z","iopub.execute_input":"2026-08-05T17:38:59.280692Z","iopub.status.idle":"2026-08-05T17:38:59.2949Z","shell.execute_reply.started":"2026-08-05T17:38:59.280675Z","shell.execute_reply":"2026-08-05T17:38:59.29418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Number of Dependents'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:59.295574Z","iopub.execute_input":"2026-08-05T17:38:59.295799Z","iopub.status.idle":"2026-08-05T17:38:59.315101Z","shell.execute_reply.started":"2026-08-05T17:38:59.295779Z","shell.execute_reply":"2026-08-05T17:38:59.314473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"null_percentage('Number of Dependents')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:59.315729Z","iopub.execute_input":"2026-08-05T17:38:59.315969Z","iopub.status.idle":"2026-08-05T17:38:59.329714Z","shell.execute_reply.started":"2026-08-05T17:38:59.31595Z","shell.execute_reply":"2026-08-05T17:38:59.328904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Number of Dependents'] = df['Number of Dependents'].astype('category')\ndf['Number of Dependents'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:59.330546Z","iopub.execute_input":"2026-08-05T17:38:59.330778Z","iopub.status.idle":"2026-08-05T17:38:59.362734Z","shell.execute_reply.started":"2026-08-05T17:38:59.330758Z","shell.execute_reply":"2026-08-05T17:38:59.361968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:59.363541Z","iopub.execute_input":"2026-08-05T17:38:59.363781Z","iopub.status.idle":"2026-08-05T17:38:59.434943Z","shell.execute_reply.started":"2026-08-05T17:38:59.363761Z","shell.execute_reply":"2026-08-05T17:38:59.434225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Occupation'] = df['Occupation'].astype(object).fillna(\"Unknown\") # Important note ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:59.435749Z","iopub.execute_input":"2026-08-05T17:38:59.436488Z","iopub.status.idle":"2026-08-05T17:38:59.522033Z","shell.execute_reply.started":"2026-08-05T17:38:59.43646Z","shell.execute_reply":"2026-08-05T17:38:59.521387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Occupation'].dtype","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:59.522805Z","iopub.execute_input":"2026-08-05T17:38:59.523062Z","iopub.status.idle":"2026-08-05T17:38:59.529035Z","shell.execute_reply.started":"2026-08-05T17:38:59.523044Z","shell.execute_reply":"2026-08-05T17:38:59.528356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:59.530053Z","iopub.execute_input":"2026-08-05T17:38:59.530262Z","iopub.status.idle":"2026-08-05T17:38:59.655042Z","shell.execute_reply.started":"2026-08-05T17:38:59.530227Z","shell.execute_reply":"2026-08-05T17:38:59.654405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Occupation'] = df['Occupation'].astype('category')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:59.655657Z","iopub.execute_input":"2026-08-05T17:38:59.655944Z","iopub.status.idle":"2026-08-05T17:38:59.741568Z","shell.execute_reply.started":"2026-08-05T17:38:59.655884Z","shell.execute_reply":"2026-08-05T17:38:59.740963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:59.742287Z","iopub.execute_input":"2026-08-05T17:38:59.74253Z","iopub.status.idle":"2026-08-05T17:38:59.812361Z","shell.execute_reply.started":"2026-08-05T17:38:59.742505Z","shell.execute_reply":"2026-08-05T17:38:59.811703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.hist(\n    df['Health Score'],\n    bins=30,                 # Plotly default-ish\n    color='#636EFA',         # Plotly default blue\n    edgecolor='white'\n)\n\nplt.xlabel('Health Score')\nplt.ylabel('Count')\nplt.title('Health Score Distribution')\n\nplt.grid(axis='y', alpha=0.3)\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:38:59.81319Z","iopub.execute_input":"2026-08-05T17:38:59.813489Z","iopub.status.idle":"2026-08-05T17:39:00.100339Z","shell.execute_reply.started":"2026-08-05T17:38:59.81347Z","shell.execute_reply":"2026-08-05T17:39:00.099703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Health Score'].median()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:39:00.101113Z","iopub.execute_input":"2026-08-05T17:39:00.101404Z","iopub.status.idle":"2026-08-05T17:39:00.122501Z","shell.execute_reply.started":"2026-08-05T17:39:00.101369Z","shell.execute_reply":"2026-08-05T17:39:00.121965Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **`If we Filled the missing values with mean or median This will happen !!!! `**","metadata":{}},{"cell_type":"code","source":"col = df['Health Score']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:39:00.123193Z","iopub.execute_input":"2026-08-05T17:39:00.123436Z","iopub.status.idle":"2026-08-05T17:39:00.126854Z","shell.execute_reply.started":"2026-08-05T17:39:00.123413Z","shell.execute_reply":"2026-08-05T17:39:00.126128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"col = col.fillna(col.mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:39:00.127549Z","iopub.execute_input":"2026-08-05T17:39:00.127781Z","iopub.status.idle":"2026-08-05T17:39:00.154484Z","shell.execute_reply.started":"2026-08-05T17:39:00.127757Z","shell.execute_reply":"2026-08-05T17:39:00.153846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nplt.hist(\n    col,\n    bins=30,                 # Plotly default-ish\n    color='#636EFA',         # Plotly default blue\n    edgecolor='white'\n)\n\nplt.xlabel('Health Score')\nplt.ylabel('Count')\nplt.title('Health Score Distribution')\n\nplt.grid(axis='y', alpha=0.3)\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:39:00.155287Z","iopub.execute_input":"2026-08-05T17:39:00.155626Z","iopub.status.idle":"2026-08-05T17:39:00.368401Z","shell.execute_reply.started":"2026-08-05T17:39:00.155586Z","shell.execute_reply":"2026-08-05T17:39:00.367759Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **`To keep the normal distribution of this column we need to fill the missing cells with random values between ( mean-std ) and ( mean + std ) `**","metadata":{}},{"cell_type":"code","source":"m = df['Health Score'].mean()\nstd = df['Health Score'].std()\nprint(m , std)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:39:00.369226Z","iopub.execute_input":"2026-08-05T17:39:00.369501Z","iopub.status.idle":"2026-08-05T17:39:00.38982Z","shell.execute_reply.started":"2026-08-05T17:39:00.369477Z","shell.execute_reply":"2026-08-05T17:39:00.389058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.random.uniform(m-std , m+std)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:39:00.390569Z","iopub.execute_input":"2026-08-05T17:39:00.39084Z","iopub.status.idle":"2026-08-05T17:39:00.395877Z","shell.execute_reply.started":"2026-08-05T17:39:00.390809Z","shell.execute_reply":"2026-08-05T17:39:00.395037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mask = df['Health Score'].isna()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:39:00.396678Z","iopub.execute_input":"2026-08-05T17:39:00.397104Z","iopub.status.idle":"2026-08-05T17:39:00.408739Z","shell.execute_reply.started":"2026-08-05T17:39:00.397087Z","shell.execute_reply":"2026-08-05T17:39:00.408185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mask","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:39:00.40954Z","iopub.execute_input":"2026-08-05T17:39:00.410009Z","iopub.status.idle":"2026-08-05T17:39:00.416555Z","shell.execute_reply.started":"2026-08-05T17:39:00.409982Z","shell.execute_reply":"2026-08-05T17:39:00.415767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"col = df['Health Score']\ncol","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:39:00.417182Z","iopub.execute_input":"2026-08-05T17:39:00.417357Z","iopub.status.idle":"2026-08-05T17:39:00.428209Z","shell.execute_reply.started":"2026-08-05T17:39:00.417343Z","shell.execute_reply":"2026-08-05T17:39:00.427516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#df['Health Score'] = df['Health Score'].fillna(np.random.uniform(m-std , m+std))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:39:00.428984Z","iopub.execute_input":"2026-08-05T17:39:00.429232Z","iopub.status.idle":"2026-08-05T17:39:00.438269Z","shell.execute_reply.started":"2026-08-05T17:39:00.429211Z","shell.execute_reply":"2026-08-05T17:39:00.437666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#df.loc[mask , col] = np.random.uniform(low = m-std , high = m+std , size=mask.sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:39:00.439141Z","iopub.execute_input":"2026-08-05T17:39:00.439293Z","iopub.status.idle":"2026-08-05T17:39:00.449027Z","shell.execute_reply.started":"2026-08-05T17:39:00.439281Z","shell.execute_reply":"2026-08-05T17:39:00.448286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Number of Dependents'] = df['Number of Dependents'].astype('float64')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:39:00.449875Z","iopub.execute_input":"2026-08-05T17:39:00.450409Z","iopub.status.idle":"2026-08-05T17:39:00.465028Z","shell.execute_reply.started":"2026-08-05T17:39:00.450393Z","shell.execute_reply":"2026-08-05T17:39:00.464378Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **`Step 3 - Data Visulaization`**","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nfor col in num:\n    plt.figure(figsize=(15,8))\n    sns.histplot(x=df[col] , kde=True , bins = 30 , data=df )\n    plt.grid(axis='y', alpha=0.3)\n    plt.tight_layout()\n    plt.title(f\"Histogram of {col}\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:39:00.465649Z","iopub.execute_input":"2026-08-05T17:39:00.465834Z","iopub.status.idle":"2026-08-05T17:39:35.330984Z","shell.execute_reply.started":"2026-08-05T17:39:00.465819Z","shell.execute_reply":"2026-08-05T17:39:35.330123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#test_data.fillna(test_data.mean() , inplace=True )\n# missing percentage\nfor col in num:\n    print(f\"Column Name : {col}\")\n    print(f\" Number of Missing Before Cleaning {df[col].isnull().mean()*100}\")\n    \n    m = df[col].mean()\n    s = df[col].std()\n    si = df[col].isna().sum()\n\n    #pd.Series\n    df[col] = df[col].fillna(pd.Series(np.random.uniform(m-s , m+s , size = int(si))\n                                                      , index = df[df[col].isna()].index ) )\n    \n    print(f\"Number of missing after Handling : {df[col].isnull().mean()*100}\")\n    print(\"________\")\n    \n    #loc , iloc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:39:35.331835Z","iopub.execute_input":"2026-08-05T17:39:35.332866Z","iopub.status.idle":"2026-08-05T17:39:35.695974Z","shell.execute_reply.started":"2026-08-05T17:39:35.332837Z","shell.execute_reply":"2026-08-05T17:39:35.695284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nfor col in num:\n    plt.figure(figsize=(15,8))\n    sns.histplot(x=df[col] , kde=True , bins = 30 , data=df )\n    plt.grid(axis='y', alpha=0.3)\n    plt.tight_layout()\n    plt.title(f\"Histogram of {col}\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:39:35.69672Z","iopub.execute_input":"2026-08-05T17:39:35.697023Z","iopub.status.idle":"2026-08-05T17:40:09.684405Z","shell.execute_reply.started":"2026-08-05T17:39:35.697004Z","shell.execute_reply":"2026-08-05T17:40:09.683637Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Handling Previous Claims \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:09.685272Z","iopub.execute_input":"2026-08-05T17:40:09.685518Z","iopub.status.idle":"2026-08-05T17:40:09.689335Z","shell.execute_reply.started":"2026-08-05T17:40:09.685486Z","shell.execute_reply":"2026-08-05T17:40:09.688363Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **` Step 4 - Encoding Categorical columns `**","metadata":{}},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:09.690236Z","iopub.execute_input":"2026-08-05T17:40:09.690552Z","iopub.status.idle":"2026-08-05T17:40:09.721593Z","shell.execute_reply.started":"2026-08-05T17:40:09.690534Z","shell.execute_reply":"2026-08-05T17:40:09.720975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:09.722552Z","iopub.execute_input":"2026-08-05T17:40:09.722838Z","iopub.status.idle":"2026-08-05T17:40:09.733347Z","shell.execute_reply.started":"2026-08-05T17:40:09.722823Z","shell.execute_reply":"2026-08-05T17:40:09.732705Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Gender']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:09.734095Z","iopub.execute_input":"2026-08-05T17:40:09.734364Z","iopub.status.idle":"2026-08-05T17:40:09.74728Z","shell.execute_reply.started":"2026-08-05T17:40:09.734349Z","shell.execute_reply":"2026-08-05T17:40:09.746398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Label Encoder \nfrom sklearn.preprocessing import LabelEncoder","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:09.748057Z","iopub.execute_input":"2026-08-05T17:40:09.748399Z","iopub.status.idle":"2026-08-05T17:40:09.827529Z","shell.execute_reply.started":"2026-08-05T17:40:09.748374Z","shell.execute_reply":"2026-08-05T17:40:09.826646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LE = LabelEncoder() #LE : Object of Class LabelEncoder\nLE","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:09.828336Z","iopub.execute_input":"2026-08-05T17:40:09.828583Z","iopub.status.idle":"2026-08-05T17:40:09.839748Z","shell.execute_reply.started":"2026-08-05T17:40:09.828563Z","shell.execute_reply":"2026-08-05T17:40:09.839147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in cat :\n    df[i] = LE.fit_transform(df[i])\ndf    \n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:09.840493Z","iopub.execute_input":"2026-08-05T17:40:09.840734Z","iopub.status.idle":"2026-08-05T17:40:11.704833Z","shell.execute_reply.started":"2026-08-05T17:40:09.840713Z","shell.execute_reply":"2026-08-05T17:40:11.703979Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 5 : Data Splitting","metadata":{}},{"cell_type":"code","source":"X = df.drop('Premium Amount' , axis=1)\ny = df['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:11.705658Z","iopub.execute_input":"2026-08-05T17:40:11.706034Z","iopub.status.idle":"2026-08-05T17:40:11.808026Z","shell.execute_reply.started":"2026-08-05T17:40:11.706015Z","shell.execute_reply":"2026-08-05T17:40:11.807433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:11.808899Z","iopub.execute_input":"2026-08-05T17:40:11.809197Z","iopub.status.idle":"2026-08-05T17:40:11.83006Z","shell.execute_reply.started":"2026-08-05T17:40:11.809174Z","shell.execute_reply":"2026-08-05T17:40:11.829377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:11.830793Z","iopub.execute_input":"2026-08-05T17:40:11.831102Z","iopub.status.idle":"2026-08-05T17:40:11.843773Z","shell.execute_reply.started":"2026-08-05T17:40:11.831052Z","shell.execute_reply":"2026-08-05T17:40:11.843197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:11.844571Z","iopub.execute_input":"2026-08-05T17:40:11.844818Z","iopub.status.idle":"2026-08-05T17:40:11.927762Z","shell.execute_reply.started":"2026-08-05T17:40:11.844798Z","shell.execute_reply":"2026-08-05T17:40:11.927252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train , X_test , y_train , y_test = train_test_split(X, y , test_size = 0.25 , random_state = 42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:11.928369Z","iopub.execute_input":"2026-08-05T17:40:11.928617Z","iopub.status.idle":"2026-08-05T17:40:12.297748Z","shell.execute_reply.started":"2026-08-05T17:40:11.928601Z","shell.execute_reply":"2026-08-05T17:40:12.297146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train # scaling ( fit_transform)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:12.298467Z","iopub.execute_input":"2026-08-05T17:40:12.298784Z","iopub.status.idle":"2026-08-05T17:40:12.318635Z","shell.execute_reply.started":"2026-08-05T17:40:12.298765Z","shell.execute_reply":"2026-08-05T17:40:12.31772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test # scaling ( transform )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:12.319386Z","iopub.execute_input":"2026-08-05T17:40:12.319901Z","iopub.status.idle":"2026-08-05T17:40:12.342369Z","shell.execute_reply.started":"2026-08-05T17:40:12.319881Z","shell.execute_reply":"2026-08-05T17:40:12.341588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:12.343063Z","iopub.execute_input":"2026-08-05T17:40:12.343307Z","iopub.status.idle":"2026-08-05T17:40:12.356509Z","shell.execute_reply.started":"2026-08-05T17:40:12.343287Z","shell.execute_reply":"2026-08-05T17:40:12.355602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:12.357385Z","iopub.execute_input":"2026-08-05T17:40:12.357649Z","iopub.status.idle":"2026-08-05T17:40:12.373031Z","shell.execute_reply.started":"2026-08-05T17:40:12.357627Z","shell.execute_reply":"2026-08-05T17:40:12.372374Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **`Step 6 - Features Scalling  `**","metadata":{}},{"cell_type":"code","source":"num","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:12.373696Z","iopub.execute_input":"2026-08-05T17:40:12.373937Z","iopub.status.idle":"2026-08-05T17:40:12.385352Z","shell.execute_reply.started":"2026-08-05T17:40:12.373899Z","shell.execute_reply":"2026-08-05T17:40:12.384639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num = num.drop('Premium Amount' )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:12.386005Z","iopub.execute_input":"2026-08-05T17:40:12.386355Z","iopub.status.idle":"2026-08-05T17:40:12.395568Z","shell.execute_reply.started":"2026-08-05T17:40:12.38633Z","shell.execute_reply":"2026-08-05T17:40:12.394975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:12.396248Z","iopub.execute_input":"2026-08-05T17:40:12.396489Z","iopub.status.idle":"2026-08-05T17:40:12.408204Z","shell.execute_reply.started":"2026-08-05T17:40:12.39647Z","shell.execute_reply":"2026-08-05T17:40:12.407406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:12.408868Z","iopub.execute_input":"2026-08-05T17:40:12.409125Z","iopub.status.idle":"2026-08-05T17:40:12.418778Z","shell.execute_reply.started":"2026-08-05T17:40:12.409109Z","shell.execute_reply":"2026-08-05T17:40:12.418245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = MinMaxScaler()\nscaler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:12.4196Z","iopub.execute_input":"2026-08-05T17:40:12.420523Z","iopub.status.idle":"2026-08-05T17:40:12.4327Z","shell.execute_reply.started":"2026-08-05T17:40:12.420506Z","shell.execute_reply":"2026-08-05T17:40:12.432076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Scalling Training Feature \nX_train[num] = scaler.fit_transform(X_train[num])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:12.433651Z","iopub.execute_input":"2026-08-05T17:40:12.434096Z","iopub.status.idle":"2026-08-05T17:40:12.571799Z","shell.execute_reply.started":"2026-08-05T17:40:12.434052Z","shell.execute_reply":"2026-08-05T17:40:12.570897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:12.572651Z","iopub.execute_input":"2026-08-05T17:40:12.57291Z","iopub.status.idle":"2026-08-05T17:40:12.590402Z","shell.execute_reply.started":"2026-08-05T17:40:12.57289Z","shell.execute_reply":"2026-08-05T17:40:12.589668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test[num] = scaler.transform(X_test[num])\nX_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:12.591148Z","iopub.execute_input":"2026-08-05T17:40:12.591809Z","iopub.status.idle":"2026-08-05T17:40:12.64681Z","shell.execute_reply.started":"2026-08-05T17:40:12.59178Z","shell.execute_reply":"2026-08-05T17:40:12.646192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T17:40:12.647604Z","iopub.execute_input":"2026-08-05T17:40:12.648165Z","iopub.status.idle":"2026-08-05T17:40:12.708559Z","shell.execute_reply.started":"2026-08-05T17:40:12.648145Z","shell.execute_reply":"2026-08-05T17:40:12.707897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = X_train.drop('Policy Start Date' , axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T18:05:36.315765Z","iopub.execute_input":"2026-08-05T18:05:36.316515Z","iopub.status.idle":"2026-08-05T18:05:36.363637Z","shell.execute_reply.started":"2026-08-05T18:05:36.316489Z","shell.execute_reply":"2026-08-05T18:05:36.362778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T18:05:42.946179Z","iopub.execute_input":"2026-08-05T18:05:42.946562Z","iopub.status.idle":"2026-08-05T18:05:42.963954Z","shell.execute_reply.started":"2026-08-05T18:05:42.946538Z","shell.execute_reply":"2026-08-05T18:05:42.963335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = X_test.drop('Policy Start Date' , axis = 1)\nX_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T18:06:06.649308Z","iopub.execute_input":"2026-08-05T18:06:06.649548Z","iopub.status.idle":"2026-08-05T18:06:06.680877Z","shell.execute_reply.started":"2026-08-05T18:06:06.649531Z","shell.execute_reply":"2026-08-05T18:06:06.6803Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **`Step 7 - Modelling `**","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T18:06:11.766298Z","iopub.execute_input":"2026-08-05T18:06:11.76718Z","iopub.status.idle":"2026-08-05T18:06:11.771314Z","shell.execute_reply.started":"2026-08-05T18:06:11.767154Z","shell.execute_reply":"2026-08-05T18:06:11.770468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LR = LinearRegression()\nLR","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T18:06:14.798313Z","iopub.execute_input":"2026-08-05T18:06:14.798668Z","iopub.status.idle":"2026-08-05T18:06:14.804479Z","shell.execute_reply.started":"2026-08-05T18:06:14.798645Z","shell.execute_reply":"2026-08-05T18:06:14.803788Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model Training","metadata":{}},{"cell_type":"code","source":"LR.fit(X_train , y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T18:06:17.73318Z","iopub.execute_input":"2026-08-05T18:06:17.733554Z","iopub.status.idle":"2026-08-05T18:06:18.710532Z","shell.execute_reply.started":"2026-08-05T18:06:17.733532Z","shell.execute_reply":"2026-08-05T18:06:18.709846Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model Testing ( Prediction )","metadata":{}},{"cell_type":"code","source":"y_pred = LR.predict(X_test)\ny_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T18:07:43.332163Z","iopub.execute_input":"2026-08-05T18:07:43.332555Z","iopub.status.idle":"2026-08-05T18:07:43.36584Z","shell.execute_reply.started":"2026-08-05T18:07:43.332532Z","shell.execute_reply":"2026-08-05T18:07:43.364956Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T18:08:04.37811Z","iopub.execute_input":"2026-08-05T18:08:04.378834Z","iopub.status.idle":"2026-08-05T18:08:04.385133Z","shell.execute_reply.started":"2026-08-05T18:08:04.378808Z","shell.execute_reply":"2026-08-05T18:08:04.384436Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model Evaluation ","metadata":{}},{"cell_type":"markdown","source":"## Error Meterics ( mae , mse , rmse )","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T18:09:54.190238Z","iopub.execute_input":"2026-08-05T18:09:54.190978Z","iopub.status.idle":"2026-08-05T18:09:54.194338Z","shell.execute_reply.started":"2026-08-05T18:09:54.190953Z","shell.execute_reply":"2026-08-05T18:09:54.193658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mae = mean_absolute_error(y_test , y_pred)\nprint(f\"Mean Absolute Error of Linear Regression : {mae}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T18:11:01.499683Z","iopub.execute_input":"2026-08-05T18:11:01.500401Z","iopub.status.idle":"2026-08-05T18:11:01.508137Z","shell.execute_reply.started":"2026-08-05T18:11:01.500372Z","shell.execute_reply":"2026-08-05T18:11:01.50713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Premium Amount'].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T18:15:37.949735Z","iopub.execute_input":"2026-08-05T18:15:37.950655Z","iopub.status.idle":"2026-08-05T18:15:38.004751Z","shell.execute_reply.started":"2026-08-05T18:15:37.950628Z","shell.execute_reply":"2026-08-05T18:15:38.004076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Premium Amount'].min()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T18:17:16.309654Z","iopub.execute_input":"2026-08-05T18:17:16.310352Z","iopub.status.idle":"2026-08-05T18:17:16.317008Z","shell.execute_reply.started":"2026-08-05T18:17:16.310324Z","shell.execute_reply":"2026-08-05T18:17:16.316144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Premium Amount'].max()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T18:17:24.06682Z","iopub.execute_input":"2026-08-05T18:17:24.067458Z","iopub.status.idle":"2026-08-05T18:17:24.074108Z","shell.execute_reply.started":"2026-08-05T18:17:24.067433Z","shell.execute_reply":"2026-08-05T18:17:24.073129Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Fitting Metrics ( R2 )","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import r2_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T18:13:41.483241Z","iopub.execute_input":"2026-08-05T18:13:41.483816Z","iopub.status.idle":"2026-08-05T18:13:41.487849Z","shell.execute_reply.started":"2026-08-05T18:13:41.483789Z","shell.execute_reply":"2026-08-05T18:13:41.486736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"r2 = r2_score(y_test , y_pred)\nprint(f\"R2 Score of Linear Regression : {r2}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-05T18:14:23.870391Z","iopub.execute_input":"2026-08-05T18:14:23.870772Z","iopub.status.idle":"2026-08-05T18:14:23.878739Z","shell.execute_reply.started":"2026-08-05T18:14:23.870748Z","shell.execute_reply":"2026-08-05T18:14:23.87812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}