{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"background-color:#fad0c4; color:#355c7d; text-align:center; padding:15px; border-radius:15px; font-size:25px; \">Insurance Prediction Analysis</div>","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\npd.set_option(\"display.max_columns\",None)\nwarnings.filterwarnings(\"ignore\")\n\n\n\ntrain_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:45:31.444705Z","iopub.execute_input":"2024-12-29T13:45:31.444935Z","iopub.status.idle":"2024-12-29T13:45:41.13941Z","shell.execute_reply.started":"2024-12-29T13:45:31.444906Z","shell.execute_reply":"2024-12-29T13:45:41.138566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Shape Of the Train Data : {train_df.shape}\")\nprint(f\"Shape Of the Test Data : {test_df.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:45:41.140441Z","iopub.execute_input":"2024-12-29T13:45:41.140752Z","iopub.status.idle":"2024-12-29T13:45:41.145717Z","shell.execute_reply.started":"2024-12-29T13:45:41.140727Z","shell.execute_reply":"2024-12-29T13:45:41.144676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:45:41.147378Z","iopub.execute_input":"2024-12-29T13:45:41.147668Z","iopub.status.idle":"2024-12-29T13:45:41.71485Z","shell.execute_reply.started":"2024-12-29T13:45:41.147643Z","shell.execute_reply":"2024-12-29T13:45:41.714091Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:45:41.716166Z","iopub.execute_input":"2024-12-29T13:45:41.716415Z","iopub.status.idle":"2024-12-29T13:45:42.075061Z","shell.execute_reply.started":"2024-12-29T13:45:41.716395Z","shell.execute_reply":"2024-12-29T13:45:42.074263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_columns  = train_df.select_dtypes(include = \"object\").columns\nfor col in categorical_columns:\n    print(f\"No of Unique Values in {col} : {train_df[col].nunique()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:45:42.075893Z","iopub.execute_input":"2024-12-29T13:45:42.076158Z","iopub.status.idle":"2024-12-29T13:45:42.938901Z","shell.execute_reply.started":"2024-12-29T13:45:42.076126Z","shell.execute_reply":"2024-12-29T13:45:42.938159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in categorical_columns:\n    print(f\"No of Unique Values in {col} : {test_df[col].nunique()}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:32:07.59913Z","iopub.execute_input":"2024-12-29T13:32:07.599459Z","iopub.status.idle":"2024-12-29T13:32:08.080338Z","shell.execute_reply.started":"2024-12-29T13:32:07.599427Z","shell.execute_reply":"2024-12-29T13:32:08.079594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"NUll values in Train Data : \")\nprint(train_df.isnull().sum())\nprint(\"*\"*100)\nprint(\"NUll values in Test Data : \")\nprint(test_df.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:45:42.93961Z","iopub.execute_input":"2024-12-29T13:45:42.939814Z","iopub.status.idle":"2024-12-29T13:45:43.824121Z","shell.execute_reply.started":"2024-12-29T13:45:42.939797Z","shell.execute_reply":"2024-12-29T13:45:43.823363Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_columns =['Age', 'Annual Income', 'Number of Dependents', 'Health Score',\n       'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration']\ncategorical_columns = ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location',\n       'Policy Type', 'Customer Feedback',\n       'Smoking Status', 'Exercise Frequency', 'Property Type']\ntrain_df = train_df.drop(columns = 'Policy Start Date')\ntest_df = test_df.drop(columns = 'Policy Start Date')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:45:43.82483Z","iopub.execute_input":"2024-12-29T13:45:43.825093Z","iopub.status.idle":"2024-12-29T13:45:44.091533Z","shell.execute_reply.started":"2024-12-29T13:45:43.825071Z","shell.execute_reply":"2024-12-29T13:45:44.090735Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"background-color:#fad0c4; color:#355c7d; text-align:center; padding:15px; border-radius:15px; font-size:25px; \">Data Analysis</div>","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12,12))\n\ni = 0\ndef hist_plot(col):\n    global i \n    i+=1\n    plt.subplot(5,2,i)\n    sns.histplot(x=train_df[col],kde=True)\n    plt.xlabel(col)\n    plt.ylabel(\"\")\n    plt.title(f\"Distribution of {col}\")\n\nfor col in numeric_columns:\n    hist_plot(col)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:32:09.183085Z","iopub.execute_input":"2024-12-29T13:32:09.183332Z","iopub.status.idle":"2024-12-29T13:32:41.767447Z","shell.execute_reply.started":"2024-12-29T13:32:09.183311Z","shell.execute_reply":"2024-12-29T13:32:41.766577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12,12))\n\ni = 0\ndef count_plot(col):\n    global i \n    i+=1\n    plt.subplot(5,2,i)\n    sns.countplot(x=train_df[col],palette=\"Set3\")\n    plt.xlabel(col)\n    plt.ylabel(\"\")\n    plt.title(f\"Distribution of {col}\")\n\nfor col in ['Gender','Education Level','Location','Policy Type','Smoking Status','Exercise Frequency','Property Type']:\n    count_plot(col)\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:32:41.768385Z","iopub.execute_input":"2024-12-29T13:32:41.768631Z","iopub.status.idle":"2024-12-29T13:32:46.216248Z","shell.execute_reply.started":"2024-12-29T13:32:41.768608Z","shell.execute_reply":"2024-12-29T13:32:46.215419Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12,10))\nfor i,col in enumerate(categorical_columns):\n    plt.subplot(3,4,i+1)\n    sns.boxplot(x=train_df[col],y=train_df['Premium Amount'],palette ='Set3')\n    plt.title(f\"Box plot of {col}\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:32:46.217376Z","iopub.execute_input":"2024-12-29T13:32:46.217703Z","iopub.status.idle":"2024-12-29T13:32:52.898326Z","shell.execute_reply.started":"2024-12-29T13:32:46.217671Z","shell.execute_reply":"2024-12-29T13:32:52.89737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12,10))\nfor i,col in enumerate(numeric_columns):\n    plt.subplot(3,4,i+1)\n    sns.boxplot(x=train_df[col],palette ='Set3')\n    plt.title(f\"Box plot of {col}\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:32:52.899372Z","iopub.execute_input":"2024-12-29T13:32:52.899702Z","iopub.status.idle":"2024-12-29T13:32:54.119389Z","shell.execute_reply.started":"2024-12-29T13:32:52.89967Z","shell.execute_reply":"2024-12-29T13:32:54.118514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n\nimputer = SimpleImputer(missing_values=np.nan, strategy='mean')\n\ntrain_df[numeric_columns] = imputer.fit_transform(train_df[numeric_columns])\ntest_df[numeric_columns] = imputer.fit_transform(test_df[numeric_columns])\n\nimputer = SimpleImputer(missing_values=np.nan, strategy='most_frequent')\n\ntrain_df[categorical_columns] = imputer.fit_transform(train_df[categorical_columns])\ntest_df[categorical_columns] = imputer.transform(test_df[categorical_columns])\n ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:45:44.09413Z","iopub.execute_input":"2024-12-29T13:45:44.094351Z","iopub.status.idle":"2024-12-29T13:45:46.928362Z","shell.execute_reply.started":"2024-12-29T13:45:44.094332Z","shell.execute_reply":"2024-12-29T13:45:46.927548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nencoder = LabelEncoder()\n\nfor col in categorical_columns:\n    train_df[col] = encoder.fit_transform(train_df[col])\n    test_df[col] = encoder.transform(test_df[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:45:46.929388Z","iopub.execute_input":"2024-12-29T13:45:46.929631Z","iopub.status.idle":"2024-12-29T13:45:49.705244Z","shell.execute_reply.started":"2024-12-29T13:45:46.92961Z","shell.execute_reply":"2024-12-29T13:45:49.704301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corr = train_df.corr()\nplt.figure(figsize=(12,8))\nsns.heatmap(corr,annot=True,cmap='Blues',fmt='.0f')\nplt.title(\"correlation of the Train Data\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:45:49.706178Z","iopub.execute_input":"2024-12-29T13:45:49.706421Z","iopub.status.idle":"2024-12-29T13:45:51.968484Z","shell.execute_reply.started":"2024-12-29T13:45:49.706399Z","shell.execute_reply":"2024-12-29T13:45:51.967496Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Highly No Correlation Between Data","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler ,StandardScaler\n\nscaler = StandardScaler()\n\ntrain_df[numeric_columns] = scaler.fit_transform(train_df[numeric_columns])\ntest_df[numeric_columns] = scaler.transform(test_df[numeric_columns]) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:45:51.969478Z","iopub.execute_input":"2024-12-29T13:45:51.969946Z","iopub.status.idle":"2024-12-29T13:45:52.288876Z","shell.execute_reply.started":"2024-12-29T13:45:51.96991Z","shell.execute_reply":"2024-12-29T13:45:52.288227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_absolute_error ,mean_squared_error\ntrain_dff = train_df.drop(columns='id')\ntest_dff = test_df.drop(columns='id')\n\nX = train_dff.drop(columns=['Premium Amount'])\ny = train_dff['Premium Amount']\nx_train,x_test,y_train,y_test = train_test_split(X,y,test_size = 0.2,random_state =42)\n\nprint(x_train.shape,y_train.shape,x_test.shape,y_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:45:52.289545Z","iopub.execute_input":"2024-12-29T13:45:52.289777Z","iopub.status.idle":"2024-12-29T13:45:52.935464Z","shell.execute_reply.started":"2024-12-29T13:45:52.289756Z","shell.execute_reply":"2024-12-29T13:45:52.934543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluation_metrics(model,y_test,y_pred):\n    mae = mean_absolute_error(y_pred,y_test)\n    mse = mean_squared_error(y_pred,y_test)\n    rmse = mse**0.5\n    print(\"ML Model : \",model)\n    print(\"Mean Absolute Error :\",mae)\n    print(\"Mean Squared Error :\",mse)\n    print(\"Root mean Squared Error :\",rmse)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:51:21.94199Z","iopub.execute_input":"2024-12-29T13:51:21.942336Z","iopub.status.idle":"2024-12-29T13:51:21.946605Z","shell.execute_reply.started":"2024-12-29T13:51:21.942312Z","shell.execute_reply":"2024-12-29T13:51:21.945773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom lightgbm import LGBMRegressor\n\nlgbm = LGBMRegressor(metric='mse',\n                    n_estimators =250,\n                    learning_rate =0.1545,\n                    )\n\nlgbm.fit(x_train,y_train)\n\ny_pred = lgbm.predict(x_test)\n\nevaluation_metrics(\"Linear Regression\",y_test,y_pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:50:26.139081Z","iopub.execute_input":"2024-12-29T13:50:26.139368Z","iopub.status.idle":"2024-12-29T13:50:34.148176Z","shell.execute_reply.started":"2024-12-29T13:50:26.139346Z","shell.execute_reply":"2024-12-29T13:50:34.147373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import SGDRegressor\n\nlr = SGDRegressor(max_iter=1241,random_state = 42)\n\nlr.fit(x_train,y_train)\n\ny_pred = lr.predict(x_test)\n\nevaluation_metrics(\"Linear Regression\",y_test,y_pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:51:03.503313Z","iopub.execute_input":"2024-12-29T13:51:03.503613Z","iopub.status.idle":"2024-12-29T13:51:14.037294Z","shell.execute_reply.started":"2024-12-29T13:51:03.50359Z","shell.execute_reply":"2024-12-29T13:51:14.035194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = lr.predict(test_dff)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:33:02.513083Z","iopub.status.idle":"2024-12-29T13:33:02.513399Z","shell.execute_reply":"2024-12-29T13:33:02.513293Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"background-color:#fad0c4; color:#355c7d; text-align:center; padding:15px; border-radius:15px; font-size:25px; \">Submission</div>","metadata":{}},{"cell_type":"code","source":"res = pd.DataFrame({\"id\":test_df[\"id\"],\n                   \"Premium Amount\":predictions})\nres = res.set_index(\"id\")\nres","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:33:02.514236Z","iopub.status.idle":"2024-12-29T13:33:02.514623Z","shell.execute_reply":"2024-12-29T13:33:02.51444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"res.to_csv(\"Submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:33:02.515273Z","iopub.status.idle":"2024-12-29T13:33:02.515643Z","shell.execute_reply":"2024-12-29T13:33:02.515475Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"background-color:#fad0c4; color:#355c7d; text-align:center; padding:15px; border-radius:15px; font-size:25px; \">Linear Regression : Hero of Machine Learning</div>","metadata":{}}]}