{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30823,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Problem statement\n\nPredict on insurance premiums","metadata":{}},{"cell_type":"markdown","source":"Import libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n\nfrom scipy.stats import ks_2samp\n\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error\n\nimport statsmodels.api as sm\nimport statsmodels.formula.api as smf\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:13:55.986101Z","iopub.execute_input":"2024-12-24T10:13:55.9864Z","iopub.status.idle":"2024-12-24T10:13:55.991116Z","shell.execute_reply.started":"2024-12-24T10:13:55.986378Z","shell.execute_reply":"2024-12-24T10:13:55.990168Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Use os to retrieve files used in competition","metadata":{}},{"cell_type":"code","source":"for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:13:55.992431Z","iopub.execute_input":"2024-12-24T10:13:55.992737Z","iopub.status.idle":"2024-12-24T10:13:56.00696Z","shell.execute_reply.started":"2024-12-24T10:13:55.992708Z","shell.execute_reply":"2024-12-24T10:13:56.006085Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Use pandas to read files","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\nsubmission = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:13:56.008475Z","iopub.execute_input":"2024-12-24T10:13:56.008671Z","iopub.status.idle":"2024-12-24T10:14:02.121784Z","shell.execute_reply.started":"2024-12-24T10:13:56.008655Z","shell.execute_reply":"2024-12-24T10:14:02.121118Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"View all columns","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:02.12302Z","iopub.execute_input":"2024-12-24T10:14:02.12335Z","iopub.status.idle":"2024-12-24T10:14:02.126805Z","shell.execute_reply.started":"2024-12-24T10:14:02.123316Z","shell.execute_reply":"2024-12-24T10:14:02.126096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:02.127594Z","iopub.execute_input":"2024-12-24T10:14:02.127869Z","iopub.status.idle":"2024-12-24T10:14:02.161497Z","shell.execute_reply.started":"2024-12-24T10:14:02.127843Z","shell.execute_reply":"2024-12-24T10:14:02.160642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:02.162498Z","iopub.execute_input":"2024-12-24T10:14:02.162816Z","iopub.status.idle":"2024-12-24T10:14:02.696886Z","shell.execute_reply.started":"2024-12-24T10:14:02.162785Z","shell.execute_reply":"2024-12-24T10:14:02.695971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.dropna(inplace=True)\ntrain.isnull().sum().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:02.697653Z","iopub.execute_input":"2024-12-24T10:14:02.697865Z","iopub.status.idle":"2024-12-24T10:14:03.503676Z","shell.execute_reply.started":"2024-12-24T10:14:02.697848Z","shell.execute_reply":"2024-12-24T10:14:03.502788Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:03.505952Z","iopub.execute_input":"2024-12-24T10:14:03.506228Z","iopub.status.idle":"2024-12-24T10:14:03.529803Z","shell.execute_reply.started":"2024-12-24T10:14:03.506206Z","shell.execute_reply":"2024-12-24T10:14:03.529134Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:03.53137Z","iopub.execute_input":"2024-12-24T10:14:03.5316Z","iopub.status.idle":"2024-12-24T10:14:03.882853Z","shell.execute_reply.started":"2024-12-24T10:14:03.53158Z","shell.execute_reply":"2024-12-24T10:14:03.88191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:03.883769Z","iopub.execute_input":"2024-12-24T10:14:03.884063Z","iopub.status.idle":"2024-12-24T10:14:03.892878Z","shell.execute_reply.started":"2024-12-24T10:14:03.884014Z","shell.execute_reply":"2024-12-24T10:14:03.892098Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Impute missing values in train and test set","metadata":{}},{"cell_type":"code","source":"#for col in train:\n#    if train[col].dtype == 'object':\n#        train[col] = train[col].fillna('Not recorded')\n#    if train[col].dtype == 'float':\n#        train[col] = train[col].fillna(-1)\n#    if train[col].dtype == 'int':\n#        train[col] = train[col].fillna(-1)\n        \nfor col in test:\n    if test[col].dtype == 'object':\n        test[col] = test[col].fillna('Not recorded')\n    if test[col].dtype == 'float':\n        test[col] = test[col].fillna(-1)\n    if test[col].dtype == 'int':\n        test[col] = test[col].fillna(-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:03.893675Z","iopub.execute_input":"2024-12-24T10:14:03.893935Z","iopub.status.idle":"2024-12-24T10:14:04.453104Z","shell.execute_reply.started":"2024-12-24T10:14:03.893916Z","shell.execute_reply":"2024-12-24T10:14:04.452411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.isna().sum().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:04.453876Z","iopub.execute_input":"2024-12-24T10:14:04.454202Z","iopub.status.idle":"2024-12-24T10:14:04.833757Z","shell.execute_reply.started":"2024-12-24T10:14:04.454172Z","shell.execute_reply":"2024-12-24T10:14:04.83301Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Compare train and test datasets to see if they come from the same distribution","metadata":{}},{"cell_type":"code","source":"to_drop = []\n\nfor col in test:\n    stat, pv = ks_2samp(train[col], test[col])\n    if pv < 0.05:\n        to_drop.append(col)\nprint(to_drop)\n\ntrain.drop(to_drop,axis=1,inplace=True)\ntest.drop(to_drop,axis=1,inplace=True)\n\nprint(train.shape, test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:04.834682Z","iopub.execute_input":"2024-12-24T10:14:04.834978Z","iopub.status.idle":"2024-12-24T10:14:39.420844Z","shell.execute_reply.started":"2024-12-24T10:14:04.834949Z","shell.execute_reply":"2024-12-24T10:14:39.419914Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Ordinal encode train and test sets","metadata":{}},{"cell_type":"code","source":"enc = OrdinalEncoder(handle_unknown='use_encoded_value', unknown_value=-1)\n\nfor col in train:\n    if train[col].dtype == 'object':\n        train[col] = enc.fit_transform(train[col].values.reshape(-1,1))\n        test[col] = enc.transform(test[col].values.reshape(-1,1))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:39.421792Z","iopub.execute_input":"2024-12-24T10:14:39.422081Z","iopub.status.idle":"2024-12-24T10:14:42.321916Z","shell.execute_reply.started":"2024-12-24T10:14:39.422033Z","shell.execute_reply":"2024-12-24T10:14:42.321266Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Analyse target","metadata":{}},{"cell_type":"code","source":"plt.hist(train['Premium Amount'])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:42.322725Z","iopub.execute_input":"2024-12-24T10:14:42.322943Z","iopub.status.idle":"2024-12-24T10:14:42.607639Z","shell.execute_reply.started":"2024-12-24T10:14:42.322925Z","shell.execute_reply":"2024-12-24T10:14:42.606811Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Heatmap","metadata":{}},{"cell_type":"code","source":"corr = train.corr()\nsns.heatmap(corr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:42.60851Z","iopub.execute_input":"2024-12-24T10:14:42.608851Z","iopub.status.idle":"2024-12-24T10:14:43.090353Z","shell.execute_reply.started":"2024-12-24T10:14:42.608815Z","shell.execute_reply":"2024-12-24T10:14:43.089537Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Define X and y variables","metadata":{}},{"cell_type":"code","source":"y = train.pop('Premium Amount')\nX = np.array(train)\nX_test = np.array(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:43.091102Z","iopub.execute_input":"2024-12-24T10:14:43.091315Z","iopub.status.idle":"2024-12-24T10:14:43.144649Z","shell.execute_reply.started":"2024-12-24T10:14:43.091297Z","shell.execute_reply":"2024-12-24T10:14:43.143906Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Split X into training and validation sets","metadata":{}},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split( X, y, test_size=0.1, shuffle=True, random_state=42)\nX_train.shape, y_train.shape, X_val.shape, y_val.shape, X_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:43.145521Z","iopub.execute_input":"2024-12-24T10:14:43.145829Z","iopub.status.idle":"2024-12-24T10:14:43.20098Z","shell.execute_reply.started":"2024-12-24T10:14:43.145792Z","shell.execute_reply":"2024-12-24T10:14:43.200252Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Define model","metadata":{}},{"cell_type":"code","source":"#model\nmodel = sm.OLS(y_train, X_train).fit()\nprint(model.summary())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:43.201727Z","iopub.execute_input":"2024-12-24T10:14:43.201955Z","iopub.status.idle":"2024-12-24T10:14:43.601833Z","shell.execute_reply.started":"2024-12-24T10:14:43.201936Z","shell.execute_reply":"2024-12-24T10:14:43.60066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#residuals greater than 3 is said to be an outlier\n#create instance of influence\ninfluence = model.get_influence()\n\n#obtain standardized residuals\nstandardized_residuals = influence.resid_studentized_internal\n\n#display standardized residuals\nprint(standardized_residuals)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:43.60283Z","iopub.execute_input":"2024-12-24T10:14:43.603233Z","iopub.status.idle":"2024-12-24T10:14:43.651988Z","shell.execute_reply.started":"2024-12-24T10:14:43.603194Z","shell.execute_reply":"2024-12-24T10:14:43.65075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sm.qqplot(standardized_residuals,line='45')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:43.652844Z","iopub.execute_input":"2024-12-24T10:14:43.653209Z","iopub.status.idle":"2024-12-24T10:14:45.751376Z","shell.execute_reply.started":"2024-12-24T10:14:43.653174Z","shell.execute_reply":"2024-12-24T10:14:45.750557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(X_val)\ny_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:45.752214Z","iopub.execute_input":"2024-12-24T10:14:45.752519Z","iopub.status.idle":"2024-12-24T10:14:45.759824Z","shell.execute_reply.started":"2024-12-24T10:14:45.752491Z","shell.execute_reply":"2024-12-24T10:14:45.758714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mse = mean_squared_error(y_val, y_pred)\nrmse = np.sqrt(mse)\nrmsle = np.log(rmse)\nrmsle","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:45.764882Z","iopub.execute_input":"2024-12-24T10:14:45.765533Z","iopub.status.idle":"2024-12-24T10:14:45.781142Z","shell.execute_reply.started":"2024-12-24T10:14:45.765492Z","shell.execute_reply":"2024-12-24T10:14:45.780145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots()\nax.scatter(y_val, y_pred, edgecolors=(0, 0, 0))\nax.plot([y.min(), y.max()], [y.min(), y.max()], 'k--', lw=4)\nax.set_xlabel('Measured')\nax.set_ylabel('Predicted')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:45.782054Z","iopub.execute_input":"2024-12-24T10:14:45.782549Z","iopub.status.idle":"2024-12-24T10:14:46.099025Z","shell.execute_reply.started":"2024-12-24T10:14:45.782387Z","shell.execute_reply":"2024-12-24T10:14:46.098232Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Predict on X_test","metadata":{}},{"cell_type":"code","source":"pred = model.predict(X_test)\npred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:46.099934Z","iopub.execute_input":"2024-12-24T10:14:46.10019Z","iopub.status.idle":"2024-12-24T10:14:46.112094Z","shell.execute_reply.started":"2024-12-24T10:14:46.100159Z","shell.execute_reply":"2024-12-24T10:14:46.11112Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Prepare submission","metadata":{}},{"cell_type":"code","source":"submission['Premium Amount'] = pred\nsubmission.to_csv('submission.csv', index=False)\nsubmission = pd.read_csv('submission.csv')\nsubmission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T10:14:46.112795Z","iopub.execute_input":"2024-12-24T10:14:46.113118Z","iopub.status.idle":"2024-12-24T10:14:47.724383Z","shell.execute_reply.started":"2024-12-24T10:14:46.113079Z","shell.execute_reply":"2024-12-24T10:14:47.723631Z"}},"outputs":[],"execution_count":null}]}