{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 📋 Table of Contents\n* [Features EDA](#eda)\n* [Target](#target)\n* [Target vs Features](#target_vs_features)\n* [Model](#model)","metadata":{}},{"cell_type":"code","source":"# packages\n\n# standard\nimport numpy as np\nimport pandas as pd\nimport time\nimport os\n\n# plots\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# machine learning tools\nimport h2o\nfrom h2o.estimators import H2OGeneralizedLinearEstimator, H2ORandomForestEstimator, H2OGradientBoostingEstimator\nfrom h2o.automl import H2OAutoML\n\n# show files\n!ls -l '/kaggle/input/playground-series-s4e12'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-23T21:04:53.154396Z","iopub.execute_input":"2024-12-23T21:04:53.154756Z","iopub.status.idle":"2024-12-23T21:04:55.970754Z","shell.execute_reply.started":"2024-12-23T21:04:53.154726Z","shell.execute_reply":"2024-12-23T21:04:55.969483Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# configs\npd.set_option('display.max_columns', None) # we want to display all columns in this notebook\npd.set_option('display.max_rows', 100) # increase rows to be displayed\npd.set_option('display.max_colwidth', None) # show full cell contents\n\n# random seed\nmy_random_seed = 111\n\n# aesthetics\ndefault_color_1 = 'darkblue'\ndefault_color_2 = 'darkgreen'\ndefault_color_3 = 'darkred'\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:04:55.973239Z","iopub.execute_input":"2024-12-23T21:04:55.97395Z","iopub.status.idle":"2024-12-23T21:04:55.981124Z","shell.execute_reply.started":"2024-12-23T21:04:55.973912Z","shell.execute_reply":"2024-12-23T21:04:55.979918Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# load data\nt1 = time.time()\ndf_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv', low_memory=False)\ndf_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv', low_memory=False)\ndf_sub = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv', low_memory=False)\nt2 = time.time()\nprint('Elapsed time [s]:', np.round(t2-t1,4))","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:04:55.982562Z","iopub.execute_input":"2024-12-23T21:04:55.982862Z","iopub.status.idle":"2024-12-23T21:05:07.755458Z","shell.execute_reply.started":"2024-12-23T21:04:55.982832Z","shell.execute_reply":"2024-12-23T21:05:07.754288Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='eda'></a>\n# Features EDA","metadata":{}},{"cell_type":"code","source":"# preview\ndf_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T21:05:07.758061Z","iopub.execute_input":"2024-12-23T21:05:07.758433Z","iopub.status.idle":"2024-12-23T21:05:07.797756Z","shell.execute_reply.started":"2024-12-23T21:05:07.758393Z","shell.execute_reply":"2024-12-23T21:05:07.796621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# date conversion\ndf_train['Policy Start Date'] = pd.to_datetime(df_train['Policy Start Date'])\ndf_test['Policy Start Date'] = pd.to_datetime(df_test['Policy Start Date'])","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:05:07.798908Z","iopub.execute_input":"2024-12-23T21:05:07.799214Z","iopub.status.idle":"2024-12-23T21:05:08.61786Z","shell.execute_reply.started":"2024-12-23T21:05:07.799183Z","shell.execute_reply":"2024-12-23T21:05:08.616548Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# show structure of data\ndf_train.info(show_counts=True, verbose=True)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-12-23T21:05:08.619139Z","iopub.execute_input":"2024-12-23T21:05:08.619501Z","iopub.status.idle":"2024-12-23T21:05:09.217014Z","shell.execute_reply.started":"2024-12-23T21:05:08.619465Z","shell.execute_reply":"2024-12-23T21:05:09.215809Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### 💡 We can see that there are a lot of missing values for some columns!","metadata":{}},{"cell_type":"code","source":"# show structure of data\ndf_test.info(show_counts=True, verbose=True)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-12-23T21:05:09.219022Z","iopub.execute_input":"2024-12-23T21:05:09.219406Z","iopub.status.idle":"2024-12-23T21:05:09.60985Z","shell.execute_reply.started":"2024-12-23T21:05:09.219368Z","shell.execute_reply":"2024-12-23T21:05:09.608705Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# time range - training\ndf_train['Policy Start Date'].describe()","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:05:09.611301Z","iopub.execute_input":"2024-12-23T21:05:09.611732Z","iopub.status.idle":"2024-12-23T21:05:09.666407Z","shell.execute_reply.started":"2024-12-23T21:05:09.611686Z","shell.execute_reply":"2024-12-23T21:05:09.66526Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# time range - test\ndf_test['Policy Start Date'].describe()","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:05:09.667761Z","iopub.execute_input":"2024-12-23T21:05:09.668194Z","iopub.status.idle":"2024-12-23T21:05:09.706279Z","shell.execute_reply.started":"2024-12-23T21:05:09.668137Z","shell.execute_reply":"2024-12-23T21:05:09.705056Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# extract date parts\ndf_train['Year'] = df_train['Policy Start Date'].apply(lambda x : x.year)\ndf_train['Month'] = df_train['Policy Start Date'].apply(lambda x : x.month)\ndf_train['Day'] = df_train['Policy Start Date'].apply(lambda x : x.day)\n\ndf_test['Year'] = df_test['Policy Start Date'].apply(lambda x : x.year)\ndf_test['Month'] = df_test['Policy Start Date'].apply(lambda x : x.month)\ndf_test['Day'] = df_test['Policy Start Date'].apply(lambda x : x.day)","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:05:09.710126Z","iopub.execute_input":"2024-12-23T21:05:09.710516Z","iopub.status.idle":"2024-12-23T21:05:22.128313Z","shell.execute_reply.started":"2024-12-23T21:05:09.710482Z","shell.execute_reply":"2024-12-23T21:05:22.127166Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# define target and predictors\ntarget = 'Premium Amount'\n\n# numerical features\nfeatures_num = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score',\n                'Previous Claims', 'Vehicle Age', 'Credit Score',\n                'Insurance Duration', 'Year', 'Month', 'Day']\n\n# categorical features\nfeatures_cat = ['Gender', 'Marital Status', 'Education Level', 'Occupation',\n                'Location', 'Policy Type', 'Customer Feedback', 'Smoking Status',\n                'Exercise Frequency', 'Property Type']","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:05:22.129854Z","iopub.execute_input":"2024-12-23T21:05:22.13031Z","iopub.status.idle":"2024-12-23T21:05:22.136973Z","shell.execute_reply.started":"2024-12-23T21:05:22.130251Z","shell.execute_reply":"2024-12-23T21:05:22.135738Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot histograms (train and test)\nn_bins = 101\nfor f in features_num:\n    plt.figure(figsize=(12,3))\n    ax1 = plt.subplot(1,2,1)\n    df_train[f].plot(kind='hist', bins=n_bins, color=default_color_1)\n    plt.title(f + ' - Train')\n    plt.grid()\n    ax2 = plt.subplot(1,2,2, sharex=ax1)\n    df_test[f].plot(kind='hist', bins=n_bins, color=default_color_2)\n    plt.title(f + ' - Test')\n    plt.grid()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T21:05:22.138504Z","iopub.execute_input":"2024-12-23T21:05:22.13895Z","iopub.status.idle":"2024-12-23T21:05:31.547769Z","shell.execute_reply.started":"2024-12-23T21:05:22.138903Z","shell.execute_reply":"2024-12-23T21:05:31.546515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# special plot for age\nplt.figure(figsize=(12,4))\ndf_train.Age.value_counts().sort_index().plot(kind='bar', color=default_color_1)\nplt.title('Age / Training - discrete plot')\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T21:05:31.549243Z","iopub.execute_input":"2024-12-23T21:05:31.549599Z","iopub.status.idle":"2024-12-23T21:05:31.995492Z","shell.execute_reply.started":"2024-12-23T21:05:31.549564Z","shell.execute_reply":"2024-12-23T21:05:31.994305Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# boxplots (train and test)\nfor f in features_num:\n    plt.figure(figsize=(14,1))\n    ax1 = plt.subplot(1,2,1)\n    df_temp = df_train[f].dropna() # boxplot does not like missings...\n    plt.boxplot(df_temp, vert=False)\n    plt.title(f + ' - Train')\n    plt.grid()\n    ax2 = plt.subplot(1,2,2, sharex=ax1)\n    df_temp = df_test[f].dropna()\n    plt.boxplot(df_temp, vert=False)\n    plt.title(f + ' - Test')\n    plt.grid()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T21:05:31.996947Z","iopub.execute_input":"2024-12-23T21:05:31.997325Z","iopub.status.idle":"2024-12-23T21:05:35.629108Z","shell.execute_reply.started":"2024-12-23T21:05:31.997292Z","shell.execute_reply":"2024-12-23T21:05:35.628084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot categorical feature distributions (train and test)\nfor f in features_cat:\n    plt.figure(figsize=(14,3))\n    ax1 = plt.subplot(1,2,1)\n    df_train[f].value_counts().sort_index().plot(kind='bar', color=default_color_1)\n    plt.title(f + ' - Train')\n    plt.grid()\n    ax2 = plt.subplot(1,2,2, sharex=ax1)\n    df_test[f].value_counts().sort_index().plot(kind='bar', color=default_color_2)\n    plt.title(f + ' - Test')\n    plt.grid()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T21:05:35.630581Z","iopub.execute_input":"2024-12-23T21:05:35.631015Z","iopub.status.idle":"2024-12-23T21:05:39.898356Z","shell.execute_reply.started":"2024-12-23T21:05:35.630969Z","shell.execute_reply":"2024-12-23T21:05:39.897237Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='target'></a>\n# Target","metadata":{}},{"cell_type":"code","source":"# plot target distribution\nplt.figure(figsize=(12,4))\ndf_train[target].plot(kind='hist', bins=100,\n                      color=default_color_3)\nplt.title(target)\nplt.grid()\nplt.show()\n# basic stats\nprint(df_train[target].describe())","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:05:39.899651Z","iopub.execute_input":"2024-12-23T21:05:39.899992Z","iopub.status.idle":"2024-12-23T21:05:40.402446Z","shell.execute_reply.started":"2024-12-23T21:05:39.899959Z","shell.execute_reply":"2024-12-23T21:05:40.40124Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# add log of target\nlog_target = 'log_target'\ndf_train[log_target] = np.log(df_train[target])\n\n# and plot\nplt.figure(figsize=(12,4))\ndf_train[log_target].plot(kind='hist', bins=100,\n                      color=default_color_3)\nplt.title(log_target)\nplt.grid()\nplt.show()\n# basic stats\nprint(df_train[log_target].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T21:05:40.403853Z","iopub.execute_input":"2024-12-23T21:05:40.404179Z","iopub.status.idle":"2024-12-23T21:05:40.934208Z","shell.execute_reply.started":"2024-12-23T21:05:40.404147Z","shell.execute_reply":"2024-12-23T21:05:40.93302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Pearson correlation\ncorr_pearson = df_train[features_num].corr(method='pearson')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T21:05:40.935562Z","iopub.execute_input":"2024-12-23T21:05:40.935922Z","iopub.status.idle":"2024-12-23T21:05:41.511618Z","shell.execute_reply.started":"2024-12-23T21:05:40.935886Z","shell.execute_reply":"2024-12-23T21:05:41.510188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# visualize\nplt.figure(figsize=(7,5))\nsns.heatmap(corr_pearson, annot=True, cmap='RdYlGn', vmin=-1, vmax=+1,\n            fmt='.2f', linewidth=0.5, linecolor='black')\nplt.title('Pearson Correlation')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T21:07:51.641806Z","iopub.status.idle":"2024-12-23T21:07:51.642208Z","shell.execute_reply.started":"2024-12-23T21:07:51.642024Z","shell.execute_reply":"2024-12-23T21:07:51.642043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot pairs with somewhat higher correlation\nsns.jointplot(data=df_train, x='Year', y='Month', \n              height=5,\n              color=default_color_1, alpha=0.01)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T21:08:17.947176Z","iopub.execute_input":"2024-12-23T21:08:17.947685Z","iopub.status.idle":"2024-12-23T21:08:22.155728Z","shell.execute_reply.started":"2024-12-23T21:08:17.947643Z","shell.execute_reply":"2024-12-23T21:08:22.15443Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### 💡 Ok, the \"correlation\" between year and month is simply caused by the missing months in 2019 and 2024.","metadata":{}},{"cell_type":"code","source":"# plot pairs with somewhat higher correlation\nsns.jointplot(data=df_train, x='Annual Income', y='Credit Score',\n              height=5,\n              color=default_color_1,\n              kind='hist')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T21:08:29.326022Z","iopub.execute_input":"2024-12-23T21:08:29.326512Z","iopub.status.idle":"2024-12-23T21:08:32.492017Z","shell.execute_reply.started":"2024-12-23T21:08:29.326469Z","shell.execute_reply":"2024-12-23T21:08:32.490775Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='target_vs_features'></a>\n# Target vs Features","metadata":{}},{"cell_type":"code","source":"# plot target vs features\nmax_levels = 20 # limit levels to plot for categorical features\nalpha_plot = 0.01 # transparency level for scatter plots\n\nfor f in features_num + features_cat:\n    if (f in features_num):\n        c = df_train[log_target].corr(df_train[f])\n        plt.figure(figsize=(10,4))\n        plt.scatter(df_train[f], df_train[target], \n                    color=default_color_1, alpha=alpha_plot)\n        plt.title('log-Target vs ' + f + ' corr=' + str(np.round(c,4)))\n        plt.grid()\n        plt.show()\n    else:\n        plt.figure(figsize=(10,4))\n        # pick only to most frequent levels\n        most_freq = df_train[f].value_counts().index[0:max_levels].tolist()\n        df_temp = df_train[df_train[f].isin(most_freq)]\n        sns.violinplot(data=df_temp, x=f, y=log_target)\n        plt.xticks(rotation=90)\n        plt.title('log-Target vs ' + f + ' (most frequent levels only)')\n        plt.grid()\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T21:05:49.266582Z","iopub.execute_input":"2024-12-23T21:05:49.266907Z","iopub.status.idle":"2024-12-23T21:07:00.310558Z","shell.execute_reply.started":"2024-12-23T21:05:49.266876Z","shell.execute_reply":"2024-12-23T21:07:00.309279Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='model'></a>\n# Model","metadata":{}},{"cell_type":"code","source":"# define predictors\npredictors = features_num + features_cat\nprint(predictors)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T21:07:00.31219Z","iopub.execute_input":"2024-12-23T21:07:00.313022Z","iopub.status.idle":"2024-12-23T21:07:00.319463Z","shell.execute_reply.started":"2024-12-23T21:07:00.312972Z","shell.execute_reply":"2024-12-23T21:07:00.318199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# start H2O\nh2o.init(max_mem_size='20G', nthreads=4)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-12-23T21:07:00.320939Z","iopub.execute_input":"2024-12-23T21:07:00.321387Z","iopub.status.idle":"2024-12-23T21:07:08.914071Z","shell.execute_reply.started":"2024-12-23T21:07:00.321341Z","shell.execute_reply":"2024-12-23T21:07:08.91272Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# prepare data for upload (combine train/test to avoid type mismatches)\ny_log = df_train[log_target]\ny = df_train[target]\ndf_train['fold'] = 'train'\ndf_test['fold'] = 'test'\ndf_train[log_target] = y_log\ndf_test[log_target] = 0\ndf_train[target] = y\ndf_test[target] = 0\n# combine train & test in one data frame\ndf = pd.concat([df_train, df_test])","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:09:24.416119Z","iopub.execute_input":"2024-12-23T21:09:24.416878Z","iopub.status.idle":"2024-12-23T21:09:24.813555Z","shell.execute_reply.started":"2024-12-23T21:09:24.416839Z","shell.execute_reply":"2024-12-23T21:09:24.812552Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# upload data in H2O environment\nt1 = time.time()\ndf_hex = h2o.H2OFrame(df)\nt2 = time.time()\nprint('Elapsed time [s]:', np.round(t2-t1,4))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T21:09:27.046837Z","iopub.execute_input":"2024-12-23T21:09:27.047362Z","iopub.status.idle":"2024-12-23T21:10:26.234806Z","shell.execute_reply.started":"2024-12-23T21:09:27.047319Z","shell.execute_reply":"2024-12-23T21:10:26.233639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# split again\ntrain_hex = df_hex[df_hex['fold']=='train']\ntest_hex = df_hex[df_hex['fold']=='test']","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:10:53.378087Z","iopub.execute_input":"2024-12-23T21:10:53.379371Z","iopub.status.idle":"2024-12-23T21:10:53.391557Z","shell.execute_reply.started":"2024-12-23T21:10:53.379318Z","shell.execute_reply":"2024-12-23T21:10:53.38987Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# setup AutoML\nmax_time_sec = 2*60*60\nfit_auto = H2OAutoML(max_runtime_secs=max_time_sec,\n                     sort_metric='RMSE',\n                     # exclude_algos=['StackedEnsemble'],\n                     # include_algos=['GLM'],\n                     seed=my_random_seed)","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:11:06.444609Z","iopub.execute_input":"2024-12-23T21:11:06.445121Z","iopub.status.idle":"2024-12-23T21:11:06.463592Z","shell.execute_reply.started":"2024-12-23T21:11:06.445078Z","shell.execute_reply":"2024-12-23T21:11:06.461839Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# and train model\nt1 = time.time()\nfit_auto.train(predictors, log_target, training_frame = train_hex)\nt2 = time.time()\nprint('Elapsed time [s]:', np.round(t2-t1,4))","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:11:12.857192Z","iopub.execute_input":"2024-12-23T21:11:12.857943Z","iopub.status.idle":"2024-12-23T21:12:23.730165Z","shell.execute_reply.started":"2024-12-23T21:11:12.857881Z","shell.execute_reply":"2024-12-23T21:12:23.728878Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# leaderboard\nlb = fit_auto.leaderboard\nlb.head(rows=lb.nrows) # show full leaderboard","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:12:25.500253Z","iopub.execute_input":"2024-12-23T21:12:25.500733Z","iopub.status.idle":"2024-12-23T21:12:25.529939Z","shell.execute_reply.started":"2024-12-23T21:12:25.500695Z","shell.execute_reply":"2024-12-23T21:12:25.528296Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# select best model\nfit_1 = fit_auto.leader","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:12:28.458922Z","iopub.execute_input":"2024-12-23T21:12:28.459455Z","iopub.status.idle":"2024-12-23T21:12:28.480698Z","shell.execute_reply.started":"2024-12-23T21:12:28.45941Z","shell.execute_reply":"2024-12-23T21:12:28.479187Z"},"trusted":true,"_kg_hide-output":false},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# show details\nfit_1","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-12-23T21:12:30.316244Z","iopub.execute_input":"2024-12-23T21:12:30.316669Z","iopub.status.idle":"2024-12-23T21:12:30.326563Z","shell.execute_reply.started":"2024-12-23T21:12:30.316637Z","shell.execute_reply":"2024-12-23T21:12:30.325276Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# variable importance\nif (fit_1.algo != 'stackedensemble'):\n    fit_1.varimp_plot(30);","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:12:34.232533Z","iopub.execute_input":"2024-12-23T21:12:34.232904Z","iopub.status.idle":"2024-12-23T21:12:34.239006Z","shell.execute_reply.started":"2024-12-23T21:12:34.232873Z","shell.execute_reply":"2024-12-23T21:12:34.237468Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# predict on train set\npred_train = fit_1.predict(train_hex)\npred_train = pred_train.as_data_frame(use_multi_thread=True)\n# and plot results\nplt.figure(figsize=(10,4))\nplt.hist(pred_train, bins=50, color=default_color_1)\nplt.title('Predictions on Training Data')\nplt.grid()\nplt.show()\n# basic stats\nprint(pred_train.describe())","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:12:36.336579Z","iopub.execute_input":"2024-12-23T21:12:36.337011Z","iopub.status.idle":"2024-12-23T21:12:43.901184Z","shell.execute_reply.started":"2024-12-23T21:12:36.336976Z","shell.execute_reply":"2024-12-23T21:12:43.899883Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# add prediction and residuals to data frame\ndf_train['prediction'] = pred_train\ndf_train['residual'] = df_train['prediction'] - df_train[log_target]","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:12:44.128863Z","iopub.execute_input":"2024-12-23T21:12:44.129297Z","iopub.status.idle":"2024-12-23T21:12:44.144084Z","shell.execute_reply.started":"2024-12-23T21:12:44.129252Z","shell.execute_reply":"2024-12-23T21:12:44.142889Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot predictions vs actuals (log transformed)\np=sns.jointplot(data=df_train, x=log_target, y='prediction',\n                color=default_color_1,\n                joint_kws={'alpha' : 0.1, 's' : 2})\np.fig.suptitle('Prediction vs Actual - Training Data')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:12:58.380791Z","iopub.execute_input":"2024-12-23T21:12:58.38132Z","iopub.status.idle":"2024-12-23T21:13:02.12571Z","shell.execute_reply.started":"2024-12-23T21:12:58.381275Z","shell.execute_reply":"2024-12-23T21:13:02.124432Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot residuals\nplt.figure(figsize=(14,3))\nplt.scatter(df_train.index, df_train.residual, color=default_color_1, \n            alpha=0.1, s=1)\nplt.title('Residuals - Training')\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:13:07.972804Z","iopub.execute_input":"2024-12-23T21:13:07.973283Z","iopub.status.idle":"2024-12-23T21:13:08.866057Z","shell.execute_reply.started":"2024-12-23T21:13:07.973245Z","shell.execute_reply":"2024-12-23T21:13:08.864786Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot residuals vs a specific feature\nf = 'Annual Income'\nplt.figure(figsize=(14,3))\nplt.scatter(df_train[f], df_train.residual, color=default_color_1, \n            alpha=0.1, s=1)\nplt.xlabel(f)\nplt.ylabel('residual')\nplt.title('Residuals - Training')\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:13:17.015565Z","iopub.execute_input":"2024-12-23T21:13:17.017385Z","iopub.status.idle":"2024-12-23T21:13:17.941548Z","shell.execute_reply.started":"2024-12-23T21:13:17.017331Z","shell.execute_reply":"2024-12-23T21:13:17.940306Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# predict on test set\npred_test = fit_1.predict(test_hex)\npred_test = pred_test.as_data_frame(use_multi_thread=True)\n# and plot results\nplt.figure(figsize=(10,4))\nplt.hist(pred_test, bins=50, color=default_color_2)\nplt.title('Predictions on Test Set - Leader')\nplt.grid()\nplt.show()\n# basic stats\nprint(pred_test.describe())","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:13:24.04348Z","iopub.execute_input":"2024-12-23T21:13:24.043926Z","iopub.status.idle":"2024-12-23T21:13:31.467555Z","shell.execute_reply.started":"2024-12-23T21:13:24.04389Z","shell.execute_reply":"2024-12-23T21:13:31.466248Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# undo log trafo + submission\ndf_sub[target] = np.exp(pred_test)\ndf_sub.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:13:33.448588Z","iopub.execute_input":"2024-12-23T21:13:33.449078Z","iopub.status.idle":"2024-12-23T21:13:33.472828Z","shell.execute_reply.started":"2024-12-23T21:13:33.449036Z","shell.execute_reply":"2024-12-23T21:13:33.471249Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# stats\ndf_sub.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T21:13:35.645607Z","iopub.execute_input":"2024-12-23T21:13:35.646082Z","iopub.status.idle":"2024-12-23T21:13:35.728096Z","shell.execute_reply.started":"2024-12-23T21:13:35.646043Z","shell.execute_reply":"2024-12-23T21:13:35.726586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# save submission file\ndf_sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-12-23T21:07:51.640419Z","iopub.status.idle":"2024-12-23T21:07:51.640798Z","shell.execute_reply.started":"2024-12-23T21:07:51.640625Z","shell.execute_reply":"2024-12-23T21:07:51.640644Z"},"trusted":true},"outputs":[],"execution_count":null}]}