{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-02T08:02:15.002772Z","iopub.execute_input":"2024-12-02T08:02:15.003473Z","iopub.status.idle":"2024-12-02T08:02:15.009468Z","shell.execute_reply.started":"2024-12-02T08:02:15.003437Z","shell.execute_reply":"2024-12-02T08:02:15.008562Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# For exploring trends in data, you can check out my EDA notebook. You will get insights on how to carry forward with imputation and transformations. \n**Upvote if you like it**🙃\n\n[EDA Notebook](http://www.kaggle.com/code/baibhavkundu2005/insurance-regression-eda)","metadata":{}},{"cell_type":"markdown","source":"# Importing Libraries","metadata":{}},{"cell_type":"code","source":"import h2o\nfrom h2o.automl import H2OAutoML\nimport numpy as np\nfrom sklearn.metrics import mean_squared_log_error\nimport pandas as pd\nimport numpy as np","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T08:01:07.643409Z","iopub.execute_input":"2024-12-02T08:01:07.643997Z","iopub.status.idle":"2024-12-02T08:01:07.648353Z","shell.execute_reply.started":"2024-12-02T08:01:07.643965Z","shell.execute_reply":"2024-12-02T08:01:07.647316Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Initializing the H2O server","metadata":{}},{"cell_type":"code","source":"h2o.init()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T08:00:40.475231Z","iopub.execute_input":"2024-12-02T08:00:40.47608Z","iopub.status.idle":"2024-12-02T08:00:47.913519Z","shell.execute_reply.started":"2024-12-02T08:00:40.476042Z","shell.execute_reply":"2024-12-02T08:00:47.91266Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Importing datasets","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\ntrain.drop('id', axis=1, inplace=True)\ntest.drop('id', axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T08:01:15.596744Z","iopub.execute_input":"2024-12-02T08:01:15.597076Z","iopub.status.idle":"2024-12-02T08:01:24.447931Z","shell.execute_reply.started":"2024-12-02T08:01:15.597047Z","shell.execute_reply":"2024-12-02T08:01:24.446815Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing the column","metadata":{}},{"cell_type":"code","source":"def date(df):\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    df['Year'] = df['Policy Start Date'].dt.year\n    df['Month'] = df['Policy Start Date'].dt.month\n    df['Day'] = df['Policy Start Date'].dt.day\n    df['Day_of_week'] = df['Policy Start Date'].dt.day_name()\n    df.drop('Policy Start Date', axis=1, inplace=True)\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T08:01:30.265185Z","iopub.execute_input":"2024-12-02T08:01:30.265905Z","iopub.status.idle":"2024-12-02T08:01:30.271021Z","shell.execute_reply.started":"2024-12-02T08:01:30.265872Z","shell.execute_reply":"2024-12-02T08:01:30.269872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = date(train)\ntest = date(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T08:01:37.511902Z","iopub.execute_input":"2024-12-02T08:01:37.51258Z","iopub.status.idle":"2024-12-02T08:01:39.197114Z","shell.execute_reply.started":"2024-12-02T08:01:37.512544Z","shell.execute_reply":"2024-12-02T08:01:39.196088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_cols = [col for col in train.columns if train[col].dtype == 'object']\nfor col in cat_cols:\n    train[col] = train[col].astype('category')\n    test[col] = test[col].astype('category')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T08:01:52.626488Z","iopub.execute_input":"2024-12-02T08:01:52.627136Z","iopub.status.idle":"2024-12-02T08:01:54.00982Z","shell.execute_reply.started":"2024-12-02T08:01:52.627104Z","shell.execute_reply":"2024-12-02T08:01:54.008809Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Creating features and target variable","metadata":{}},{"cell_type":"code","source":"X = train.drop('Premium Amount', axis=1)\ny = train['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T08:02:11.859961Z","iopub.execute_input":"2024-12-02T08:02:11.860659Z","iopub.status.idle":"2024-12-02T08:02:11.903311Z","shell.execute_reply.started":"2024-12-02T08:02:11.860623Z","shell.execute_reply":"2024-12-02T08:02:11.902209Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"AutoML requires a H2O dataframe for learning","metadata":{}},{"cell_type":"code","source":"h2o_train = h2o.H2OFrame(pd.concat([X, y], axis=1))\nh2o_train['Premium Amount'] = h2o_train['Premium Amount'].asnumeric()\nh2o_test = h2o.H2OFrame(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T08:04:07.227394Z","iopub.execute_input":"2024-12-02T08:04:07.228215Z","iopub.status.idle":"2024-12-02T08:04:41.931568Z","shell.execute_reply.started":"2024-12-02T08:04:07.228184Z","shell.execute_reply":"2024-12-02T08:04:41.930791Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training AutoML","metadata":{}},{"cell_type":"code","source":"automl = H2OAutoML(max_runtime_secs=600, seed=42)\nautoml.train(y='Premium Amount', training_frame=h2o_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T08:04:46.114837Z","iopub.execute_input":"2024-12-02T08:04:46.115554Z","iopub.status.idle":"2024-12-02T08:14:56.619801Z","shell.execute_reply.started":"2024-12-02T08:04:46.11552Z","shell.execute_reply":"2024-12-02T08:14:56.618703Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Selecting the best model**","metadata":{}},{"cell_type":"code","source":"best_h2o_model = automl.leader","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T08:15:27.649736Z","iopub.execute_input":"2024-12-02T08:15:27.650573Z","iopub.status.idle":"2024-12-02T08:15:27.662796Z","shell.execute_reply.started":"2024-12-02T08:15:27.650538Z","shell.execute_reply":"2024-12-02T08:15:27.66193Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Metrics","metadata":{}},{"cell_type":"code","source":"val_predictions = best_h2o_model.predict(h2o_train).as_data_frame()['predict']\nrmsle = np.sqrt(mean_squared_log_error(y, val_predictions))\nprint(f\"H2O AutoML RMSLE: {rmsle}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T08:15:39.861402Z","iopub.execute_input":"2024-12-02T08:15:39.862064Z","iopub.status.idle":"2024-12-02T08:16:47.770455Z","shell.execute_reply.started":"2024-12-02T08:15:39.86203Z","shell.execute_reply":"2024-12-02T08:16:47.769431Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Predicting on the \"test\" dataset","metadata":{}},{"cell_type":"code","source":"test_predictions = best_h2o_model.predict(h2o_test).as_data_frame()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T08:17:06.873536Z","iopub.execute_input":"2024-12-02T08:17:06.873884Z","iopub.status.idle":"2024-12-02T08:17:53.112017Z","shell.execute_reply.started":"2024-12-02T08:17:06.873854Z","shell.execute_reply":"2024-12-02T08:17:53.111042Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\nsubmission['Premium Amount'] = test_predictions\nsubmission.to_csv(\"submission.csv\",index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T08:22:31.075048Z","iopub.execute_input":"2024-12-02T08:22:31.075417Z","iopub.status.idle":"2024-12-02T08:22:32.83548Z","shell.execute_reply.started":"2024-12-02T08:22:31.075384Z","shell.execute_reply":"2024-12-02T08:22:32.834574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"h2o.shutdown(prompt=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T08:22:49.894402Z","iopub.execute_input":"2024-12-02T08:22:49.894742Z","iopub.status.idle":"2024-12-02T08:22:49.919401Z","shell.execute_reply.started":"2024-12-02T08:22:49.894715Z","shell.execute_reply":"2024-12-02T08:22:49.918393Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n* This was a pretty simple implementation of data preprocessing and AutoML.\n* I also tried boosting models but they were producing negative values and I don't know why.\n* I will be doing another AutoML and RandomForestRegressor notebook with more preprocessing to see how the results compare.\n* If you liked it, do give an upvote!\n* I'm open to suggestions. Please put them in the comments.\n* Thank youu!","metadata":{}},{"cell_type":"markdown","source":"![](http://i.imgflip.com/478xve.jpg)","metadata":{}}]}