{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-18T00:59:44.784709Z","iopub.execute_input":"2024-12-18T00:59:44.785162Z","iopub.status.idle":"2024-12-18T00:59:45.422475Z","shell.execute_reply.started":"2024-12-18T00:59:44.785111Z","shell.execute_reply":"2024-12-18T00:59:45.421292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train= pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T00:59:45.425289Z","iopub.execute_input":"2024-12-18T00:59:45.425839Z","iopub.status.idle":"2024-12-18T00:59:54.738911Z","shell.execute_reply.started":"2024-12-18T00:59:45.42579Z","shell.execute_reply":"2024-12-18T00:59:54.737532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.drop([\"id\"], axis =1)\ntest = test.drop([\"id\"], axis =1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T00:59:54.740282Z","iopub.execute_input":"2024-12-18T00:59:54.740642Z","iopub.status.idle":"2024-12-18T00:59:55.07011Z","shell.execute_reply.started":"2024-12-18T00:59:54.740604Z","shell.execute_reply":"2024-12-18T00:59:55.068417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_columns = train.loc[:, train.isnull().any()].columns.tolist()\nprint(missing_columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T00:59:55.071503Z","iopub.execute_input":"2024-12-18T00:59:55.071958Z","iopub.status.idle":"2024-12-18T00:59:55.75971Z","shell.execute_reply.started":"2024-12-18T00:59:55.071909Z","shell.execute_reply":"2024-12-18T00:59:55.758291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntarget = 'Premium Amount'\nX = train.drop(columns=[target])\ny = train[target]\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T00:59:55.761004Z","iopub.execute_input":"2024-12-18T00:59:55.761336Z","iopub.status.idle":"2024-12-18T00:59:56.443443Z","shell.execute_reply.started":"2024-12-18T00:59:55.761302Z","shell.execute_reply":"2024-12-18T00:59:56.442093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert 'Policy Start Date' to datetime explicitly\nX['Policy Start Date'] = pd.to_datetime(X['Policy Start Date'], errors='coerce')\ntest['Policy Start Date'] = pd.to_datetime(test['Policy Start Date'], errors='coerce')\n\n\nX['day_sin'] = np.sin(2* np.pi *X['Policy Start Date'].dt.day/31)\nX['day_cos']= np.cos(2* np.pi *X['Policy Start Date'].dt.day/31)\nX['month_sin'] = np.sin(2 * np.pi * X['Policy Start Date'].dt.month / 12)\nX['month_cos'] = np.cos(2 * np.pi * X['Policy Start Date'].dt.month / 12)\nX.drop(columns=['Policy Start Date'], inplace=True)\n\ntest['day_sin'] = np.sin(2* np.pi *test['Policy Start Date'].dt.day/31)\ntest['day_cos']= np.cos(2* np.pi *test['Policy Start Date'].dt.day/31)\ntest['month_sin'] = np.sin(2 * np.pi * test['Policy Start Date'].dt.month / 12)\ntest['month_cos'] = np.cos(2 * np.pi * test['Policy Start Date'].dt.month / 12)\ntest.drop(columns=['Policy Start Date'], inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T00:59:56.44497Z","iopub.execute_input":"2024-12-18T00:59:56.445725Z","iopub.status.idle":"2024-12-18T00:59:58.150265Z","shell.execute_reply.started":"2024-12-18T00:59:56.445667Z","shell.execute_reply":"2024-12-18T00:59:58.148992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.head()\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T00:59:58.154682Z","iopub.execute_input":"2024-12-18T00:59:58.155785Z","iopub.status.idle":"2024-12-18T00:59:58.188415Z","shell.execute_reply.started":"2024-12-18T00:59:58.15574Z","shell.execute_reply":"2024-12-18T00:59:58.187089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import KFold, train_test_split, cross_val_score\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import mean_squared_log_error, make_scorer\nfrom xgboost import XGBRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T00:59:58.190019Z","iopub.execute_input":"2024-12-18T00:59:58.19042Z","iopub.status.idle":"2024-12-18T00:59:58.36302Z","shell.execute_reply.started":"2024-12-18T00:59:58.190375Z","shell.execute_reply":"2024-12-18T00:59:58.362058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ncategorical_columns = X.select_dtypes(include=['object', 'category']).columns\nprint(categorical_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T00:59:58.364591Z","iopub.execute_input":"2024-12-18T00:59:58.365Z","iopub.status.idle":"2024-12-18T00:59:58.94129Z","shell.execute_reply.started":"2024-12-18T00:59:58.364943Z","shell.execute_reply":"2024-12-18T00:59:58.94019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_cols= ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location',\n       'Policy Type', 'Customer Feedback', 'Smoking Status',\n       'Exercise Frequency', 'Property Type']\n\nlabel_encoders = {}\nfor col in categorical_cols:\n    le = LabelEncoder()\n    X[col] = le.fit_transform(X[col].astype(str))  # Fit and transform train data\n    label_encoders[col] = le  # Store the fitted encoder\n\n# Transform test data using already-fitted LabelEncoders\nfor col in categorical_cols:\n    test[col] = label_encoders[col].transform(test[col].astype(str))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T00:59:58.942752Z","iopub.execute_input":"2024-12-18T00:59:58.943101Z","iopub.status.idle":"2024-12-18T01:00:03.108771Z","shell.execute_reply.started":"2024-12-18T00:59:58.943067Z","shell.execute_reply":"2024-12-18T01:00:03.107596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T01:00:03.110378Z","iopub.execute_input":"2024-12-18T01:00:03.110876Z","iopub.status.idle":"2024-12-18T01:00:03.144789Z","shell.execute_reply.started":"2024-12-18T01:00:03.110822Z","shell.execute_reply":"2024-12-18T01:00:03.143057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T01:00:03.146374Z","iopub.execute_input":"2024-12-18T01:00:03.147145Z","iopub.status.idle":"2024-12-18T01:00:03.178181Z","shell.execute_reply.started":"2024-12-18T01:00:03.14708Z","shell.execute_reply":"2024-12-18T01:00:03.176833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T01:00:03.180389Z","iopub.execute_input":"2024-12-18T01:00:03.180926Z","iopub.status.idle":"2024-12-18T01:00:03.978154Z","shell.execute_reply.started":"2024-12-18T01:00:03.180868Z","shell.execute_reply":"2024-12-18T01:00:03.97694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = XGBRegressor(random_state=42)\nmodel.fit(X_train, y_train)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T01:00:03.979809Z","iopub.execute_input":"2024-12-18T01:00:03.980206Z","iopub.status.idle":"2024-12-18T01:00:11.931644Z","shell.execute_reply.started":"2024-12-18T01:00:03.980168Z","shell.execute_reply":"2024-12-18T01:00:11.930575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ensure target values are non-negative\ny_train = y_train.clip(0)\ny_val = y_val.clip(0)\n\n# Define RMSLE function with clipping for predictions\ndef rmsle(y_true, y_pred):\n    y_true = np.clip(y_true, 0, None)  # Clip y_true to be >= 0\n    y_pred = np.clip(y_pred, 0, None)  # Clip y_pred to be >= 0\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))\n\n# Predict and calculate RMSLE\ny_pred = model.predict(X_val)\nrmsle_score = rmsle(y_val, y_pred)\nprint(\"RMSLE on Validation Set:\", rmsle_score)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T01:00:11.933125Z","iopub.execute_input":"2024-12-18T01:00:11.93359Z","iopub.status.idle":"2024-12-18T01:00:12.299288Z","shell.execute_reply.started":"2024-12-18T01:00:11.933539Z","shell.execute_reply":"2024-12-18T01:00:12.297901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test=model.predict(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T01:00:12.301218Z","iopub.execute_input":"2024-12-18T01:00:12.301733Z","iopub.status.idle":"2024-12-18T01:00:13.365606Z","shell.execute_reply.started":"2024-12-18T01:00:12.301663Z","shell.execute_reply":"2024-12-18T01:00:13.364581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\nsubmission_df['Premium Amount']= y_test\nsubmission_df.to_csv('submission.csv', index=False)\nprint(\"Submission file saved as submission.csv!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T01:00:13.367276Z","iopub.execute_input":"2024-12-18T01:00:13.367775Z","iopub.status.idle":"2024-12-18T01:00:14.833421Z","shell.execute_reply.started":"2024-12-18T01:00:13.367727Z","shell.execute_reply":"2024-12-18T01:00:14.832208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T01:43:41.826802Z","iopub.execute_input":"2024-12-18T01:43:41.828065Z","iopub.status.idle":"2024-12-18T01:43:41.841969Z","shell.execute_reply.started":"2024-12-18T01:43:41.828014Z","shell.execute_reply":"2024-12-18T01:43:41.840689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}