{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30805,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:12:13.091311Z","iopub.execute_input":"2024-12-05T17:12:13.092148Z","iopub.status.idle":"2024-12-05T17:12:13.418141Z","shell.execute_reply.started":"2024-12-05T17:12:13.092113Z","shell.execute_reply":"2024-12-05T17:12:13.417285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import KNNImputer\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.base import BaseEstimator, TransformerMixin\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:12:16.022592Z","iopub.execute_input":"2024-12-05T17:12:16.023058Z","iopub.status.idle":"2024-12-05T17:12:17.013162Z","shell.execute_reply.started":"2024-12-05T17:12:16.023022Z","shell.execute_reply":"2024-12-05T17:12:17.0122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df=pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv').drop(['id', 'Policy Start Date'], axis = 1)\ntest_df = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv').drop(['id', 'Policy Start Date'], axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:12:19.774332Z","iopub.execute_input":"2024-12-05T17:12:19.774805Z","iopub.status.idle":"2024-12-05T17:12:28.162829Z","shell.execute_reply.started":"2024-12-05T17:12:19.774772Z","shell.execute_reply":"2024-12-05T17:12:28.161839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:12:28.164249Z","iopub.execute_input":"2024-12-05T17:12:28.164533Z","iopub.status.idle":"2024-12-05T17:12:28.18759Z","shell.execute_reply.started":"2024-12-05T17:12:28.164501Z","shell.execute_reply":"2024-12-05T17:12:28.186767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"g = sns.displot(train_df['Premium Amount'], bins=10, kde=True)\ng.fig.suptitle('Distribution of Premium Amount', fontsize=16)\ng.set_axis_labels('Premium Amount', 'Frequency')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:12:32.767378Z","iopub.execute_input":"2024-12-05T17:12:32.767684Z","iopub.status.idle":"2024-12-05T17:12:37.617816Z","shell.execute_reply.started":"2024-12-05T17:12:32.767659Z","shell.execute_reply":"2024-12-05T17:12:37.617042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DateFeatureExtractor(BaseEstimator, TransformerMixin):\n    \"\"\"Custom transformer to extract sine and cosine features from date.\"\"\"\n    \n    def fit(self, X, y=None):\n        return self\n    \n    def transform(self, X):\n        # Ensure 'Policy Start Date' is in datetime format\n        X['Policy Start Date'] = pd.to_datetime(X['Policy Start Date'])\n        \n        # Extract day and month\n        X['Policy Start Day'] = X['Policy Start Date'].dt.day\n        X['Policy Start Month'] = X['Policy Start Date'].dt.month\n        \n        # Create sine and cosine features\n        X['Policy Start Day_Sin'] = np.sin(2 * np.pi * X['Policy Start Day'] / 31)\n        X['Policy Start Day_Cos'] = np.cos(2 * np.pi * X['Policy Start Day'] / 31)\n        X['Policy Start Month_Sin'] = np.sin(2 * np.pi * X['Policy Start Month'] / 12)\n        X['Policy Start Month_Cos'] = np.cos(2 * np.pi * X['Policy Start Month'] / 12)\n        \n        # Drop the original date column\n        return X.drop(columns=['Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:12:40.210953Z","iopub.execute_input":"2024-12-05T17:12:40.211715Z","iopub.status.idle":"2024-12-05T17:12:40.218024Z","shell.execute_reply.started":"2024-12-05T17:12:40.211678Z","shell.execute_reply":"2024-12-05T17:12:40.217212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.select_dtypes('object').columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:12:45.507037Z","iopub.execute_input":"2024-12-05T17:12:45.507379Z","iopub.status.idle":"2024-12-05T17:12:45.640041Z","shell.execute_reply.started":"2024-12-05T17:12:45.507352Z","shell.execute_reply":"2024-12-05T17:12:45.639046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_pipeline():\n    categorical_cols = ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location',\n                        'Policy Type', 'Customer Feedback','Smoking Status',\n                        'Exercise Frequency', 'Property Type']\n    \n    # Create a pipeline for preprocessing\n    pipeline = Pipeline(steps=[\n        # ('date_features', DateFeatureExtractor()),  # Extract date features\n        ('encoder', ColumnTransformer(transformers=[\n            ('cat', OrdinalEncoder(), categorical_cols)  # Ordinally encode categorical features\n        ], remainder='passthrough')), # Leave other columns unchanged\n        # ('imputer', KNNImputer(n_neighbors=5)),  # Impute missing values\n    ])\n    \n    return pipeline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:12:49.010993Z","iopub.execute_input":"2024-12-05T17:12:49.012031Z","iopub.status.idle":"2024-12-05T17:12:49.019521Z","shell.execute_reply.started":"2024-12-05T17:12:49.011957Z","shell.execute_reply":"2024-12-05T17:12:49.017444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_data(train_df, test_df):\n    pipeline = create_pipeline()\n    train_processed = pipeline.fit_transform(train_df)\n    test_processed = pipeline.transform(test_df)\n    \n    return train_processed, test_processed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:12:53.968389Z","iopub.execute_input":"2024-12-05T17:12:53.96874Z","iopub.status.idle":"2024-12-05T17:12:53.973352Z","shell.execute_reply.started":"2024-12-05T17:12:53.96871Z","shell.execute_reply":"2024-12-05T17:12:53.972372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target = train_df.pop('Premium Amount')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:12:56.925825Z","iopub.execute_input":"2024-12-05T17:12:56.926527Z","iopub.status.idle":"2024-12-05T17:12:56.930714Z","shell.execute_reply.started":"2024-12-05T17:12:56.926494Z","shell.execute_reply":"2024-12-05T17:12:56.929841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_processed, test_processed = preprocess_data(train_df, test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:12:59.224483Z","iopub.execute_input":"2024-12-05T17:12:59.225274Z","iopub.status.idle":"2024-12-05T17:13:03.384132Z","shell.execute_reply.started":"2024-12-05T17:12:59.225239Z","shell.execute_reply":"2024-12-05T17:13:03.383434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBRegressor\n\nmodel = XGBRegressor().fit(train_processed, target.values)\npredictions = model.predict(test_processed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:13:05.876586Z","iopub.execute_input":"2024-12-05T17:13:05.87742Z","iopub.status.idle":"2024-12-05T17:13:13.703693Z","shell.execute_reply.started":"2024-12-05T17:13:05.877384Z","shell.execute_reply":"2024-12-05T17:13:13.702944Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\nsubmission['Premium Amount'] = predictions\nsubmission.to_csv('submission.csv', index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:14:00.034627Z","iopub.execute_input":"2024-12-05T17:14:00.034988Z","iopub.status.idle":"2024-12-05T17:14:01.184313Z","shell.execute_reply.started":"2024-12-05T17:14:00.034945Z","shell.execute_reply":"2024-12-05T17:14:01.183302Z"}},"outputs":[],"execution_count":null}]}