{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.impute import SimpleImputer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T16:44:38.816889Z","iopub.execute_input":"2024-12-13T16:44:38.817288Z","iopub.status.idle":"2024-12-13T16:44:40.221265Z","shell.execute_reply.started":"2024-12-13T16:44:38.817251Z","shell.execute_reply":"2024-12-13T16:44:40.220064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_file_path = '/kaggle/input/playground-series-s4e12/train.csv'\ntest_file_path = '/kaggle/input/playground-series-s4e12/test.csv'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T16:44:40.223919Z","iopub.execute_input":"2024-12-13T16:44:40.224495Z","iopub.status.idle":"2024-12-13T16:44:40.229918Z","shell.execute_reply.started":"2024-12-13T16:44:40.224458Z","shell.execute_reply":"2024-12-13T16:44:40.228686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = pd.read_csv(train_file_path)\ntest_data = pd.read_csv(test_file_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T16:44:40.231441Z","iopub.execute_input":"2024-12-13T16:44:40.231901Z","iopub.status.idle":"2024-12-13T16:44:50.502697Z","shell.execute_reply.started":"2024-12-13T16:44:40.231855Z","shell.execute_reply":"2024-12-13T16:44:50.501529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.head(), test_data.head()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-13T16:44:50.504361Z","iopub.execute_input":"2024-12-13T16:44:50.504809Z","iopub.status.idle":"2024-12-13T16:44:50.546308Z","shell.execute_reply.started":"2024-12-13T16:44:50.50476Z","shell.execute_reply":"2024-12-13T16:44:50.545092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_data(data, is_train=True):\n    data = data.copy()\n    \n    # Handle missing values for specific columns\n    imputer = SimpleImputer(strategy='most_frequent')\n    for col in ['Gender', 'Marital Status', 'Occupation', 'Credit Score', 'Previous Claims']:\n        data[col] = imputer.fit_transform(data[[col]]).ravel()  # Use `.ravel()` to convert to 1D\n\n    # Handle missing values for numerical columns (e.g., fill NaN with column mean)\n    num_imputer = SimpleImputer(strategy='mean')\n    num_cols = data.select_dtypes(include=['float64', 'int64']).columns\n    data[num_cols] = num_imputer.fit_transform(data[num_cols])\n    \n    # Convert Policy Start Date to numerical feature (days since start)\n    data['Policy Start Date'] = pd.to_datetime(data['Policy Start Date'])\n    data['Days Since Policy Start'] = (pd.Timestamp('2024-12-31') - data['Policy Start Date']).dt.days\n    \n    # Drop original Policy Start Date and id columns\n    data = data.drop(columns=['Policy Start Date', 'id'])\n    \n    # Encode categorical variables\n    categorical_cols = data.select_dtypes(include=['object']).columns\n    for col in categorical_cols:\n        data[col] = LabelEncoder().fit_transform(data[col])\n    \n    # Separate target for training data\n    if is_train:\n        X = data.drop(columns=['Premium Amount'])\n        y = data['Premium Amount']\n        return X, y\n    else:\n        return data\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T16:44:50.54879Z","iopub.execute_input":"2024-12-13T16:44:50.549149Z","iopub.status.idle":"2024-12-13T16:44:50.558019Z","shell.execute_reply.started":"2024-12-13T16:44:50.549081Z","shell.execute_reply":"2024-12-13T16:44:50.556733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, y_train = preprocess_data(train_data)\nX_test = preprocess_data(test_data, is_train=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T16:44:50.559523Z","iopub.execute_input":"2024-12-13T16:44:50.559945Z","iopub.status.idle":"2024-12-13T16:44:58.498336Z","shell.execute_reply.started":"2024-12-13T16:44:50.559901Z","shell.execute_reply":"2024-12-13T16:44:58.497422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = RandomForestRegressor(random_state=42)\nmodel.fit(X_train, y_train)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = model.predict(X_test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"output = pd.DataFrame({'id': test_data['id'], 'Premium Amount': predictions})","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"output_file_path = \"/kaggle/working/submission.csv\"\noutput.to_csv(output_file_path, index=False)\n\noutput_file_path","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}