{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-19T00:29:23.092956Z","iopub.execute_input":"2024-12-19T00:29:23.09343Z","iopub.status.idle":"2024-12-19T00:29:23.10017Z","shell.execute_reply.started":"2024-12-19T00:29:23.0934Z","shell.execute_reply":"2024-12-19T00:29:23.098849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T00:34:58.647018Z","iopub.execute_input":"2024-12-19T00:34:58.647415Z","iopub.status.idle":"2024-12-19T00:34:58.651908Z","shell.execute_reply.started":"2024-12-19T00:34:58.647382Z","shell.execute_reply":"2024-12-19T00:34:58.650665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file = open(\"/kaggle/input/playground-series-s4e12/train.csv\")\ndf = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T00:35:51.742773Z","iopub.execute_input":"2024-12-19T00:35:51.743142Z","iopub.status.idle":"2024-12-19T00:35:58.47569Z","shell.execute_reply.started":"2024-12-19T00:35:51.743109Z","shell.execute_reply":"2024-12-19T00:35:58.474658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T00:36:02.699427Z","iopub.execute_input":"2024-12-19T00:36:02.699824Z","iopub.status.idle":"2024-12-19T00:36:02.740176Z","shell.execute_reply.started":"2024-12-19T00:36:02.699794Z","shell.execute_reply":"2024-12-19T00:36:02.739015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.metrics import classification_report, confusion_matrix, accuracy_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T00:37:14.791292Z","iopub.execute_input":"2024-12-19T00:37:14.7917Z","iopub.status.idle":"2024-12-19T00:37:16.74182Z","shell.execute_reply.started":"2024-12-19T00:37:14.791667Z","shell.execute_reply":"2024-12-19T00:37:16.740688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split the data into training and testing sets\nX = df.iloc[:, :-1]\ny = df.iloc[:, -1]\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T00:39:34.498205Z","iopub.execute_input":"2024-12-19T00:39:34.498633Z","iopub.status.idle":"2024-12-19T00:39:34.726935Z","shell.execute_reply.started":"2024-12-19T00:39:34.498601Z","shell.execute_reply":"2024-12-19T00:39:34.725984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_squared_error, r2_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T00:40:38.407059Z","iopub.execute_input":"2024-12-19T00:40:38.407446Z","iopub.status.idle":"2024-12-19T00:40:38.412463Z","shell.execute_reply.started":"2024-12-19T00:40:38.407417Z","shell.execute_reply":"2024-12-19T00:40:38.411377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import necessary libraries\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.tree import DecisionTreeRegressor, export_text, plot_tree\nfrom sklearn.metrics import mean_squared_error, r2_score\nimport matplotlib.pyplot as plt","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2️⃣ Handle Missing Values\ndf.fillna(df.mean(numeric_only=True), inplace=True)\nfor col in df.select_dtypes(include=['object']).columns:\n    df[col].fillna('Unknown', inplace=True)\n\n# 3️⃣ Identify Columns for Processing\nexclude_columns = ['id', 'Premium Amount']  # Exclude identifiers and target\nX = df.drop(columns=exclude_columns, errors='ignore')\ny = df['Premium Amount']  # Target variable\n\n# 4️⃣ Encode Categorical Variables\ncategorical_columns = X.select_dtypes(include=['object', 'bool']).columns\nX = pd.get_dummies(X, columns=categorical_columns, drop_first=True)\n\n# 5️⃣ Split the Data into Training and Testing Sets (80% train, 20% test)\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# 6️⃣ Train a Decision Tree Regressor\nregressor = DecisionTreeRegressor(max_depth=10, random_state=42)\nregressor.fit(X_train, y_train)\n\n# 7️⃣ Predict on Test Data\ny_pred = regressor.predict(X_test)\nmse = mean_squared_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\n\n# print(f\"Mean Squared Error (MSE): {mse}\")\n# print(f\"R-squared (R2) Score: {r2}\")\n\n# # 8️⃣ Visualize the Decision Tree\n# plt.figure(figsize=(12, 8))\n# plot_tree(regressor, feature_names=X.columns, filled=True, rounded=True, fontsize=8)\n# plt.title(\"Decision Tree for Premium Amount Prediction\")\n# plt.show()\n\n# # 🔟 Print Tree Rules (Optional)\n# tree_rules = export_text(regressor, feature_names=list(X.columns))\n# # print(tree_rules)\n\n# 8️⃣ Prepare the Submission File\n# Assume 'id' exists in the test set and was excluded earlier, so we pull it back from df\nX_test_with_ids = df.loc[X_test.index, ['id']].copy()  # Get 'id' from original df using X_test's index\nX_test_with_ids['Premium Amount'] = y_pred  # Add predictions as \"Premium Amount\"\n\n# 9️⃣ Save the Submission File\nX_test_with_ids.to_csv('/kaggle/working/submission.csv', index=False)\n\nprint(f\"Submission file 'submission.csv' created successfully.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T01:48:38.130812Z","iopub.execute_input":"2024-12-19T01:48:38.131161Z","iopub.status.idle":"2024-12-19T01:48:51.27665Z","shell.execute_reply.started":"2024-12-19T01:48:38.131134Z","shell.execute_reply":"2024-12-19T01:48:51.275613Z"}},"outputs":[],"execution_count":null}]}