{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30823,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:34:31.640845Z","iopub.execute_input":"2025-01-12T04:34:31.641204Z","iopub.status.idle":"2025-01-12T04:34:31.655863Z","shell.execute_reply.started":"2025-01-12T04:34:31.641175Z","shell.execute_reply":"2025-01-12T04:34:31.65526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd \n\nsampled_df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\n\ntest_df = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:40:09.94352Z","iopub.execute_input":"2025-01-12T04:40:09.943824Z","iopub.status.idle":"2025-01-12T04:40:15.636427Z","shell.execute_reply.started":"2025-01-12T04:40:09.943793Z","shell.execute_reply":"2025-01-12T04:40:15.635748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fraction = 0.1  # 10% of the data\nsampled_df = sampled_df.sample(frac=fraction, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:34:35.178013Z","iopub.execute_input":"2025-01-12T04:34:35.178326Z","iopub.status.idle":"2025-01-12T04:34:35.348526Z","shell.execute_reply.started":"2025-01-12T04:34:35.178302Z","shell.execute_reply":"2025-01-12T04:34:35.347838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sampled_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:34:35.349671Z","iopub.execute_input":"2025-01-12T04:34:35.349874Z","iopub.status.idle":"2025-01-12T04:34:35.423051Z","shell.execute_reply.started":"2025-01-12T04:34:35.349856Z","shell.execute_reply":"2025-01-12T04:34:35.422387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sampled_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:34:35.423773Z","iopub.execute_input":"2025-01-12T04:34:35.424056Z","iopub.status.idle":"2025-01-12T04:34:35.444189Z","shell.execute_reply.started":"2025-01-12T04:34:35.424019Z","shell.execute_reply":"2025-01-12T04:34:35.443256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sampled_df.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:34:35.445148Z","iopub.execute_input":"2025-01-12T04:34:35.445433Z","iopub.status.idle":"2025-01-12T04:34:35.588454Z","shell.execute_reply.started":"2025-01-12T04:34:35.445411Z","shell.execute_reply":"2025-01-12T04:34:35.587575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sampled_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:34:35.589344Z","iopub.execute_input":"2025-01-12T04:34:35.589576Z","iopub.status.idle":"2025-01-12T04:34:35.657641Z","shell.execute_reply.started":"2025-01-12T04:34:35.589557Z","shell.execute_reply":"2025-01-12T04:34:35.656997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Nan_Cols = []\n\nfor col in sampled_df.columns:\n\n    if sampled_df[col].isnull().any() == True:\n\n        print(f\"{col} dtype : {sampled_df[col].dtype}\")\n\n        ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:34:35.659792Z","iopub.execute_input":"2025-01-12T04:34:35.660154Z","iopub.status.idle":"2025-01-12T04:34:35.729215Z","shell.execute_reply.started":"2025-01-12T04:34:35.660134Z","shell.execute_reply":"2025-01-12T04:34:35.72856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in sampled_df.columns:\n    if sampled_df[col].dtype == 'object':  \n        mode_value = sampled_df[col].mode()[0]  \n        sampled_df[col].fillna(mode_value, inplace=True)\n    else:  \n        mean_value = sampled_df[col].mean()  # Calculate mean\n        sampled_df[col].fillna(mean_value, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:34:35.730626Z","iopub.execute_input":"2025-01-12T04:34:35.73083Z","iopub.status.idle":"2025-01-12T04:34:35.923759Z","shell.execute_reply.started":"2025-01-12T04:34:35.730812Z","shell.execute_reply":"2025-01-12T04:34:35.923113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in test_df.columns:\n    if test_df[col].dtype == 'object':  \n        mode_value = test_df[col].mode()[0]  \n        test_df[col].fillna(mode_value, inplace=True)\n    else:  \n        mean_value = test_df[col].mean()  # Calculate mean\n        test_df[col].fillna(mean_value, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:46:12.079166Z","iopub.execute_input":"2025-01-12T04:46:12.079493Z","iopub.status.idle":"2025-01-12T04:46:12.154442Z","shell.execute_reply.started":"2025-01-12T04:46:12.079468Z","shell.execute_reply":"2025-01-12T04:46:12.153494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in sampled_df.columns:\n\n    if sampled_df[col].isnull().any() == True:\n\n        print(f\"{col} dtype : {sampled_df[col].dtype}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:34:35.924515Z","iopub.execute_input":"2025-01-12T04:34:35.924736Z","iopub.status.idle":"2025-01-12T04:34:35.985577Z","shell.execute_reply.started":"2025-01-12T04:34:35.924716Z","shell.execute_reply":"2025-01-12T04:34:35.985014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sampled_df.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:34:35.986234Z","iopub.execute_input":"2025-01-12T04:34:35.986419Z","iopub.status.idle":"2025-01-12T04:34:36.047084Z","shell.execute_reply.started":"2025-01-12T04:34:35.986402Z","shell.execute_reply":"2025-01-12T04:34:36.046444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\ndef numerical_plots(data):\n\n    for col in data.columns:\n\n        if data[col].dtype != 'object':\n\n            sns.histplot(data[col], kde=True, bins=20)\n            plt.title(f'{col} Distribution')\n            plt.show()\n\nnumerical_plots(sampled_df)\n\n\n            ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:34:36.047762Z","iopub.execute_input":"2025-01-12T04:34:36.047944Z","iopub.status.idle":"2025-01-12T04:34:42.674774Z","shell.execute_reply.started":"2025-01-12T04:34:36.047928Z","shell.execute_reply":"2025-01-12T04:34:42.67377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def categorical_plots(data):\n    \n    for col in data.columns:\n        \n        if data[col].dtype == 'object':  \n            \n            sns.countplot(x=data[col])\n            plt.title(f'{col} Distribution')\n            plt.show()\n\ncategorical_plots(sampled_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:34:42.6758Z","iopub.execute_input":"2025-01-12T04:34:42.676153Z","iopub.status.idle":"2025-01-12T04:35:01.536845Z","shell.execute_reply.started":"2025-01-12T04:34:42.676119Z","shell.execute_reply":"2025-01-12T04:35:01.535317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sampled_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:35:08.355449Z","iopub.execute_input":"2025-01-12T04:35:08.35574Z","iopub.status.idle":"2025-01-12T04:35:08.428279Z","shell.execute_reply.started":"2025-01-12T04:35:08.355717Z","shell.execute_reply":"2025-01-12T04:35:08.427563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"constant_cols = [col for col in sampled_df.columns if sampled_df[col].nunique() == 1]\n# sampled_df = sampled_df.drop(constant_cols, axis=1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:35:11.594701Z","iopub.execute_input":"2025-01-12T04:35:11.594981Z","iopub.status.idle":"2025-01-12T04:35:11.714161Z","shell.execute_reply.started":"2025-01-12T04:35:11.59496Z","shell.execute_reply":"2025-01-12T04:35:11.713413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"constant_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:35:12.733649Z","iopub.execute_input":"2025-01-12T04:35:12.73392Z","iopub.status.idle":"2025-01-12T04:35:12.738803Z","shell.execute_reply.started":"2025-01-12T04:35:12.733899Z","shell.execute_reply":"2025-01-12T04:35:12.737897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in sampled_df.columns:\n\n    if sampled_df[col].dtype == 'object':\n\n        print(f\"{col} has {sampled_df[col].value_counts()} uniq values\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:35:14.413678Z","iopub.execute_input":"2025-01-12T04:35:14.413958Z","iopub.status.idle":"2025-01-12T04:35:14.554672Z","shell.execute_reply.started":"2025-01-12T04:35:14.413938Z","shell.execute_reply":"2025-01-12T04:35:14.553825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nbinary_cols = ['Gender', 'Smoking Status']\nfor col in binary_cols:\n    le = LabelEncoder()\n    sampled_df[col] = le.fit_transform(sampled_df[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:35:18.114822Z","iopub.execute_input":"2025-01-12T04:35:18.115149Z","iopub.status.idle":"2025-01-12T04:35:18.157784Z","shell.execute_reply.started":"2025-01-12T04:35:18.115124Z","shell.execute_reply":"2025-01-12T04:35:18.157143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"binary_cols = ['Gender', 'Smoking Status']\nfor col in binary_cols:\n    le = LabelEncoder()\n    test_df[col] = le.fit_transform(test_df[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:44:50.561568Z","iopub.execute_input":"2025-01-12T04:44:50.561943Z","iopub.status.idle":"2025-01-12T04:44:50.807733Z","shell.execute_reply.started":"2025-01-12T04:44:50.561913Z","shell.execute_reply":"2025-01-12T04:44:50.806915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sampled_df = pd.get_dummies(sampled_df, columns=[\n    'Marital Status', 'Education Level', 'Occupation', \n    'Location', 'Policy Type', 'Customer Feedback', \n    'Exercise Frequency', 'Property Type'\n], drop_first=True) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:35:19.609342Z","iopub.execute_input":"2025-01-12T04:35:19.609665Z","iopub.status.idle":"2025-01-12T04:35:19.695058Z","shell.execute_reply.started":"2025-01-12T04:35:19.609635Z","shell.execute_reply":"2025-01-12T04:35:19.694177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = pd.get_dummies(test_df, columns=[\n    'Marital Status', 'Education Level', 'Occupation', \n    'Location', 'Policy Type', 'Customer Feedback', \n    'Exercise Frequency', 'Property Type'\n], drop_first=True) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:44:52.804451Z","iopub.execute_input":"2025-01-12T04:44:52.804734Z","iopub.status.idle":"2025-01-12T04:44:53.485285Z","shell.execute_reply.started":"2025-01-12T04:44:52.804714Z","shell.execute_reply":"2025-01-12T04:44:53.484353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sampled_df['Policy Start Year'] = pd.to_datetime(sampled_df['Policy Start Date']).dt.year\nsampled_df['Policy Start Month'] = pd.to_datetime(sampled_df['Policy Start Date']).dt.month\nsampled_df['Policy Start Day'] = pd.to_datetime(sampled_df['Policy Start Date']).dt.day\nsampled_df['Policy Start DayofWeek'] = pd.to_datetime(sampled_df['Policy Start Date']).dt.dayofweek\nsampled_df['Policy Duration (Days)'] = (pd.Timestamp('now') - pd.to_datetime(sampled_df['Policy Start Date'])).dt.days","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:35:21.147395Z","iopub.execute_input":"2025-01-12T04:35:21.147682Z","iopub.status.idle":"2025-01-12T04:35:21.385343Z","shell.execute_reply.started":"2025-01-12T04:35:21.147661Z","shell.execute_reply":"2025-01-12T04:35:21.384634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df['Policy Start Year'] = pd.to_datetime(test_df['Policy Start Date']).dt.year\ntest_df['Policy Start Month'] = pd.to_datetime(test_df['Policy Start Date']).dt.month\ntest_df['Policy Start Day'] = pd.to_datetime(test_df['Policy Start Date']).dt.day\ntest_df['Policy Start DayofWeek'] = pd.to_datetime(test_df['Policy Start Date']).dt.dayofweek\ntest_df['Policy Duration (Days)'] = (pd.Timestamp('now') - pd.to_datetime(test_df['Policy Start Date'])).dt.days","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:44:56.158577Z","iopub.execute_input":"2025-01-12T04:44:56.158898Z","iopub.status.idle":"2025-01-12T04:44:57.402451Z","shell.execute_reply.started":"2025-01-12T04:44:56.15887Z","shell.execute_reply":"2025-01-12T04:44:57.401501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sampled_df = sampled_df.drop(columns=['Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:35:22.659209Z","iopub.execute_input":"2025-01-12T04:35:22.659525Z","iopub.status.idle":"2025-01-12T04:35:22.673231Z","shell.execute_reply.started":"2025-01-12T04:35:22.659498Z","shell.execute_reply":"2025-01-12T04:35:22.672403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = test_df.drop(columns=['Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:45:00.172394Z","iopub.execute_input":"2025-01-12T04:45:00.172716Z","iopub.status.idle":"2025-01-12T04:45:00.224631Z","shell.execute_reply.started":"2025-01-12T04:45:00.172689Z","shell.execute_reply":"2025-01-12T04:45:00.223705Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sampled_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:35:24.149659Z","iopub.execute_input":"2025-01-12T04:35:24.150058Z","iopub.status.idle":"2025-01-12T04:35:24.168641Z","shell.execute_reply.started":"2025-01-12T04:35:24.150011Z","shell.execute_reply":"2025-01-12T04:35:24.167776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:35:26.953299Z","iopub.execute_input":"2025-01-12T04:35:26.953618Z","iopub.status.idle":"2025-01-12T04:35:26.957527Z","shell.execute_reply.started":"2025-01-12T04:35:26.953591Z","shell.execute_reply":"2025-01-12T04:35:26.956599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = sampled_df.drop(columns=['Premium Amount', 'id'])  \ny = sampled_df['Premium Amount']\n\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:35:33.772222Z","iopub.execute_input":"2025-01-12T04:35:33.772522Z","iopub.status.idle":"2025-01-12T04:35:33.804972Z","shell.execute_reply.started":"2025-01-12T04:35:33.772501Z","shell.execute_reply":"2025-01-12T04:35:33.804341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = RandomForestRegressor(n_estimators=100, random_state=42)\n\nmodel.fit(X_train, y_train)\n\ny_pred = model.predict(X_val)\n\nrmse = np.sqrt(mean_squared_error(y_val, y_pred))\nprint(f\"Validation RMSE: {rmse}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:35:35.81039Z","iopub.execute_input":"2025-01-12T04:35:35.810688Z","iopub.status.idle":"2025-01-12T04:37:34.962947Z","shell.execute_reply.started":"2025-01-12T04:35:35.810666Z","shell.execute_reply":"2025-01-12T04:37:34.962011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:45:27.18837Z","iopub.execute_input":"2025-01-12T04:45:27.188691Z","iopub.status.idle":"2025-01-12T04:45:27.219437Z","shell.execute_reply.started":"2025-01-12T04:45:27.188664Z","shell.execute_reply":"2025-01-12T04:45:27.218704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_features = test_df.drop(columns=['id']) \ntest_predictions = model.predict(test_features)\n\nsubmission = pd.DataFrame({\n    'id': test_df['id'],  \n    'Premium Amount': test_predictions\n})\n\n# Save to CSV\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission file saved as 'submission.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-12T04:46:28.934476Z","iopub.execute_input":"2025-01-12T04:46:28.934789Z","iopub.status.idle":"2025-01-12T04:46:57.790135Z","shell.execute_reply.started":"2025-01-12T04:46:28.934762Z","shell.execute_reply":"2025-01-12T04:46:57.789227Z"}},"outputs":[],"execution_count":null}]}