{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:19:23.868862Z","iopub.execute_input":"2024-12-31T12:19:23.869138Z","iopub.status.idle":"2024-12-31T12:19:24.29398Z","shell.execute_reply.started":"2024-12-31T12:19:23.869108Z","shell.execute_reply":"2024-12-31T12:19:24.292852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndf_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\ndf_submit = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\n\nclass get_summary:\n    def __init__(self, x):\n        self.x = x\n    def data_set(self):\n        #checks for duplicate\n        duplicate = self.x.duplicated().any()\n        #drop duplicates \n        if duplicate == True:\n            self.x.drop_duplicates(inplace=True)\n            self.x.reset_index(drop=True)\n        #checks for empty values\n        null = self.x.isna().sum().any()\n        #missing values\n        total_missing = self.x.isnull().sum().sum()\n        #data types\n        data_type = self.x.dtypes\n        #shape\n        shapes = self.x.shape\n        return f\"Duplicate: {duplicate}\\nNull: {null}\\nMissing_value: {total_missing}\\nTypes:\\n{data_type}\\nShape: {shapes}\"\n    \n    def total_missing(self):\n        missing_vals = self.x.isnull().sum()\n        cols_with_missing = missing_vals[missing_vals > 0]\n        return cols_with_missing.to_dict()\nprint(f\"Training dataset:\\n{get_summary(df_train).data_set()}\\nTest dataset:\\n{get_summary(df_test).data_set()}\")\nprint(f\"columns with missing values train\\n{get_summary(df_train).total_missing()}\\ncolumns with missing values test\\n{get_summary(df_test).total_missing()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:19:49.134695Z","iopub.execute_input":"2024-12-31T12:19:49.135265Z","iopub.status.idle":"2024-12-31T12:20:05.561331Z","shell.execute_reply.started":"2024-12-31T12:19:49.135216Z","shell.execute_reply":"2024-12-31T12:20:05.56022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.preprocessing import OrdinalEncoder\nimport lightgbm as lgb","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:20:33.020592Z","iopub.execute_input":"2024-12-31T12:20:33.021336Z","iopub.status.idle":"2024-12-31T12:20:35.013778Z","shell.execute_reply.started":"2024-12-31T12:20:33.021299Z","shell.execute_reply":"2024-12-31T12:20:35.012756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def pie_chat(x):\n    plt.pie(x.value_counts(),\n            labels=x.unique(),\n            autopct='%.2f%%',\n            shadow=True,\n            pctdistance=0.5)\n    plt.show()\npie_chat(df_train['Education Level'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:20:48.657894Z","iopub.execute_input":"2024-12-31T12:20:48.658828Z","iopub.status.idle":"2024-12-31T12:20:49.02981Z","shell.execute_reply.started":"2024-12-31T12:20:48.658787Z","shell.execute_reply":"2024-12-31T12:20:49.028342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def pie_chat(x):\n    plt.pie(x.value_counts(),\n            labels=x.unique(),\n            autopct='%.2f%%',\n            shadow=True,\n            pctdistance=0.5,\n            colors=('skyblue', 'gold'),\n            explode=[0.01, 0.2])\n    plt.show()\npie_chat(df_train['Gender'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:21:06.126671Z","iopub.execute_input":"2024-12-31T12:21:06.127635Z","iopub.status.idle":"2024-12-31T12:21:06.411049Z","shell.execute_reply.started":"2024-12-31T12:21:06.12759Z","shell.execute_reply":"2024-12-31T12:21:06.409442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def make_balance(df_train):\n    ord_enc = OrdinalEncoder()\n    #filling missing values 'int'\n    df_train.fillna(df_train.select_dtypes(include='number').mean().iloc[0], inplace=True)\n    #filling missing values 'object'\n    df_train.fillna(df_train.select_dtypes(include='object').mode().iloc[0], inplace=True)\n    #datetime\n    df_train['Policy Start Date'] = pd.to_datetime(df_train['Policy Start Date'])\n    datetime_col = df_train.select_dtypes(include=['datetime64']).columns\n    df_train[datetime_col] = df_train[datetime_col].astype(str)\n    #encoding\n    for cols in df_train.columns:\n        if df_train[cols].dtype in ['float64', 'float32']:\n            df_train[[cols]] = df_train[[cols]].astype(int)\n        if df_train[cols].apply(lambda x: isinstance(x, (str, float))).any():\n            df_train[cols] = df_train[cols].astype(str)\n            df_train[[cols]] = ord_enc.fit_transform(df_train[[cols]])\n    return df_train.head(3)\nmake_balance(df_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:23:43.708758Z","iopub.execute_input":"2024-12-31T12:23:43.709632Z","iopub.status.idle":"2024-12-31T12:24:02.673734Z","shell.execute_reply.started":"2024-12-31T12:23:43.70959Z","shell.execute_reply":"2024-12-31T12:24:02.672691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"make_balance(df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:24:12.633473Z","iopub.execute_input":"2024-12-31T12:24:12.633891Z","iopub.status.idle":"2024-12-31T12:24:25.240669Z","shell.execute_reply.started":"2024-12-31T12:24:12.633855Z","shell.execute_reply":"2024-12-31T12:24:25.239555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df_train.drop(['id', 'Premium Amount'], axis=1)\ny = df_train['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:25:14.939748Z","iopub.execute_input":"2024-12-31T12:25:14.940675Z","iopub.status.idle":"2024-12-31T12:25:15.074587Z","shell.execute_reply.started":"2024-12-31T12:25:14.940632Z","shell.execute_reply":"2024-12-31T12:25:15.073636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_X = df_test.drop('id', axis=1)\ntest_X.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:25:36.303794Z","iopub.execute_input":"2024-12-31T12:25:36.304225Z","iopub.status.idle":"2024-12-31T12:25:36.430014Z","shell.execute_reply.started":"2024-12-31T12:25:36.304185Z","shell.execute_reply":"2024-12-31T12:25:36.428755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:25:46.417063Z","iopub.execute_input":"2024-12-31T12:25:46.417945Z","iopub.status.idle":"2024-12-31T12:25:46.958004Z","shell.execute_reply.started":"2024-12-31T12:25:46.417887Z","shell.execute_reply":"2024-12-31T12:25:46.956832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"light_gb = lgb.LGBMRegressor(random_state=10, \n                             n_estimators=2000,\n                             n_jobs=-1,\n                             max_depth=10,\n                             num_leaves=10,\n                             learning_rate=0.07,\n                             shrinkage_rate=0.12,\n                             metric='rmse')\n\nlight_gb.fit(X_train, y_train)\n\nX_predict = light_gb.predict(X_val)\nLoss = np.sqrt(mean_squared_error(y_val, X_predict))\nprint(f\"RMSE: {Loss}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:32:42.667424Z","iopub.execute_input":"2024-12-31T12:32:42.667906Z","iopub.status.idle":"2024-12-31T12:34:03.331914Z","shell.execute_reply.started":"2024-12-31T12:32:42.667871Z","shell.execute_reply":"2024-12-31T12:34:03.330612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"prediction = light_gb.predict(test_X)\nsubmit = df_submit\nsubmit['Premium Amount'] = prediction\nsubmit.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:34:25.143344Z","iopub.execute_input":"2024-12-31T12:34:25.143715Z","iopub.status.idle":"2024-12-31T12:35:40.990732Z","shell.execute_reply.started":"2024-12-31T12:34:25.143685Z","shell.execute_reply":"2024-12-31T12:35:40.989634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:36:01.620044Z","iopub.execute_input":"2024-12-31T12:36:01.620449Z","iopub.status.idle":"2024-12-31T12:36:03.306198Z","shell.execute_reply.started":"2024-12-31T12:36:01.620417Z","shell.execute_reply":"2024-12-31T12:36:03.3053Z"}},"outputs":[],"execution_count":null}]}