{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":10218729,"sourceType":"datasetVersion","datasetId":6316802},{"sourceId":200378,"sourceType":"modelInstanceVersion","modelInstanceId":170942,"modelId":193254},{"sourceId":207625,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":176997,"modelId":199292}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:17:41.056935Z","iopub.execute_input":"2024-12-23T10:17:41.05738Z","iopub.status.idle":"2024-12-23T10:17:42.391826Z","shell.execute_reply.started":"2024-12-23T10:17:41.05733Z","shell.execute_reply":"2024-12-23T10:17:42.390077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport math\n!pip install -q scikit-learn==1.4.2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:17:42.394527Z","iopub.execute_input":"2024-12-23T10:17:42.395038Z","iopub.status.idle":"2024-12-23T10:17:59.370021Z","shell.execute_reply.started":"2024-12-23T10:17:42.395002Z","shell.execute_reply":"2024-12-23T10:17:59.368393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!pip install miceforest --no-cache-dir","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:17:59.371688Z","iopub.execute_input":"2024-12-23T10:17:59.372114Z","iopub.status.idle":"2024-12-23T10:17:59.377836Z","shell.execute_reply.started":"2024-12-23T10:17:59.372074Z","shell.execute_reply":"2024-12-23T10:17:59.376566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install cleanlab[datalab]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:17:59.37923Z","iopub.execute_input":"2024-12-23T10:17:59.379584Z","iopub.status.idle":"2024-12-23T10:18:10.501311Z","shell.execute_reply.started":"2024-12-23T10:17:59.379551Z","shell.execute_reply":"2024-12-23T10:18:10.499863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import platform\nimport sklearn\nprint(f\"Bits: {platform.architecture()[0]} | Sklearn Version : {sklearn.__version__}\")\nif sklearn.__version__ != \"1.4.2\":\n   print(\"Please use sklearn 1.4.2\")\n   \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:18:10.50297Z","iopub.execute_input":"2024-12-23T10:18:10.503343Z","iopub.status.idle":"2024-12-23T10:18:10.994521Z","shell.execute_reply.started":"2024-12-23T10:18:10.503307Z","shell.execute_reply":"2024-12-23T10:18:10.993259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rmsle (predicted,target):\n    return round(np.sqrt(np.mean((np.log(predicted + 1) - np.log(target + 1)) ** 2)),6)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:18:10.997466Z","iopub.execute_input":"2024-12-23T10:18:10.998061Z","iopub.status.idle":"2024-12-23T10:18:11.003445Z","shell.execute_reply.started":"2024-12-23T10:18:10.998025Z","shell.execute_reply":"2024-12-23T10:18:11.002192Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Load Packages","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport platform\nimport sys\nfrom pathlib import Path\nfrom sklearn.preprocessing import LabelEncoder, OneHotEncoder, StandardScaler\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nimport seaborn as sbn\nfrom itertools import product\nfrom matplotlib.cbook import boxplot_stats \nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nfrom sklearn.linear_model import BayesianRidge, LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier,DecisionTreeRegressor\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.metrics import root_mean_squared_log_error\nfrom sklearn.ensemble import HistGradientBoostingRegressor\nimport lightgbm as lgbm\n#import miceforest as mf\nfrom cleanlab.regression.learn import CleanLearning\nimport joblib\nimport gc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:18:11.004875Z","iopub.execute_input":"2024-12-23T10:18:11.005249Z","iopub.status.idle":"2024-12-23T10:18:13.602109Z","shell.execute_reply.started":"2024-12-23T10:18:11.005216Z","shell.execute_reply":"2024-12-23T10:18:13.600881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PATH = \"/kaggle/input/playground-series-s4e12/train.csv\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:18:13.603584Z","iopub.execute_input":"2024-12-23T10:18:13.604422Z","iopub.status.idle":"2024-12-23T10:18:13.609388Z","shell.execute_reply.started":"2024-12-23T10:18:13.60437Z","shell.execute_reply":"2024-12-23T10:18:13.608231Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Load Model","metadata":{}},{"cell_type":"code","source":"MODEL_PATH = \"/kaggle/input/sav/other/default/1/insurance_data.sav\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:18:13.610885Z","iopub.execute_input":"2024-12-23T10:18:13.611315Z","iopub.status.idle":"2024-12-23T10:18:13.625375Z","shell.execute_reply.started":"2024-12-23T10:18:13.611266Z","shell.execute_reply":"2024-12-23T10:18:13.624143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = joblib.load(MODEL_PATH)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:18:13.626924Z","iopub.execute_input":"2024-12-23T10:18:13.627424Z","iopub.status.idle":"2024-12-23T10:18:14.207147Z","shell.execute_reply.started":"2024-12-23T10:18:13.627359Z","shell.execute_reply":"2024-12-23T10:18:14.206031Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Load Test Data","metadata":{}},{"cell_type":"code","source":"%%time\nTEST_PATH = \"/kaggle/input/playground-series-s4e12/test.csv\"\ntest = pd.read_csv(TEST_PATH)\ntest.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:18:14.208625Z","iopub.execute_input":"2024-12-23T10:18:14.209358Z","iopub.status.idle":"2024-12-23T10:18:18.901208Z","shell.execute_reply.started":"2024-12-23T10:18:14.209317Z","shell.execute_reply":"2024-12-23T10:18:18.900102Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Preprocessing of Test Data","metadata":{}},{"cell_type":"code","source":"id = test[\"id\"]\nid","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:18:18.902348Z","iopub.execute_input":"2024-12-23T10:18:18.902639Z","iopub.status.idle":"2024-12-23T10:18:18.912864Z","shell.execute_reply.started":"2024-12-23T10:18:18.902611Z","shell.execute_reply":"2024-12-23T10:18:18.911052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_cat_col = [col for col in test.select_dtypes(exclude='number')]\nencoder = LabelEncoder()\nfor col in test_cat_col:\n    if col not in ['Policy Start Date']:\n        test[col] = encoder.fit_transform(test[col])\ntest['Policy Start Date'] = pd.to_datetime(test['Policy Start Date'])\ntest['Year'] = test['Policy Start Date'].dt.year\ntest['Month'] = test['Policy Start Date'].dt.month\ntest['Day'] = test['Policy Start Date'].dt.day\ntest['Annual Income'] = np.log1p(test['Annual Income'])\n#test['Credit Score'] = np.log1p(test['Credit Score'])\n#test['Health Score'] = np.log1p(test['Health Score'])\nmod_ = test['Number of Dependents'].mode().iloc[0]\ntest['Number of Dependents'] = test['Number of Dependents'].fillna(mod_)\ntest.drop(['Policy Start Date','id'],axis=1,inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:18:18.91422Z","iopub.execute_input":"2024-12-23T10:18:18.914612Z","iopub.status.idle":"2024-12-23T10:18:21.055425Z","shell.execute_reply.started":"2024-12-23T10:18:18.914528Z","shell.execute_reply":"2024-12-23T10:18:21.054192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predict = model.predict(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:18:21.056763Z","iopub.execute_input":"2024-12-23T10:18:21.057107Z","iopub.status.idle":"2024-12-23T10:18:24.304853Z","shell.execute_reply.started":"2024-12-23T10:18:21.057076Z","shell.execute_reply":"2024-12-23T10:18:24.303623Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Prediction of Test Data","metadata":{}},{"cell_type":"code","source":"predict = np.exp(predict)-1\npredict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:18:24.306241Z","iopub.execute_input":"2024-12-23T10:18:24.306589Z","iopub.status.idle":"2024-12-23T10:18:24.326183Z","shell.execute_reply.started":"2024-12-23T10:18:24.306551Z","shell.execute_reply":"2024-12-23T10:18:24.324872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame({\"id\":id,\"Premium Amount\":predict})\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:18:24.327675Z","iopub.execute_input":"2024-12-23T10:18:24.328055Z","iopub.status.idle":"2024-12-23T10:18:24.346554Z","shell.execute_reply.started":"2024-12-23T10:18:24.328021Z","shell.execute_reply":"2024-12-23T10:18:24.345195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\",index=False,sep=',')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T10:18:24.348488Z","iopub.execute_input":"2024-12-23T10:18:24.348879Z","iopub.status.idle":"2024-12-23T10:18:25.979932Z","shell.execute_reply.started":"2024-12-23T10:18:24.348831Z","shell.execute_reply":"2024-12-23T10:18:25.978827Z"}},"outputs":[],"execution_count":null}]}