{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!python --version","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:16:33.780677Z","iopub.execute_input":"2024-12-30T21:16:33.781186Z","iopub.status.idle":"2024-12-30T21:16:33.914724Z","shell.execute_reply.started":"2024-12-30T21:16:33.781128Z","shell.execute_reply":"2024-12-30T21:16:33.913374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !sudo apt-get install python3.9","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:16:33.916373Z","iopub.execute_input":"2024-12-30T21:16:33.91676Z","iopub.status.idle":"2024-12-30T21:16:33.921237Z","shell.execute_reply.started":"2024-12-30T21:16:33.916714Z","shell.execute_reply":"2024-12-30T21:16:33.920361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !sudo update-alternatives --install /usr/bin/python3 python3 /usr/bin/python3.9 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:16:33.923036Z","iopub.execute_input":"2024-12-30T21:16:33.923528Z","iopub.status.idle":"2024-12-30T21:16:33.939846Z","shell.execute_reply.started":"2024-12-30T21:16:33.923471Z","shell.execute_reply":"2024-12-30T21:16:33.938776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!python --version","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-12-30T21:16:33.941989Z","iopub.execute_input":"2024-12-30T21:16:33.942324Z","iopub.status.idle":"2024-12-30T21:16:34.077897Z","shell.execute_reply.started":"2024-12-30T21:16:33.942297Z","shell.execute_reply":"2024-12-30T21:16:34.076593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport xgboost as xgb\nfrom xgboost import XGBRegressor\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.metrics import mean_squared_log_error as MSLE\n# from sklearn.metrics import root_mean_squared_error as RMSE\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.linear_model import LinearRegression\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-30T21:16:34.07901Z","iopub.execute_input":"2024-12-30T21:16:34.079332Z","iopub.status.idle":"2024-12-30T21:16:35.7528Z","shell.execute_reply.started":"2024-12-30T21:16:34.079304Z","shell.execute_reply":"2024-12-30T21:16:35.751756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train=pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\nsample=pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\ndata_test=pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\ndata_train.drop(columns=['id','Policy Start Date'],inplace=True)\nsubmission_ids=data_test['id']\ndata_test.drop(columns=['id','Policy Start Date'],inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:16:35.753877Z","iopub.execute_input":"2024-12-30T21:16:35.754429Z","iopub.status.idle":"2024-12-30T21:16:46.926387Z","shell.execute_reply.started":"2024-12-30T21:16:35.75439Z","shell.execute_reply":"2024-12-30T21:16:46.925401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train['Age']=data_train['Age'].apply(lambda x: 'young' if (x>=18) and (x<=30) else 'adult' if (x>30) and (x<=50) else 'grown' if (x>50) and (x<=60) else 'old' )\n\ndata_train['Vehicle Age']=data_train['Vehicle Age'].apply(lambda x: 'new' if (x<=3) else 'mid_age' if (x>3) and (x<=7) else 'old' if (x>7) and (x<=12) else 'very old' )\n\n\ndata_train['Number of Dependents']=data_train['Number of Dependents'].apply(lambda x: 'no_kids' if (x<=0) else 'one_kid' if (x>0) and (x<=1) else 'few kids' if (x>1) and (x<=3) else 'big_family' )\ndata_train['Annual Income']=data_train['Annual Income'].apply(lambda x: 'poor' if (x<=20000) else 'low' if (x>20000) and (x<=35000) else 'enough' if (x>35000) and (x<=80000) else 'luxury' )\ndata_train['Health Score']=data_train['Health Score'].apply(lambda x: 'low' if (x<=15) else 'medium' if (x>15) and (x<=35) else 'luxury')\ndata_train['Insurance Duration']=data_train['Insurance Duration'].apply(lambda x: 'low' if (x<=3) else 'medium' if (x>3) and (x<=5) else 'luxury')\ndata_train['Credit Score']=data_train['Credit Score'].apply(lambda x: 'low' if (x<=468) else 'medium' if (x>468) and (x<=700) else 'luxury')\n\ndata_test['Age']=data_test['Age'].apply(lambda x: 'young' if (x>=18) and (x<=30) else 'adult' if (x>30) and (x<=50) else 'grown' if (x>50) and (x<=60) else 'old' )\n\ndata_test['Vehicle Age']=data_test['Vehicle Age'].apply(lambda x: 'new' if (x<=3) else 'mid_age' if (x>3) and (x<=7) else 'old' if (x>7) and (x<=12) else 'very old' )\n\ndata_test['Annual Income']=data_test['Annual Income'].apply(lambda x: 'poor' if (x<=20000) else 'low' if (x>20000) and (x<=35000) else 'enough' if (x>35000) and (x<=80000) else 'luxury' )\ndata_test['Health Score']=data_test['Health Score'].apply(lambda x: 'low' if (x<=15) else 'medium' if (x>15) and (x<=35) else 'luxury')\ndata_test['Insurance Duration']=data_test['Insurance Duration'].apply(lambda x: 'low' if (x<=3) else 'medium' if (x>3) and (x<=5) else 'luxury')\ndata_test['Credit Score']=data_test['Credit Score'].apply(lambda x: 'low' if (x<=468) else 'medium' if (x>468) and (x<=700) else 'luxury')\n\ndata_test['Number of Dependents']=data_test['Number of Dependents'].apply(lambda x: 'no_kids' if (x<=0) else 'one_kid' if (x>0) and (x<=1) else 'few kids' if (x>1) and (x<=3) else 'big_family' )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:16:46.927366Z","iopub.execute_input":"2024-12-30T21:16:46.927623Z","iopub.status.idle":"2024-12-30T21:16:50.840286Z","shell.execute_reply.started":"2024-12-30T21:16:46.927598Z","shell.execute_reply":"2024-12-30T21:16:50.839263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfor column in data_train.columns.values:\n\n    if data_train.dtypes[f'{column}'] ==np.float64:\n        mean=data_train[f'{column}'].mean()\n        data_train[f'{column}'].fillna(mean,inplace=True)\n        data_train[f'{column}']=data_train[f'{column}'].astype(\"float\")\n    else:\n        mean = data_train[f'{column}'].value_counts().index.values[0]\n        data_train[f'{column}'].fillna(mean, inplace=True)\n        data_train[f'{column}']=data_train[f'{column}'].astype(\"category\")\n\n\nfor column in data_test.columns.values:\n    if data_test.dtypes[f'{column}'] ==np.float64:\n        mean=data_test[f'{column}'].mean()\n        data_test[f'{column}'].fillna(mean,inplace=True)\n        data_test[f'{column}']=data_test[f'{column}'].astype(\"float\")\n    else:\n        mean = data_test[f'{column}'].value_counts().index.values[0]\n        data_test[f'{column}'].fillna(mean, inplace=True)\n        data_test[f'{column}']=data_test[f'{column}'].astype(\"category\")\n\nprint('a')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:16:50.842531Z","iopub.execute_input":"2024-12-30T21:16:50.842814Z","iopub.status.idle":"2024-12-30T21:16:56.584294Z","shell.execute_reply.started":"2024-12-30T21:16:50.842788Z","shell.execute_reply":"2024-12-30T21:16:56.583039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ohe=OneHotEncoder()\nnan_count=data_train.isna().sum()\ndata_train=pd.get_dummies(data_train,drop_first=True,dtype=float)\ndata_test=pd.get_dummies(data_test,drop_first=True,dtype=float)\n\nnan_count=data_test.isna().sum()\n\nnan_count=data_train.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:16:56.585577Z","iopub.execute_input":"2024-12-30T21:16:56.585851Z","iopub.status.idle":"2024-12-30T21:16:57.424621Z","shell.execute_reply.started":"2024-12-30T21:16:56.585825Z","shell.execute_reply":"2024-12-30T21:16:57.423835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nX_train, X_test, y_train, y_test = data_train.drop(columns=['Premium Amount']),data_test,data_train['Premium Amount'],sample['Premium Amount']\ny_test=np.log1p(y_test)\ny_train=np.log1p(y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:16:57.425813Z","iopub.execute_input":"2024-12-30T21:16:57.426197Z","iopub.status.idle":"2024-12-30T21:16:57.543229Z","shell.execute_reply.started":"2024-12-30T21:16:57.42616Z","shell.execute_reply":"2024-12-30T21:16:57.542449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lightgbm import LGBMRegressor\nlgbm = LGBMRegressor(metric='rmsle')\n\nfrom sklearn.model_selection import RandomizedSearchCV\ndistributions = {\n    'max_depth': [3,6,9,15,20,50] ,\n'min_child_samples':[5,10,20,50,100],\n'learning_rate':[0.1,0.01,0.5,0.001],\n'reg_alpha':[0,1],\n'reg_lambda':[0,1],\n'num_leaves':[15,31,50]}\nclf = RandomizedSearchCV(lgbm, distributions, random_state=0)\n\nsearch = clf.fit(X_train, y_train)\nsearch.best_params_\nmodel=search.best_estimator_\n\nmodel.fit(X_train, y_train)\n\npreds_train = model.predict(X_train)\npreds_test = model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:16:57.544071Z","iopub.execute_input":"2024-12-30T21:16:57.544332Z","iopub.status.idle":"2024-12-30T21:22:47.797761Z","shell.execute_reply.started":"2024-12-30T21:16:57.544309Z","shell.execute_reply":"2024-12-30T21:22:47.796876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ny_train=np.expm1(y_train)\npreds_train=np.expm1(preds_train)\ny_test=np.expm1(y_test)\npreds_test=np.expm1(preds_test)\n\nsubmission=pd.concat([submission_ids,pd.Series(preds_test,name='Premium Amount')],names=['id','Premium Amount'],axis=1)\nprint(submission)\nsubmission.to_csv('submission.csv',index=False)\nrmsle_train=np.sqrt(MSLE(y_train,preds_train))\nrmsle_test=np.sqrt(MSLE(y_test,preds_test))\nprint(f'RMSLE train:{rmsle_train}')\nprint(f'RMSLE test:{rmsle_test}')\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:22:47.798761Z","iopub.execute_input":"2024-12-30T21:22:47.799407Z","iopub.status.idle":"2024-12-30T21:22:49.556923Z","shell.execute_reply.started":"2024-12-30T21:22:47.799379Z","shell.execute_reply":"2024-12-30T21:22:49.556042Z"}},"outputs":[],"execution_count":null}]}