{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Model Tuning\n\n- Version 1 - 10 trials LGBMRegressor\n- Version 2 - 100 trials LGBMRegressor\n- Version 3 - 10 trials HistGradientBoostingRegressor\n- Version 4 ->\n- Version 5 - change to use Ordinal Encoder on all object types (otherwise same as four)\n- Version 6 - use log1p transformer on target (only on tuning, not on final)\n- Version 7 - use log1p on final prediction\n- Version 8 - use XGBRegressor, 30 trials\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\npd.set_option('display.max_columns', 100)\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.compose import TransformedTargetRegressor\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import OneHotEncoder\n#from sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.decomposition import PCA\n\n#import lightgbm as lgb\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import HistGradientBoostingRegressor\n\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import r2_score\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport optuna\n\nclass CFG:\n    random_state = 81\n    target = 'Premium Amount'  # target feature to predict","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-25T20:36:26.768279Z","iopub.execute_input":"2024-12-25T20:36:26.768686Z","iopub.status.idle":"2024-12-25T20:36:30.332093Z","shell.execute_reply.started":"2024-12-25T20:36:26.768648Z","shell.execute_reply":"2024-12-25T20:36:30.329808Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load Data","metadata":{}},{"cell_type":"code","source":"def load_data(input_path='../input/playground-series-s4e12'):\n    \"\"\" loads the train, test and submissions as dataframes \"\"\"\n    params = {\n        'index_col': 0,\n        'parse_dates': ['Policy Start Date'],\n        'date_parser': lambda x: pd.to_datetime(x).date\n    }\n    train_ = pd.read_csv(f'{input_path}/train.csv', **params)\n    test_ = pd.read_csv(f'{input_path}/test.csv', **params)\n    submission_ = pd.read_csv(f'{input_path}/sample_submission.csv')\n\n    return train_, test_, submission_\n\ntrain_df, test_df, submission_df = load_data()\nprint('train_df shape', train_df.shape)\nprint('test_df shape', test_df.shape)\ndisplay(train_df.sample(3, random_state=CFG.random_state))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T20:36:30.334963Z","iopub.execute_input":"2024-12-25T20:36:30.335626Z","iopub.status.idle":"2024-12-25T20:36:50.373669Z","shell.execute_reply.started":"2024-12-25T20:36:30.335584Z","shell.execute_reply":"2024-12-25T20:36:50.372375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_type_lists(df_):\n    \"\"\" returns categorical/object, number, and datetime64 column names \"\"\"\n    object_cols = df_.select_dtypes(include=['object']).columns\n    numerical_cols = df_.select_dtypes(include=['number']).columns\n    datetime_cols = df_.select_dtypes(include=['datetime64']).columns\n    return object_cols, numerical_cols, datetime_cols\n\nobject_features, numerical_features, datetime_features = feature_type_lists(test_df)\nprint('Object Column Names\\n', object_features.tolist())\nprint('\\nNumerical Column Names\\n', numerical_features.tolist())\nprint('\\nDatetime Column Names\\n', datetime_features.tolist())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T20:36:50.375169Z","iopub.execute_input":"2024-12-25T20:36:50.375515Z","iopub.status.idle":"2024-12-25T20:36:50.504364Z","shell.execute_reply.started":"2024-12-25T20:36:50.375482Z","shell.execute_reply":"2024-12-25T20:36:50.503088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['Education Level'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T20:36:50.505887Z","iopub.execute_input":"2024-12-25T20:36:50.506392Z","iopub.status.idle":"2024-12-25T20:36:50.612699Z","shell.execute_reply.started":"2024-12-25T20:36:50.506332Z","shell.execute_reply":"2024-12-25T20:36:50.611468Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Preprocessor","metadata":{}},{"cell_type":"code","source":"# numerical processing\n# use 0 for Annual Income, Number of Dependents, and Previous Claims, other fields use mean\nnum_zero_impute = ['Annual Income', 'Number of Dependents', 'Previous Claims']\nnum_mean_impute = [feat for feat in numerical_features if feat not in num_zero_impute]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T20:36:50.615323Z","iopub.execute_input":"2024-12-25T20:36:50.615805Z","iopub.status.idle":"2024-12-25T20:36:50.624619Z","shell.execute_reply.started":"2024-12-25T20:36:50.615755Z","shell.execute_reply":"2024-12-25T20:36:50.623368Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"for categorical we could do a Enumerate Encoding instead of One-Hot in the case where there are only 2 values this will reduce the number of features by 1 for each case","metadata":{}},{"cell_type":"code","source":"ordinal_cats = ['Gender', 'Smoking Status']\nonehot_cats = [feat for feat in object_features if feat not in ordinal_cats]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T20:36:50.62621Z","iopub.execute_input":"2024-12-25T20:36:50.626694Z","iopub.status.idle":"2024-12-25T20:36:50.637729Z","shell.execute_reply.started":"2024-12-25T20:36:50.626642Z","shell.execute_reply":"2024-12-25T20:36:50.636393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# TODO - update https://www.kaggle.com/code/francescoalbanesephd/eda-mi-tree-selection-xgb-optuna-1-06147-v1\n\n\n# set up the preprocessor to use\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num-zero', Pipeline(steps=[\n            ('impute', SimpleImputer(strategy='constant', fill_value=0)),\n            ('scale', MinMaxScaler())\n        ]), num_zero_impute),\n        ('num-mean', Pipeline(steps=[\n            ('impute', SimpleImputer(strategy='mean')),\n            ('scale', MinMaxScaler()),\n        ]), num_mean_impute),\n        ('cat-one', Pipeline(steps=[\n            ('ohe', OneHotEncoder())\n        ]), onehot_cats),\n        ('cat-ord', Pipeline(steps=[\n            ('ordinal', OrdinalEncoder())\n        ]), ordinal_cats),\n    ])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T20:36:50.639143Z","iopub.execute_input":"2024-12-25T20:36:50.639892Z","iopub.status.idle":"2024-12-25T20:36:50.651385Z","shell.execute_reply.started":"2024-12-25T20:36:50.639846Z","shell.execute_reply":"2024-12-25T20:36:50.650259Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Tuning","metadata":{}},{"cell_type":"code","source":"# setup X, y data\nX = train_df.copy()\ny = X.pop(CFG.target)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T20:36:50.652981Z","iopub.execute_input":"2024-12-25T20:36:50.653628Z","iopub.status.idle":"2024-12-25T20:36:50.814974Z","shell.execute_reply.started":"2024-12-25T20:36:50.653573Z","shell.execute_reply":"2024-12-25T20:36:50.813899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def root_mean_squared_log_error(y_true, y_pred):\n    y_pred = np.maximum(0, y_pred)\n    return np.sqrt(np.mean((np.log1p(y_true) - np.log1p(y_pred)) ** 2))\n\n\n\nclass Tuning():\n    \"\"\"\n    \"\"\"\n    def __init__(self, model, preprocessor, tuning_parameters):\n        \"\"\"\n        Sets up a Tuning Object for the model, preprocessor and tuning parameters\n        \"\"\"\n        self.model = model\n        self.preprocessor = preprocessor\n        self.tuning_parameters = tuning_parameters\n\n        self.study = optuna.create_study(direction='minimize')\n\n\n    def __param_fn(self, trial):\n        \"\"\"\n        creates a mapping of trial functions to suggest the parameters to tune\n        expects self.tuning_parameters value to be a tuple (str, min, max)\n        \"\"\"\n\n        def process_entry(key_, value):\n            \"\"\"\n            value is either tuple (type_, minval, maxval) or constant\n            \"\"\"\n            if isinstance(value, tuple):\n                return getattr(trial, f'suggest_{value[0]}')(key_, value[1], value[2])\n            return value\n        \n        return {\n            varkey: process_entry(varkey, value)\n            for varkey, value in self.tuning_parameters.items()\n        }\n    \n    def tune(self, X, y, n_trials=30):\n        \"\"\"\n        tunes the model with the provided data\n        \n        \"\"\"\n        X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=.2, random_state=CFG.random_state)\n\n\n        \n        def objective_fn():\n            \"\"\"\n            creates an objective(trial) function\n            :param model: an sklearn estimator (catogorical)\n            :param params: this is a function of \"trial\"\n            :return: a function of trial that is evalating the model using accuracy_score()\n            \"\"\"\n            def objective(trial):\n\n                model1 = Pipeline(steps=[\n                    ('preprocessor', self.preprocessor),\n                    ('model', self.model(**self.__param_fn(trial)))\n                ])\n                model = TransformedTargetRegressor(\n                    regressor=model1,\n                    func=np.log1p,            # Transformation for the target variable\n                    inverse_func=np.expm1     # Inverse transformation for predictions\n                )\n    \n                model.fit(X_train, y_train)\n                y_pred = model.predict(X_test)\n                return root_mean_squared_log_error(y_test, y_pred)\n\n            return objective\n\n        self.study.optimize(\n            objective_fn(),\n            n_trials=n_trials\n        )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T20:36:50.816424Z","iopub.execute_input":"2024-12-25T20:36:50.816743Z","iopub.status.idle":"2024-12-25T20:36:50.828733Z","shell.execute_reply.started":"2024-12-25T20:36:50.816712Z","shell.execute_reply":"2024-12-25T20:36:50.82738Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#pipeline = evaluate_model(HistGradientBoostingRegressor())\n#best_model = LGBMRegressor(**best_params, verbose=-1, n_jobs=-1)\n#best_model.fit(X, np.log1p(y))\n#pipeline = evaluate_model(LGBMRegressor(verbose=-1, n_jobs=-1))\n\n\n#pipeline = evaluate_model(RandomForestRegressor())\n\n#best_params = {'n_estimators': 550, 'learning_rate': 0.011815382331813444, 'num_leaves': 116, 'max_depth': 12, 'min_child_samples': 16, 'subsample': 0.6354134229005893, 'colsample_bytree': 0.8861944216276492, 'reg_alpha': 1.7789343457859361, 'reg_lambda': 0.4341060744803095}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T20:36:50.830226Z","iopub.execute_input":"2024-12-25T20:36:50.830628Z","iopub.status.idle":"2024-12-25T20:36:50.848928Z","shell.execute_reply.started":"2024-12-25T20:36:50.830594Z","shell.execute_reply":"2024-12-25T20:36:50.847865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tuner = Tuning(\n    LGBMRegressor,\n    preprocessor,\n    {\n        'n_estimators': ('int', 100, 1000),\n        'learning_rate': ('loguniform', 1e-2, 0.2),\n        'num_leaves': ('int', 50, 300),\n        'min_child_samples': ('int', 10, 50),\n        'max_depth': ('int', 5, 15),\n        'subsample': ('float', 0.5, 1.0),\n        'colsample_bytree': ('float', 0.5, 0.95),\n        'reg_alpha': ('loguniform', 1e-1, 5.0),\n        'reg_lambda': ('loguniform', 1e-2, 1.0),\n        'verbose': -1,\n        'objective': 'regression',\n        'metric': 'rmse',\n        'boosting_type': 'gbdt',  \n    }\n)\n\n#tuner.tune(X, y, n_trials=30)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T20:36:50.850453Z","iopub.execute_input":"2024-12-25T20:36:50.850904Z","iopub.status.idle":"2024-12-25T20:36:50.864217Z","shell.execute_reply.started":"2024-12-25T20:36:50.85085Z","shell.execute_reply":"2024-12-25T20:36:50.862837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tuner2 = Tuning(\n    HistGradientBoostingRegressor,\n    preprocessor,\n    {\n        'quantile': ('float', 0.0, 1.0),\n        'learning_rate': ('loguniform', 1e-4, 1),\n        'max_iter': ('int', 80, 200),\n        'max_leaf_nodes': ('int', 20, 40),\n        'max_depth': ('int', 5, 15),\n        'min_samples_leaf': ('int', 10, 30),\n        'l2_regularization': ('float', 1e-4, 5.0),\n    }\n)\n\n#tuner.tune(X, y, n_trials=10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T20:36:50.865975Z","iopub.execute_input":"2024-12-25T20:36:50.866365Z","iopub.status.idle":"2024-12-25T20:36:50.878196Z","shell.execute_reply.started":"2024-12-25T20:36:50.86633Z","shell.execute_reply":"2024-12-25T20:36:50.876957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cat_params = {\n#     'iterations': trial.suggest_int('iterations', 2000, 4000),\n#     'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.1),\n#     'depth': trial.suggest_int('depth', 4, 10),\n#     'l2_leaf_reg': trial.suggest_float('l2_leaf_reg', 1e-8, 5.0),\n#     'bagging_temperature': trial.suggest_float('bagging_temperature', 0.0, 1.0),\n#     'task_type': 'GPU',\n#     'devices': '0',\n# }\n\ntuner = Tuning(\n    XGBRegressor,\n    preprocessor,\n    {\n        'n_estimators': ('int', 100, 3000),\n        'learning_rate': ('float', 0.01, 0.1),\n        'max_depth': ('int', 3, 12),\n        'min_child_weight': ('int', 1, 10),\n        'subsample': ('float', 0.5, 1.0),\n        'colsample_bytree': ('float', 0.5, 1.0),\n        'reg_alpha': ('float', 1e-8, 10.0),\n        'reg_lambda': ('float', 1e-8, 10.0),\n        #'tree_method': 'gpu_hist',\n        #'predictor': 'gpu_predictor',\n        'verbose': 0\n    }\n)\ntuner.tune(X, y, n_trials=30)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T20:37:55.151637Z","iopub.execute_input":"2024-12-25T20:37:55.152081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('best params', tuner.study.best_params)\nprint('best value', tuner.study.best_value)\n\n#optuna.visualization.plot_optimization_history(study)\ntrials_df = tuner.study.trials_dataframe()\ndisplay(trials_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T20:36:51.290566Z","iopub.status.idle":"2024-12-25T20:36:51.290985Z","shell.execute_reply.started":"2024-12-25T20:36:51.290778Z","shell.execute_reply":"2024-12-25T20:36:51.290807Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submit Model\nUsing **pipeline** from previous step, and **test_df** and **subsmission_df** from initial load, setup submission file","metadata":{}},{"cell_type":"code","source":"pipeline1 = Pipeline(steps=[\n    ('preprocessor', preprocessor),\n    ('model', tuner.model(**tuner.study.best_params))\n])\n\npipeline = TransformedTargetRegressor(\n    regressor=pipeline1,\n    func=np.log1p,            # Transformation for the target variable\n    inverse_func=np.expm1     # Inverse transformation for predictions\n)\n\npipeline.fit(X, y)\ny_pred = pipeline.predict(test_df)\nsubmission_df[CFG.target] = y_pred\n\nsubmission_df.to_csv('submission.csv', index=False)\nsubmission_df.sample(5, random_state=CFG.random_state)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T20:36:51.292614Z","iopub.status.idle":"2024-12-25T20:36:51.292987Z","shell.execute_reply.started":"2024-12-25T20:36:51.292817Z","shell.execute_reply":"2024-12-25T20:36:51.292836Z"}},"outputs":[],"execution_count":null}]}