{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"---\n### Upvote, please, if you liked this notebook\n\n---","metadata":{}},{"cell_type":"markdown","source":"# Table of Content\n---\n- <a href='#c1'> 1. Import modules, building classes and read CSV </a>\n\n- <a href='#c4'> 2. Model building</a>\n\n- <a href='#c5'> 3. Validation</a>\n\n- <a href='#c6'> 4. Results and submissions</a>\n---","metadata":{}},{"cell_type":"markdown","source":"<a id='c1'></a>\n# <div style=\"text-align:center; border-radius:15px 15px; padding:15px; color:#333333; margin:0; ; padding:15px; font-size:100%; font:'Verdana'; background-color:#F5F5F5;border: 1px; overflow:hidden\"><b>1. Import modules, building classes and read CSV</b></div>","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder, LabelEncoder, OrdinalEncoder\nfrom sklearn.model_selection import cross_val_score, KFold\nfrom sklearn.metrics import make_scorer, accuracy_score, mean_squared_error, mean_squared_log_error\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import FunctionTransformer\nfrom sklearn.compose import ColumnTransformer\n\nimport optuna\nimport lightgbm as lgb\n\nfrom tqdm import tqdm\n\nimport warnings\nimport logging\n\n# Redirect warnings to a log file\nlogging.basicConfig(filename='warnings.log', level=logging.WARNING)\nlogging.captureWarnings(True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T20:13:09.820722Z","iopub.execute_input":"2024-12-04T20:13:09.821059Z","iopub.status.idle":"2024-12-04T20:13:09.827099Z","shell.execute_reply.started":"2024-12-04T20:13:09.821028Z","shell.execute_reply":"2024-12-04T20:13:09.826161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T20:13:09.83901Z","iopub.execute_input":"2024-12-04T20:13:09.839542Z","iopub.status.idle":"2024-12-04T20:13:15.478187Z","shell.execute_reply.started":"2024-12-04T20:13:09.839505Z","shell.execute_reply":"2024-12-04T20:13:15.477505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target = 'Premium Amount'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T20:13:15.479482Z","iopub.execute_input":"2024-12-04T20:13:15.479762Z","iopub.status.idle":"2024-12-04T20:13:15.483706Z","shell.execute_reply.started":"2024-12-04T20:13:15.479736Z","shell.execute_reply":"2024-12-04T20:13:15.482823Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c4'></a>\n# <div style=\"text-align:center; border-radius:15px 15px; padding:15px; color:#333333; margin:0; ; padding:15px; font-size:100%; font:'Verdana'; background-color:#F5F5F5;border: 1px; overflow:hidden\"><b> 2. Model Building </b></div>","metadata":{}},{"cell_type":"markdown","source":"## DataPreprocessor Class","metadata":{}},{"cell_type":"code","source":"class DataPreprocessor:\n    \"\"\"\n    A class for preprocessing data using ColumnTransformer and Pipelines.\n    Automatically handles numerical and categorical features with customizable pipelines.\n\n    Params\n    ----------\n    numerical_columns : list of str\n        List of numerical feature names.\n    one_hot_columns : list of str\n        List of categorical feature names for one-hot encoding.\n    label_columns : list of str\n        List of categorical feature names for label encoding.\n\n    Attributes\n    ----------\n    preprocessor : ColumnTransformer\n        Combines numerical, one-hot, and label pipelines for data transformation.\n\n    Methods\n    -------\n    fit(X)\n        Fits the preprocessor on the provided DataFrame.\n    transform(X)\n        Transforms the DataFrame using the fitted preprocessor.\n    fit_transform(X)\n        Fits the preprocessor and transforms the DataFrame.\n    \"\"\"\n\n    def __init__(self, numerical_columns, one_hot_columns, label_columns):\n        self.numerical_columns = numerical_columns\n        self.one_hot_columns = one_hot_columns\n        self.label_columns = label_columns\n\n        # Define preprocessing pipelines\n        self.numerical_pipeline = Pipeline(steps=[\n            ('imputer', SimpleImputer(strategy='median')),\n            ('scaler', StandardScaler()),\n            ('convert_to_float32', FunctionTransformer(lambda x: x.astype(np.float32)))\n        ])\n\n        self.one_hot_pipeline = Pipeline(steps=[\n            ('imputer', SimpleImputer(strategy='constant', fill_value='missing')),\n            ('one_hot', OneHotEncoder(drop='first', sparse=False, handle_unknown='ignore'))\n        ])\n\n        self.label_pipeline = Pipeline(steps=[\n            ('imputer', SimpleImputer(strategy='constant', fill_value='missing')),\n            ('ordinal', OrdinalEncoder(dtype=np.int32, handle_unknown='use_encoded_value', unknown_value=-1))\n        ])\n\n        # Combine the pipelines into a ColumnTransformer\n        self.preprocessor = ColumnTransformer(\n            transformers=[\n                ('num', self.numerical_pipeline, self.numerical_columns),\n                ('one_hot', self.one_hot_pipeline, self.one_hot_columns),\n                ('label', self.label_pipeline, self.label_columns)\n            ]\n        )\n\n    def fit(self, X):\n        self.preprocessor.fit(X)\n\n    def transform(self, X):\n        return self.preprocessor.transform(X)\n\n    def fit_transform(self, X):\n        return self.preprocessor.fit_transform(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T20:13:15.485083Z","iopub.execute_input":"2024-12-04T20:13:15.485341Z","iopub.status.idle":"2024-12-04T20:13:15.498074Z","shell.execute_reply.started":"2024-12-04T20:13:15.485294Z","shell.execute_reply":"2024-12-04T20:13:15.497191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.sample(frac=1, random_state=42).reset_index(drop=True)\ntest = test.sample(frac=1, random_state=42).reset_index(drop=True)\n\n# Drop the 'id' column from the dataset (if present)\ntrain = train.drop('id', axis=1)\ntest_ids = test['id']\ntest = test.drop('id', axis=1)\n\n# Separate features and target for the training set\nX_train = train.drop(columns=[target])\ny_train = train[target]\n\n# Test set should only contain features\nX_test = test.copy()\n\n# Automatically determine columns for preprocessing\nnumerical_columns = X_train.select_dtypes(include=['float64', 'int64']).columns.tolist()\ncategorical_columns = X_train.select_dtypes(include=['object']).columns.tolist()\n\n# Further classify categorical columns into one-hot and label encoding based on cardinality\nhigh_cardinality_threshold = 10\none_hot_columns = [col for col in categorical_columns if X_train[col].nunique() <= high_cardinality_threshold]\nlabel_columns = [col for col in categorical_columns if X_train[col].nunique() > high_cardinality_threshold]\n\n# Initialize the preprocessor\npreprocessor = DataPreprocessor(numerical_columns, one_hot_columns, label_columns)\n\n# Fit the preprocessor on the training features\npreprocessor.fit(X_train)\n\n# Transform both training and test datasets\nX_encoded_full = preprocessor.transform(X_train)  \nX_test_encoded_full = preprocessor.transform(X_test)\n\nX_encoded = X_encoded_full[:100000]\nX_test_encoded = X_test_encoded_full[:100000]\ny_train_full = y_train\ny_train = y_train[:100000]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T20:13:15.50024Z","iopub.execute_input":"2024-12-04T20:13:15.500598Z","iopub.status.idle":"2024-12-04T20:13:36.228008Z","shell.execute_reply.started":"2024-12-04T20:13:15.500563Z","shell.execute_reply":"2024-12-04T20:13:36.227064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Custom scorer for RMSLE\ndef rmsle(y_true, y_pred):\n    return np.sqrt(mean_squared_log_error(y_true, np.maximum(y_pred, 0)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T20:13:36.22917Z","iopub.execute_input":"2024-12-04T20:13:36.22954Z","iopub.status.idle":"2024-12-04T20:13:36.234554Z","shell.execute_reply.started":"2024-12-04T20:13:36.229504Z","shell.execute_reply":"2024-12-04T20:13:36.23351Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c5'></a>\n# <div style=\"text-align:center; border-radius:15px 15px; padding:15px; color:#333333; margin:0; ; padding:15px; font-size:100%; font:'Verdana'; background-color:#F5F5F5;border: 1px; overflow:hidden\"><b>3. Validation</b>\n</div>","metadata":{}},{"cell_type":"code","source":"# Suppress Optuna's default progress logs\noptuna.logging.set_verbosity(optuna.logging.WARNING)\n\n# Define the objective function\ndef objective(trial):\n    # Parameter space\n    params = {\n        'learning_rate': trial.suggest_loguniform('learning_rate', 0.001, 0.1),\n        'n_estimators': trial.suggest_int('n_estimators', 100, 1000),\n        'max_depth': trial.suggest_int('max_depth', 3, 12),\n        'num_leaves': trial.suggest_int('num_leaves', 20, 40),\n        'min_child_samples': trial.suggest_int('min_child_samples', 10, 100),\n        'subsample': trial.suggest_uniform('subsample', 0.5, 1.0),\n        'colsample_bytree': trial.suggest_uniform('colsample_bytree', 0.5, 1.0),\n        'reg_alpha': trial.suggest_loguniform('reg_alpha', 1e-3, 1.0),\n        'reg_lambda': trial.suggest_loguniform('reg_lambda', 1e-3, 1.0),\n        'device': 'gpu'\n    }\n\n    # Use precomputed folds\n    scores = []\n    for train_index, valid_index in folds:\n        X_train_fold, X_valid_fold = X_encoded[train_index], X_encoded[valid_index]\n        y_train_fold, y_valid_fold = y_train[train_index], y_train[valid_index]\n\n        # Define the LightGBM model\n        model_lgb = lgb.LGBMRegressor(**params)\n        model_lgb.fit(\n            X_train_fold, y_train_fold,\n            eval_set=[(X_valid_fold, y_valid_fold)],\n            eval_metric='rmse',\n            callbacks=[lgb.early_stopping(stopping_rounds=50, verbose=False)]\n        )\n\n        # Predict and calculate RMSE\n        preds_lgb = model_lgb.predict(X_valid_fold)\n        score = mean_squared_error(y_valid_fold, preds_lgb, squared=False)  # RMSE\n        scores.append(score)\n\n    return np.mean(scores)\n\n# Precompute K-Fold splits for reusability\n#kf = KFold(n_splits=5, shuffle=True, random_state=42)\n#folds = list(kf.split(X_encoded))\n\n# Start Optuna optimization with tqdm for progress bar\n#n_trials = 50\n#study = optuna.create_study(direction=\"minimize\")\n#with tqdm(total=n_trials, desc=\"Optuna Trials\") as pbar:\n#    for _ in range(n_trials):\n#        study.optimize(objective, n_trials=1, catch=(Exception,))\n#        pbar.update(1)\n\n# Display best parameters and score\n#print(\"Best Parameters:\", study.best_params)\n#print(\"Best RMSE:\", study.best_value)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T20:13:36.235902Z","iopub.execute_input":"2024-12-04T20:13:36.236571Z","iopub.status.idle":"2024-12-04T20:13:36.246596Z","shell.execute_reply.started":"2024-12-04T20:13:36.236544Z","shell.execute_reply":"2024-12-04T20:13:36.245754Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_params = {'learning_rate': 0.03434906957798577, \n               'n_estimators': 570, \n               'max_depth': 11, \n               'num_leaves': 32, \n               'min_child_samples': 53, \n               'subsample': 0.9400016806622432, \n               'colsample_bytree': 0.998061100209949, \n               'reg_alpha': 0.6049024305670194, \n               'reg_lambda': 0.04845141168149618}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T20:13:36.24764Z","iopub.execute_input":"2024-12-04T20:13:36.247923Z","iopub.status.idle":"2024-12-04T20:13:36.257161Z","shell.execute_reply.started":"2024-12-04T20:13:36.247899Z","shell.execute_reply":"2024-12-04T20:13:36.256501Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c6'></a>\n# <div style=\"text-align:center; border-radius:15px 15px; padding:15px; color:#333333; margin:0; ; padding:15px; font-size:100%; font:'Verdana'; background-color:#F5F5F5;border: 1px; overflow:hidden\"><b>4. Results and submission</b>\n</div>","metadata":{}},{"cell_type":"code","source":"model_lgb = lgb.LGBMRegressor(**best_params)\n\nmodel_lgb.fit(X_encoded_full, y_train_full)\n\npreds_lgb = model_lgb.predict(X_test_encoded_full)\n\nsubmit = pd.DataFrame({\n    'id': test_ids,\n    'Premium Amount': preds_lgb.flatten()\n})\n\nsubmit.to_csv(\"../working/submission.csv\", index=False)\n\nprint(submit)\nprint(submit['Premium Amount'].describe())\n\nprint(\"Best Parameters:\", best_params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T20:14:34.595099Z","iopub.execute_input":"2024-12-04T20:14:34.595472Z","iopub.status.idle":"2024-12-04T20:15:08.842032Z","shell.execute_reply.started":"2024-12-04T20:14:34.595439Z","shell.execute_reply":"2024-12-04T20:15:08.84097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.histplot(submit['Premium Amount'], kde=True, bins=30)\nplt.title(\"Distribution of Predictions\")\nplt.xlabel(\"Prediction\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T20:15:08.843887Z","iopub.execute_input":"2024-12-04T20:15:08.844546Z","iopub.status.idle":"2024-12-04T20:15:12.331193Z","shell.execute_reply.started":"2024-12-04T20:15:08.844502Z","shell.execute_reply":"2024-12-04T20:15:12.330332Z"}},"outputs":[],"execution_count":null}]}