{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.simplefilter(\"ignore\")\n\nimport lightgbm as lgb\n\nimport optuna\nfrom sklearn.model_selection import RepeatedKFold, cross_val_score\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import SimpleImputer\n\ntarget, seed = 'Premium Amount', 42","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:07:07.389419Z","iopub.execute_input":"2024-12-28T16:07:07.389702Z","iopub.status.idle":"2024-12-28T16:07:11.548738Z","shell.execute_reply.started":"2024-12-28T16:07:07.38968Z","shell.execute_reply":"2024-12-28T16:07:11.547671Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loading dataset","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\", index_col=\"id\")\ntest_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\", index_col=\"id\")\nsample_submission = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:07:11.549738Z","iopub.execute_input":"2024-12-28T16:07:11.550421Z","iopub.status.idle":"2024-12-28T16:07:20.618297Z","shell.execute_reply.started":"2024-12-28T16:07:11.550355Z","shell.execute_reply":"2024-12-28T16:07:20.617512Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Split data (Stratified by Annual Income)","metadata":{}},{"cell_type":"code","source":"median = train_df[\"Annual Income\"].median()\ntrain_df[\"Annual Income\"].fillna(median, inplace=True)\ntrain_df[\"Annual Income\"].hist(bins=50, figsize=(5, 3))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:07:20.619044Z","iopub.execute_input":"2024-12-28T16:07:20.61929Z","iopub.status.idle":"2024-12-28T16:07:20.977636Z","shell.execute_reply.started":"2024-12-28T16:07:20.619268Z","shell.execute_reply":"2024-12-28T16:07:20.976795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[\"Annual Income (log1p)\"] = np.log1p(train_df[\"Annual Income\"])\nnp.log1p(train_df[\"Annual Income\"]).hist(bins=50, figsize=(5, 3))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:07:20.979824Z","iopub.execute_input":"2024-12-28T16:07:20.980057Z","iopub.status.idle":"2024-12-28T16:07:21.386206Z","shell.execute_reply.started":"2024-12-28T16:07:20.980038Z","shell.execute_reply":"2024-12-28T16:07:21.385485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['income_cat'] = pd.cut(train_df[\"Annual Income (log1p)\"],\n                                 bins=[0., 6., 7.5, 9., 10.5, np.inf],\n                                 labels=[1, 2, 3, 4, 5])\ntrain_df['income_cat'].hist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:07:21.387541Z","iopub.execute_input":"2024-12-28T16:07:21.387789Z","iopub.status.idle":"2024-12-28T16:07:21.633551Z","shell.execute_reply.started":"2024-12-28T16:07:21.387767Z","shell.execute_reply":"2024-12-28T16:07:21.632676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedShuffleSplit\nspliter = StratifiedShuffleSplit(n_splits=1, test_size=0.15, random_state=42)\n\nfor train_index, test_index in spliter.split(train_df, train_df['income_cat']):\n    train_stratified_train_df, test_stratified_train_df = train_df.loc[train_index], train_df.loc[test_index]\n\nprint(f\"stratified train shape: {train_stratified_train_df.shape}\\nstratified test shape: {test_stratified_train_df.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:07:21.634347Z","iopub.execute_input":"2024-12-28T16:07:21.634658Z","iopub.status.idle":"2024-12-28T16:07:22.972223Z","shell.execute_reply.started":"2024-12-28T16:07:21.634636Z","shell.execute_reply":"2024-12-28T16:07:22.971451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for _set in (train_stratified_train_df, test_stratified_train_df):\n    _set.drop('income_cat', axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:07:22.973041Z","iopub.execute_input":"2024-12-28T16:07:22.973322Z","iopub.status.idle":"2024-12-28T16:07:23.199292Z","shell.execute_reply.started":"2024-12-28T16:07:22.9733Z","shell.execute_reply":"2024-12-28T16:07:23.198622Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Pipeline Preprocess","metadata":{}},{"cell_type":"code","source":"def reduce_memory_usage(dataframe):\n    df = dataframe.copy()\n    start_mem = df.memory_usage().sum() / 1024**2\n    print(f\"Memory ini: {start_mem:.2f} MB\")\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            \n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n    \n    end_mem = df.memory_usage().sum() / 1024**2\n    print(f\"Memory final: {end_mem:.2f} MB\")\n    print(f\"Reduce of {100 * (start_mem - end_mem) / start_mem:.1f}%\")\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:07:23.200116Z","iopub.execute_input":"2024-12-28T16:07:23.200471Z","iopub.status.idle":"2024-12-28T16:07:23.208595Z","shell.execute_reply.started":"2024-12-28T16:07:23.200439Z","shell.execute_reply":"2024-12-28T16:07:23.207633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ordinals = {\n    'Customer Feedback': {'Poor': 1, 'Average': 2, 'Good': 3},\n    'Education Level': {'High School': 1, 'Bachelor\\'s': 2, 'Master\\'s': 3, 'PhD': 4},\n    'Exercise Frequency': {'Daily': 1, 'Weekly': 2, 'Monthly': 3, 'Rarely': 4},\n    'Policy Type': {'Basic': 1, 'Premium': 2, 'Comprehensive': 3}\n}\n\n# Função para mapear ordinais\ndef map_ordinals(df):\n    df_ = df.copy()\n    for col, mapping in ordinals.items():\n        df_[col] = df_[col].map(mapping)\n    return df_\n\n# Função para extrair características de data\ndef extract_date_features(df):\n    df_ = df.copy()\n    df_['Policy Start Date'] = pd.to_datetime(df_['Policy Start Date'])\n    df_['Year'] = df_['Policy Start Date'].dt.year\n    df_['Month'] = df_['Policy Start Date'].dt.month\n    df_['Day'] = df_['Policy Start Date'].dt.day\n    df_['DayOfWeek'] = df_['Policy Start Date'].dt.dayofweek\n    return df_.drop(columns=['Policy Start Date'])\n\ndef Preprocess_insurance_data(train, test, target, final=True):\n    X_train = train.drop(target, axis=1)\n    y_train = train[target]\n    if final:\n        X_test = test.copy()\n    else:\n        X_test = test.copy().drop(target, axis=1)\n        y_test = test[target]\n    \n    X_train['Previous Claims'].fillna(0, inplace=True)\n    X_test['Previous Claims'].fillna(0, inplace=True)\n\n    if 'Policy Start Date' in X_train.columns:\n        X_train = extract_date_features(X_train)\n        X_test = extract_date_features(X_test)\n    \n    imputer = SimpleImputer(strategy=\"median\")\n\n    date_cols = ['Year', 'Month', 'Day', 'DayOfWeek']\n    X_train_dates = X_train[date_cols]\n    X_test_dates = X_test[date_cols]\n    \n    X_train_num = X_train.select_dtypes(include=['int64', 'float64'])\n    X_test_num = X_test.select_dtypes(include=['int64', 'float64'])\n    \n    X_train_imputed = imputer.fit_transform(X_train_num)\n    X_test_imputed = imputer.transform(X_test_num)\n    \n    X_train_nums_imputed = pd.DataFrame(\n        X_train_imputed,\n        columns=X_train_num.columns,\n        index=X_train_num.index)\n    X_test_nums_imputed = pd.DataFrame(\n        X_test_imputed,\n        columns=X_test_num.columns,\n        index=X_test_num.index)\n    \n    X_train_cat = X_train.select_dtypes(include='object')\n    X_test_cat = X_test.select_dtypes(include='object')\n    \n    X_train_cat_encoded = X_train_cat.copy()\n    X_test_cat_encoded = X_test_cat.copy()\n    \n    X_train_cat_encoded = map_ordinals(X_train_cat_encoded)\n    X_test_cat_encoded = map_ordinals(X_test_cat_encoded)\n\n    X_train_cat_encoded[\"Customer Feedback\"] = X_train_cat_encoded[\"Customer Feedback\"].fillna(0)\n    X_test_cat_encoded[\"Customer Feedback\"] = X_test_cat_encoded[\"Customer Feedback\"].fillna(0)\n    \n    cols_one_hot = [\"Gender\", \"Marital Status\", \"Occupation\", \"Location\", \"Property Type\", \"Smoking Status\"]\n    \n    train_df_one_hot = pd.get_dummies(X_train_cat_encoded[cols_one_hot], columns=cols_one_hot)\n    test_df_one_hot = pd.get_dummies(X_test_cat_encoded[cols_one_hot], columns=cols_one_hot)\n    \n    train_df_one_hot, test_df_one_hot = train_df_one_hot.align(test_df_one_hot, join='inner', axis=1)\n    \n    new_X_train = pd.concat([X_train_nums_imputed,\n                             X_train_dates,\n                             X_train_cat_encoded.drop(cols_one_hot, axis=1), \n                             train_df_one_hot], axis=1)\n    new_X_test = pd.concat([X_test_nums_imputed,\n                            X_test_dates,\n                            X_test_cat_encoded.drop(cols_one_hot, axis=1), \n                            test_df_one_hot], axis=1)\n    \n    final_train = new_X_train.copy()\n    final_train[target] = np.log1p(y_train)\n\n    if final:\n        final_test = new_X_test.copy()\n    else:\n        final_test = new_X_test.copy()\n        final_test[target] = np.log1p(y_test)\n\n    return final_train, final_test\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:07:23.209414Z","iopub.execute_input":"2024-12-28T16:07:23.209727Z","iopub.status.idle":"2024-12-28T16:07:23.231077Z","shell.execute_reply.started":"2024-12-28T16:07:23.209704Z","shell.execute_reply":"2024-12-28T16:07:23.230228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_train_df_processed, test_train_df_processed = Preprocess_insurance_data(\n    train_stratified_train_df,\n    test_stratified_train_df,\n    target,\n    final=False # No test final\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:07:23.231892Z","iopub.execute_input":"2024-12-28T16:07:23.232098Z","iopub.status.idle":"2024-12-28T16:07:27.712133Z","shell.execute_reply.started":"2024-12-28T16:07:23.23208Z","shell.execute_reply":"2024-12-28T16:07:27.711475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_train_df_processed = reduce_memory_usage(train_train_df_processed)\ntest_train_df_processed = reduce_memory_usage(test_train_df_processed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:07:27.712881Z","iopub.execute_input":"2024-12-28T16:07:27.713085Z","iopub.status.idle":"2024-12-28T16:07:28.186923Z","shell.execute_reply.started":"2024-12-28T16:07:27.713068Z","shell.execute_reply":"2024-12-28T16:07:28.1861Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training and tune model","metadata":{}},{"cell_type":"code","source":"X_train, y_train_log1p = train_train_df_processed.drop(target, axis=1), train_train_df_processed[target]\nX_test, y_test_log1p = test_train_df_processed.drop(target, axis=1), test_train_df_processed[target]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:07:28.187845Z","iopub.execute_input":"2024-12-28T16:07:28.188073Z","iopub.status.idle":"2024-12-28T16:07:28.304848Z","shell.execute_reply.started":"2024-12-28T16:07:28.188053Z","shell.execute_reply":"2024-12-28T16:07:28.304176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def root_mean_squared_log_error(y_true, y_pred):\n    y_true = np.maximum(0, y_true)\n    y_pred = np.maximum(0, y_pred)\n    return np.sqrt(np.mean(np.square(np.log1p(y_pred) - np.log1p(y_true))))\n\ndef model_training(model, X_train, y_train):\n    rkf = RepeatedKFold(n_splits=5, n_repeats=2, random_state=seed)\n\n    rmsle_scores = []\n    \n    for train_index, valid_index in rkf.split(X_train):\n        X_train_fold, X_valid_fold = X_train.iloc[train_index], X_train.iloc[valid_index]\n        y_train_log1p_fold, y_valid_log1p_fold = y_train.iloc[train_index], y_train.iloc[valid_index]\n        \n        model.fit(X_train_fold, y_train_log1p_fold)\n        \n        y_pred_log1p = model.predict(X_valid_fold)\n\n        # Invert\n        y_valid_fold = np.expm1(y_valid_log1p_fold)\n        y_pred = np.expm1(y_pred_log1p)\n        \n        rmsle = root_mean_squared_log_error(y_valid_fold, y_pred)\n        rmsle_scores.append(rmsle)\n    return rmsle_scores\n\ndef objective(trial):\n    params = {\n        'objective': 'regression',\n        'metric': 'rmse',\n        'device': 'gpu',\n        'boosting_type': trial.suggest_categorical('boosting_type', ['gbdt', 'dart', 'goss']),\n        'num_leaves': trial.suggest_int('num_leaves', 20, 100),\n        'max_depth': trial.suggest_int('max_depth', 3, 12),\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3),\n        'n_estimators': trial.suggest_int('n_estimators', 50, 200),\n        'subsample': trial.suggest_float('subsample', 0.6, 1.0),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.6, 1.0),\n        'min_child_samples': trial.suggest_int('min_child_samples', 10, 100),\n        'reg_alpha': trial.suggest_float('reg_alpha', 0.0, 1.0),\n        'reg_lambda': trial.suggest_float('reg_lambda', 0.0, 1.0),\n    }\n\n    rmsle_scores = model_training(\n        model = lgb.LGBMRegressor(**params, verbose=0, random_state=seed),\n        X_train = X_train,\n        y_train = y_train_log1p, # with log1p\n    )\n\n    mean_rmsle = np.mean(rmsle_scores)\n    return mean_rmsle","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:07:28.306931Z","iopub.execute_input":"2024-12-28T16:07:28.307146Z","iopub.status.idle":"2024-12-28T16:07:28.314668Z","shell.execute_reply.started":"2024-12-28T16:07:28.307127Z","shell.execute_reply":"2024-12-28T16:07:28.313791Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Importance","metadata":{}},{"cell_type":"code","source":"best_params = {\n'boosting_type': 'gbdt', \n'num_leaves': 55, \n'max_depth': 12, \n'learning_rate': 0.07072056371353222, \n'n_estimators': 167, \n'subsample': 0.8381378812322983, \n'colsample_bytree': 0.9909744221417157, \n'min_child_samples': 85, \n'reg_alpha': 0.6262119810221626, \n'reg_lambda': 0.10141907431792432,\n'device': 'gpu',\n'random_state': seed,\n}\n\nbest_model = lgb.LGBMRegressor(**best_params)\nbest_model.fit(X_train, y_train_log1p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:10:13.672059Z","iopub.execute_input":"2024-12-28T16:10:13.672477Z","iopub.status.idle":"2024-12-28T16:10:18.49577Z","shell.execute_reply.started":"2024-12-28T16:10:13.672441Z","shell.execute_reply":"2024-12-28T16:10:18.494812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_features_importance = pd.DataFrame(best_model.feature_importances_, X_train.columns.tolist()).sort_values(by=0, ascending=True)\ndf_features_importance","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:11:16.603314Z","iopub.execute_input":"2024-12-28T16:11:16.603778Z","iopub.status.idle":"2024-12-28T16:11:16.613081Z","shell.execute_reply.started":"2024-12-28T16:11:16.603741Z","shell.execute_reply":"2024-12-28T16:11:16.612295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# Score without drop\nscores_ = model_training(\n    model = best_model,\n    X_train = X_train,\n    y_train = y_train_log1p\n)\nprint(f\"\\n\\nMean Scores: {np.mean(scores_)}\")\nprint(f\"Std Scores: {np.std(scores_)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:10:22.8353Z","iopub.execute_input":"2024-12-28T16:10:22.835816Z","iopub.status.idle":"2024-12-28T16:11:16.601976Z","shell.execute_reply.started":"2024-12-28T16:10:22.835773Z","shell.execute_reply":"2024-12-28T16:11:16.601021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nfeatures_drop = df_features_importance.iloc[1:2,:].index.tolist()\n\n# Score with drop\nscores_ = model_training(\n    model = best_model,\n    X_train = X_train.drop(features_drop, # DROP!\n                           axis=1),\n    y_train = y_train_log1p\n)\nprint(f\"\\n\\nMean Scores: {np.mean(scores_)}\")\nprint(f\"Std Scores: {np.std(scores_)}\")\n\n\"\"\"\ndf_features_importance.iloc[:1,:].index.tolist()\nMean Scores: 1.046 38 7702921352\nStd Scores: 0.00 17 84 7399616884236\nCPU times: user 1min 38s, sys: 2.5 s, total: 1min 41s\nWall time: 52.6 s\n\ndf_features_importance.iloc[1:2,:].index.tolist()\nMean Scores: 1.046 38 77594331965\nStd Scores: 0.0017 84 7465548372985\nCPU times: user 1min 38s, sys: 2.48 s, total: 1min 40s\nWall time: 52.4 s\n\nBEST WITHOUT DROP\nMean Scores: 1.0463876145861106\nStd Scores: 0.0017 84 7212933946833\nCPU times: user 1min 40s, sys: 2.43 s, total: 1min 43s\nWall time: 53.8 s\n\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:15:08.834283Z","iopub.execute_input":"2024-12-28T16:15:08.834666Z","iopub.status.idle":"2024-12-28T16:16:01.225739Z","shell.execute_reply.started":"2024-12-28T16:15:08.834625Z","shell.execute_reply":"2024-12-28T16:16:01.224819Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Predict test df","metadata":{}},{"cell_type":"code","source":"train_final = train_df.copy()\ntest_final = test_df.copy()\n\ntrain_final[\"Annual Income (log1p)\"] = np.log1p(train_final[\"Annual Income\"])\ntest_final[\"Annual Income (log1p)\"] = np.log1p(test_final[\"Annual Income\"])\n\ntrain_final_processed, test_final_processed = Preprocess_insurance_data(train_df, test_final, target)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:38:27.630445Z","iopub.execute_input":"2024-12-28T16:38:27.630771Z","iopub.status.idle":"2024-12-28T16:38:34.281868Z","shell.execute_reply.started":"2024-12-28T16:38:27.630746Z","shell.execute_reply":"2024-12-28T16:38:34.281172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train_final_processed.drop(target, axis=1)\ny = train_final_processed[target]\n\nmodel_final = best_model\nmodel_final.fit(X, y)\n\nfinal_preds = np.expm1(model_final.predict(test_final_processed))\nprint(final_preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:39:51.233138Z","iopub.execute_input":"2024-12-28T16:39:51.233448Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import joblib\n\njoblib.dump(model_final, \"final_lgbm_tuned.pkl\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:38:40.039772Z","iopub.status.idle":"2024-12-28T16:38:40.040108Z","shell.execute_reply":"2024-12-28T16:38:40.039935Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Output submission","metadata":{}},{"cell_type":"code","source":"sample_submission[target] = final_preds\nsample_submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T15:59:36.109725Z","iopub.status.idle":"2024-12-28T15:59:36.110061Z","shell.execute_reply":"2024-12-28T15:59:36.109884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission.to_csv(\"ss_lgbm_optuned_dropFI.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T15:59:36.110822Z","iopub.status.idle":"2024-12-28T15:59:36.111084Z","shell.execute_reply":"2024-12-28T15:59:36.110982Z"}},"outputs":[],"execution_count":null}]}