{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"text-align: center;\">\n<h1> Easy to get starts, Thanks for all voting 🙏❤️ <h1/>\n<h2> Hope it will be helpful for everyone🪴 <h2/>\n</div>","metadata":{}},{"cell_type":"markdown","source":"## CAUTIONS \n- Submission notebook allows only submission.csv with corrected format\n- No duplicate on prediction_id when submit\n- No internet connection \n- Keep calm and get gold 👊🥇","metadata":{}},{"cell_type":"markdown","source":"## LOGING VERSION\n* Version 6: Train, Test spliting 1 folds + Optuna Tuning\n    + local F1 macro 0.49xxx LB: 0.01\n* Version 8: Adding 5 folds\n    + local F1 macro 0.49630 LB: 0.01\n* Version 11: Adding Weight Class\n    + local F1 macro 0.42079 LB: 0.04 (Woww Huge step on Handling Imbalanced Dataset) +0.03x\n* Version 12: Adding laterality features (Forget since starts !)\n    + local F1 macro 0.41401 LB: 0.04 (seem a bit lower than version 10.) \n* Version 13: Handling with ADASYN instead of class weight\n    + local F1 macro 0.50148 LB: 0.03 (ADASYN leads to overfit !) \n* Version 15: Weight Class + Average 2 side of memogram\n    + local F1 macro 0.42079 LB: 0.04 (Improve from version 10)\n* Version 16: Adding Ensemble with Classical Model\n    + local F1 macro 0.42079 | 0.3903 LB: Error\n* Version 21: Re Tune with Probabilistic F1 score, beta = 1, LGBM model\n    + local F1 prob  0.06291 LB: xx\n* Version 20: Prob F1 score + Ensemble\n    + local F1 prob  0.06726 |  LB: xx\n* Version 20: Re Tune with Probabilistic F1 score, beta = 1, LGBM model\n    + local F1 prob  xxxx LB: xx","metadata":{}},{"cell_type":"markdown","source":"# 1. Import Library","metadata":{}},{"cell_type":"code","source":"import os, gc\nimport numpy as np\nimport pandas as pd\nimport pickle\nimport sys\n\nimport lightgbm as lgb\nimport optuna\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction import DictVectorizer\nfrom sklearn.model_selection import StratifiedKFold, GroupKFold, KFold\n\nfrom imblearn.over_sampling import ADASYN\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.svm import SVC\n\n\nfrom sklearn.ensemble import VotingClassifier\nimport warnings\nwarnings.simplefilter('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-03T01:09:24.93774Z","iopub.execute_input":"2022-12-03T01:09:24.938719Z","iopub.status.idle":"2022-12-03T01:09:28.663023Z","shell.execute_reply.started":"2022-12-03T01:09:24.938563Z","shell.execute_reply":"2022-12-03T01:09:28.661559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.1 Parameter Setup","metadata":{}},{"cell_type":"code","source":"is_tune_params = False\nCATEGORICAL_COL = [\"view\", \"implant\",\"machine_id\"]\nNUMERICAL_COL = [\"age\"]\nTARGET_COLS = [\"cancer\"]\nRANDOM_SEED = 42\n\ndef set_seed(seed=2022):\n    np.random.seed(seed)\n    #tf.random.set_seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    #os.environ['TF_DETERMINISTIC_OPS'] = '1'\n    \nset_seed(RANDOM_SEED)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:09:28.666522Z","iopub.execute_input":"2022-12-03T01:09:28.667098Z","iopub.status.idle":"2022-12-03T01:09:28.677309Z","shell.execute_reply.started":"2022-12-03T01:09:28.66703Z","shell.execute_reply":"2022-12-03T01:09:28.675363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/rsna-breast-cancer-detection/train.csv\")\ndf_test = pd.read_csv(\"../input/rsna-breast-cancer-detection/test.csv\")\ndf_sub = pd.read_csv(\"../input/rsna-breast-cancer-detection/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:09:28.680213Z","iopub.execute_input":"2022-12-03T01:09:28.680948Z","iopub.status.idle":"2022-12-03T01:09:28.882949Z","shell.execute_reply.started":"2022-12-03T01:09:28.680868Z","shell.execute_reply":"2022-12-03T01:09:28.881241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_FOLD = 5\nskf = StratifiedKFold(n_splits=N_FOLD, shuffle=True, random_state=RANDOM_SEED)\nfor n, (train_index, val_index) in enumerate(skf.split(df_train, df_train[TARGET_COLS])):\n    df_train.loc[val_index, 'fold'] = int(n)\ndf_train['fold'] = df_train['fold'].astype(int)\ndf_train[\"age\"] = df_train[\"age\"].fillna(df_train[\"age\"].median())","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:09:28.886384Z","iopub.execute_input":"2022-12-03T01:09:28.886931Z","iopub.status.idle":"2022-12-03T01:09:28.951536Z","shell.execute_reply.started":"2022-12-03T01:09:28.886878Z","shell.execute_reply":"2022-12-03T01:09:28.95044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:09:28.952868Z","iopub.execute_input":"2022-12-03T01:09:28.953769Z","iopub.status.idle":"2022-12-03T01:09:28.981833Z","shell.execute_reply.started":"2022-12-03T01:09:28.953718Z","shell.execute_reply":"2022-12-03T01:09:28.980241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.2 Check target Distribution ","metadata":{}},{"cell_type":"code","source":"df_train[TARGET_COLS].value_counts() * 100 / len(df_train)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:09:28.983758Z","iopub.execute_input":"2022-12-03T01:09:28.984573Z","iopub.status.idle":"2022-12-03T01:09:29.014416Z","shell.execute_reply.started":"2022-12-03T01:09:28.984508Z","shell.execute_reply":"2022-12-03T01:09:29.012749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.3 Check class weight each class","metadata":{}},{"cell_type":"code","source":"len(df_train) / (2 * np.bincount(df_train['cancer'].values))","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:09:29.016406Z","iopub.execute_input":"2022-12-03T01:09:29.017203Z","iopub.status.idle":"2022-12-03T01:09:29.027495Z","shell.execute_reply.started":"2022-12-03T01:09:29.017118Z","shell.execute_reply":"2022-12-03T01:09:29.026469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Prepare Dataset","metadata":{}},{"cell_type":"code","source":"# Skip this part","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:09:29.029244Z","iopub.execute_input":"2022-12-03T01:09:29.029792Z","iopub.status.idle":"2022-12-03T01:09:29.04813Z","shell.execute_reply.started":"2022-12-03T01:09:29.029739Z","shell.execute_reply":"2022-12-03T01:09:29.046581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Tuning Model\n- If you want to try tuning use this part and try varying range of parameters\n- Thanks PFBETA by https://www.kaggle.com/code/sohier/probabilistic-f-score","metadata":{}},{"cell_type":"code","source":"def pfbeta(labels, predictions, beta):\n    y_true_count = 0\n    ctp = 0\n    cfp = 0\n\n    for idx in range(len(labels)):\n        prediction = min(max(predictions[idx], 0), 1)\n        if (labels[idx]):\n            y_true_count += 1\n            ctp += prediction\n        else:\n            cfp += prediction\n\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp)\n    c_recall = ctp / y_true_count\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n        return result\n    else:\n        return 0","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:09:29.050264Z","iopub.execute_input":"2022-12-03T01:09:29.051205Z","iopub.status.idle":"2022-12-03T01:09:29.062378Z","shell.execute_reply.started":"2022-12-03T01:09:29.051144Z","shell.execute_reply":"2022-12-03T01:09:29.061367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.1 Train Test Tune 1 Fold","metadata":{}},{"cell_type":"code","source":"# fold = 1\n\n# train_df = df_train[df_train['fold'] != fold].reset_index(drop=True)\n# valid_df = df_train[df_train['fold'] == fold].reset_index(drop=True)\n\n# dv = DictVectorizer(sparse=False)\n\n# train_dict = train_df[CATEGORICAL_COL + NUMERICAL_COL].to_dict(\"records\")\n# X_train = dv.fit_transform(train_dict)\n# y_train = train_df[TARGET_COLS].values\n\n# val_dict = valid_df[CATEGORICAL_COL + NUMERICAL_COL].to_dict(\"records\")\n# X_val = dv.transform(val_dict)\n# y_val = valid_df[TARGET_COLS].values","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:09:29.068314Z","iopub.execute_input":"2022-12-03T01:09:29.068827Z","iopub.status.idle":"2022-12-03T01:09:29.082119Z","shell.execute_reply.started":"2022-12-03T01:09:29.068785Z","shell.execute_reply":"2022-12-03T01:09:29.080943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def objective(trial):\n#     params = {\n#         'metric': 'f1',\n#         'random_state': RANDOM_SEED,\n#         'n_estimators': 300,\n#         'learning_rate': 0.1,\n#         'reg_alpha': trial.suggest_loguniform('reg_alpha', 1e-3, 10.0),\n#         'reg_lambda': trial.suggest_loguniform('reg_lambda', 1e-3, 10.0),\n#         'colsample_bytree': trial.suggest_categorical('colsample_bytree', [0.3, 0.4, 0.5, 0.6, 0.7, 0.8, 0.9, 1.0]),\n#         'subsample': trial.suggest_categorical('subsample', [0.4, 0.5, 0.6, 0.7, 0.8, 1.0]),\n#         'max_depth': trial.suggest_categorical('max_depth', [10, 20, 100]),\n#         'num_leaves' : trial.suggest_int('num_leaves', 1, 1000),\n#         'min_child_samples': trial.suggest_int('min_child_samples', 1, 300),\n#         'cat_smooth' : trial.suggest_int('min_data_per_groups', 1, 100)\n#     }\n#     model = lgb.LGBMClassifier(**params, zero_as_missing=True)\n\n#     model.fit(X_train, y_train)\n\n#     y_va_pred = model.predict(X_val)\n#     f1 = pfbeta(y_val, y_va_pred, 1)\n    \n#     return f1","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:09:29.084042Z","iopub.execute_input":"2022-12-03T01:09:29.084871Z","iopub.status.idle":"2022-12-03T01:09:29.104208Z","shell.execute_reply.started":"2022-12-03T01:09:29.084812Z","shell.execute_reply":"2022-12-03T01:09:29.103097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# is_tune_params = True\n# if is_tune_params:\n#     study = optuna.create_study(\n#         direction='maximize', \n#         pruner=optuna.pruners.MedianPruner(n_warmup_steps=20),\n#         study_name='RSNA')\n#     study.optimize(objective, n_trials=20)\n#     print(study.best_params)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:09:29.106418Z","iopub.execute_input":"2022-12-03T01:09:29.107419Z","iopub.status.idle":"2022-12-03T01:09:29.122563Z","shell.execute_reply.started":"2022-12-03T01:09:29.107338Z","shell.execute_reply":"2022-12-03T01:09:29.121089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Select and tune best one\n- Part for getting running 1 fold, you can skip ","metadata":{}},{"cell_type":"code","source":"# if not is_tune_params:\n#     params_tuned = {\n#         'reg_alpha': 0.0028731193020013765,\n#         'reg_lambda': 0.04370710510459441,\n#         'colsample_bytree': 0.6,\n#         'subsample': 0.7,\n#         'max_depth': 20,\n#         'num_leaves': 594,\n#         'min_child_samples': 12,\n#         'min_data_per_groups': 65\n#     }\n#     tuned_model = lgb.LGBMClassifier(**params_tuned, zero_as_missing=True)\n#     tuned_model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:09:29.12521Z","iopub.execute_input":"2022-12-03T01:09:29.126631Z","iopub.status.idle":"2022-12-03T01:09:29.136041Z","shell.execute_reply.started":"2022-12-03T01:09:29.126566Z","shell.execute_reply":"2022-12-03T01:09:29.134991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5 Folds Training and Inference (LGBM)\n- Prepare Train, Valid, and Test\n- Adding ADASYN processing << SKIP >> ","metadata":{}},{"cell_type":"code","source":"## Old tunned params\n# params_tuned = {\n#     'reg_alpha': 0.0028731193020013765,\n#     'reg_lambda': 0.04370710510459441,\n#     'colsample_bytree': 0.6,\n#     'subsample': 0.7,\n#     'max_depth': 20,\n#     'num_leaves': 594,\n#     'min_child_samples': 12,\n#     'min_data_per_groups': 65\n# }","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:09:29.137602Z","iopub.execute_input":"2022-12-03T01:09:29.139783Z","iopub.status.idle":"2022-12-03T01:09:29.150723Z","shell.execute_reply.started":"2022-12-03T01:09:29.139722Z","shell.execute_reply":"2022-12-03T01:09:29.148569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params_tuned = {\n    'reg_alpha': 0.0023893357750573454, \n    'reg_lambda': 0.03380052356488591, \n    'colsample_bytree': 1.0, \n    'subsample': 0.5, \n    'max_depth': 10, \n    'num_leaves': 936, \n    'min_child_samples': 142, \n    'min_data_per_groups': 37\n}\n\nvalid_f1_scores = []\nlgbm_preds = []\n\nfor fold in range(N_FOLD):\n    print(f'\\n-----------FOLD {fold} ------------')\n    print('Data prepared.')\n    train_df = df_train[df_train['fold'] != fold].reset_index(drop=True)\n    valid_df = df_train[df_train['fold'] == fold].reset_index(drop=True)\n\n    dv = DictVectorizer(sparse=False)\n\n    train_dict = train_df[CATEGORICAL_COL + NUMERICAL_COL].to_dict(\"records\")\n    X_train = dv.fit_transform(train_dict)\n    y_train = train_df[TARGET_COLS].values\n\n    val_dict = valid_df[CATEGORICAL_COL + NUMERICAL_COL].to_dict(\"records\")\n    X_val = dv.transform(val_dict)\n    y_val = valid_df[TARGET_COLS].values\n\n    test_dict = df_test[CATEGORICAL_COL + NUMERICAL_COL].to_dict(\"records\")\n    X_test = dv.transform(test_dict)\n\n#     save_dict = f\"./dict_fold{fold}.pkl\"\n#     pickle.dump(dv, open(save_dict, 'wb'))\n\n    class_weight_arr = len(train_df) / (2 * np.bincount(train_df['cancer'].values))\n\n    print('Model Training.')\n    tuned_model = lgb.LGBMClassifier(**params_tuned, \n                                    zero_as_missing=True, \n                                    class_weight={0: class_weight_arr[0], \n                                                  1: class_weight_arr[1]})\n    tuned_model.fit(X_train, y_train)\n\n    print('Model Inferencing.')\n    y_val_pred = tuned_model.predict(X_val)\n    y_test_pred = tuned_model.predict_proba(X_test)[:, 1]\n    \n    valid_f1_score = pfbeta(y_val, y_val_pred, 1)\n    #valid_f1_score = f1_score(y_val, y_val_pred, pos_label=1, average='macro')\n    print('Finished.')\n\n    valid_f1_scores.append(valid_f1_score)\n    lgbm_preds.append(y_test_pred)\n\n    del tuned_model\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:09:29.155133Z","iopub.execute_input":"2022-12-03T01:09:29.156073Z","iopub.status.idle":"2022-12-03T01:09:36.053453Z","shell.execute_reply.started":"2022-12-03T01:09:29.156022Z","shell.execute_reply":"2022-12-03T01:09:36.052322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5.1 Check Local CV","metadata":{}},{"cell_type":"code","source":"print(f'{len(valid_f1_scores)} Folds validation F1:\\n{valid_f1_scores}')\nprint(f'Local CV Average F1 score: {np.mean(valid_f1_scores)}')","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:09:36.055392Z","iopub.execute_input":"2022-12-03T01:09:36.055785Z","iopub.status.idle":"2022-12-03T01:09:36.061996Z","shell.execute_reply.started":"2022-12-03T01:09:36.055746Z","shell.execute_reply":"2022-12-03T01:09:36.0608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5.2. Average Prediction","metadata":{}},{"cell_type":"code","source":"avg_lgbm_preds = np.mean(lgbm_preds, axis=0)\navg_lgbm_preds","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:09:36.063938Z","iopub.execute_input":"2022-12-03T01:09:36.064326Z","iopub.status.idle":"2022-12-03T01:09:36.082281Z","shell.execute_reply.started":"2022-12-03T01:09:36.064289Z","shell.execute_reply":"2022-12-03T01:09:36.080721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. Ensemble with Clasical Model\n- KNN learns nothing","metadata":{}},{"cell_type":"code","source":"f1_score_ensembles = []\nensemble_all_preds = []\n\nfor fold in range(N_FOLD):\n    ensemble_preds_buffers = []\n\n    print(f'\\n-----------FOLD {fold} ------------')\n    dv = DictVectorizer(sparse=False)\n\n    train_dict = train_df[CATEGORICAL_COL + NUMERICAL_COL].to_dict(\"records\")\n    X_train = dv.fit_transform(train_dict)\n    y_train = train_df[TARGET_COLS].values\n\n    val_dict = valid_df[CATEGORICAL_COL + NUMERICAL_COL].to_dict(\"records\")\n    X_val = dv.transform(val_dict)\n    y_val = valid_df[TARGET_COLS].values\n\n    test_dict = df_test[CATEGORICAL_COL + NUMERICAL_COL].to_dict(\"records\")\n    X_test = dv.transform(test_dict)\n    \n    print(f'Model Training fold {fold}')\n    class_weight_arr = len(train_df) / (2 * np.bincount(train_df['cancer'].values))\n    \n    clf1 = LogisticRegression(\n        random_state=RANDOM_SEED, \n        max_iter=200, \n        class_weight={0: class_weight_arr[0], 1: class_weight_arr[1]})\n    clf2 = RandomForestClassifier(n_estimators=50, random_state=RANDOM_SEED, class_weight={0: class_weight_arr[0], 1: class_weight_arr[1]})\n    clf3 = GaussianNB()\n\n\n    eclf = VotingClassifier(\n        estimators=[('lr', clf1), ('rf', clf2), ('gnb', clf3)],\n        voting='soft', weights=[1, 1.2, 1.5])\n    \n\n    eclf.fit(X_train, y_train)\n\n    print(f'Model Inferencing fold {fold}.')\n    y_val_pred = eclf.predict(X_val)\n    y_test_pred = eclf.predict_proba(X_test)[:, 1]\n\n    valid_f1_score = pfbeta(y_val, y_val_pred, 1)\n    f1_score_ensembles.append(valid_f1_score)\n    ensemble_preds_buffers.append(y_test_pred)\n    print('Finished.')\n\n    print(f\"F1 prob fold {fold} = {valid_f1_score:.4f}\")\n    \n    ensemble_all_preds.append(ensemble_preds_buffers)\n    \nprint(f\"Ensemble model - Average F1 prob: {np.mean(f1_score_ensembles):.4f}\")","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:10:37.345967Z","iopub.execute_input":"2022-12-03T01:10:37.346468Z","iopub.status.idle":"2022-12-03T01:10:50.399795Z","shell.execute_reply.started":"2022-12-03T01:10:37.34643Z","shell.execute_reply":"2022-12-03T01:10:50.398443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 7. Average prediction result","metadata":{}},{"cell_type":"code","source":"avg_ens_preds = np.mean(ensemble_all_preds, axis=0)[0]\navg_ens_preds","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:11:07.10331Z","iopub.execute_input":"2022-12-03T01:11:07.104768Z","iopub.status.idle":"2022-12-03T01:11:07.116708Z","shell.execute_reply.started":"2022-12-03T01:11:07.104703Z","shell.execute_reply":"2022-12-03T01:11:07.114767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"avg_lgbm_preds = np.mean(lgbm_preds, axis=0)\navg_lgbm_preds","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:11:07.271863Z","iopub.execute_input":"2022-12-03T01:11:07.273709Z","iopub.status.idle":"2022-12-03T01:11:07.28413Z","shell.execute_reply.started":"2022-12-03T01:11:07.273638Z","shell.execute_reply":"2022-12-03T01:11:07.282774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_preds = np.mean([avg_ens_preds, avg_lgbm_preds], axis=0)\nfinal_preds","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:11:07.415054Z","iopub.execute_input":"2022-12-03T01:11:07.415636Z","iopub.status.idle":"2022-12-03T01:11:07.426664Z","shell.execute_reply.started":"2022-12-03T01:11:07.415581Z","shell.execute_reply":"2022-12-03T01:11:07.42517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 8. Submission","metadata":{}},{"cell_type":"code","source":"final_sub = pd.DataFrame()\nfinal_sub[\"prediction_id\"] = df_test['prediction_id']\nfinal_sub[\"cancer\"] = final_preds\nfinal_sub = final_sub.groupby([\"prediction_id\"])[[\"cancer\"]].mean().reset_index()\nfinal_sub.to_csv('submission.csv', index=False)\nfinal_sub.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-03T01:11:10.179617Z","iopub.execute_input":"2022-12-03T01:11:10.180119Z","iopub.status.idle":"2022-12-03T01:11:10.217396Z","shell.execute_reply.started":"2022-12-03T01:11:10.180075Z","shell.execute_reply":"2022-12-03T01:11:10.215842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}