{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.11"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"},{"sourceId":11875525,"sourceType":"datasetVersion","datasetId":7463358},{"sourceId":12209434,"sourceType":"datasetVersion","datasetId":7691376}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":182.308437,"end_time":"2025-06-18T10:01:52.601214","environment_variables":{},"exception":true,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-06-18T09:58:50.292777","version":"2.6.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"23cabac816fa4e68b1268b275d386ea2":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"2e3bbdf386474de7af18b6da1fd481d9":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_f2f7c24c94b94527928b873559660e25","placeholder":"​","style":"IPY_MODEL_dd3e7fe8952d49a3ba49f795d0972e90","tabbable":null,"tooltip":null,"value":"100%"}},"320351550b1d4375b41ad80167586e0a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_82be3b20339049a7a41874db4a63f748","placeholder":"​","style":"IPY_MODEL_23cabac816fa4e68b1268b275d386ea2","tabbable":null,"tooltip":null,"value":" 54706/54706 [00:05&lt;00:00, 10741.63it/s]"}},"53a5f4fadff548dab266bc574e7a54b0":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"7b5dbf4a7b3a43c49322cc18e7e61f48":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_2e3bbdf386474de7af18b6da1fd481d9","IPY_MODEL_ea637d12a2c047438fe6dc58af27ed86","IPY_MODEL_320351550b1d4375b41ad80167586e0a"],"layout":"IPY_MODEL_ec1cd0e045ff4a09aa20196771764a3f","tabbable":null,"tooltip":null}},"82be3b20339049a7a41874db4a63f748":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"c600fd7c0f234682bc12da73ff1fb91b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"dd3e7fe8952d49a3ba49f795d0972e90":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"ea637d12a2c047438fe6dc58af27ed86":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_53a5f4fadff548dab266bc574e7a54b0","max":54706,"min":0,"orientation":"horizontal","style":"IPY_MODEL_c600fd7c0f234682bc12da73ff1fb91b","tabbable":null,"tooltip":null,"value":54706}},"ec1cd0e045ff4a09aa20196771764a3f":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"f2f7c24c94b94527928b873559660e25":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install \\\n  numpy==1.26.4 \\\n  scipy==1.13.0 \\\n  scikit-learn==1.3.2 \\\n  albumentations==1.4.3 \\\n  transformers==4.44.2 \\\n  efficientnet_pytorch==0.7.1 \\\n  dicomsdl==0.109.2 \\\n  wandb==0.16.6 \\\n  --no-cache-dir","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-13T05:57:14.216551Z","iopub.execute_input":"2025-08-13T05:57:14.217261Z","iopub.status.idle":"2025-08-13T05:58:41.12814Z","shell.execute_reply.started":"2025-08-13T05:57:14.217236Z","shell.execute_reply":"2025-08-13T05:58:41.127323Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Handle datasets\nimport io\nimport os\nimport copy\nimport gc\nimport cv2\nimport time\nimport math\nimport random\nimport pydicom\nimport dicomsdl\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nfrom glob import glob\nimport tifffile as tiff\nimport imageio.v3 as iio\nimport SimpleITK as sitk\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nimport multiprocessing as mp\nfrom collections import Counter\nfrom joblib import Parallel, delayed\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\n# 2. Visualize datasets\nimport datetime as dtime\nfrom datetime import datetime\nimport itertools\nimport matplotlib.pyplot as plt \nimport seaborn as sns \nimport plotly.express as px\nimport plotly.figure_factory as pff\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nfrom matplotlib.offsetbox import AnnotationBbox, OffsetImage\nfrom matplotlib.colors import ListedColormap, LinearSegmentedColormap\nfrom matplotlib.patches import Rectangle\nfrom IPython.display import display_html\n\n# 3. Preprocess datasets\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, LabelEncoder\nfrom sklearn.impute import SimpleImputer, KNNImputer\n## import iterative impute\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\n## fastai\n# from fastai.data.all import *\n# from fastai.vision.all import *\n\n# 4. machine learning\nfrom sklearn.model_selection import train_test_split, GridSearchCV, cross_val_score\nfrom sklearn.model_selection import StratifiedKFold, GroupKFold\n## for classification\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier, RandomForestRegressor\nfrom sklearn.ensemble import AdaBoostClassifier, GradientBoostingClassifier\nfrom sklearn.preprocessing import KBinsDiscretizer\nfrom xgboost import XGBClassifier\n\n# 5. Deep Learning\n## Augmentation\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nfrom transformers import ViTModel, ViTFeatureExtractor, ViTForImageClassification\n\n## Torch\nimport torch\nimport torchvision\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch import FloatTensor, LongTensor\nfrom torch.utils.data import Dataset, DataLoader, Subset\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\nfrom efficientnet_pytorch import EfficientNet\nfrom torchvision.models import resnet34, resnet50, ResNet50_Weights\nfrom torchvision import datasets, transforms\n\n# 6. metrics\nimport optuna\nfrom sklearn.metrics import f1_score, r2_score, classification_report\nfrom sklearn.metrics import accuracy_score, roc_auc_score\nfrom sklearn.metrics import roc_curve\nfrom sklearn.metrics import auc\n# 7. ignore warnings   \nimport warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\n\n# 8. For displaying and wandb\nimport wandb\n# wandb.login()\nplt.style.use(\"Solarize_Light2\")\nplt.rcParams.update({'font.size': 16})\n\n# 0. Customization\nparent_dir = \"/kaggle/input/rsna-breast-cancer-detection\"\nroi_dir ='/kaggle/input/roi-extract'\npre_train ='/kaggle/input/pre-train'\nWANDB_PROJ_NAME = \"RSNA_Breast_Cancer_Detection\"\nCONFIG = {\n    'competition': 'RSNA_Breast_Cancer',\n    '_wandb_kernel': 'aot'\n}\n\nmy_colors = ['#517664', '#73AA90', '#94DDBC', '#DAB06C',\n             '#DF928E', '#C97973', '#B25F57']\nCMAP1 = ListedColormap(my_colors)\nprint(\"Notebook Color Scheme: \")\nsns.palplot(sns.color_palette(my_colors))\nplt.show()","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":45.798914,"end_time":"2025-06-18T10:01:35.797217","exception":false,"start_time":"2025-06-18T10:00:49.998303","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-13T05:59:32.201299Z","iopub.execute_input":"2025-08-13T05:59:32.201947Z","iopub.status.idle":"2025-08-13T05:59:57.953069Z","shell.execute_reply.started":"2025-08-13T05:59:32.201915Z","shell.execute_reply":"2025-08-13T05:59:57.952465Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\nsecret_value_0 = user_secrets.get_secret(\"wandb\")\n\n! wandb login $secret_value_0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-13T06:01:58.382168Z","iopub.execute_input":"2025-08-13T06:01:58.382408Z","iopub.status.idle":"2025-08-13T06:02:00.744889Z","shell.execute_reply.started":"2025-08-13T06:01:58.382391Z","shell.execute_reply":"2025-08-13T06:02:00.743866Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === Here lies some general functionalities ===\n# === Seeds and Visualization ===\ndef set_seed(seed = 1234):\n    np.random.seed(seed)\n    random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    \n    # On CuDNN we need 2 further options\n    torch.backends.cudnn.deterministic = True\n    os.environ['PYTHONHASHSEED'] = str(seed)\n\ndef show_values_on_bars(axs, h_v = 'v', space = 0.4):\n    def _show_on_single_plot(ax):\n        if h_v == 'v':\n            for p in ax.patches:\n                _x = p.get_x() + p.get_width() / 2\n                _y = p.get_y() + p.get_height()\n                \n                value = int(p.get_height())\n                ax.text(_x, _y, format(value, ','), ha='center')\n        elif h_v == 'h':\n            for p in ax.patches:\n                _x = p.get_x() + p.get_width() + float(space)\n                _y = p.get_x() + p.get_height()\n                \n                value = int(p.get_width())\n                ax.text(_x, _y, format(value, ','), ha='left')\n    \n    if isinstance(axs, np.ndarray):\n        for i, ax in np.ndenumerate(axs):\n            _show_on_single_plot(ax)\n    else:\n        _show_on_single_plot(axs)\n","metadata":{"papermill":{"duration":0.038432,"end_time":"2025-06-18T10:01:35.860373","exception":false,"start_time":"2025-06-18T10:01:35.821941","status":"completed"},"tags":[],"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2025-08-13T06:02:01.025964Z","iopub.execute_input":"2025-08-13T06:02:01.026808Z","iopub.status.idle":"2025-08-13T06:02:01.037832Z","shell.execute_reply.started":"2025-08-13T06:02:01.02677Z","shell.execute_reply":"2025-08-13T06:02:01.036964Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def save_dataset_artifact(run_name, artifact_name, path, \n                          projectName = None, config = None, data_type = \"dataset\"):\n    run = wandb.init(project=projectName,\n                     name = run_name,\n                     config= config)\n\n    artifact = wandb.Artifact(name = artifact_name,\n                              type = data_type)\n    artifact.add_file(path)\n    \n    wandb.log_artifact(artifact)\n    wandb.finish()\n    print(\"Artifact has been save successfully!\")\n\ndef create_wandb_plot(x_data = None, y_data = None, x_name = None, y_name = None,\n                      title = None, log = None, plot = \"line\"):\n    data = [\n        [label, val] for (label, val) in zip(x_data, y_data)\n    ]\n    table = wandb.Table(data = data, columns = [x_name, y_name])\n    \n    if plot == \"line\":\n        wandb.log({ log: wandb.plot.line(table, x_name, y_name, title=title) })\n    elif plot == \"bar\":\n        wandb.log({ log: wandb.plot.bar(table, x_name, y_name, title=title) })\n    elif plot == \"scatter\":\n        wandb.log({ log: wandb.plot.scatter(table, x_name, y_name, title=title) })\n\ndef create_wandb_hist(x_data = None, x_name = None, title = None, log = None):\n    data = [[x] for x in x_data]\n    table = wandb.Table(data = data, columns=[x_name])\n    wandb.log({ log: wandb.plot.histogram(table, x_name, title=title) })\n    \n\ndef show_stacked_images(image_tensor_batch, target_labels=None):\n    \n    num_images = image_tensor_batch.size(0)\n    sqrt_n = int(math.sqrt(num_images))\n    ncols = sqrt_n\n    nrows = math.ceil(num_images / ncols)\n    fig, axis = plt.subplots(\n        nrows=nrows, ncols=ncols, figsize=(12 * ncols, 8 * nrows)\n    )\n    axis = axis.flatten()\n    \n    for i in range(num_images):\n        image_tensor = image_tensor_batch[i]\n        image = image_tensor.cpu().numpy().transpose((1, 2, 0))     # Transpose to HWC\n        \n        mean = np.array([0.485, 0.456, 0.406])\n        std = np.array([0.229, 0.224, 0.225])\n    \n        image = std * image + mean                                  # Unnormalize\n        image = np.clip(image, 0, 1)\n        \n         # Convert RGB to grayscale\n        image_gray = cv2.cvtColor((image * 255).astype(np.uint8), cv2.COLOR_RGB2GRAY)\n        image_gray = image_gray / 255.0\n        \n        axis[i].imshow(image_gray, cmap=\"bone\")\n        axis[i].axis('off')\n        if target_labels is not None:\n            axis[i].set_title(f\"Target: {target_labels[i].item()}\")\n    \n    plt.tight_layout()\n    plt.axis('off')\n    plt.show()","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2025-08-13T06:02:01.603934Z","iopub.execute_input":"2025-08-13T06:02:01.604214Z","iopub.status.idle":"2025-08-13T06:02:01.615374Z","shell.execute_reply.started":"2025-08-13T06:02:01.604192Z","shell.execute_reply":"2025-08-13T06:02:01.614548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(os.path.join(parent_dir, \"train.csv\"))\ntest = pd.read_csv(os.path.join(parent_dir, \"test.csv\"))\n\ntrain_image_path = os.path.join(roi_dir, \"train_image_ROI_processed_jp2000_512\")\n# train_image_path = os.path.join(parent_dir, \"train_image_processed_cv2_512\")\n\nall_paths = []\nfor k in tqdm(range(len(train))):\n    row = train.iloc[k, :]\n    all_paths.append(\n        os.path.join(\n            train_image_path, str(row.patient_id), f\"{str(row.image_id)}.jp2\"\n        ) \n    )\n    \ntrain['path'] = all_paths\n\ntrain = train[\n    ['patient_id', 'image_id', 'laterality', 'view', 'age', 'implant',\n     \"cancer\", \"path\"]\n]\n\nle_laterality = LabelEncoder()\nle_view = LabelEncoder()\n\ntrain['laterality'] = le_laterality.fit_transform(train['laterality'])\ntrain['view']       = le_view.fit_transform(train['view'])\n\ntrain['age'] = train['age'].fillna(58)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-13T06:02:02.032267Z","iopub.execute_input":"2025-08-13T06:02:02.032537Z","iopub.status.idle":"2025-08-13T06:02:06.101296Z","shell.execute_reply.started":"2025-08-13T06:02:02.032517Z","shell.execute_reply":"2025-08-13T06:02:06.100576Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Seed\nset_seed()\nDEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint('Device available now: ', DEVICE)\n\n# Read in Data\ntrain_df = pd.read_csv(os.path.join(pre_train, \"train_preprocessed.csv\"))\n# ====== GLOBAL PARAMS =======\ncsv_columns = [\"laterality\", \"view\", \"age\", \"implant\"]\nno_columns = len(csv_columns)\noutput_size = 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-13T06:02:07.023533Z","iopub.execute_input":"2025-08-13T06:02:07.023793Z","iopub.status.idle":"2025-08-13T06:02:07.124428Z","shell.execute_reply.started":"2025-08-13T06:02:07.023774Z","shell.execute_reply":"2025-08-13T06:02:07.123767Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def transforms(isTrain=False, isMinority=False):\n    aug_list = []\n    SIZE = 224\n    \n    # === Spatial: Isotropic Scaling ===\n    # For handling different size images\n    aug_list += [\n        A.LongestMaxSize(max_size=SIZE, p=1),\n        A.PadIfNeeded(min_height=SIZE, min_width=SIZE, border_mode=cv2.BORDER_CONSTANT, value=0, p=1),  # Padding to make the image size consistent\n    ]\n    \n    if isTrain:\n        if isMinority:          # Stronger augmentations for the minority class\n            \n            aug_list += [\n                \n                A.ShiftScaleRotate(shift_limit=0.1, scale_limit=0.1, rotate_limit=15, p=0.7),\n                A.ISONoise(color_shift=(0.01, 0.03), intensity=(0.05, 0.15), p=0.2),\n                A.ElasticTransform(alpha=1, sigma=50, alpha_affine=50, p=0.2),\n                A.RandomBrightnessContrast(p=0.2),\n                A.HorizontalFlip(p=0.3),\n                A.VerticalFlip(p=0.3),\n                A.GridDistortion(p=0.3),\n            ]\n            \n        else:                   # Mild augmentations for the majority class\n            \n            aug_list += [\n                \n                # === Photometric Augmentations - Brightness / Contrast / Gamma — very mild ===\n                A.OneOf([\n                    A.RandomToneCurve(scale=0.2, p=0.3),\n                    A.RandomGamma(gamma_limit=(90, 110), p=0.2),\n                    A.RandomBrightnessContrast(brightness_limit=(-0.075, 0.075), contrast_limit=(-0.4, 0.5), p=0.3)\n                ], p=0.5),\n                \n                # # === Mild Contrast Enhancer (safe CLAHE) ===\n                A.CLAHE(clip_limit=2.0, tile_grid_size=(8, 8), p=0.2),\n                \n                # == Downscaling - Blurring (slight only) ==\n                A.OneOf([   \n                    A.MotionBlur(blur_limit=3, p=0.3),\n                    A.Downscale(scale_min=0.98, scale_max=0.999, interpolation=dict(\n                        upscale=cv2.INTER_LINEAR, downscale=cv2.INTER_AREA), p=0.5),\n                ], p=0.2),\n                \n                # # === Occlusion-style Augmentation ===\n                A.OneOf([\n                    A.GridDropout(ratio=0.2, unit_size_min=16, unit_size_max=32, random_offset=True, p=0.2),\n                    A.CoarseDropout(max_holes=6, max_height=0.15, max_width=0.25, min_holes=1, min_height=0.05, min_width=0.1, fill_value=0, mask_fill_value=None, p=0.25),\n                ], p=0.3),\n                \n                # # === Flips ===\n                A.HorizontalFlip(p=0.5) if isTrain else A.NoOp(),\n                A.VerticalFlip(p=0.5) if isTrain else A.NoOp(),\n                \n            ]\n    \n    # === Normalize & ToTensor ===\n    aug_list += [\n        A.Normalize(),\n        ToTensorV2()\n    ]\n    \n    return A.Compose(aug_list)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-13T06:02:07.514539Z","iopub.execute_input":"2025-08-13T06:02:07.515171Z","iopub.status.idle":"2025-08-13T06:02:07.523763Z","shell.execute_reply.started":"2025-08-13T06:02:07.515149Z","shell.execute_reply":"2025-08-13T06:02:07.523078Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def data_to_device(data):\n    values = list(data.values())\n    \n    image = values[0].to(DEVICE)\n    meta = values[1].to(DEVICE)\n    \n    if len(values) == 3:\n        targets = values[2].to(DEVICE)\n        return image, meta, targets\n    else:\n        return image, meta\n\nclass RSNADataset(Dataset):\n    \n    def __init__(self, dataFrame, isTrain = True, transforms=None):\n        self.dataFrame, self.isTrain = dataFrame, isTrain\n        self.metaData = self.dataFrame[csv_columns].to_numpy(dtype=np.float32)\n        self.transforms = transforms  ## Data Augmentation\n    \n    def __len__(self):\n        return len(self.dataFrame)\n    \n    def __getitem__(self, index):\n        try:\n            # Dealing with tabular\n            csv_data = self.metaData[index]\n            csv_data = torch.tensor(csv_data, dtype=torch.float32)\n            \n            # Dealing with images\n            image_path = self.dataFrame['path'][index]\n            if not os.path.exists(image_path):  # Check if image exists\n                return self.__getitem__((index + 1) % len(self))\n            \n            image = cv2.imread(image_path)\n            if image is None:\n                return self.__getitem__((index + 1) % len(self))  # Pass if img is None\n            image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n            \n            # == AUGMENTATION ==\n            # Dealing with imbalanced data\n            label = self.dataFrame.loc[index]['cancer']\n            isMinority = (label == 1)\n            \n            # == AUGMENTATION ==\n            if self.isTrain:\n                image = self.transforms(\n                    isMinority=isMinority, isTrain=True\n                )(image=image)['image']\n            else:\n                image = self.transforms(isTrain=False)(image=image)['image']\n            \n            if self.isTrain:\n                return {\n                    \"image\": image, \n                    \"meta\": csv_data, \n                    \"target\": self.dataFrame['cancer'][index]\n                }\n            else:\n                return {\n                    \"image\": image, \n                    \"meta\": csv_data, \n                }\n        except Exception as e:\n            print(f\"[ERROR] Failed at index {index}: {e}\")\n            if self.isTrain:\n                return {\n                    \"image\": torch.zeros(3, 224, 224), \n                    \"meta\": torch.zeros(no_columns),\n                    \"target\": torch.tensor(0)\n                }\n            else:\n                return {\n                    \"image\": torch.zeros(3, 224, 224), \n                    \"meta\": torch.zeros(no_columns), \n                }\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-13T06:02:08.013938Z","iopub.execute_input":"2025-08-13T06:02:08.014221Z","iopub.status.idle":"2025-08-13T06:02:08.023263Z","shell.execute_reply.started":"2025-08-13T06:02:08.014201Z","shell.execute_reply":"2025-08-13T06:02:08.022488Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# == Sanity Check ==\ntrain_data = train_df.reset_index(drop=True).head(12)\ndataset = RSNADataset(\n    dataFrame = train_data, \n    isTrain = True, \n    transforms=transforms\n)\ndataLoader = DataLoader(dataset, batch_size=64, shuffle=False, pin_memory=True)\n\nfor i, data in enumerate(dataLoader):\n    image, meta, targets = data_to_device(data)\n    print(f\"Batch: {i} \\n Image: {image.shape} \\n Meta: {meta} \\n Targets: {targets}\")\n    print(\"=\"*50)\n    \n    show_stacked_images(image, targets)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-13T06:02:08.731273Z","iopub.execute_input":"2025-08-13T06:02:08.731557Z","iopub.status.idle":"2025-08-13T06:02:10.904366Z","shell.execute_reply.started":"2025-08-13T06:02:08.731535Z","shell.execute_reply":"2025-08-13T06:02:10.903335Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class EffNetV2Network(nn.Module):\n    \n    def __init__(self, outputSize, no_columns):\n        super().__init__()\n        self.no_columns, self.outputSize = no_columns, outputSize\n        \n        # Load EfficientNetV2 (using correct model name)\n    \n        self.features = torchvision.models.efficientnet_v2_s(weights='IMAGENET1K_V1')\n        self.features = nn.Sequential(*list(self.features.children())[:-1])\n        \n        self.csv = nn.Sequential(\n            nn.Linear(self.no_columns, 250), \n            nn.BatchNorm1d(250),\n            nn.ReLU(),\n            nn.Dropout(p=0.2),\n            \n            nn.Linear(250, 250),\n            nn.BatchNorm1d(250),\n            nn.ReLU(),\n            nn.Dropout(p=0.2),\n        )\n        \n        self.classification = nn.Sequential(nn.Linear(1280 + 250, self.outputSize),\n                                            nn.Dropout(p=0.2))  # 1280 comes from EfficientNetV2-B0 output dimension\n        \n    def forward(self, image, meta, prints=False):\n        if prints: \n            print(f\"Input Image Shape: {image.shape} \\n Input Metadata Shape: {meta.shape}\")\n        \n        # == Image CNN == (EfficientNetV2 Feature Extraction)\n        image = self.features(image) \n        image = F.avg_pool2d(image, image.size()[2:]).reshape(-1, 1280)  # 1280 is the output feature size for EfficientNetV2-B0\n        if prints: \n            print(f'Features image shape: {image.shape}')\n        \n        # == CSV FNN == (Tabular data processing)\n        meta = self.csv(meta)\n        if prints: \n            print(f'Metadata shape: {meta.shape}')\n        \n        # Concatenate image features with metadata features\n        image_meta_data = torch.cat((image, meta), dim=1)\n        if prints: \n            print(f'Concatenated data: {image_meta_data.shape}')\n       \n        # == Final Classification ==\n        out = self.classification(image_meta_data)\n        if prints: \n            print(f'Out shape: {out.shape}')\n        \n        return out","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-13T06:04:42.370902Z","iopub.execute_input":"2025-08-13T06:04:42.371711Z","iopub.status.idle":"2025-08-13T06:04:42.378964Z","shell.execute_reply.started":"2025-08-13T06:04:42.371688Z","shell.execute_reply":"2025-08-13T06:04:42.378339Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# == Sanity Check ==\nmodel_example_2 = EffNetV2Network(outputSize=output_size,\n                                no_columns=no_columns).to(DEVICE)\n\nout = model_example_2(image, meta, prints=True)\ncriterion_example = nn.BCEWithLogitsLoss()\nloss = criterion_example(out, targets.unsqueeze(1).float())\n\nprint(\"=\"*50)\nprint(f\"Loss = {loss.item()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-13T06:04:42.676246Z","iopub.execute_input":"2025-08-13T06:04:42.676477Z","iopub.status.idle":"2025-08-13T06:04:43.18891Z","shell.execute_reply.started":"2025-08-13T06:04:42.67646Z","shell.execute_reply":"2025-08-13T06:04:43.188187Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ResNet50Network(nn.Module):\n    \n    def __init__(self, outSize, no_columns):\n        super().__init__()\n        self.no_columns, self.outSize = no_columns, outSize\n        \n        backbone = resnet50(weights=ResNet50_Weights.DEFAULT)\n        \n        # output before modifications: [B, 2048, 7, 7] for a 224x224 input\n        # Remove the final classification layer to extract features (2048-dim)\n        modules = list(backbone.children())[:-2]\n        modules.append(nn.AdaptiveAvgPool2d(output_size=(1, 1)))\n        self.features = nn.Sequential(*modules)     # output after AvgPool: [B, 2048, 1, 1]\n        \n        # == metadata ==\n        self.csv = nn.Sequential(\n            nn.Linear(self.no_columns, 500),\n            nn.BatchNorm1d(500),\n            nn.ReLU(),\n            nn.Dropout(p=0.2)\n        )\n        \n        # Classification\n        self.classification = nn.Linear(2048 + 500, self.outSize)\n    \n    def forward(self, image, meta, prints=False):\n        if prints: \n            print(f\"Input Image Shape: {image.shape} \\n Input Metadata Shape: {meta.shape}\")\n        \n        # == Image CNN ==\n        image = self.features(image)\n        image = image.view(image.size(0), -1)   # Flatten → [B, 2048]\n        if prints: print(f'Features image shape: {image.shape}')\n        \n        # == CSV FNN ==\n        meta = self.csv(meta)\n        if prints: print(f'Metadata shape: {meta.shape}')\n        \n        # Concatenate layers from image with layers from csv_data\n        image_meta_data = torch.cat((image, meta), dim=1)\n        if prints: print(f'Concatenated data: {image_meta_data.shape}')\n        \n        # == CLASSIF ==\n        out = self.classification(image_meta_data)\n        if prints: print(f'Out shape: {out.shape}')\n        \n        return out","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-13T06:00:09.94767Z","iopub.execute_input":"2025-08-13T06:00:09.947917Z","iopub.status.idle":"2025-08-13T06:00:09.954952Z","shell.execute_reply.started":"2025-08-13T06:00:09.94789Z","shell.execute_reply":"2025-08-13T06:00:09.954304Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# == Sanity Check =#\nmodel_example = ResNet50Network(\n    outSize=output_size,\n    no_columns=no_columns\n).to(DEVICE)\n\nout = model_example(image, meta, prints=True)\n\ncriterion_example = nn.BCEWithLogitsLoss()\nloss = criterion_example(out, targets.unsqueeze(1).float())\n\nprint(\"=\"*50)\nprint(f\"Loss = {loss.item()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T09:17:32.323714Z","iopub.execute_input":"2025-08-12T09:17:32.324Z","iopub.status.idle":"2025-08-12T09:17:33.420562Z","shell.execute_reply.started":"2025-08-12T09:17:32.323981Z","shell.execute_reply":"2025-08-12T09:17:33.419973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_pos = train_df[train_df['cancer'] == 1]\ndf_neg = train_df[train_df['cancer'] == 0]\n\n# Oversample positives to match negatives\ndf_pos_oversampled = df_pos.sample(n=len(df_neg), replace=True, random_state=42)\n\n# Combine and shuffle\ntrain_df_balanced = pd.concat([df_neg, df_pos_oversampled], ignore_index=True)\ntrain_df_balanced = train_df_balanced.sample(frac=1, random_state=42).reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T09:17:34.650473Z","iopub.execute_input":"2025-08-12T09:17:34.650766Z","iopub.status.idle":"2025-08-12T09:17:34.681652Z","shell.execute_reply.started":"2025-08-12T09:17:34.650746Z","shell.execute_reply":"2025-08-12T09:17:34.680816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df_balanced['cancer'].value_counts(normalize=True))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T09:17:36.665524Z","iopub.execute_input":"2025-08-12T09:17:36.666256Z","iopub.status.idle":"2025-08-12T09:17:36.676267Z","shell.execute_reply.started":"2025-08-12T09:17:36.666222Z","shell.execute_reply":"2025-08-12T09:17:36.675645Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training","metadata":{"papermill":{"duration":0.038915,"end_time":"2025-06-18T10:01:49.197278","exception":false,"start_time":"2025-06-18T10:01:49.158363","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def add_in_file(text, f, trial):\n    with open(f\"logs_{VERSION}_trial_{trial}.txt\", \"a+\") as f:\n        print(text, file=f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-13T06:00:38.010944Z","iopub.execute_input":"2025-08-13T06:00:38.011513Z","iopub.status.idle":"2025-08-13T06:00:38.01518Z","shell.execute_reply.started":"2025-08-13T06:00:38.011487Z","shell.execute_reply":"2025-08-13T06:00:38.014476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"FOLDS = 2\nEPOCHS = 40\nMIN_THRESHOLD_EPOCHS = 5\nPATIENCE = 3\nWD = 1e-4\nLR_PATIENCE = 5\nLR_FACTOR = 0.6","metadata":{"trusted":true,"execution":{"execution_failed":"2025-08-13T06:07:40.952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom torch.cuda.amp import autocast, GradScaler\n\n\ndef pfbeta_torch(labels, preds, beta=1.0, eps=1e-5):\n    \"\"\"Compute probabilistic F-beta (pF1) score.\"\"\"\n    preds = preds.clamp(0, 1)\n    labels = labels.float()\n\n    y_true_count = labels.sum()\n    ctp = preds[labels == 1].sum()  # true positives weighted by predicted probs\n    cfp = preds[labels == 0].sum()  # false positives weighted by predicted probs\n\n    beta_squared = beta ** 2\n    c_precision = ctp / (ctp + cfp + eps)\n    c_recall = ctp / (y_true_count + eps)\n\n    if c_precision > 0 and c_recall > 0:\n        return (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall + eps)\n    else:\n        return torch.tensor(0.0, device=preds.device)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-08-13T06:07:40.953Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def pfbeta_best_threshold(labels, probs, beta=1.0, step=0.05):\n    thresholds = torch.arange(0.0, 1.0, step, device=probs.device)\n    best_score = torch.tensor(0.0, device=probs.device)\n    \n    for thr in thresholds:\n        bin_preds = (probs > thr).float()\n        score = pfbeta_torch(labels, bin_preds, beta=beta)\n        if score > best_score:\n            best_score = score\n    return best_score","metadata":{"trusted":true,"execution":{"execution_failed":"2025-08-13T06:07:40.953Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import GroupKFold, GroupShuffleSplit\n\nset_seed(42)\n\ngss = GroupShuffleSplit(n_splits=1, test_size=0.05, random_state=42)\nholdout_idx, cv_idx = next(\n    gss.split(\n        X=train_df,\n        y=train_df['cancer'],\n        groups=train_df['patient_id']\n    )\n)\n\nholdout_df = train_df.iloc[holdout_idx].reset_index(drop=True)\nremaining_df = train_df.iloc[cv_idx].reset_index(drop=True)\n\nholdout_ds = RSNADataset(holdout_df, isTrain=False, transforms=transforms)\nholdout_loader = DataLoader(\n    holdout_ds, batch_size=64, shuffle=False, pin_memory=True\n)\n# =======================================\n# 1. Define Ensembling Function (Majority Voting / Soft Voting)\n# - Purpose: \n#       + Designed to work with K-Fold Cross-Validation.\n#       + After N folds, we have N best, independent models, working as a 'judge'\n#       + These N judges output N independent predictions for each 'unseen' data.\n#       + For ex: Model1: 0.98 - Model2: 0.99 - Model3: 0.975\n#       + Finally, calculate the final average probability (soft voting)\n# =======================================\n\ndef ensemble_predict(models, loader, voting_type='soft'):\n    for m in models: m.eval().to(DEVICE)\n    all_outputs = []\n    with torch.no_grad():\n        for m in models:\n            outputs = []\n            for batch in loader:\n                img, meta, *_ = data_to_device(batch)\n                with torch.cuda.amp.autocast():\n                    logits = m(img, meta)\n                output = torch.sigmoid(logits).detach()\n                outputs.append(output)\n\n                print(' --------------------------') # Loading on Kaggle\n            all_outputs.append(torch.cat(outputs, dim=0).cpu())\n    print('8888888888888888888')        \n    all_outputs = torch.stack(all_outputs)\n    print('9999999999999999999')\n    if voting_type=='soft':\n        # either average logits or average sigmoid\n        all_probs = [torch.sigmoid(o) for o in all_outputs]\n        avg_probs = torch.stack(all_probs).mean(dim=0)\n        return avg_probs","metadata":{"trusted":true,"execution":{"execution_failed":"2025-08-13T06:07:40.953Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EfficientNet","metadata":{}},{"cell_type":"code","source":"# =======================================\n# 2. Modify `train_folds` to save top 3 models\n# - Purpose:\n#\n#   + saveBestModel():\n#       > stores N best models' weights for later uses.\n#       > outputs a list of names of those N models (topModels).\n#\n#   + trainFunction():\n#       > uses defined model to train. U can just replace model for diff results.\n#       > return training_accuracy and training_loss.\n#\n#   + validFunction():\n#       > for medical imaging, this function uses both ROC and F1 score\n#       > to decide the best model.\n#       > logs out result metrics, and returns best_overall_score.\n#\n#   + after validFunction()...:\n#       > defines K-folds.\n#       > uses all functions to train and some tricks for better scores.\n#       > clear cache and cleaning process.\n#\n# =======================================\n\ndef train_folds(train_original, lr, batch_size, trial, epochs=3):\n    \n    topModelsPath = []\n\n    def saveBestModel(foldNum, epochNum, valid_acc, best_f1, best_roc, best_overall_score):\n        model_name = f\"{VERSION}_Fold{foldNum}_Epoch{epochNum}_F1{best_f1:.3f}_ROC{best_roc:.3f}_SCORE{best_overall_score:.3f}.pth\"\n        torch.save(current_model.state_dict(), model_name)\n        \n        topModelsPath.append((model_name, best_overall_score))\n        topModelsPath.sort(key=lambda x: x[1], reverse=True)\n        \n        if len(topModelsPath) > 3:\n            removedModel = topModelsPath.pop()\n            os.remove(removedModel[0])\n            print(f\"Deleted worst model: {removedModel[0]}\")\n\n    def create_dataloader(dataset,batch_size, shuffle=False, sampler=None):\n        return DataLoader(dataset, batch_size=batch_size, shuffle=shuffle if sampler is None else False, sampler=sampler, pin_memory=True)\n\n    def trainFunction(epoch):\n        current_model.train()\n        total_loss, correct = 0.0, 0.0\n        progBar = tqdm(train_loader, desc=f\"Epoch {epoch+1} [Train]\", leave=False)\n        scaler = GradScaler()\n        for data in progBar:\n            image, meta, targets = data_to_device(data)\n            \n            optimizer.zero_grad()\n            with autocast():\n                out = current_model(image, meta)\n                loss = criterion(out, targets.unsqueeze(1).float())\n\n            scaler.scale(loss).backward()\n            scaler.step(optimizer)\n            scaler.update()\n            \n            total_loss += loss.item()\n            train_preds = torch.round(torch.sigmoid(out))\n            \n            correct += (train_preds.cpu() == targets.cpu().unsqueeze(1)).sum().item()\n            wandb.log({'train_loss': total_loss})\n            progBar.set_description(f'loss: {total_loss:.2f}')\n\n        train_acc = correct / len(train_index)\n        \n        wandb.log({\"train_acc\": train_acc})\n        return train_acc, total_loss\n    \n    def validFunction(epoch, train_acc, total_loss):\n        nonlocal best_roc, best_f1, best_overall_score, patience_f\n        current_model.eval()\n        \n        # Pre-allocate output buffer\n        preds_tensor = torch.zeros(len(valid_data), 1, device=DEVICE, dtype=torch.float32)\n        offset = 0\n        \n        progBar = tqdm(valid_loader, desc=f\"Epoch {epoch+1} [Valid]\", leave=False)\n        with torch.no_grad():\n            for data in progBar:\n                image, meta = data_to_device(data)\n                out = current_model(image, meta)\n                \n                probs = torch.sigmoid(out).cpu()\n                bs = probs.size(0)\n                preds_tensor[offset: offset + bs] = probs\n                offset += bs\n                \n                \n        labels = valid_data['cancer'].values.astype(np.float32)\n        probs = preds_tensor.cpu().numpy().reshape(-1)\n\n        probs_torch = torch.tensor(probs, dtype=torch.float32, device=DEVICE)\n        labels_torch = torch.tensor(labels, dtype=torch.float32, device=DEVICE)\n\n        pf1 = pfbeta_torch(labels_torch, probs_torch).item()\n        pf1_thr = pfbeta_best_threshold(labels_torch, probs_torch).item()\n        \n        fpr, tpr, thresholds = roc_curve(labels, probs, pos_label=1)\n        roc = auc(fpr, tpr)\n        \n\n\n        wandb.log({\"valid_roc\": roc})\n        wandb.log({\"valid_F1Score\":pf1})\n        \n        duration = str(time.time() - start_time)[:7]\n        log = f'{duration} | Epoch: {epoch+1}/{epochs} | Loss: {total_loss:.4f} | Acc_tr: {train_acc:.3f} | ROC: {roc:.3f} | F1: {pf1:.3f}'\n        print(log)\n        add_in_file(log, f, trial=trial)\n\n        # scheduler.step(aupr) \n        scheduler.step(roc)\n\n        if best_roc == 0 and best_f1 == 0:\n            best_roc = roc\n            best_f1 = pf1\n            \n       \n        current_score = 0.5 * roc + 0.5 * pf1\n        if best_overall_score is None or current_score > best_overall_score:\n            best_overall_score = current_score\n            best_roc = roc\n            best_f1 = pf1\n            \n            saveBestModel(\n                foldNum=i+1, epochNum=epoch+1, valid_acc=acc, \n                best_f1=best_f1, best_roc=best_roc, best_overall_score=best_overall_score\n            )\n            print(f\"Saved new best model at Epoch {epoch+1}\")\n            return best_overall_score, PATIENCE\n        else:\n            return best_overall_score, patience_f - 1\n    \n    with open(f\"logs_{VERSION}.txt\", \"w+\") as f:\n        print(f\"Training with learning rate: {lr} and batch size: {batch_size}\")\n        f.write(f\"Training with learning rate: {lr} and batch size: {batch_size}\\n\")\n    \n    train_original = train_original.sample(frac=1, random_state=42).reset_index(drop=True)\n\n    group_fold = GroupKFold(n_splits=FOLDS)\n    k_folds = group_fold.split(X=np.zeros(len(train_original)), y=train_original['cancer'], groups=train_original['patient_id'].tolist())\n    \n    for i, (train_index, valid_index) in enumerate(k_folds):\n        print(f\"-------- Fold: {i+1} --------\")\n        add_in_file(f\"-------- Fold: {i+1} --------\", f, trial=trial)\n        \n        RUN_CONFIG = CONFIG.copy()\n        params = dict(model=MODEL, version=VERSION, fold=i, epochs=epochs, batch=batch_size, lr=lr, weight_decay=WD)\n        RUN_CONFIG.update(params)\n        wandb.init(project=WANDB_PROJ_NAME, config=RUN_CONFIG)\n        \n        best_roc = 0\n        best_f1 = 0\n        best_overall_score = None\n        patience_f = PATIENCE\n        current_model = EffNetV2Network(outputSize=output_size, no_columns=no_columns).to(DEVICE)\n\n        \n        wandb.watch(current_model, log_freq=100)\n        \n        train_data = train_original.iloc[train_index].reset_index(drop=True)\n        valid_data = train_original.iloc[valid_index].reset_index(drop=True)\n        \n        neg, pos = train_data['cancer'].value_counts()\n        pos_weight = torch.tensor([neg / pos], dtype=torch.float).to(DEVICE)\n\n        optimizer = torch.optim.AdamW(current_model.parameters(), lr=lr, weight_decay=WD)\n        scheduler = ReduceLROnPlateau(optimizer=optimizer, mode=\"max\", patience=LR_PATIENCE, factor=LR_FACTOR)\n        criterion = nn.BCEWithLogitsLoss(pos_weight=pos_weight)\n    \n        train = RSNADataset(train_data, isTrain=True, transforms=transforms)\n        valid = RSNADataset(valid_data, isTrain=False, transforms=transforms)\n\n        class_counts = train_df['cancer'].value_counts().to_dict()  # e.g., {0: 53548, 1: 1158}\n        class_weights = {cls: 1.0 / count for cls, count in class_counts.items()}\n        sample_weights = train_data['cancer'].map(lambda x: class_weights[x])\n        \n        sampler = torch.utils.data.WeightedRandomSampler(\n            weights=sample_weights.values, \n            num_samples=len(sample_weights), \n            replacement=True\n        )\n        train_loader = create_dataloader(train,batch_size=batch_size, shuffle=True, sampler=sampler)\n        valid_loader = create_dataloader(valid,batch_size=batch_size, shuffle=False)\n        \n        for epoch in range(epochs):\n            start_time = time.time()\n            train_acc, total_loss = trainFunction(epoch)\n            best_overall_score, patience_f = validFunction(epoch, train_acc, total_loss)\n            if epoch + 1 > MIN_THRESHOLD_EPOCHS and patience_f == 0:\n                msg = f\"Early stopping | Best ROC: {best_roc:.4f}\"\n                print(msg)\n                add_in_file(msg, f, trial=trial)\n                break\n        \n        del train_loader, valid_loader, train, valid\n                \n        gc.collect()\n        torch.cuda.empty_cache()\n        # wandb.finish()\n\n    return topModelsPath","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-13T06:05:58.504847Z","iopub.execute_input":"2025-08-13T06:05:58.50514Z","iopub.status.idle":"2025-08-13T06:05:58.526157Z","shell.execute_reply.started":"2025-08-13T06:05:58.505119Z","shell.execute_reply":"2025-08-13T06:05:58.525456Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"###### # =======================================\n# 3. Utilize all and train\n# - Purpose:\n#   + hyperparameter optimization using Optuna with learning_rate and batch_size.\n#   + model can be changed freely, RestNet or EffNet or Transformer.\n#   + reuse the earlier holdout set for final evaluation.\n# =======================================\n\nVERSION = 'v6'\nMODEL = 'EffNetV2'\n\ntop_models = []\n# top_models_path = None    \n\n# Optimization using Optuna\ndef objective(trial):\n    trial_number = trial.number\n    lr = trial.suggest_loguniform(\"lr\", 1e-5, 1e-2)\n    batch_size = trial.suggest_categorical('batch_size', [32,64])\n\n\n    \n    # Train the model and get N top models from N folds\n    top_models_path = train_folds(\n        train_original=train_df, \n        lr=lr, batch_size=batch_size, trial=trial_number, epochs=EPOCHS\n    )\n    \n    top_models = [path for (path,roc) in top_models_path]\n    ''' \n        Below we will calculate the final (average) score from \n        N best models of N_folds by using freshly new data.\n        The data can be prepared by either:\n        \n        1. You have new data, or\n        2. using the 5% data - holdout dataset that I have saved earlier\n    '''\n    \n    BASE_DIR = '/kaggle/working/'\n    loaded_models = []\n    for model_path, _ in top_models_path:\n        model_instance = EffNetV2Network(outputSize=output_size, no_columns=no_columns)\n        dict_path = os.path.join(BASE_DIR, model_path)\n        \n        model_instance.load_state_dict(torch.load(dict_path, map_location=DEVICE))\n        model_instance.to(DEVICE)\n        loaded_models.append(model_instance)\n        \n\n    holdout_logits = ensemble_predict(\n        loaded_models, \n        holdout_loader, \n        voting_type='soft'\n    )\n    roc = roc_auc_score(holdout_df['cancer'], holdout_logits.numpy())\n    print(\"Hold-out ROC AUC:\", roc)\n\n    # == Clean up ==\n    for m in loaded_models:\n        del m\n    torch.cuda.empty_cache()\n    return roc\n\nstudy = optuna.create_study(direction='maximize')\nstudy.optimize(objective, n_trials=2)\nprint(\"Best hyperparameters: \", study.best_params)\n\nbest_trial = study.best_trial.number\nwith open(f\"logs_{VERSION}_trial_{best_trial}.txt\", \"r\") as f:\n    contents = f.read()\n    print(contents)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-08-13T06:07:40.951Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## ResNet","metadata":{}},{"cell_type":"code","source":"def train_folds(train_original, lr, batch_size, trial, epochs=3):\n    \n    topModelsPath = []\n\n    def saveBestModel(foldNum, epochNum, valid_acc, best_f1, best_roc, best_overall_score):\n        model_name = f\"{VERSION}_Fold{foldNum}_Epoch{epochNum}_F1{best_f1:.3f}_ROC{best_roc:.3f}_SCORE{best_overall_score:.3f}.pth\"\n        torch.save(current_model.state_dict(), model_name)\n        \n        topModelsPath.append((model_name, best_overall_score))\n        topModelsPath.sort(key=lambda x: x[1], reverse=True)\n        \n        if len(topModelsPath) > 3:\n            removedModel = topModelsPath.pop()\n            os.remove(removedModel[0])\n            print(f\"Deleted worst model: {removedModel[0]}\")\n\n    def create_dataloader(dataset,batch_size, shuffle=False, sampler=None):\n        return DataLoader(dataset, batch_size=batch_size, shuffle=shuffle if sampler is None else False, sampler=sampler, pin_memory=True)\n\n    def trainFunction(epoch):\n        current_model.train()\n        total_loss, correct = 0.0, 0.0\n        progBar = tqdm(train_loader, desc=f\"Epoch {epoch+1} [Train]\", leave=False)\n        scaler = GradScaler()\n        for data in progBar:\n            image, meta, targets = data_to_device(data)\n            \n            optimizer.zero_grad()\n            with autocast():\n                out = current_model(image, meta)\n                loss = criterion(out, targets.unsqueeze(1).float())\n\n            scaler.scale(loss).backward()\n            scaler.step(optimizer)\n            scaler.update()\n            \n            total_loss += loss.item()\n            train_preds = torch.round(torch.sigmoid(out))\n            \n            correct += (train_preds.cpu() == targets.cpu().unsqueeze(1)).sum().item()\n            wandb.log({'train_loss': total_loss})\n            progBar.set_description(f'loss: {total_loss:.2f}')\n\n        train_acc = correct / len(train_index)\n        \n        wandb.log({\"train_acc\": train_acc})\n        return train_acc, total_loss\n    \n    def validFunction(epoch, train_acc, total_loss):\n        nonlocal best_roc, best_f1, best_overall_score, patience_f\n        current_model.eval()\n        \n        # Pre-allocate output buffer\n        preds_tensor = torch.zeros(len(valid_data), 1, device=DEVICE, dtype=torch.float32)\n        offset = 0\n        \n        progBar = tqdm(valid_loader, desc=f\"Epoch {epoch+1} [Valid]\", leave=False)\n        with torch.no_grad():\n            for data in progBar:\n                image, meta = data_to_device(data)\n                out = current_model(image, meta)\n                \n                probs = torch.sigmoid(out).cpu()\n                bs = probs.size(0)\n                preds_tensor[offset: offset + bs] = probs\n                offset += bs\n                \n                \n        labels = valid_data['cancer'].values.astype(np.float32)\n        probs = preds_tensor.cpu().numpy().reshape(-1)\n\n        probs_torch = torch.tensor(probs, dtype=torch.float32, device=DEVICE)\n        labels_torch = torch.tensor(labels, dtype=torch.float32, device=DEVICE)\n\n        pf1 = pfbeta_torch(labels_torch, probs_torch).item()\n        pf1_thr = pfbeta_best_threshold(labels_torch, probs_torch).item()\n        \n        fpr, tpr, thresholds = roc_curve(labels, probs, pos_label=1)\n        roc = auc(fpr, tpr)\n        \n\n        wandb.log({\"valid_roc\": roc})\n        wandb.log({\"valid_F1Score\": pf1})\n        \n        duration = str(time.time() - start_time)[:7]\n        log = f'{duration} | Epoch: {epoch+1}/{epochs} | Loss: {total_loss:.4f} | Acc_tr: {train_acc:.3f} | ROC: {roc:.3f} | F1: {pf1:.3f}'\n        print(log)\n        add_in_file(log, f, trial=trial)\n\n        # scheduler.step(aupr) \n        scheduler.step(roc)\n\n        if best_roc == 0 and best_f1 == 0:\n            best_roc = roc\n            best_f1 = pf1\n            \n       \n        current_score = 0.5 * roc + 0.5 * pf1\n        if best_overall_score is None or current_score > best_overall_score:\n            best_overall_score = current_score\n            best_roc = roc\n            best_f1 = pf1\n            \n            saveBestModel(\n                foldNum=i+1, epochNum=epoch+1, valid_acc=acc, \n                best_f1=best_f1, best_roc=best_roc, best_overall_score=best_overall_score\n            )\n            print(f\"Saved new best model at Epoch {epoch+1}\")\n            return best_overall_score, PATIENCE\n        else:\n            return best_overall_score, patience_f - 1\n    \n    with open(f\"logs_{VERSION}.txt\", \"w+\") as f:\n        print(f\"Training with learning rate: {lr} and batch size: {batch_size}\")\n        f.write(f\"Training with learning rate: {lr} and batch size: {batch_size}\\n\")\n    \n    train_original = train_original.sample(frac=1, random_state=42).reset_index(drop=True)\n\n    group_fold = GroupKFold(n_splits=FOLDS)\n    k_folds = group_fold.split(X=np.zeros(len(train_original)), y=train_original['cancer'], groups=train_original['patient_id'].tolist())\n    \n    for i, (train_index, valid_index) in enumerate(k_folds):\n        print(f\"-------- Fold: {i+1} --------\")\n        add_in_file(f\"-------- Fold: {i+1} --------\", f, trial=trial)\n        \n        RUN_CONFIG = CONFIG.copy()\n        params = dict(model=MODEL, version=VERSION, fold=i, epochs=epochs, batch=batch_size, lr=lr, weight_decay=WD)\n        RUN_CONFIG.update(params)\n        wandb.init(project=WANDB_PROJ_NAME, config=RUN_CONFIG)\n        \n        best_roc = 0\n        best_f1 = 0\n        best_overall_score = None\n        patience_f = PATIENCE\n        current_model = ResNet50Network(outSize=output_size, no_columns=no_columns).to(DEVICE)\n\n        \n        wandb.watch(current_model, log_freq=100)\n        \n        train_data = train_original.iloc[train_index].reset_index(drop=True)\n        valid_data = train_original.iloc[valid_index].reset_index(drop=True)\n        \n        neg, pos = train_data['cancer'].value_counts()\n        pos_weight = torch.tensor([neg / pos], dtype=torch.float).to(DEVICE)\n\n        optimizer = torch.optim.AdamW(current_model.parameters(), lr=lr, weight_decay=WD)\n        scheduler = ReduceLROnPlateau(optimizer=optimizer, mode=\"max\", patience=LR_PATIENCE, factor=LR_FACTOR)\n        criterion = nn.BCEWithLogitsLoss(pos_weight=pos_weight)\n    \n        train = RSNADataset(train_data, isTrain=True, transforms=transforms)\n        valid = RSNADataset(valid_data, isTrain=False, transforms=transforms)\n\n        class_counts = train_df['cancer'].value_counts().to_dict()  # e.g., {0: 53548, 1: 1158}\n        class_weights = {cls: 1.0 / count for cls, count in class_counts.items()}\n        sample_weights = train_data['cancer'].map(lambda x: class_weights[x])\n        \n        sampler = torch.utils.data.WeightedRandomSampler(\n            weights=sample_weights.values, \n            num_samples=len(sample_weights), \n            replacement=True\n        )\n        train_loader = create_dataloader(train,batch_size=batch_size, shuffle=True, sampler=sampler)\n        valid_loader = create_dataloader(valid,batch_size=batch_size, shuffle=False)\n        \n        for epoch in range(epochs):\n            start_time = time.time()\n            train_acc, total_loss = trainFunction(epoch)\n            best_overall_score, patience_f = validFunction(epoch, train_acc, total_loss)\n            if epoch + 1 > MIN_THRESHOLD_EPOCHS and patience_f == 0:\n                msg = f\"Early stopping | Best ROC: {best_roc:.4f}\"\n                print(msg)\n                add_in_file(msg, f, trial=trial)\n                break\n        \n        del train_loader, valid_loader, train, valid\n                \n        gc.collect()\n        torch.cuda.empty_cache()\n        wandb.finish()\n\n    return topModelsPath","metadata":{"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"VERSION = 'v1'\nMODEL = 'resnet50'\n\ntop_models = []\n# top_models_path = None    \n\n# Optimization using Optuna\ndef objective(trial):\n    trial_number = trial.number\n    lr = trial.suggest_loguniform(\"lr\", 1e-5, 1e-2)\n    batch_size = trial.suggest_categorical('batch_size', [32,64])\n\n    \n    # Train the model and get N top models from N folds\n    top_models_path = train_folds( \n        train_original=train_df, \n        lr=lr, batch_size=batch_size, trial=trial_number, epochs=EPOCHS\n    )\n    \n    top_models = [path for (path,roc) in top_models_path]\n    ''' \n        Below we will calculate the final (average) score from \n        N best models of N_folds by using freshly new data.\n        The data can be prepared by either:\n        \n        1. You have new data, or\n        2. using the 5% data - holdout dataset that I have saved earlier\n    '''\n    \n    BASE_DIR = '/kaggle/working/'\n    loaded_models = []\n    \n    for model_path in top_models:\n        model_instance = ResNet50Network(outSize=output_size, no_columns=no_columns)\n        dict_path = os.path.join(BASE_DIR, model_path)\n        \n        model_instance.load_state_dict(torch.load(dict_path, map_location=DEVICE))\n        model_instance.to(DEVICE)\n        loaded_models.append(model_instance)\n        \n    \n    holdout_logits = ensemble_predict(\n        loaded_models, \n        holdout_loader, \n        voting_type='soft'\n    )\n    print('---------------------------------------------------------------------------------------')\n    roc = roc_auc_score(holdout_df['cancer'], holdout_logits.numpy())\n    print(\"Hold-out ROC AUC:\", roc)\n\n    # == Clean up ==\n    for m in loaded_models:\n        del m\n    torch.cuda.empty_cache()\n    return roc\n\nstudy = optuna.create_study(direction='maximize')\nstudy.optimize(objective, n_trials=2)\nprint(\"Best hyperparameters: \", study.best_params)\n\nbest_trial = study.best_trial.number\nwith open(f\"logs_{VERSION}_trial_{best_trial}.txt\", \"r\") as f:\n    contents = f.read()\n    print(contents)\n","metadata":{"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\n\ntotal = torch.cuda.get_device_properties(0).total_memory / (1024**3)\nreserved = torch.cuda.memory_reserved(0) / (1024**3)\nallocated = torch.cuda.memory_allocated(0) / (1024**3)\nfree = reserved - allocated\n\nprint(f\"Total Memory:     {total:.2f} GiB\")\nprint(f\"Reserved Memory:  {reserved:.2f} GiB\")\nprint(f\"Allocated Memory: {allocated:.2f} GiB\")\nprint(f\"Free (available): {free:.2f} GiB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-31T10:00:25.358892Z","iopub.execute_input":"2025-07-31T10:00:25.359458Z","iopub.status.idle":"2025-07-31T10:00:25.3646Z","shell.execute_reply.started":"2025-07-31T10:00:25.359436Z","shell.execute_reply":"2025-07-31T10:00:25.363879Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os._exit(00)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-31T10:00:34.789Z"}},"outputs":[],"execution_count":null}]}