{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nimport random\nimport warnings\nimport re\nimport math\n\nwarnings.filterwarnings(\"ignore\")\n\nimport numpy as np\nimport pandas as pd\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport cv2\nimport pydicom\n\nfrom tqdm.auto import tqdm\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score\n\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\n\nimport torchvision\nfrom torchvision import transforms\n\nfrom torchvision.models import (\n    resnet50,\n    ResNet50_Weights\n)\n\nprint(\"PyTorch:\", torch.__version__)\nprint(\"Torchvision:\", torchvision.__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:27.903897Z","iopub.execute_input":"2026-08-22T08:34:27.90416Z","iopub.status.idle":"2026-08-22T08:34:41.499329Z","shell.execute_reply.started":"2026-08-22T08:34:27.904129Z","shell.execute_reply":"2026-08-22T08:34:41.498659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DEVICE = torch.device(\n    \"cuda\" if torch.cuda.is_available() else \"cpu\"\n)\n\nprint(\"Device:\", DEVICE)\n\nif torch.cuda.is_available():\n\n    print(\n        \"GPU:\",\n        torch.cuda.get_device_name(0)\n    )\n\n    print(\n        \"GPU Memory:\",\n        round(\n            torch.cuda.get_device_properties(\n                0\n            ).total_memory / 1024**3,\n            2\n        ),\n        \"GB\"\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:41.501207Z","iopub.execute_input":"2026-08-22T08:34:41.501747Z","iopub.status.idle":"2026-08-22T08:34:41.841281Z","shell.execute_reply.started":"2026-08-22T08:34:41.50172Z","shell.execute_reply":"2026-08-22T08:34:41.840633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"INPUT_ROOT = \"/kaggle/input\"\n\nprint(\n    os.listdir(INPUT_ROOT)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:41.842137Z","iopub.execute_input":"2026-08-22T08:34:41.842517Z","iopub.status.idle":"2026-08-22T08:34:41.847188Z","shell.execute_reply.started":"2026-08-22T08:34:41.842474Z","shell.execute_reply":"2026-08-22T08:34:41.846473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = (\n    \"/kaggle/input/competitions/\"\n    \"rsna-knee-abnormality-detection\"\n)\n\nprint(\n    os.listdir(DATA_DIR)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:41.84814Z","iopub.execute_input":"2026-08-22T08:34:41.848488Z","iopub.status.idle":"2026-08-22T08:34:41.869125Z","shell.execute_reply.started":"2026-08-22T08:34:41.848463Z","shell.execute_reply":"2026-08-22T08:34:41.863117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(\n    os.path.join(\n        DATA_DIR,\n        \"train.csv\"\n    )\n)\n\ntrain_series = pd.read_csv(\n    os.path.join(\n        DATA_DIR,\n        \"train_series.csv\"\n    )\n)\n\ntest = pd.read_csv(\n    os.path.join(\n        DATA_DIR,\n        \"test.csv\"\n    )\n)\n\ntest_series = pd.read_csv(\n    os.path.join(\n        DATA_DIR,\n        \"test_series.csv\"\n    )\n)\n\nsample_submission = pd.read_csv(\n    os.path.join(\n        DATA_DIR,\n        \"sample_submission.csv\"\n    )\n)\n\nprint(\n    \"Train:\",\n    train.shape\n)\n\nprint(\n    \"Train series:\",\n    train_series.shape\n)\n\nprint(\n    \"Test:\",\n    test.shape\n)\n\nprint(\n    \"Test series:\",\n    test_series.shape\n)\n\nprint(\n    \"Sample submission:\",\n    sample_submission.shape\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:41.869823Z","iopub.status.idle":"2026-08-22T08:34:41.87025Z","shell.execute_reply.started":"2026-08-22T08:34:41.870102Z","shell.execute_reply":"2026-08-22T08:34:41.87013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TARGETS = [\n    \"ACL\",\n    \"MCL\",\n    \"Medial Meniscus\",\n    \"Lateral Meniscus\",\n    \"Medial OA\",\n    \"Lateral OA\",\n    \"PF OA\",\n    \"Effusion\",\n    \"Synovitis\",\n    \"Baker's\",\n    \"Contusion\",\n    \"Fracture\"\n]\n\nNUM_CLASSES = len(TARGETS)\n\nprint(\n    \"Number of targets:\",\n    NUM_CLASSES\n)\n\nfor i, target in enumerate(TARGETS):\n    print(i, target)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:41.871284Z","iopub.status.idle":"2026-08-22T08:34:41.871644Z","shell.execute_reply.started":"2026-08-22T08:34:41.871404Z","shell.execute_reply":"2026-08-22T08:34:41.871417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\n    \"Total studies:\",\n    len(train)\n)\n\nprint(\n    \"\\nNon-null labels:\"\n)\n\nprint(\n    train[TARGETS].notna().sum()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:41.872975Z","iopub.status.idle":"2026-08-22T08:34:41.873365Z","shell.execute_reply.started":"2026-08-22T08:34:41.873159Z","shell.execute_reply":"2026-08-22T08:34:41.873184Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labeled_mask = (\n    train[TARGETS]\n    .notna()\n    .all(axis=1)\n)\n\nunlabeled_mask = (\n    train[TARGETS]\n    .isna()\n    .all(axis=1)\n)\n\nlabeled_data = train[\n    labeled_mask\n].copy()\n\nunlabeled_data = train[\n    unlabeled_mask\n].copy()\n\nprint(\n    \"Officially labeled studies:\",\n    len(labeled_data)\n)\n\nprint(\n    \"Unlabeled studies:\",\n    len(unlabeled_data)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:41.875231Z","iopub.status.idle":"2026-08-22T08:34:41.875454Z","shell.execute_reply.started":"2026-08-22T08:34:41.875347Z","shell.execute_reply":"2026-08-22T08:34:41.875359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_pattern = (\n    train[TARGETS]\n    .isna()\n    .astype(int)\n    .value_counts()\n)\n\ndisplay(\n    missing_pattern\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:41.876432Z","iopub.status.idle":"2026-08-22T08:34:41.876796Z","shell.execute_reply.started":"2026-08-22T08:34:41.87662Z","shell.execute_reply":"2026-08-22T08:34:41.876642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"official_counts = (\n    labeled_data[TARGETS]\n    .sum()\n    .sort_values(\n        ascending=False\n    )\n)\n\ndisplay(\n    official_counts.to_frame(\n        \"Positive Cases\"\n    )\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:41.878108Z","iopub.status.idle":"2026-08-22T08:34:41.878321Z","shell.execute_reply.started":"2026-08-22T08:34:41.878218Z","shell.execute_reply":"2026-08-22T08:34:41.87823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(\n    figsize=(14, 6)\n)\n\nofficial_counts.plot(\n    kind=\"bar\"\n)\n\nplt.title(\n    \"Officially Labeled MRI Studies\"\n)\n\nplt.ylabel(\n    \"Positive Cases\"\n)\n\nplt.xticks(\n    rotation=45,\n    ha=\"right\"\n)\n\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:41.879528Z","iopub.status.idle":"2026-08-22T08:34:41.879905Z","shell.execute_reply.started":"2026-08-22T08:34:41.879776Z","shell.execute_reply":"2026-08-22T08:34:41.879793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display(\n    train_series.head()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:41.881214Z","iopub.status.idle":"2026-08-22T08:34:41.881488Z","shell.execute_reply.started":"2026-08-22T08:34:41.881367Z","shell.execute_reply":"2026-08-22T08:34:41.881381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\n    train_series[\n        \"Anatomical_Plane\"\n    ].value_counts()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.072138Z","iopub.status.idle":"2026-08-22T08:34:42.072433Z","shell.execute_reply.started":"2026-08-22T08:34:42.072305Z","shell.execute_reply":"2026-08-22T08:34:42.07232Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\n    train_series[\n        \"Fluid_Sensitive\"\n    ].value_counts()\n)\n\nprint()\n\nprint(\n    train_series[\n        \"Fat_Suppression\"\n    ].value_counts()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.07356Z","iopub.status.idle":"2026-08-22T08:34:42.074004Z","shell.execute_reply.started":"2026-08-22T08:34:42.073796Z","shell.execute_reply":"2026-08-22T08:34:42.073834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_SERIES_DIR = os.path.join(\n    DATA_DIR,\n    \"train_series\"\n)\n\nTEST_SERIES_DIR = os.path.join(\n    DATA_DIR,\n    \"test_series\"\n)\n\nprint(\n    TRAIN_SERIES_DIR\n)\n\nprint(\n    TEST_SERIES_DIR\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.075306Z","iopub.status.idle":"2026-08-22T08:34:42.075574Z","shell.execute_reply.started":"2026-08-22T08:34:42.075416Z","shell.execute_reply":"2026-08-22T08:34:42.075428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_dicom_paths(\n    study_id,\n    series_id,\n    base_dir\n):\n\n    series_path = os.path.join(\n        base_dir,\n        str(study_id),\n        str(series_id)\n    )\n\n    if not os.path.exists(\n        series_path\n    ):\n        return []\n\n    paths = []\n\n    for filename in os.listdir(\n        series_path\n    ):\n\n        if filename.lower().endswith(\n            \".dcm\"\n        ):\n\n            paths.append(\n                os.path.join(\n                    series_path,\n                    filename\n                )\n            )\n\n    return paths","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.076811Z","iopub.status.idle":"2026-08-22T08:34:42.077105Z","shell.execute_reply.started":"2026-08-22T08:34:42.07698Z","shell.execute_reply":"2026-08-22T08:34:42.076999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_sorted_dicom_paths(\n    study_id,\n    series_id,\n    base_dir\n):\n\n    paths = get_dicom_paths(\n        study_id,\n        series_id,\n        base_dir\n    )\n\n    records = []\n\n    for path in paths:\n\n        try:\n\n            dcm = pydicom.dcmread(\n                path,\n                stop_before_pixels=True\n            )\n\n            instance_number = getattr(\n                dcm,\n                \"InstanceNumber\",\n                0\n            )\n\n            records.append(\n                (\n                    instance_number,\n                    path\n                )\n            )\n\n        except Exception:\n            continue\n\n    records.sort(\n        key=lambda x: x[0]\n    )\n\n    return [\n        path\n        for _, path in records\n    ]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.078486Z","iopub.status.idle":"2026-08-22T08:34:42.078821Z","shell.execute_reply.started":"2026-08-22T08:34:42.078656Z","shell.execute_reply":"2026-08-22T08:34:42.078674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_dicom(\n    path\n):\n\n    dcm = pydicom.dcmread(\n        path\n    )\n\n    image = dcm.pixel_array.astype(\n        np.float32\n    )\n\n    if getattr(\n        dcm,\n        \"PhotometricInterpretation\",\n        \"\"\n    ) == \"MONOCHROME1\":\n\n        image = (\n            image.max() - image\n        )\n\n    # Apply rescale slope/intercept\n    slope = float(\n        getattr(\n            dcm,\n            \"RescaleSlope\",\n            1\n        )\n    )\n\n    intercept = float(\n        getattr(\n            dcm,\n            \"RescaleIntercept\",\n            0\n        )\n    )\n\n    image = (\n        image * slope\n        + intercept\n    )\n\n    # Percentile normalization\n    low = np.percentile(\n        image,\n        1\n    )\n\n    high = np.percentile(\n        image,\n        99\n    )\n\n    if high > low:\n\n        image = (\n            image - low\n        ) / (\n            high - low\n        )\n\n    else:\n\n        image = np.zeros_like(\n            image\n        )\n\n    image = np.clip(\n        image,\n        0,\n        1\n    )\n\n    return image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.079839Z","iopub.status.idle":"2026-08-22T08:34:42.0802Z","shell.execute_reply.started":"2026-08-22T08:34:42.080008Z","shell.execute_reply":"2026-08-22T08:34:42.08003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def select_representative_slices(\n    paths,\n    num_slices=8\n):\n\n    if len(paths) == 0:\n        return []\n\n    if len(paths) <= num_slices:\n\n        return paths\n\n    indices = np.linspace(\n        0,\n        len(paths) - 1,\n        num_slices\n    ).astype(int)\n\n    return [\n        paths[i]\n        for i in indices\n    ]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.081558Z","iopub.status.idle":"2026-08-22T08:34:42.081968Z","shell.execute_reply.started":"2026-08-22T08:34:42.081787Z","shell.execute_reply":"2026-08-22T08:34:42.081807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def select_series(\n    study_id,\n    series_df,\n    max_series=3\n):\n\n    df = series_df[\n        series_df[\n            \"StudyInstanceUID\"\n        ] == study_id\n    ].copy()\n\n    if len(df) == 0:\n        return df\n\n    # Priority:\n    # 1. Fluid-sensitive\n    # 2. Fat suppression\n    # 3. Anatomical plane\n\n    df[\"priority\"] = (\n        df[\"Fluid_Sensitive\"].astype(int)\n        * 2\n        +\n        df[\"Fat_Suppression\"].astype(int)\n    )\n\n    df = df.sort_values(\n        \"priority\",\n        ascending=False\n    )\n\n    return df.head(\n        max_series\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.083328Z","iopub.status.idle":"2026-08-22T08:34:42.083655Z","shell.execute_reply.started":"2026-08-22T08:34:42.083483Z","shell.execute_reply":"2026-08-22T08:34:42.083499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_study = labeled_data.iloc[0][\n    \"StudyInstanceUID\"\n]\n\nprint(\n    \"Sample study:\",\n    sample_study\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.085484Z","iopub.status.idle":"2026-08-22T08:34:42.085824Z","shell.execute_reply.started":"2026-08-22T08:34:42.085699Z","shell.execute_reply":"2026-08-22T08:34:42.085715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_series = select_series(\n    sample_study,\n    train_series,\n    max_series=3\n)\n\ndisplay(\n    sample_series\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.086877Z","iopub.status.idle":"2026-08-22T08:34:42.087177Z","shell.execute_reply.started":"2026-08-22T08:34:42.087047Z","shell.execute_reply":"2026-08-22T08:34:42.087073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"series_id = sample_series.iloc[0][\n    \"SeriesInstanceUID\"\n]\n\npaths = get_sorted_dicom_paths(\n    sample_study,\n    series_id,\n    TRAIN_SERIES_DIR\n)\n\nprint(\n    \"Number of slices:\",\n    len(paths)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.088813Z","iopub.status.idle":"2026-08-22T08:34:42.089158Z","shell.execute_reply.started":"2026-08-22T08:34:42.08904Z","shell.execute_reply":"2026-08-22T08:34:42.089055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_paths = (\n    select_representative_slices(\n        paths,\n        num_slices=8\n    )\n)\n\nfig, axes = plt.subplots(\n    2,\n    4,\n    figsize=(16, 8)\n)\n\nfor ax, path in zip(\n    axes.ravel(),\n    selected_paths\n):\n\n    image = read_dicom(\n        path\n    )\n\n    ax.imshow(\n        image,\n        cmap=\"gray\"\n    )\n\n    ax.axis(\"off\")\n\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.090511Z","iopub.status.idle":"2026-08-22T08:34:42.090881Z","shell.execute_reply.started":"2026-08-22T08:34:42.090704Z","shell.execute_reply":"2026-08-22T08:34:42.09075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in range(\n    min(10, len(train))\n):\n\n    print(\n        \"=\" * 80\n    )\n\n    print(\n        train.iloc[i][\"Report\"]\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.091991Z","iopub.status.idle":"2026-08-22T08:34:42.092746Z","shell.execute_reply.started":"2026-08-22T08:34:42.092507Z","shell.execute_reply":"2026-08-22T08:34:42.092533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def normalize_text(\n    text\n):\n\n    if pd.isna(text):\n        return \"\"\n\n    text = str(text).lower()\n\n    # Normalize accents\n    import unicodedata\n\n    text = unicodedata.normalize(\n        \"NFKD\",\n        text\n    )\n\n    text = \"\".join(\n        char\n        for char in text\n        if not unicodedata.combining(\n            char\n        )\n    )\n\n    # Replace line breaks\n    text = re.sub(\n        r\"\\s+\",\n        \" \",\n        text\n    )\n\n    return text.strip()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.093874Z","iopub.status.idle":"2026-08-22T08:34:42.094279Z","shell.execute_reply.started":"2026-08-22T08:34:42.094154Z","shell.execute_reply":"2026-08-22T08:34:42.09417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"KEYWORDS = {\n\n    \"ACL\": [\n        \"acl\",\n        \"anterior cruciate\",\n        \"ligament croise anterieur\",\n        \"ligamento cruzado anterior\",\n        \"vorderes kreuzband\"\n    ],\n\n    \"MCL\": [\n        \"mcl\",\n        \"medial collateral\",\n        \"ligament collateral medial\",\n        \"ligamento colateral medial\",\n        \"innenband\"\n    ],\n\n    \"Medial Meniscus\": [\n        \"medial meniscus\",\n        \"menisque medial\",\n        \"menisco medial\",\n        \"medialmeniskus\"\n    ],\n\n    \"Lateral Meniscus\": [\n        \"lateral meniscus\",\n        \"menisque lateral\",\n        \"menisco lateral\",\n        \"lateralmeniskus\"\n    ],\n\n    \"Medial OA\": [\n        \"medial osteoarthritis\",\n        \"medial compartment osteoarthritis\",\n        \"medial compartment\",\n        \"osteoarthritis medial\",\n        \"arthrose medial\",\n        \"gonarthrose medial\"\n    ],\n\n    \"Lateral OA\": [\n        \"lateral osteoarthritis\",\n        \"lateral compartment osteoarthritis\",\n        \"osteoarthritis lateral\",\n        \"arthrose lateral\"\n    ],\n\n    \"PF OA\": [\n        \"patellofemoral osteoarthritis\",\n        \"patellofemoral arthrosis\",\n        \"patellofemoral arthritis\",\n        \"pf osteoarthritis\",\n        \"patellofemoral\"\n    ],\n\n    \"Effusion\": [\n        \"joint effusion\",\n        \"effusion\",\n        \"knee effusion\",\n        \"epanchement\",\n        \"derrame\",\n        \"gelenkerguss\"\n    ],\n\n    \"Synovitis\": [\n        \"synovitis\",\n        \"synovial inflammation\",\n        \"synoviale inflammation\"\n    ],\n\n    \"Baker's\": [\n        \"baker cyst\",\n        \"baker's cyst\",\n        \"popliteal cyst\",\n        \"kyste de baker\",\n        \"quiste de baker\",\n        \"bakerzyste\"\n    ],\n\n    \"Contusion\": [\n        \"bone contusion\",\n        \"bone bruise\",\n        \"osseous contusion\",\n        \"bone marrow edema\",\n        \"bone oedema\",\n        \"bone edema\",\n        \"contusion osseuse\"\n    ],\n\n    \"Fracture\": [\n        \"fracture\",\n        \"fractura\",\n        \"fracture osseuse\",\n        \"fraktur\"\n    ]\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.096327Z","iopub.status.idle":"2026-08-22T08:34:42.096581Z","shell.execute_reply.started":"2026-08-22T08:34:42.096473Z","shell.execute_reply":"2026-08-22T08:34:42.096487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"NEGATION_WORDS = [\n    \"no\",\n    \"not\",\n    \"without\",\n    \"normal\",\n    \"intact\",\n    \"negative\",\n    \"absent\",\n    \"aucune\",\n    \"aucun\",\n    \"sans\",\n    \"normal\",\n    \"normale\",\n    \"keine\",\n    \"kein\",\n    \"keinen\",\n    \"keiner\",\n    \"nein\",\n    \"sin\",\n    \"sano\",\n    \"sana\"\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.100748Z","iopub.status.idle":"2026-08-22T08:34:42.101061Z","shell.execute_reply.started":"2026-08-22T08:34:42.100943Z","shell.execute_reply":"2026-08-22T08:34:42.100958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def detect_finding(\n    text,\n    keywords,\n    window=100\n):\n\n    text = normalize_text(\n        text\n    )\n\n    for keyword in keywords:\n\n        keyword = normalize_text(\n            keyword\n        )\n\n        start = 0\n\n        while True:\n\n            position = text.find(\n                keyword,\n                start\n            )\n\n            if position == -1:\n                break\n\n            context_start = max(\n                0,\n                position - window\n            )\n\n            context_end = min(\n                len(text),\n                position + len(keyword) + window\n            )\n\n            context = text[\n                context_start:context_end\n            ]\n\n            # Check negation words\n            negative = any(\n                re.search(\n                    r\"\\b\" +\n                    re.escape(word) +\n                    r\"\\b\",\n                    context\n                )\n                for word in NEGATION_WORDS\n            )\n\n            if not negative:\n\n                return 1\n\n            start = (\n                position\n                + len(keyword)\n            )\n\n    return 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.10234Z","iopub.status.idle":"2026-08-22T08:34:42.103261Z","shell.execute_reply.started":"2026-08-22T08:34:42.103055Z","shell.execute_reply":"2026-08-22T08:34:42.10308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_weak_labels(\n    dataframe\n):\n\n    result = dataframe[\n        [\n            \"StudyInstanceUID\",\n            \"Report\"\n        ]\n    ].copy()\n\n    for target in TARGETS:\n\n        keywords = KEYWORDS[\n            target\n        ]\n\n        result[target] = (\n            result[\"Report\"]\n            .apply(\n                lambda x:\n                detect_finding(\n                    x,\n                    keywords\n                )\n            )\n        )\n\n    return result","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.104455Z","iopub.status.idle":"2026-08-22T08:34:42.10474Z","shell.execute_reply.started":"2026-08-22T08:34:42.10457Z","shell.execute_reply":"2026-08-22T08:34:42.104582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"weak_labels = create_weak_labels(\n    train\n)\n\ndisplay(\n    weak_labels.head()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.105491Z","iopub.status.idle":"2026-08-22T08:34:42.105844Z","shell.execute_reply.started":"2026-08-22T08:34:42.105665Z","shell.execute_reply":"2026-08-22T08:34:42.105686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"weak_counts = (\n    weak_labels[TARGETS]\n    .sum()\n    .sort_values(\n        ascending=False\n    )\n)\n\ndisplay(\n    weak_counts.to_frame(\n        \"Weak Positive Count\"\n    )\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.107287Z","iopub.status.idle":"2026-08-22T08:34:42.107553Z","shell.execute_reply.started":"2026-08-22T08:34:42.107438Z","shell.execute_reply":"2026-08-22T08:34:42.107452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"comparison = pd.DataFrame(\n    index=TARGETS\n)\n\ncomparison[\n    \"Official Positive\"\n] = (\n    labeled_data[TARGETS]\n    .sum()\n)\n\ncomparison[\n    \"Weak Positive\"\n] = (\n    weak_labels[\n        weak_labels[\n            \"StudyInstanceUID\"\n        ].isin(\n            labeled_data[\n                \"StudyInstanceUID\"\n            ]\n        )\n    ][TARGETS]\n    .sum()\n)\n\ndisplay(\n    comparison\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.108778Z","iopub.status.idle":"2026-08-22T08:34:42.109356Z","shell.execute_reply.started":"2026-08-22T08:34:42.10915Z","shell.execute_reply":"2026-08-22T08:34:42.109191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"official = labeled_data[\n    TARGETS\n].copy()\n\nweak_official = weak_labels[\n    weak_labels[\n        \"StudyInstanceUID\"\n    ].isin(\n        labeled_data[\n            \"StudyInstanceUID\"\n        ]\n    )\n].copy()\n\nweak_official = (\n    weak_official\n    .set_index(\n        \"StudyInstanceUID\"\n    )\n    .loc[\n        labeled_data[\n            \"StudyInstanceUID\"\n        ]]\n)\n\nofficial.index = (\n    labeled_data[\n        \"StudyInstanceUID\"\n    ]\n)\n\nagreement = (\n    official[TARGETS]\n    .values\n    ==\n    weak_official[TARGETS]\n    .values\n)\n\nagreement_rate = (\n    agreement.mean(\n        axis=0\n    )\n)\n\nagreement_table = pd.DataFrame({\n    \"Target\": TARGETS,\n    \"Agreement\": agreement_rate\n})\n\ndisplay(\n    agreement_table\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.110311Z","iopub.status.idle":"2026-08-22T08:34:42.110681Z","shell.execute_reply.started":"2026-08-22T08:34:42.110518Z","shell.execute_reply":"2026-08-22T08:34:42.110532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_data = train[\n    [\n        \"StudyInstanceUID\",\n        \"Report\"\n    ]\n].copy()\n\ntraining_data = training_data.merge(\n    weak_labels,\n    on=[\n        \"StudyInstanceUID\",\n        \"Report\"\n    ],\n    how=\"left\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.112006Z","iopub.status.idle":"2026-08-22T08:34:42.112338Z","shell.execute_reply.started":"2026-08-22T08:34:42.112217Z","shell.execute_reply":"2026-08-22T08:34:42.112234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"official_map = labeled_data.set_index(\n    \"StudyInstanceUID\"\n)[TARGETS]\n\nfor target in TARGETS:\n\n    training_data[target] = (\n        training_data[target]\n        .astype(float)\n    )\n\n    for study_id in official_map.index:\n\n        official_value = (\n            official_map.loc[\n                study_id,\n                target\n            ]\n        )\n\n        training_data.loc[\n            training_data[\n                \"StudyInstanceUID\"\n            ] == study_id,\n            target\n        ] = official_value","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.113395Z","iopub.status.idle":"2026-08-22T08:34:42.113753Z","shell.execute_reply.started":"2026-08-22T08:34:42.113542Z","shell.execute_reply":"2026-08-22T08:34:42.113556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\n    training_data[\n        TARGETS\n    ].isna().sum()\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.11497Z","iopub.status.idle":"2026-08-22T08:34:42.115247Z","shell.execute_reply.started":"2026-08-22T08:34:42.115125Z","shell.execute_reply":"2026-08-22T08:34:42.115143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMAGE_SIZE = 224\n\ntrain_transform = transforms.Compose([\n\n    transforms.ToPILImage(),\n\n    transforms.Resize(\n        (\n            IMAGE_SIZE,\n            IMAGE_SIZE\n        )\n    ),\n\n    transforms.RandomHorizontalFlip(\n        p=0.5\n    ),\n\n    transforms.RandomRotation(\n        8\n    ),\n\n    transforms.ToTensor(),\n\n    transforms.Normalize(\n        mean=[\n            0.485,\n            0.456,\n            0.406\n        ],\n        std=[\n            0.229,\n            0.224,\n            0.225\n        ]\n    )\n])\n\nvalid_transform = transforms.Compose([\n\n    transforms.ToPILImage(),\n\n    transforms.Resize(\n        (\n            IMAGE_SIZE,\n            IMAGE_SIZE\n        )\n    ),\n\n    transforms.ToTensor(),\n\n    transforms.Normalize(\n        mean=[\n            0.485,\n            0.456,\n            0.406\n        ],\n        std=[\n            0.229,\n            0.224,\n            0.225\n        ]\n    )\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.116359Z","iopub.status.idle":"2026-08-22T08:34:42.116587Z","shell.execute_reply.started":"2026-08-22T08:34:42.116479Z","shell.execute_reply":"2026-08-22T08:34:42.116492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prepare_image(\n    image,\n    transform\n):\n\n    image = (\n        image * 255\n    ).clip(\n        0,\n        255\n    ).astype(\n        np.uint8\n    )\n\n    image = np.stack(\n        [\n            image,\n            image,\n            image\n        ],\n        axis=-1\n    )\n\n    return transform(\n        image\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.117901Z","iopub.status.idle":"2026-08-22T08:34:42.11842Z","shell.execute_reply.started":"2026-08-22T08:34:42.11828Z","shell.execute_reply":"2026-08-22T08:34:42.118306Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class KneeDataset(\n    Dataset\n):\n\n    def __init__(\n        self,\n        dataframe,\n        series_dataframe,\n        base_dir,\n        targets,\n        transform=None,\n        num_slices=8,\n        max_series=2\n    ):\n\n        self.df = (\n            dataframe\n            .reset_index(\n                drop=True\n            )\n        )\n\n        self.series_df = (\n            series_dataframe\n        )\n\n        self.base_dir = base_dir\n\n        self.targets = targets\n\n        self.transform = transform\n\n        self.num_slices = (\n            num_slices\n        )\n\n        self.max_series = (\n            max_series\n        )\n\n    def __len__(\n        self\n    ):\n\n        return len(\n            self.df\n        )\n\n    def load_images(\n        self,\n        study_id\n    ):\n\n        series_info = (\n            select_series(\n                study_id,\n                self.series_df,\n                self.max_series\n            )\n        )\n\n        images = []\n\n        for _, row in (\n            series_info.iterrows()\n        ):\n\n            series_id = row[\n                \"SeriesInstanceUID\"\n            ]\n\n            paths = (\n                get_sorted_dicom_paths(\n                    study_id,\n                    series_id,\n                    self.base_dir\n                )\n            )\n\n            selected_paths = (\n                select_representative_slices(\n                    paths,\n                    self.num_slices\n                )\n            )\n\n            for path in selected_paths:\n\n                try:\n\n                    image = read_dicom(\n                        path\n                    )\n\n                    if self.transform:\n\n                        image = (\n                            prepare_image(\n                                image,\n                                self.transform\n                            )\n                        )\n\n                    images.append(\n                        image\n                    )\n\n                except Exception:\n                    continue\n\n        return images\n\n    def __getitem__(\n        self,\n        idx\n    ):\n\n        row = self.df.iloc[\n            idx\n        ]\n\n        study_id = row[\n            \"StudyInstanceUID\"\n        ]\n\n        images = self.load_images(\n            study_id\n        )\n\n        if len(images) == 0:\n\n            image = torch.zeros(\n                3,\n                IMAGE_SIZE,\n                IMAGE_SIZE\n            )\n\n        else:\n\n            image = random.choice(\n                images\n            )\n\n        labels = torch.tensor(\n            row[\n                self.targets\n            ].values.astype(\n                np.float32\n            ),\n            dtype=torch.float32\n        )\n\n        return (\n            image,\n            labels\n        )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.11997Z","iopub.status.idle":"2026-08-22T08:34:42.120334Z","shell.execute_reply.started":"2026-08-22T08:34:42.120134Z","shell.execute_reply":"2026-08-22T08:34:42.120153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"official_train, official_valid = (\n    train_test_split(\n        labeled_data,\n        test_size=0.2,\n        random_state=42\n    )\n)\n\nprint(\n    \"Official train:\",\n    len(official_train)\n)\n\nprint(\n    \"Official validation:\",\n    len(official_valid)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.121909Z","iopub.status.idle":"2026-08-22T08:34:42.122277Z","shell.execute_reply.started":"2026-08-22T08:34:42.122065Z","shell.execute_reply":"2026-08-22T08:34:42.122084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"official_train_ids = set(\n    official_train[\n        \"StudyInstanceUID\"\n    ]\n)\n\nofficial_valid_ids = set(\n    official_valid[\n        \"StudyInstanceUID\"\n    ]\n)\n\n# Remove validation studies\n# from pseudo-labelled training\n\npseudo_training = training_data[\n    ~training_data[\n        \"StudyInstanceUID\"\n    ].isin(\n        official_valid_ids\n    )\n].copy()\n\nprint(\n    \"Training studies:\",\n    len(pseudo_training)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.123479Z","iopub.status.idle":"2026-08-22T08:34:42.123832Z","shell.execute_reply.started":"2026-08-22T08:34:42.123659Z","shell.execute_reply":"2026-08-22T08:34:42.123679Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = KneeDataset(\n    dataframe=pseudo_training,\n    series_dataframe=train_series,\n    base_dir=TRAIN_SERIES_DIR,\n    targets=TARGETS,\n    transform=train_transform,\n    num_slices=8,\n    max_series=2\n)\n\nvalid_dataset = KneeDataset(\n    dataframe=official_valid,\n    series_dataframe=train_series,\n    base_dir=TRAIN_SERIES_DIR,\n    targets=TARGETS,\n    transform=valid_transform,\n    num_slices=8,\n    max_series=2\n)\n\nprint(\n    \"Training dataset:\",\n    len(train_dataset)\n)\n\nprint(\n    \"Validation dataset:\",\n    len(valid_dataset)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.12491Z","iopub.status.idle":"2026-08-22T08:34:42.125129Z","shell.execute_reply.started":"2026-08-22T08:34:42.125022Z","shell.execute_reply":"2026-08-22T08:34:42.125035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BATCH_SIZE = 8\n\ntrain_loader = DataLoader(\n    train_dataset,\n    batch_size=BATCH_SIZE,\n    shuffle=True,\n    num_workers=2,\n    pin_memory=True\n)\n\nvalid_loader = DataLoader(\n    valid_dataset,\n    batch_size=BATCH_SIZE,\n    shuffle=False,\n    num_workers=2,\n    pin_memory=True\n)\n\nprint(\n    \"DataLoaders ready.\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.126365Z","iopub.status.idle":"2026-08-22T08:34:42.126873Z","shell.execute_reply.started":"2026-08-22T08:34:42.126687Z","shell.execute_reply":"2026-08-22T08:34:42.126718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"images, labels = next(\n    iter(train_loader)\n)\n\nprint(\n    \"Images:\",\n    images.shape\n)\n\nprint(\n    \"Labels:\",\n    labels.shape\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.127687Z","iopub.status.idle":"2026-08-22T08:34:42.128118Z","shell.execute_reply.started":"2026-08-22T08:34:42.127927Z","shell.execute_reply":"2026-08-22T08:34:42.127949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class KneeResNet(\n    nn.Module\n):\n\n    def __init__(\n        self,\n        num_classes=12\n    ):\n\n        super().__init__()\n\n        self.backbone = resnet50(\n            weights=(\n                ResNet50_Weights.DEFAULT\n            )\n        )\n\n        features = (\n            self.backbone.fc.in_features\n        )\n\n        self.backbone.fc = nn.Linear(\n            features,\n            num_classes\n        )\n\n    def forward(\n        self,\n        x\n    ):\n\n        return self.backbone(\n            x\n        )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.129329Z","iopub.status.idle":"2026-08-22T08:34:42.129557Z","shell.execute_reply.started":"2026-08-22T08:34:42.129447Z","shell.execute_reply":"2026-08-22T08:34:42.129461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = KneeResNet(\n    num_classes=NUM_CLASSES\n)\n\nmodel = model.to(\n    DEVICE\n)\n\nprint(\n    \"Model ready.\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.130455Z","iopub.status.idle":"2026-08-22T08:34:42.130839Z","shell.execute_reply.started":"2026-08-22T08:34:42.130632Z","shell.execute_reply":"2026-08-22T08:34:42.130654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"criterion = (\n    nn.BCEWithLogitsLoss()\n)\n\noptimizer = torch.optim.AdamW(\n    model.parameters(),\n    lr=1e-4,\n    weight_decay=1e-4\n)\n\nscheduler = (\n    torch.optim.lr_scheduler\n    .ReduceLROnPlateau(\n        optimizer,\n        mode=\"max\",\n        factor=0.5,\n        patience=1\n    )\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.131851Z","iopub.status.idle":"2026-08-22T08:34:42.132074Z","shell.execute_reply.started":"2026-08-22T08:34:42.131963Z","shell.execute_reply":"2026-08-22T08:34:42.131976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_one_epoch(\n    model,\n    loader,\n    criterion,\n    optimizer,\n    device\n):\n\n    model.train()\n\n    total_loss = 0\n\n    progress = tqdm(\n        loader,\n        desc=\"Training\"\n    )\n\n    for images, labels in progress:\n\n        images = images.to(\n            device,\n            non_blocking=True\n        )\n\n        labels = labels.to(\n            device,\n            non_blocking=True\n        )\n\n        optimizer.zero_grad(\n            set_to_none=True\n        )\n\n        logits = model(\n            images\n        )\n\n        loss = criterion(\n            logits,\n            labels\n        )\n\n        loss.backward()\n\n        optimizer.step()\n\n        total_loss += (\n            loss.item()\n        )\n\n        progress.set_postfix(\n            loss=round(\n                loss.item(),\n                4\n            )\n        )\n\n    return (\n        total_loss /\n        len(loader)\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.132971Z","iopub.status.idle":"2026-08-22T08:34:42.133198Z","shell.execute_reply.started":"2026-08-22T08:34:42.133088Z","shell.execute_reply":"2026-08-22T08:34:42.133101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def validate(\n    model,\n    loader,\n    criterion,\n    device\n):\n\n    model.eval()\n\n    total_loss = 0\n\n    all_predictions = []\n\n    all_labels = []\n\n    with torch.no_grad():\n\n        for images, labels in tqdm(\n            loader,\n            desc=\"Validation\"\n        ):\n\n            images = images.to(\n                device,\n                non_blocking=True\n            )\n\n            labels = labels.to(\n                device,\n                non_blocking=True\n            )\n\n            logits = model(\n                images\n            )\n\n            loss = criterion(\n                logits,\n                labels\n            )\n\n            total_loss += (\n                loss.item()\n            )\n\n            probabilities = (\n                torch.sigmoid(\n                    logits\n                )\n            )\n\n            all_predictions.append(\n                probabilities\n                .cpu()\n                .numpy()\n            )\n\n            all_labels.append(\n                labels\n                .cpu()\n                .numpy()\n            )\n\n    predictions = np.concatenate(\n        all_predictions,\n        axis=0\n    )\n\n    labels = np.concatenate(\n        all_labels,\n        axis=0\n    )\n\n    auc_scores = []\n\n    for i in range(\n        NUM_CLASSES\n    ):\n\n        try:\n\n            auc = roc_auc_score(\n                labels[:, i],\n                predictions[:, i]\n            )\n\n        except ValueError:\n\n            auc = np.nan\n\n        auc_scores.append(\n            auc\n        )\n\n    mean_auc = np.nanmean(\n        auc_scores\n    )\n\n    avg_loss = (\n        total_loss /\n        len(loader)\n    )\n\n    return (\n        avg_loss,\n        mean_auc,\n        auc_scores,\n        predictions,\n        labels\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.134115Z","iopub.status.idle":"2026-08-22T08:34:42.13445Z","shell.execute_reply.started":"2026-08-22T08:34:42.134274Z","shell.execute_reply":"2026-08-22T08:34:42.134294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"EPOCHS = 3\n\nbest_auc = -np.inf\n\nhistory = {\n    \"train_loss\": [],\n    \"valid_loss\": [],\n    \"valid_auc\": []\n}\n\nfor epoch in range(\n    EPOCHS\n):\n\n    print(\n        \"\\n\"\n        + \"=\" * 60\n    )\n\n    print(\n        f\"Epoch {epoch + 1}/{EPOCHS}\"\n    )\n\n    print(\n        \"=\" * 60\n    )\n\n    train_loss = train_one_epoch(\n        model,\n        train_loader,\n        criterion,\n        optimizer,\n        DEVICE\n    )\n\n    (\n        valid_loss,\n        valid_auc,\n        auc_scores,\n        predictions,\n        labels\n    ) = validate(\n        model,\n        valid_loader,\n        criterion,\n        DEVICE\n    )\n\n    scheduler.step(\n        valid_auc\n    )\n\n    history[\n        \"train_loss\"\n    ].append(\n        train_loss\n    )\n\n    history[\n        \"valid_loss\"\n    ].append(\n        valid_loss\n    )\n\n    history[\n        \"valid_auc\"\n    ].append(\n        valid_auc\n    )\n\n    print(\n        f\"Train Loss: \"\n        f\"{train_loss:.4f}\"\n    )\n\n    print(\n        f\"Validation Loss: \"\n        f\"{valid_loss:.4f}\"\n    )\n\n    print(\n        f\"Mean ROC-AUC: \"\n        f\"{valid_auc:.4f}\"\n    )\n\n    if valid_auc > best_auc:\n\n        best_auc = valid_auc\n\n        torch.save(\n            model.state_dict(),\n            \"/kaggle/working/\"\n            \"best_knee_model.pth\"\n        )\n\n        print(\n            \"Best model saved.\"\n        )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.135666Z","iopub.status.idle":"2026-08-22T08:34:42.135897Z","shell.execute_reply.started":"2026-08-22T08:34:42.135789Z","shell.execute_reply":"2026-08-22T08:34:42.135803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"auc_table = pd.DataFrame({\n\n    \"Abnormality\": TARGETS,\n\n    \"ROC_AUC\": auc_scores\n\n})\n\ndisplay(\n    auc_table.sort_values(\n        \"ROC_AUC\",\n        ascending=False\n    )\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.136984Z","iopub.status.idle":"2026-08-22T08:34:42.137346Z","shell.execute_reply.started":"2026-08-22T08:34:42.137213Z","shell.execute_reply":"2026-08-22T08:34:42.137238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(\n    figsize=(10, 5)\n)\n\nplt.plot(\n    history[\"train_loss\"],\n    marker=\"o\",\n    label=\"Train Loss\"\n)\n\nplt.plot(\n    history[\"valid_loss\"],\n    marker=\"o\",\n    label=\"Validation Loss\"\n)\n\nplt.xlabel(\n    \"Epoch\"\n)\n\nplt.ylabel(\n    \"Loss\"\n)\n\nplt.title(\n    \"Training History\"\n)\n\nplt.legend()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.138379Z","iopub.status.idle":"2026-08-22T08:34:42.138782Z","shell.execute_reply.started":"2026-08-22T08:34:42.138655Z","shell.execute_reply":"2026-08-22T08:34:42.138671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(\n    figsize=(10, 5)\n)\n\nplt.plot(\n    history[\"valid_auc\"],\n    marker=\"o\"\n)\n\nplt.xlabel(\n    \"Epoch\"\n)\n\nplt.ylabel(\n    \"ROC-AUC\"\n)\n\nplt.title(\n    \"Validation ROC-AUC\"\n)\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.139997Z","iopub.status.idle":"2026-08-22T08:34:42.140345Z","shell.execute_reply.started":"2026-08-22T08:34:42.140145Z","shell.execute_reply":"2026-08-22T08:34:42.140159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.load_state_dict(\n    torch.load(\n        \"/kaggle/working/\"\n        \"best_knee_model.pth\",\n        map_location=DEVICE\n    )\n)\n\nmodel.eval()\n\nprint(\n    \"Best model loaded.\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.142268Z","iopub.status.idle":"2026-08-22T08:34:42.142499Z","shell.execute_reply.started":"2026-08-22T08:34:42.142385Z","shell.execute_reply":"2026-08-22T08:34:42.142398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class KneeTestDataset(\n    Dataset\n):\n\n    def __init__(\n        self,\n        dataframe,\n        series_dataframe,\n        base_dir,\n        transform=None,\n        num_slices=8,\n        max_series=2\n    ):\n\n        self.df = (\n            dataframe\n            .reset_index(\n                drop=True\n            )\n        )\n\n        self.series_df = (\n            series_dataframe\n        )\n\n        self.base_dir = base_dir\n\n        self.transform = transform\n\n        self.num_slices = (\n            num_slices\n        )\n\n        self.max_series = (\n            max_series\n        )\n\n    def __len__(\n        self\n    ):\n\n        return len(\n            self.df\n        )\n\n    def __getitem__(\n        self,\n        idx\n    ):\n\n        row = self.df.iloc[\n            idx\n        ]\n\n        study_id = row[\n            \"StudyInstanceUID\"\n        ]\n\n        series_info = select_series(\n            study_id,\n            self.series_df,\n            self.max_series\n        )\n\n        images = []\n\n        for _, series_row in (\n            series_info.iterrows()\n        ):\n\n            series_id = (\n                series_row[\n                    \"SeriesInstanceUID\"\n                ]\n            )\n\n            paths = (\n                get_sorted_dicom_paths(\n                    study_id,\n                    series_id,\n                    self.base_dir\n                )\n            )\n\n            selected_paths = (\n                select_representative_slices(\n                    paths,\n                    self.num_slices\n                )\n            )\n\n            for path in selected_paths:\n\n                try:\n\n                    image = read_dicom(\n                        path\n                    )\n\n                    image = prepare_image(\n                        image,\n                        self.transform\n                    )\n\n                    images.append(\n                        image\n                    )\n\n                except Exception:\n                    continue\n\n        if len(images) == 0:\n\n            image = torch.zeros(\n                3,\n                IMAGE_SIZE,\n                IMAGE_SIZE\n            )\n\n        else:\n\n            image = random.choice(\n                images\n            )\n\n        return (\n            image,\n            study_id\n        )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.144219Z","iopub.status.idle":"2026-08-22T08:34:42.144727Z","shell.execute_reply.started":"2026-08-22T08:34:42.144524Z","shell.execute_reply":"2026-08-22T08:34:42.144547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_dataset = KneeTestDataset(\n    dataframe=test,\n    series_dataframe=test_series,\n    base_dir=TEST_SERIES_DIR,\n    transform=valid_transform,\n    num_slices=8,\n    max_series=2\n)\n\ntest_loader = DataLoader(\n    test_dataset,\n    batch_size=8,\n    shuffle=False,\n    num_workers=2,\n    pin_memory=True\n)\n\nprint(\n    \"Test studies:\",\n    len(test_dataset)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.145355Z","iopub.status.idle":"2026-08-22T08:34:42.145714Z","shell.execute_reply.started":"2026-08-22T08:34:42.14552Z","shell.execute_reply":"2026-08-22T08:34:42.14554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.eval()\n\ntest_predictions = []\n\ntest_ids = []\n\nwith torch.no_grad():\n\n    for images, study_ids in tqdm(\n        test_loader,\n        desc=\"Predicting\"\n    ):\n\n        images = images.to(\n            DEVICE,\n            non_blocking=True\n        )\n\n        logits = model(\n            images\n        )\n\n        probabilities = (\n            torch.sigmoid(\n                logits\n            )\n        )\n\n        test_predictions.append(\n            probabilities\n            .cpu()\n            .numpy()\n        )\n\n        test_ids.extend(\n            list(study_ids)\n        )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.146483Z","iopub.status.idle":"2026-08-22T08:34:42.147019Z","shell.execute_reply.started":"2026-08-22T08:34:42.146832Z","shell.execute_reply":"2026-08-22T08:34:42.146855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_predictions = np.concatenate(\n    test_predictions,\n    axis=0\n)\n\nsubmission = pd.DataFrame(\n    test_predictions,\n    columns=TARGETS\n)\n\nsubmission.insert(\n    0,\n    \"StudyInstanceUID\",\n    test_ids\n)\n\ndisplay(\n    submission.head()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.148002Z","iopub.status.idle":"2026-08-22T08:34:42.148326Z","shell.execute_reply.started":"2026-08-22T08:34:42.148158Z","shell.execute_reply":"2026-08-22T08:34:42.148179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\n    \"Minimum prediction:\",\n    submission[TARGETS]\n    .min()\n    .min()\n)\n\nprint(\n    \"Maximum prediction:\",\n    submission[TARGETS]\n    .max()\n    .max()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.149568Z","iopub.status.idle":"2026-08-22T08:34:42.149918Z","shell.execute_reply.started":"2026-08-22T08:34:42.149746Z","shell.execute_reply":"2026-08-22T08:34:42.149766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\n    \"Sample columns:\"\n)\n\nprint(\n    sample_submission.columns.tolist()\n)\n\nprint(\n    \"\\nOur columns:\"\n)\n\nprint(\n    submission.columns.tolist()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.151078Z","iopub.status.idle":"2026-08-22T08:34:42.151405Z","shell.execute_reply.started":"2026-08-22T08:34:42.151235Z","shell.execute_reply":"2026-08-22T08:34:42.151256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = submission[\n    sample_submission.columns\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.152796Z","iopub.status.idle":"2026-08-22T08:34:42.153145Z","shell.execute_reply.started":"2026-08-22T08:34:42.152963Z","shell.execute_reply":"2026-08-22T08:34:42.152984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SUBMISSION_PATH = (\n    \"/kaggle/working/\"\n    \"submission.csv\"\n)\n\nsubmission.to_csv(\n    SUBMISSION_PATH,\n    index=False\n)\n\nprint(\n    \"Submission saved:\"\n)\n\nprint(\n    SUBMISSION_PATH\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.154276Z","iopub.status.idle":"2026-08-22T08:34:42.154651Z","shell.execute_reply.started":"2026-08-22T08:34:42.154476Z","shell.execute_reply":"2026-08-22T08:34:42.154491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_check = pd.read_csv(\n    SUBMISSION_PATH\n)\n\nassert list(\n    submission_check.columns\n) == list(\n    sample_submission.columns\n)\n\nassert len(\n    submission_check\n) == len(\n    sample_submission\n)\n\nassert (\n    submission_check[\n        TARGETS\n    ].isna()\n    .sum()\n    .sum()\n    == 0\n)\n\nassert (\n    submission_check[\n        TARGETS\n    ].values >= 0\n).all()\n\nassert (\n    submission_check[\n        TARGETS\n    ].values <= 1\n).all()\n\nprint(\n    \"================================\"\n)\n\nprint(\n    \"SUBMISSION VALIDATION PASSED\"\n)\n\nprint(\n    \"================================\"\n)\n\ndisplay(\n    submission_check.head()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.155436Z","iopub.status.idle":"2026-08-22T08:34:42.155787Z","shell.execute_reply.started":"2026-08-22T08:34:42.15557Z","shell.execute_reply":"2026-08-22T08:34:42.155597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# FINAL SUBMISSION\n\nsubmission = submission[\n    sample_submission.columns\n].copy()\n\nsubmission.to_csv(\n    \"/kaggle/working/submission.csv\",\n    index=False\n)\n\nprint(\"submission.csv created successfully!\")\nprint(submission.shape)\ndisplay(submission.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.156734Z","iopub.status.idle":"2026-08-22T08:34:42.157018Z","shell.execute_reply.started":"2026-08-22T08:34:42.156894Z","shell.execute_reply":"2026-08-22T08:34:42.156912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\npath = \"/kaggle/working/submission.csv\"\n\nprint(\"Exists:\", os.path.exists(path))\n\nif os.path.exists(path):\n    print(\"File size:\", os.path.getsize(path), \"bytes\")\n    print(\"Ready for Kaggle submission!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:42.157975Z","iopub.status.idle":"2026-08-22T08:34:42.158274Z","shell.execute_reply.started":"2026-08-22T08:34:42.15816Z","shell.execute_reply":"2026-08-22T08:34:42.158175Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}