{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# # Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# # Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\n# import kagglehub\n# # kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:33.174662Z","iopub.execute_input":"2026-08-17T09:48:33.175295Z","iopub.status.idle":"2026-08-17T09:48:33.180702Z","shell.execute_reply.started":"2026-08-17T09:48:33.175265Z","shell.execute_reply":"2026-08-17T09:48:33.179694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from kaggle_secrets import UserSecretsClient\nimport os\n\n# Fetch the token from Kaggle Secrets\nuser_secrets = UserSecretsClient()\nhf_token = user_secrets.get_secret(\"HF_TOKEN\")\n\n# Set it as an environment variable so Hugging Face detects it automatically\nos.environ[\"HF_TOKEN\"] = hf_token","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:33.18219Z","iopub.execute_input":"2026-08-17T09:48:33.18267Z","iopub.status.idle":"2026-08-17T09:48:33.277771Z","shell.execute_reply.started":"2026-08-17T09:48:33.182607Z","shell.execute_reply":"2026-08-17T09:48:33.277258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nDATA_DIR = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\nos.listdir(DATA_DIR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:33.278544Z","iopub.execute_input":"2026-08-17T09:48:33.279094Z","iopub.status.idle":"2026-08-17T09:48:33.284667Z","shell.execute_reply.started":"2026-08-17T09:48:33.279053Z","shell.execute_reply":"2026-08-17T09:48:33.283945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ntrain = pd.read_csv(f\"{DATA_DIR}/train.csv\")\nseries = pd.read_csv(f\"{DATA_DIR}/train_series.csv\")\n\nprint(train.shape)\nprint(series.shape)\n\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:33.286723Z","iopub.execute_input":"2026-08-17T09:48:33.287084Z","iopub.status.idle":"2026-08-17T09:48:33.728333Z","shell.execute_reply.started":"2026-08-17T09:48:33.287052Z","shell.execute_reply":"2026-08-17T09:48:33.727647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install pydicom","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:33.729263Z","iopub.execute_input":"2026-08-17T09:48:33.729657Z","iopub.status.idle":"2026-08-17T09:48:37.147649Z","shell.execute_reply.started":"2026-08-17T09:48:33.729618Z","shell.execute_reply":"2026-08-17T09:48:37.146941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nimport matplotlib.pyplot as plt\nfrom pathlib import Path","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:37.148958Z","iopub.execute_input":"2026-08-17T09:48:37.149748Z","iopub.status.idle":"2026-08-17T09:48:37.430896Z","shell.execute_reply.started":"2026-08-17T09:48:37.149701Z","shell.execute_reply":"2026-08-17T09:48:37.42996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"study = study = train.iloc[0][\"StudyInstanceUID\"]\n\nstudy_path = Path(DATA_DIR) / \"train_series\" / study\nprint(study_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:37.431889Z","iopub.execute_input":"2026-08-17T09:48:37.432243Z","iopub.status.idle":"2026-08-17T09:48:37.437747Z","shell.execute_reply.started":"2026-08-17T09:48:37.43222Z","shell.execute_reply":"2026-08-17T09:48:37.436719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"series_folder = list(study_path.iterdir())[0]\nprint(series_folder)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:37.43869Z","iopub.execute_input":"2026-08-17T09:48:37.439024Z","iopub.status.idle":"2026-08-17T09:48:37.456533Z","shell.execute_reply.started":"2026-08-17T09:48:37.438988Z","shell.execute_reply":"2026-08-17T09:48:37.455821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dicom_file = list(series_folder.glob(\"*.dcm\"))\ndicom_file","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:37.457537Z","iopub.execute_input":"2026-08-17T09:48:37.457807Z","iopub.status.idle":"2026-08-17T09:48:37.473075Z","shell.execute_reply.started":"2026-08-17T09:48:37.457777Z","shell.execute_reply":"2026-08-17T09:48:37.472294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\n\ndcm = pydicom.dcmread(dicom_file[0])\n\nprint(dcm)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:37.475469Z","iopub.execute_input":"2026-08-17T09:48:37.475704Z","iopub.status.idle":"2026-08-17T09:48:37.488507Z","shell.execute_reply.started":"2026-08-17T09:48:37.475644Z","shell.execute_reply":"2026-08-17T09:48:37.487718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# dicom_file = list(series_folder.glob(\"*.dcm\"))[0]\n# print(dicom_file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:37.489423Z","iopub.execute_input":"2026-08-17T09:48:37.489631Z","iopub.status.idle":"2026-08-17T09:48:37.493044Z","shell.execute_reply.started":"2026-08-17T09:48:37.489613Z","shell.execute_reply":"2026-08-17T09:48:37.492207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# len(list(series_folder.glob(\"*.dcm\")))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:37.493948Z","iopub.execute_input":"2026-08-17T09:48:37.494206Z","iopub.status.idle":"2026-08-17T09:48:37.507102Z","shell.execute_reply.started":"2026-08-17T09:48:37.494185Z","shell.execute_reply":"2026-08-17T09:48:37.506272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ds = pydicom.dcmread(dicom_file)\n# plt.imshow(ds.pixel_array, cmap=\"gray\")\n# plt.axis(\"off\")\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:37.508106Z","iopub.execute_input":"2026-08-17T09:48:37.508707Z","iopub.status.idle":"2026-08-17T09:48:37.521455Z","shell.execute_reply.started":"2026-08-17T09:48:37.508657Z","shell.execute_reply":"2026-08-17T09:48:37.520357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ntrain = pd.read_csv(f\"{DATA_DIR}/train.csv\")\ntrain_series = pd.read_csv(f\"{DATA_DIR}/train_series.csv\")\n\ntest = pd.read_csv(f\"{DATA_DIR}/test.csv\")\ntest_series = pd.read_csv(f\"{DATA_DIR}/test_series.csv\")\n\nprint(f\"Train -> {train.shape}\")\nprint(f\"Test  -> {test.shape}\")\nprint(f\"Train Series -> {train_series.shape}\")\nprint(f\"Test Series  -> {test_series.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:37.522404Z","iopub.execute_input":"2026-08-17T09:48:37.522596Z","iopub.status.idle":"2026-08-17T09:48:37.691164Z","shell.execute_reply.started":"2026-08-17T09:48:37.522576Z","shell.execute_reply":"2026-08-17T09:48:37.690272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"study = train.iloc[0][\"StudyInstanceUID\"]\n\nstudy_path = Path(DATA_DIR) / \"train_series\" / study\n\nseries_folder = list(study_path.iterdir())[0]\n\ndicom_file = list(series_folder.glob(\"*.dcm\"))\nprint(f\"series{1} ->\" ,len(dicom_file))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:37.692265Z","iopub.execute_input":"2026-08-17T09:48:37.692574Z","iopub.status.idle":"2026-08-17T09:48:37.7005Z","shell.execute_reply.started":"2026-08-17T09:48:37.692552Z","shell.execute_reply":"2026-08-17T09:48:37.699771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"list(study_path.iterdir())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:37.701627Z","iopub.execute_input":"2026-08-17T09:48:37.701991Z","iopub.status.idle":"2026-08-17T09:48:37.718257Z","shell.execute_reply.started":"2026-08-17T09:48:37.701892Z","shell.execute_reply":"2026-08-17T09:48:37.717493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"K = list(study_path.iterdir())\nfor i in range(len(K)):\n    series_folder = list(study_path.iterdir())[i]\n    dicom_file = list(series_folder.glob(\"*.dcm\"))\n    print(f\"series{i} ->\" ,len(dicom_file))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:37.719228Z","iopub.execute_input":"2026-08-17T09:48:37.720009Z","iopub.status.idle":"2026-08-17T09:48:37.740827Z","shell.execute_reply.started":"2026-08-17T09:48:37.719985Z","shell.execute_reply":"2026-08-17T09:48:37.740181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\n\nfor j in range(10):\n    x = random.randint(0, 4407)\n    study = train.iloc[x][\"StudyInstanceUID\"]\n    print(f\"Study---->{x}\")\n    study_path = Path(DATA_DIR) / \"train_series\" / study\n    K = list(study_path.iterdir())\n    for i in range(len(K)):\n        series_folder = list(study_path.iterdir())[i]\n        dicom_file = list(series_folder.glob(\"*.dcm\"))\n        print(f\"series{i} ->\" ,len(dicom_file))\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:37.741637Z","iopub.execute_input":"2026-08-17T09:48:37.742332Z","iopub.status.idle":"2026-08-17T09:48:38.248063Z","shell.execute_reply.started":"2026-08-17T09:48:37.742302Z","shell.execute_reply":"2026-08-17T09:48:38.247313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_series.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.248889Z","iopub.execute_input":"2026-08-17T09:48:38.249258Z","iopub.status.idle":"2026-08-17T09:48:38.259421Z","shell.execute_reply.started":"2026-08-17T09:48:38.249235Z","shell.execute_reply":"2026-08-17T09:48:38.258561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nfrom pathlib import Path\n\nfor i, series_folder in enumerate(study_path.iterdir()):\n\n    dicom_files = list(series_folder.glob(\"*.dcm\"))\n\n    if len(dicom_files) == 0:\n        continue\n\n    ds = pydicom.dcmread(dicom_files[0])\n\n    print(\"=\" * 80)\n    print(f\"Series {i}\")\n    print(\"Folder:\", series_folder.name)\n    print(\"Number of slices:\", len(dicom_files))\n\n    print(\"Series Number:\",\n          getattr(ds, \"SeriesNumber\", None))\n\n    print(\"Series Description:\",\n          getattr(ds, \"SeriesDescription\", None))\n\n    print(\"Modality:\",\n          getattr(ds, \"Modality\", None))\n\n    print(\"MR Acquisition Type:\",\n          getattr(ds, \"MRAcquisitionType\", None))\n\n    print(\"Rows:\",\n          getattr(ds, \"Rows\", None))\n\n    print(\"Columns:\",\n          getattr(ds, \"Columns\", None))\n\n    print(\"Pixel Spacing:\",\n          getattr(ds, \"PixelSpacing\", None))\n\n    print(\"Slice Thickness:\",\n          getattr(ds, \"SliceThickness\", None))\n\n    print(\"Spacing Between Slices:\",\n          getattr(ds, \"SpacingBetweenSlices\", None))\n\n    print(\"Image Orientation:\",\n          getattr(ds, \"ImageOrientationPatient\", None))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.260294Z","iopub.execute_input":"2026-08-17T09:48:38.260541Z","iopub.status.idle":"2026-08-17T09:48:38.324684Z","shell.execute_reply.started":"2026-08-17T09:48:38.260519Z","shell.execute_reply":"2026-08-17T09:48:38.323886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.shape)\ntrain.sample(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.325678Z","iopub.execute_input":"2026-08-17T09:48:38.325982Z","iopub.status.idle":"2026-08-17T09:48:38.339711Z","shell.execute_reply.started":"2026-08-17T09:48:38.325952Z","shell.execute_reply":"2026-08-17T09:48:38.338948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_series.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.340982Z","iopub.execute_input":"2026-08-17T09:48:38.341289Z","iopub.status.idle":"2026-08-17T09:48:38.357669Z","shell.execute_reply.started":"2026-08-17T09:48:38.341254Z","shell.execute_reply":"2026-08-17T09:48:38.35697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"common = train_series['SeriesInstanceUID'].isin(test_series['SeriesInstanceUID']).any()\n\nprint(common)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.358537Z","iopub.execute_input":"2026-08-17T09:48:38.358835Z","iopub.status.idle":"2026-08-17T09:48:38.374675Z","shell.execute_reply.started":"2026-08-17T09:48:38.358809Z","shell.execute_reply":"2026-08-17T09:48:38.373867Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.375695Z","iopub.execute_input":"2026-08-17T09:48:38.376734Z","iopub.status.idle":"2026-08-17T09:48:38.39275Z","shell.execute_reply.started":"2026-08-17T09:48:38.376708Z","shell.execute_reply":"2026-08-17T09:48:38.391871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.sample(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.393642Z","iopub.execute_input":"2026-08-17T09:48:38.394413Z","iopub.status.idle":"2026-08-17T09:48:38.411586Z","shell.execute_reply.started":"2026-08-17T09:48:38.39436Z","shell.execute_reply":"2026-08-17T09:48:38.410795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TARGETS = [\n    \"ACL\",\n    \"MCL\",\n    \"Medial Meniscus\",\n    \"Lateral Meniscus\",\n    \"Medial OA\",\n    \"Lateral OA\",\n    \"PF OA\",\n    \"Effusion\",\n    \"Synovitis\",\n    \"Baker's\",\n    \"Contusion\",\n    \"Fracture\"\n]\n\nprint(\"Number of target labels:\", len(TARGETS))\n\nprint(\"\\nLabeled studies:\")\nprint(train[TARGETS].notna().any(axis=1).sum())\n\nprint(\"\\nCompletely unlabeled studies:\")\nprint(train[TARGETS].isna().all(axis=1).sum())\n\nprint(\"\\nLabel coverage:\")\nfor target in TARGETS:\n    print(\n        f\"{target:20s} -> \"\n        f\"{train[target].notna().sum()} labeled\"\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.412619Z","iopub.execute_input":"2026-08-17T09:48:38.412864Z","iopub.status.idle":"2026-08-17T09:48:38.431869Z","shell.execute_reply.started":"2026-08-17T09:48:38.412842Z","shell.execute_reply":"2026-08-17T09:48:38.430946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labeled_train = train[\n    train[TARGETS].notna().any(axis=1)\n].copy()\n\nprint(labeled_train.shape)\n\nlabeled_train[TARGETS].sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.432768Z","iopub.execute_input":"2026-08-17T09:48:38.433077Z","iopub.status.idle":"2026-08-17T09:48:38.447519Z","shell.execute_reply.started":"2026-08-17T09:48:38.433043Z","shell.execute_reply":"2026-08-17T09:48:38.446722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for target in TARGETS:\n    print(\"\\n\", target)\n    print(labeled_train[target].value_counts(dropna=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.451005Z","iopub.execute_input":"2026-08-17T09:48:38.451367Z","iopub.status.idle":"2026-08-17T09:48:38.470721Z","shell.execute_reply.started":"2026-08-17T09:48:38.451342Z","shell.execute_reply":"2026-08-17T09:48:38.469794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labeled_df = train[\n    train[TARGETS].notna().any(axis=1)\n].copy()\n\nunlabeled_df = train[\n    train[TARGETS].isna().all(axis=1)\n].copy()\n\nprint(\"Labeled:\", labeled_df.shape)\nprint(\"Unlabeled:\", unlabeled_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.471878Z","iopub.execute_input":"2026-08-17T09:48:38.472201Z","iopub.status.idle":"2026-08-17T09:48:38.491798Z","shell.execute_reply.started":"2026-08-17T09:48:38.472169Z","shell.execute_reply":"2026-08-17T09:48:38.491022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option(\"display.max_colwidth\", 1000)\n\nprint(\"LABELED REPORT\")\nprint(labeled_df[[\"StudyInstanceUID\", \"Report\"]].head(3).to_string(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.492893Z","iopub.execute_input":"2026-08-17T09:48:38.493284Z","iopub.status.idle":"2026-08-17T09:48:38.504775Z","shell.execute_reply.started":"2026-08-17T09:48:38.49325Z","shell.execute_reply":"2026-08-17T09:48:38.503989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\nUNLABELED REPORT\")\nprint(unlabeled_df[[\"StudyInstanceUID\", \"Report\"]].head(3).to_string(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.505638Z","iopub.execute_input":"2026-08-17T09:48:38.505978Z","iopub.status.idle":"2026-08-17T09:48:38.521978Z","shell.execute_reply.started":"2026-08-17T09:48:38.505947Z","shell.execute_reply":"2026-08-17T09:48:38.521206Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Train for unlabled data: -**","metadata":{}},{"cell_type":"code","source":"# import os\n# import requests\n\n# print(\"HF endpoint test:\")\n\n# url = \"https://huggingface.co/joeddav/xlm-roberta-large-xnli/resolve/main/config.json\"\n\n# try:\n#     r = requests.get(url, timeout=30)\n#     print(\"Status:\", r.status_code)\n#     print(\"Size:\", len(r.content))\n# except Exception as e:\n#     print(\"ERROR:\", repr(e))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.522934Z","iopub.execute_input":"2026-08-17T09:48:38.523225Z","iopub.status.idle":"2026-08-17T09:48:38.535313Z","shell.execute_reply.started":"2026-08-17T09:48:38.523193Z","shell.execute_reply":"2026-08-17T09:48:38.534609Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import pandas as pd\n# import numpy as np\n# from transformers import pipeline\n# from sklearn.metrics import accuracy_score, roc_auc_score\n# from tqdm.auto import tqdm\n\n# # 1. Initialize the multilingual zero-shot classifier \n# print(\"Loading XLM-RoBERTa Multilingual Classifier...\")\n# classifier = pipeline(\n#     \"zero-shot-classification\", \n#     model=\"joeddav/xlm-roberta-large-xnli\", \n#     device=0 \n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.536206Z","iopub.execute_input":"2026-08-17T09:48:38.536454Z","iopub.status.idle":"2026-08-17T09:48:38.548419Z","shell.execute_reply.started":"2026-08-17T09:48:38.536434Z","shell.execute_reply":"2026-08-17T09:48:38.547728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # 2. Define the exact semantic concepts for our 12 targets\n# # The model will map the foreign languages to these English concepts automatically\n# abnormality_prompts = {\n#     'ACL': ['anterior cruciate ligament tear or injury', 'intact anterior cruciate ligament'],\n#     'MCL': ['medial collateral ligament tear or injury', 'intact medial collateral ligament'],\n#     'Medial Meniscus': ['medial meniscus tear', 'normal medial meniscus'],\n#     'Lateral Meniscus': ['lateral meniscus tear', 'normal lateral meniscus'],\n#     'Medial OA': ['medial compartment osteoarthritis', 'no medial osteoarthritis'],\n#     'Lateral OA': ['lateral compartment osteoarthritis', 'no lateral osteoarthritis'],\n#     'PF OA': ['patellofemoral osteoarthritis', 'no patellofemoral osteoarthritis'],\n#     'Effusion': ['joint effusion or fluid', 'no joint effusion'],\n#     'Synovitis': ['synovitis or joint inflammation', 'no synovitis'],\n#     \"Baker's\": [\"Baker's cyst or popliteal cyst\", \"no Baker's cyst\"],\n#     'Contusion': ['bone contusion or bruise', 'no bone contusion'],\n#     'Fracture': ['bone fracture', 'no fracture']\n# }\n\n# def predict_report(report_text):\n#     \"\"\"Runs zero-shot classification on a single report.\"\"\"\n#     if not isinstance(report_text, str):\n#         return {target: 0.5 for target in abnormality_prompts.keys()}\n        \n#     predictions = {}\n#     for target, candidate_labels in abnormality_prompts.items():\n#         # multi_label=False forces the model to choose between the two options\n#         res = classifier(report_text, candidate_labels, multi_label=False)\n        \n#         # We want the confidence score of the FIRST label (the abnormality)\n#         abnormality_label = candidate_labels[0]\n#         idx = res['labels'].index(abnormality_label)\n#         predictions[target] = res['scores'][idx]\n        \n#     return predictions\n\n# # 3. Run validation on the 58 labeled examples\n# print(f\"Validating on {len(labeled_df)} ground-truth reports...\")\n# tqdm.pandas(desc=\"Evaluating NLP Pipeline\")\n\n# # Get predictions as probabilities\n# val_predictions = labeled_df['Report'].progress_apply(predict_report).apply(pd.Series)\n\n# # 4. Calculate accuracy and AUC for each category\n# print(\"\\n--- NLP Pipeline Validation Results ---\")\n# for target in abnormality_prompts.keys():\n#     y_true = labeled_df[target].values\n#     y_prob = val_predictions[target].values\n#     y_pred = (y_prob > 0.5).astype(int) # Threshold at 0.5\n    \n#     acc = accuracy_score(y_true, y_pred)\n#     # Only calculate AUC if both classes (0 and 1) are present in the validation set\n#     try:\n#         auc = roc_auc_score(y_true, y_prob)\n#     except ValueError:\n#         auc = np.nan\n        \n#     print(f\"{target:<20} | Accuracy: {acc:.3f} | AUC: {auc:.3f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.549251Z","iopub.execute_input":"2026-08-17T09:48:38.549573Z","iopub.status.idle":"2026-08-17T09:48:38.561824Z","shell.execute_reply.started":"2026-08-17T09:48:38.549543Z","shell.execute_reply":"2026-08-17T09:48:38.561116Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### FEW SHOT:_","metadata":{}},{"cell_type":"code","source":"# # Install ONLY bitsandbytes for 4-bit quantization. \n# # Keep Kaggle's stable versions of transformers and accelerate!\n# !pip install -q bitsandbytes \n\n# import pandas as pd\n# import torch\n# import json\n# import re\n# from transformers import AutoTokenizer, AutoModelForCausalLM, BitsAndBytesConfig\n# from tqdm.auto import tqdm\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.562728Z","iopub.execute_input":"2026-08-17T09:48:38.562991Z","iopub.status.idle":"2026-08-17T09:48:38.577417Z","shell.execute_reply.started":"2026-08-17T09:48:38.562957Z","shell.execute_reply":"2026-08-17T09:48:38.576599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# # We will use Qwen2.5-7B-Instruct (No Hugging Face token required, excellent at multilingual JSON)\n# MODEL_NAME = \"Qwen/Qwen2.5-7B-Instruct\" \n\n# bnb_config = BitsAndBytesConfig(\n#     load_in_4bit=True,\n#     bnb_4bit_use_double_quant=True,\n#     bnb_4bit_quant_type=\"nf4\",\n#     bnb_4bit_compute_dtype=torch.bfloat16\n# )\n\n# print(\"Loading tokenizer and model...\")\n# tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME)\n# model = AutoModelForCausalLM.from_pretrained(\n#     MODEL_NAME,\n#     quantization_config=bnb_config,\n#     device_map=\"auto\" \n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.578299Z","iopub.execute_input":"2026-08-17T09:48:38.579268Z","iopub.status.idle":"2026-08-17T09:48:38.591131Z","shell.execute_reply.started":"2026-08-17T09:48:38.579237Z","shell.execute_reply":"2026-08-17T09:48:38.590098Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import numpy as np\n# from sklearn.metrics import accuracy_score, roc_auc_score\n\n# TARGETS = [\n#     \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\", \n#     \"Medial OA\", \"Lateral OA\", \"PF OA\", \"Effusion\", \n#     \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\"\n# ]\n\n# def extract_labels_llm(report_text):\n#     if not isinstance(report_text, str) or len(report_text.strip()) == 0:\n#         return {k: 0.0 for k in TARGETS}\n\n#     prompt = create_prompt(report_text)\n#     inputs = tokenizer(prompt, return_tensors=\"pt\").to(\"cuda\")\n    \n#     with torch.no_grad():\n#         outputs = model.generate(\n#             **inputs, \n#             max_new_tokens=150,\n#             temperature=0.1,\n#             do_sample=False,\n#             pad_token_id=tokenizer.eos_token_id\n#         )\n    \n#     generated_text = tokenizer.decode(outputs[0][inputs['input_ids'].shape[1]:], skip_special_tokens=True)\n    \n#     try:\n#         json_match = re.search(r'\\{.*?\\}', generated_text, re.DOTALL)\n#         if json_match:\n#             parsed = json.loads(json_match.group(0))\n#             return {k: int(parsed.get(k, 0)) for k in TARGETS}\n#     except Exception:\n#         pass\n    \n#     return {k: 0.0 for k in TARGETS}\n\n# # Run extraction on the 58 labeled cases\n# print(\"Evaluating Qwen on 58 ground-truth reports...\")\n# tqdm.pandas(desc=\"Validation\")\n# val_results = labeled_df['Report'].progress_apply(extract_labels_llm).apply(pd.Series)\n\n# # Print metrics comparison\n# print(\"\\n\" + \"=\" * 55)\n# print(f\"{'Condition':<20} | {'Accuracy':<10} | {'AUC':<10}\")\n# print(\"=\" * 55)\n\n# for target in TARGETS:\n#     y_true = labeled_df[target].values.astype(int)\n#     y_pred = val_results[target].values.astype(int)\n#     acc = accuracy_score(y_true, y_pred)\n#     try:\n#         auc = roc_auc_score(y_true, y_pred)\n#     except ValueError:\n#         auc = np.nan\n#     print(f\"{target:<20} | {acc:<10.3f} | {auc:<10.3f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.592301Z","iopub.execute_input":"2026-08-17T09:48:38.592797Z","iopub.status.idle":"2026-08-17T09:48:38.60506Z","shell.execute_reply.started":"2026-08-17T09:48:38.592723Z","shell.execute_reply":"2026-08-17T09:48:38.603985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q bitsandbytes \n!pip install -q vllm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:48:38.605964Z","iopub.execute_input":"2026-08-17T09:48:38.607085Z","iopub.status.idle":"2026-08-17T09:48:47.058371Z","shell.execute_reply.started":"2026-08-17T09:48:38.607054Z","shell.execute_reply":"2026-08-17T09:48:47.057438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport torch\nimport json\nimport re\nfrom transformers import AutoTokenizer, AutoModelForCausalLM\nfrom tqdm.auto import tqdm\n\nprint(\"Loading Qwen2.5-3B-Instruct in Native PyTorch...\")\n\nMODEL_NAME = \"Qwen/Qwen2.5-3B-Instruct\"\n\n# 1. Load Tokenizer (Left padding is required for batched generation)\ntokenizer = AutoTokenizer.from_pretrained(MODEL_NAME, padding_side=\"left\")\nif tokenizer.pad_token is None:\n    tokenizer.pad_token = tokenizer.eos_token\n\n# 2. Load Model in pure float16 (Fits easily on one T4 GPU)\nmodel = AutoModelForCausalLM.from_pretrained(\n    MODEL_NAME,\n    torch_dtype=torch.float16,\n    device_map=\"auto\"\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:56:14.691265Z","iopub.execute_input":"2026-08-17T09:56:14.692021Z","iopub.status.idle":"2026-08-17T09:56:19.927031Z","shell.execute_reply.started":"2026-08-17T09:56:14.691991Z","shell.execute_reply":"2026-08-17T09:56:19.926293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_prompt(report_text):\n    prompt = f\"\"\"<|im_start|>system\nYou are an expert, multilingual radiologist. Read knee MRI reports in any language and extract the presence (1) or absence (0) of 12 abnormalities.\nOutput ONLY a valid JSON dictionary with the exact keys provided. Do not write anything else.\n\nTarget Keys: \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\", \"Medial OA\", \"Lateral OA\", \"PF OA\", \"Effusion\", \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\"\n<|im_end|>\n<|im_start|>user\nReport:\nTécnica: RMN de la rodilla. Resultados: Rotura de menisco interno. Signo de necrosis avascular subcondral en el cóndilo femoral medial. Artrosis femorotibial medial. Derrame.\n<|im_end|>\n<|im_start|>assistant\n{{\"ACL\": 0, \"MCL\": 0, \"Medial Meniscus\": 1, \"Lateral Meniscus\": 0, \"Medial OA\": 1, \"Lateral OA\": 0, \"PF OA\": 0, \"Effusion\": 1, \"Synovitis\": 0, \"Baker's\": 0, \"Contusion\": 0, \"Fracture\": 0}}<|im_end|>\n<|im_start|>user\nReport:\nThe study reveals normal knee joint alignment. No fracture is seen. ACL is intact. PCL is preserved. The MCL is intact. Medial meniscus is not torn. Horizontal tear at anterior horn of the lateral meniscus is noted. Moderate joint effusion.\n<|im_end|>\n<|im_start|>assistant\n{{\"ACL\": 0, \"MCL\": 0, \"Medial Meniscus\": 0, \"Lateral Meniscus\": 1, \"Medial OA\": 0, \"Lateral OA\": 0, \"PF OA\": 0, \"Effusion\": 1, \"Synovitis\": 0, \"Baker's\": 0, \"Contusion\": 0, \"Fracture\": 0}}<|im_end|>\n<|im_start|>user\nReport:\n{report_text}\n<|im_end|>\n<|im_start|>assistant\n\"\"\"\n    return prompt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:56:27.919804Z","iopub.execute_input":"2026-08-17T09:56:27.920588Z","iopub.status.idle":"2026-08-17T09:56:27.925308Z","shell.execute_reply.started":"2026-08-17T09:56:27.920556Z","shell.execute_reply":"2026-08-17T09:56:27.9242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nTARGETS = [\"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\", \"Medial OA\", \"Lateral OA\", \"PF OA\", \"Effusion\", \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\"]\n\n\n\ndef native_batch_extract(reports, batch_size=4):\n    all_results = []\n    \n    # Process in batches of 8 to prevent Out-Of-Memory errors\n    for i in tqdm(range(0, len(reports), batch_size), desc=\"Extracting Labels\"):\n        batch_reports = reports[i:i + batch_size]\n        prompts = [create_prompt(r) for r in batch_reports]\n        \n        inputs = tokenizer(prompts, return_tensors=\"pt\", padding=True, truncation=True, max_length=512).to(\"cuda\")\n        \n        with torch.no_grad():\n            outputs = model.generate(\n                **inputs,\n                max_new_tokens=100,\n                temperature=0.01,\n                do_sample=False,\n                pad_token_id=tokenizer.pad_token_id\n            )\n            \n        input_len = inputs[\"input_ids\"].shape[1]\n        generated_tokens = outputs[:, input_len:]\n        decoded_batch = tokenizer.batch_decode(generated_tokens, skip_special_tokens=True)\n        \n        for text in decoded_batch:\n            try:\n                json_match = re.search(r'\\{.*?\\}', text, re.DOTALL)\n                if json_match:\n                    parsed = json.loads(json_match.group(0))\n                    all_results.append({k: int(parsed.get(k, 0)) for k in TARGETS})\n                    continue\n            except Exception:\n                pass\n            all_results.append({k: 0 for k in TARGETS})\n            \n    return pd.DataFrame(all_results)\n\n# 3. Run the extraction on the unlabeled data\nprint(f\"Starting extraction for {len(unlabeled_df)} reports. Grab a coffee, this will take ~30 mins...\")\nunlabeled_extracted_df = native_batch_extract(unlabeled_df['Report'].tolist(), batch_size=8)\n\n# 4. Merge and Save\nfor target in TARGETS:\n    unlabeled_df[target] = unlabeled_extracted_df[target].values\n\nfully_labeled_train = pd.concat([labeled_df, unlabeled_df], ignore_index=True)\nfully_labeled_train.to_csv(\"/kaggle/working/fully_labeled_train.csv\", index=False)\n\nprint(\"SUCCESS! Saved fully_labeled_train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T09:59:07.992551Z","iopub.execute_input":"2026-08-17T09:59:07.993022Z","iopub.status.idle":"2026-08-17T11:07:55.830987Z","shell.execute_reply.started":"2026-08-17T09:59:07.992992Z","shell.execute_reply":"2026-08-17T11:07:55.830051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport numpy as np\n\n# Assuming fully_labeled_train is your final dataframe\nTARGETS = [\"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\", \"Medial OA\", \"Lateral OA\", \"PF OA\", \"Effusion\", \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\"]\n\n# Calculate positive weights\npos_weights = []\ntotal_samples = len(fully_labeled_train)\n\nfor target in TARGETS:\n    positives = fully_labeled_train[target].sum()\n    negatives = total_samples - positives\n    \n    # Avoid division by zero just in case\n    weight = negatives / (positives + 1e-5)\n    pos_weights.append(weight)\n\n# Convert to a PyTorch tensor and move to GPU\npos_weights_tensor = torch.tensor(pos_weights, dtype=torch.float32).to(\"cuda\")\n\nprint(\"Calculated Positive Weights for the 12 Classes:\")\nfor t, w in zip(TARGETS, pos_weights):\n    print(f\"{t:<20}: {w:.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-17T11:24:57.97362Z","iopub.execute_input":"2026-08-17T11:24:57.974094Z","iopub.status.idle":"2026-08-17T11:24:57.984225Z","shell.execute_reply.started":"2026-08-17T11:24:57.974061Z","shell.execute_reply":"2026-08-17T11:24:57.983229Z"}},"outputs":[],"execution_count":null}]}