{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-03T14:43:42.778064Z","iopub.execute_input":"2026-09-03T14:43:42.778871Z","iopub.status.idle":"2026-09-03T14:43:42.782625Z","shell.execute_reply.started":"2026-09-03T14:43:42.778832Z","shell.execute_reply":"2026-09-03T14:43:42.781643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nprint(\"Available datasets:\")\nprint(os.listdir(\"/kaggle/input\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:50:00.793983Z","iopub.execute_input":"2026-09-04T04:50:00.794943Z","iopub.status.idle":"2026-09-04T04:50:00.799972Z","shell.execute_reply.started":"2026-09-04T04:50:00.794891Z","shell.execute_reply":"2026-09-04T04:50:00.799068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nDATA_DIR = \"/kaggle/input/competitions\"\nprint(\"Files/folders inside competitions:\")\nprint(os.listdir(DATA_DIR))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:50:03.503937Z","iopub.execute_input":"2026-09-04T04:50:03.504694Z","iopub.status.idle":"2026-09-04T04:50:03.509599Z","shell.execute_reply.started":"2026-09-04T04:50:03.504661Z","shell.execute_reply":"2026-09-04T04:50:03.508696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\nprint(os.listdir(DATA_DIR))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:50:06.836056Z","iopub.execute_input":"2026-09-04T04:50:06.837167Z","iopub.status.idle":"2026-09-04T04:50:06.84703Z","shell.execute_reply.started":"2026-09-04T04:50:06.837122Z","shell.execute_reply":"2026-09-04T04:50:06.846095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the competition CSV files\ntrain = pd.read_csv(f\"{DATA_DIR}/train.csv\")\ntest = pd.read_csv(f\"{DATA_DIR}/test.csv\")\ntrain_series = pd.read_csv(f\"{DATA_DIR}/train_series.csv\")\ntest_series = pd.read_csv(f\"{DATA_DIR}/test_series.csv\")\nsample_submission = pd.read_csv(f\"{DATA_DIR}/sample_submission.csv\")\n\nprint(\"Files loaded successfully!\")\nprint()\nprint(\"train:\", train.shape)\nprint(\"test:\", test.shape)\nprint(\"train_series:\", train_series.shape)\nprint(\"test_series:\", test_series.shape)\nprint(\"sample_submission:\", sample_submission.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:50:09.509779Z","iopub.execute_input":"2026-09-04T04:50:09.510142Z","iopub.status.idle":"2026-09-04T04:50:09.818116Z","shell.execute_reply.started":"2026-09-04T04:50:09.510108Z","shell.execute_reply":"2026-09-04T04:50:09.817101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"TRAIN COLUMNS:\")\nprint(train.columns.tolist())\nprint(\"\\nFIRST 5 ROWS:\")\ndisplay(train.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:50:15.635604Z","iopub.execute_input":"2026-09-04T04:50:15.635877Z","iopub.status.idle":"2026-09-04T04:50:15.673603Z","shell.execute_reply.started":"2026-09-04T04:50:15.635854Z","shell.execute_reply":"2026-09-04T04:50:15.672867Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_cols = [\n    \"ACL\",\n    \"MCL\",\n    \"Medial Meniscus\",\n    \"Lateral Meniscus\",\n    \"Medial OA\",\n    \"Lateral OA\",\n    \"PF OA\",\n    \"Effusion\",\n    \"Synovitis\",\n    \"Baker's\",\n    \"Contusion\",\n    \"Fracture\"\n]\n\nprint(\"Total training studies:\", len(train))\nprint(\"\\nNon-missing labels:\")\nprint(train[target_cols].notna().sum())\n\nprint(\"\\nMissing labels:\")\nprint(train[target_cols].isna().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:50:19.355416Z","iopub.execute_input":"2026-09-04T04:50:19.356102Z","iopub.status.idle":"2026-09-04T04:50:19.36816Z","shell.execute_reply.started":"2026-09-04T04:50:19.356069Z","shell.execute_reply":"2026-09-04T04:50:19.367347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labelled_mask = train[target_cols].notna().all(axis=1)\n\nlabelled_train = train[labelled_mask].copy()\n\nprint(\"Total training studies:\", len(train))\nprint(\"Fully labelled studies:\", len(labelled_train))\nprint(\"Unlabelled/partially labelled studies:\", len(train) - len(labelled_train))\n\nprint(\"\\nLabelled data shape:\")\nprint(labelled_train.shape)\n\ndisplay(labelled_train.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:50:24.437124Z","iopub.execute_input":"2026-09-04T04:50:24.437697Z","iopub.status.idle":"2026-09-04T04:50:24.462991Z","shell.execute_reply.started":"2026-09-04T04:50:24.437663Z","shell.execute_reply":"2026-09-04T04:50:24.461969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Unique values for each abnormality:\\n\")\nfor col in target_cols:\n    print(f\"{col}: {train[col].dropna().unique()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:50:28.905508Z","iopub.execute_input":"2026-09-04T04:50:28.906367Z","iopub.status.idle":"2026-09-04T04:50:28.919256Z","shell.execute_reply.started":"2026-09-04T04:50:28.906331Z","shell.execute_reply":"2026-09-04T04:50:28.918453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Label availability and distribution:\\n\")\nfor col in target_cols:\n    available = train[col].notna().sum()\n    missing = train[col].isna().sum()\n    positive = (train[col] == 1).sum()\n    negative = (train[col] == 0).sum()\n\n    print(\n        f\"{col:20s} | \"\n        f\"Available: {available:4d} | \"\n        f\"Missing: {missing:4d} | \"\n        f\"Positive: {positive:4d} | \"\n        f\"Negative: {negative:4d}\"\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:50:33.003014Z","iopub.execute_input":"2026-09-04T04:50:33.003295Z","iopub.status.idle":"2026-09-04T04:50:33.01979Z","shell.execute_reply.started":"2026-09-04T04:50:33.003273Z","shell.execute_reply":"2026-09-04T04:50:33.018859Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Number of available labels per study:\\n\")\nlabel_count_per_study = train[target_cols].notna().sum(axis=1)\nprint(label_count_per_study.value_counts().sort_index())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:50:38.267597Z","iopub.execute_input":"2026-09-04T04:50:38.268303Z","iopub.status.idle":"2026-09-04T04:50:38.284546Z","shell.execute_reply.started":"2026-09-04T04:50:38.268258Z","shell.execute_reply":"2026-09-04T04:50:38.283723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"studies_with_labels = train[target_cols].notna().any(axis=1)\n\nprint(\"Total studies:\", len(train))\nprint(\"Studies with at least one label:\", studies_with_labels.sum())\nprint(\"Studies with no labels:\", (~studies_with_labels).sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:50:41.8979Z","iopub.execute_input":"2026-09-04T04:50:41.898693Z","iopub.status.idle":"2026-09-04T04:50:41.906726Z","shell.execute_reply.started":"2026-09-04T04:50:41.89866Z","shell.execute_reply":"2026-09-04T04:50:41.905959Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"partial_mask = (\n    train[target_cols].notna().any(axis=1)\n    & ~train[target_cols].notna().all(axis=1)\n)\n\npartial_train = train[partial_mask].copy()\nprint(\"Partially labelled studies:\", len(partial_train))\ndisplay(partial_train[[\"StudyInstanceUID\"] + target_cols].head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:50:45.743725Z","iopub.execute_input":"2026-09-04T04:50:45.744442Z","iopub.status.idle":"2026-09-04T04:50:45.767251Z","shell.execute_reply.started":"2026-09-04T04:50:45.744374Z","shell.execute_reply":"2026-09-04T04:50:45.766334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"TRAIN SERIES COLUMNS:\")\nprint(train_series.columns.tolist())\nprint(\"\\nTRAIN SERIES SHAPE:\")\nprint(train_series.shape)\nprint(\"\\nFIRST 10 ROWS:\")\ndisplay(train_series.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:50:49.562543Z","iopub.execute_input":"2026-09-04T04:50:49.563002Z","iopub.status.idle":"2026-09-04T04:50:49.57942Z","shell.execute_reply.started":"2026-09-04T04:50:49.562959Z","shell.execute_reply":"2026-09-04T04:50:49.578224Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labelled_study_ids = train.loc[\n    train[target_cols].notna().all(axis=1),\n    \"StudyInstanceUID\"\n]\n\nprint(\"Number of labelled studies:\", len(labelled_study_ids))\n\nprint(\"\\nLabelled studies found in train_series:\")\nprint(\n    train_series[\"StudyInstanceUID\"]\n    .isin(labelled_study_ids)\n    .sum()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:50:54.22507Z","iopub.execute_input":"2026-09-04T04:50:54.225798Z","iopub.status.idle":"2026-09-04T04:50:54.236714Z","shell.execute_reply.started":"2026-09-04T04:50:54.225765Z","shell.execute_reply":"2026-09-04T04:50:54.235987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labelled_series = train_series[\n    train_series[\"StudyInstanceUID\"].isin(labelled_study_ids)\n].copy()\n\nseries_per_study = (\n    labelled_series\n    .groupby(\"StudyInstanceUID\")\n    .size()\n)\nprint(\"Number of MRI series per labelled study:\")\nprint(series_per_study.value_counts().sort_index())\nprint(\"\\nFirst 10 labelled studies:\")\ndisplay(series_per_study.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:50:58.090854Z","iopub.execute_input":"2026-09-04T04:50:58.091151Z","iopub.status.idle":"2026-09-04T04:50:58.104581Z","shell.execute_reply.started":"2026-09-04T04:50:58.091127Z","shell.execute_reply":"2026-09-04T04:50:58.103652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select the first labelled study\nstudy_id = labelled_study_ids.iloc[0]\nprint(\"Selected StudyInstanceUID:\")\nprint(study_id)\nprint(\"\\nMRI series belonging to this study:\")\nstudy_series = train_series[\n    train_series[\"StudyInstanceUID\"] == study_id\n].copy()\ndisplay(study_series)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:51:02.669517Z","iopub.execute_input":"2026-09-04T04:51:02.670387Z","iopub.status.idle":"2026-09-04T04:51:02.684322Z","shell.execute_reply.started":"2026-09-04T04:51:02.670352Z","shell.execute_reply":"2026-09-04T04:51:02.683395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_DIR = os.path.join(DATA_DIR, \"train_series\")\nprint(\"Train MRI directory:\")\nprint(TRAIN_DIR)\nprint(\"\\nDoes the selected study folder exist?\")\nstudy_path = os.path.join(TRAIN_DIR, str(study_id))\nprint(study_path)\nprint(os.path.exists(study_path))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:51:06.522868Z","iopub.execute_input":"2026-09-04T04:51:06.523222Z","iopub.status.idle":"2026-09-04T04:51:06.529453Z","shell.execute_reply.started":"2026-09-04T04:51:06.523195Z","shell.execute_reply":"2026-09-04T04:51:06.528396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Contents of selected study folder:\")\nstudy_contents = os.listdir(study_path)\nprint(\"Number of items:\", len(study_contents))\n\nfor item in study_contents:\n    item_path = os.path.join(study_path, item)\n\n    if os.path.isdir(item_path):\n        print(\"Fol\", item)\n    else:\n        print(\"F\", item)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:51:09.914121Z","iopub.execute_input":"2026-09-04T04:51:09.91495Z","iopub.status.idle":"2026-09-04T04:51:09.922969Z","shell.execute_reply.started":"2026-09-04T04:51:09.914902Z","shell.execute_reply":"2026-09-04T04:51:09.922173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"MRI files in each series:\\n\")\n\nfor series_id in study_series[\"SeriesInstanceUID\"]:\n    \n    series_path = os.path.join(\n        study_path,\n        str(series_id)\n    )\n    if os.path.exists(series_path):\n        files = os.listdir(series_path)\n        print(\n            \"Series:\",\n            series_id,\n            \"| Number of files:\",\n            len(files)\n        )\n    else:\n        print(\n            \"Series:\",\n            series_id,\n            \"| Folder NOT FOUND\"\n        )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:51:13.363935Z","iopub.execute_input":"2026-09-04T04:51:13.364292Z","iopub.status.idle":"2026-09-04T04:51:13.409506Z","shell.execute_reply.started":"2026-09-04T04:51:13.364263Z","shell.execute_reply":"2026-09-04T04:51:13.408506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select the first MRI series from the selected labelled study\nseries_id = study_series[\"SeriesInstanceUID\"].iloc[0]\nseries_path = os.path.join(\n    study_path,\n    str(series_id)\n)\nprint(\"Selected SeriesInstanceUID:\")\nprint(series_id)\nprint(\"\\nSeries path:\")\nprint(series_path)\n\n# List the files in this series\nseries_files = os.listdir(series_path)\nprint(\"\\nNumber of files:\", len(series_files))\nprint(\"\\nFirst 10 files:\")\nfor file in series_files[:10]:\n    print(file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:51:17.140594Z","iopub.execute_input":"2026-09-04T04:51:17.141308Z","iopub.status.idle":"2026-09-04T04:51:17.14762Z","shell.execute_reply.started":"2026-09-04T04:51:17.141276Z","shell.execute_reply":"2026-09-04T04:51:17.14699Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nimport os\nimport matplotlib.pyplot as plt\n\n# Select first MRI series\nseries_id = study_series[\"SeriesInstanceUID\"].iloc[0]\nseries_path = os.path.join(study_path, str(series_id))\n\n# Select first DICOM file\ndicom_file = os.listdir(series_path)[0]\ndicom_path = os.path.join(series_path, dicom_file)\n\n# Read DICOM\nds = pydicom.dcmread(dicom_path)\nimage = ds.pixel_array\nprint(\"Series:\", series_id)\nprint(\"DICOM file:\", dicom_file)\nprint(\"Image shape:\", image.shape)\nprint(\"Modality:\", getattr(ds, \"Modality\", \"N/A\"))\n\n# Display MRI\nplt.figure(figsize=(6, 6))\nplt.imshow(image, cmap=\"gray\")\nplt.axis(\"off\")\nplt.title(\"Knee MRI Slice\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:51:22.074773Z","iopub.execute_input":"2026-09-04T04:51:22.075183Z","iopub.status.idle":"2026-09-04T04:51:23.00366Z","shell.execute_reply.started":"2026-09-04T04:51:22.075137Z","shell.execute_reply":"2026-09-04T04:51:23.002766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nimport os\nimport matplotlib.pyplot as plt\n\n# Read all DICOM files in the selected series\ndicoms = []\nfor f in os.listdir(series_path):\n    path = os.path.join(series_path, f)\n    try:\n        ds = pydicom.dcmread(path)\n        if hasattr(ds, \"PixelData\"):\n            dicoms.append(ds)\n    except:\n        pass\n\n# Sort by anatomical position if available\nif all(hasattr(ds, \"ImagePositionPatient\") for ds in dicoms):\n    dicoms.sort(key=lambda x: float(x.ImagePositionPatient[2]))\nelse:\n    dicoms.sort(key=lambda x: int(getattr(x, \"InstanceNumber\", 0)))\nprint(\"Total MRI slices:\", len(dicoms))\n\n# Display up to 12 evenly spaced slices\nn = min(12, len(dicoms))\nindices = [int(i * (len(dicoms)-1) / (n-1)) for i in range(n)] if n > 1 else [0]\n\nfig, axes = plt.subplots(3, 4, figsize=(12, 9))\naxes = axes.ravel()\n\nfor ax, idx in zip(axes, indices):\n    ax.imshow(dicoms[idx].pixel_array, cmap=\"gray\")\n    ax.set_title(f\"Slice {idx + 1}\")\n    ax.axis(\"off\")\n\nfor ax in axes[n:]:\n    ax.axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:51:29.130617Z","iopub.execute_input":"2026-09-04T04:51:29.131398Z","iopub.status.idle":"2026-09-04T04:51:30.936656Z","shell.execute_reply.started":"2026-09-04T04:51:29.131365Z","shell.execute_reply":"2026-09-04T04:51:30.935461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Inspect MRI series characteristics\nprint(\"=== Fluid-Sensitive Series ===\")\nprint(train_series[\"Fluid_Sensitive\"].value_counts(dropna=False))\n\nprint(\"\\n=== Fat-Suppression Series ===\")\nprint(train_series[\"Fat_Suppression\"].value_counts(dropna=False))\n\nprint(\"\\n=== Anatomical Plane ===\")\nprint(train_series[\"Anatomical_Plane\"].value_counts(dropna=False))\n\nprint(\"\\n=== Combined MRI Series Information ===\")\ndisplay(\n    train_series[\n        [\n            \"StudyInstanceUID\",\n            \"SeriesInstanceUID\",\n            \"Fluid_Sensitive\",\n            \"Fat_Suppression\",\n            \"Anatomical_Plane\"\n        ]\n    ].head(20)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:51:36.754828Z","iopub.execute_input":"2026-09-04T04:51:36.755773Z","iopub.status.idle":"2026-09-04T04:51:36.776207Z","shell.execute_reply.started":"2026-09-04T04:51:36.755742Z","shell.execute_reply":"2026-09-04T04:51:36.775227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pydicom\nimport pandas as pd\n# Get labels for this study\nlabels = train.loc[train[\"StudyInstanceUID\"] == study_id, target_cols].iloc[0]\n\n# Read all 36 slices from the selected series\nrows = []\nfor f in os.listdir(series_path):\n    try:\n        ds = pydicom.dcmread(os.path.join(series_path, f))\n        if hasattr(ds, \"PixelData\"):\n            rows.append({\n                \"File\": f,\n                \"Slice\": getattr(ds, \"InstanceNumber\", 0),\n                **labels.to_dict()\n            })\n    except:\n        pass\n\nslice_df = pd.DataFrame(rows).sort_values(\"Slice\").reset_index(drop=True)\nprint(\"Total MRI slices:\", len(slice_df))\ndisplay(slice_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:51:42.923176Z","iopub.execute_input":"2026-09-04T04:51:42.923473Z","iopub.status.idle":"2026-09-04T04:51:43.03308Z","shell.execute_reply.started":"2026-09-04T04:51:42.923448Z","shell.execute_reply":"2026-09-04T04:51:43.032291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Inspect every MRI series in this labelled study\nfor _, row in study_series.iterrows():\n    sid = row[\"SeriesInstanceUID\"]\n    path = os.path.join(study_path, str(sid))\n\n    files = os.listdir(path) if os.path.exists(path) else []\n\n    print(\n        f\"Plane: {row['Anatomical_Plane']:8} | \"\n        f\"Fluid: {row['Fluid_Sensitive']} | \"\n        f\"Fat-suppressed: {row['Fat_Suppression']} | \"\n        f\"Slices: {len(files)}\"\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:51:49.611427Z","iopub.execute_input":"2026-09-04T04:51:49.612316Z","iopub.status.idle":"2026-09-04T04:51:49.625315Z","shell.execute_reply.started":"2026-09-04T04:51:49.612281Z","shell.execute_reply":"2026-09-04T04:51:49.624267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, pydicom\nimport matplotlib.pyplot as plt\nfig, ax = plt.subplots(len(study_series), 4, figsize=(10, 12))\nfor r, (_, row) in enumerate(study_series.iterrows()):\n    path = os.path.join(study_path, str(row[\"SeriesInstanceUID\"]))\n    ds = [pydicom.dcmread(os.path.join(path, f))\n          for f in os.listdir(path)]\n\n    ds = [x for x in ds if hasattr(x, \"PixelData\")]\n    ds.sort(key=lambda x: getattr(x, \"InstanceNumber\", 0))\n    for c, i in enumerate([0, len(ds)//3, 2*len(ds)//3, len(ds)-1]):\n        ax[r,c].imshow(ds[i].pixel_array, cmap=\"gray\")\n        ax[r,c].set_title(f\"{row['Anatomical_Plane']} | {i+1}\")\n        ax[r,c].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:51:53.553281Z","iopub.execute_input":"2026-09-04T04:51:53.554189Z","iopub.status.idle":"2026-09-04T04:51:55.916712Z","shell.execute_reply.started":"2026-09-04T04:51:53.554156Z","shell.execute_reply":"2026-09-04T04:51:55.915765Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, pydicom\ndef load_study(study_id):\n    data = {}\n    study_path = os.path.join(TRAIN_DIR, str(study_id))\n\n    for _, row in train_series[train_series.StudyInstanceUID == study_id].iterrows():\n        path = os.path.join(study_path, str(row.SeriesInstanceUID))\n        imgs = []\n\n        for f in os.listdir(path):\n            try:\n                ds = pydicom.dcmread(os.path.join(path, f))\n                if hasattr(ds, \"PixelData\"):\n                    imgs.append((getattr(ds, \"InstanceNumber\", 0), ds.pixel_array))\n            except:\n                pass\n        imgs.sort()\n        data[row.SeriesInstanceUID] = [x[1] for x in imgs]\n    return data\nstudy_data = load_study(study_id)\nfor sid, imgs in study_data.items():\n    print(sid, \"→\", len(imgs), \"slices\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:52:02.813888Z","iopub.execute_input":"2026-09-04T04:52:02.814735Z","iopub.status.idle":"2026-09-04T04:52:03.16106Z","shell.execute_reply.started":"2026-09-04T04:52:02.814689Z","shell.execute_reply":"2026-09-04T04:52:03.16012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\n\nIMG_SIZE = 224\n\ndef preprocess(images):\n    return np.array([\n        cv2.resize(img, (IMG_SIZE, IMG_SIZE)).astype(\"float32\") /\n        (img.max() if img.max() > 0 else 1)\n        for img in images\n    ])\n\nprocessed_data = {\n    sid: preprocess(imgs)\n    for sid, imgs in study_data.items()\n}\n\nfor sid, imgs in processed_data.items():\n    print(sid, \"→\", imgs.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:52:08.415633Z","iopub.execute_input":"2026-09-04T04:52:08.415967Z","iopub.status.idle":"2026-09-04T04:52:08.80256Z","shell.execute_reply.started":"2026-09-04T04:52:08.415887Z","shell.execute_reply":"2026-09-04T04:52:08.801605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, pydicom, cv2\nimport numpy as np\nIMG_SIZE, N_SLICES = 224, 16\ndef prepare_study(sid):\n    study_path = os.path.join(TRAIN_DIR, str(sid))\n    series_data = []\n\n    for _, row in train_series[train_series.StudyInstanceUID == sid].iterrows():\n        path = os.path.join(study_path, str(row.SeriesInstanceUID))\n        imgs = []\n\n        for f in os.listdir(path):\n            try:\n                ds = pydicom.dcmread(os.path.join(path, f))\n                if hasattr(ds, \"PixelData\"):\n                    img = cv2.resize(ds.pixel_array.astype(\"float32\"),\n                                     (IMG_SIZE, IMG_SIZE))\n                    img /= max(img.max(), 1)\n                    imgs.append((getattr(ds, \"InstanceNumber\", 0), img))\n            except:\n                pass\n        imgs.sort(key=lambda x: x[0])\n        imgs = np.array([x[1] for x in imgs])\n        idx = np.linspace(0, len(imgs)-1, N_SLICES).astype(int)\n        series_data.append(imgs[idx])\n    label = train.loc[\n        train.StudyInstanceUID == sid, target_cols\n    ].iloc[0].values\n    return series_data, label\n\n\n# Test first 3 labelled studies\nstudies = [prepare_study(sid) for sid in labelled_study_ids[:3]]\nfor i, (series, label) in enumerate(studies, 1):\n    print(f\"Study {i}: {len(series)} series\")\n    print(\"  Series shapes:\", [x.shape for x in series])\n    print(\"  Labels:\", label)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:52:12.063236Z","iopub.execute_input":"2026-09-04T04:52:12.063603Z","iopub.status.idle":"2026-09-04T04:52:18.402601Z","shell.execute_reply.started":"2026-09-04T04:52:12.063574Z","shell.execute_reply":"2026-09-04T04:52:18.401831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_ids, val_ids = train_test_split(\n    labelled_study_ids,\n    test_size=0.2,\n    random_state=42\n)\nprint(\"Training studies:\", len(train_ids))\nprint(\"Validation studies:\", len(val_ids))\n\n# Check number of MRI series per labelled study\nseries_counts = (\n    train_series[train_series.StudyInstanceUID.isin(labelled_study_ids)]\n    .groupby(\"StudyInstanceUID\")\n    .size()\n)\nprint(\"Minimum series:\", series_counts.min())\nprint(\"Maximum series:\", series_counts.max())\nprint(\"Average series:\", round(series_counts.mean(), 2))\nprint(\"\\nDistribution:\")\nprint(series_counts.value_counts().sort_index())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:52:23.831091Z","iopub.execute_input":"2026-09-04T04:52:23.831385Z","iopub.status.idle":"2026-09-04T04:52:23.866336Z","shell.execute_reply.started":"2026-09-04T04:52:23.83136Z","shell.execute_reply":"2026-09-04T04:52:23.865311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"types = (\n    train_series[train_series.StudyInstanceUID.isin(labelled_study_ids)]\n    .groupby([\"Anatomical_Plane\", \"Fluid_Sensitive\", \"Fat_Suppression\"])\n    .size()\n    .reset_index(name=\"Count\")\n)\ndisplay(types.sort_values(\"Count\", ascending=False))\n\nmeta = train_series[train_series.StudyInstanceUID.isin(labelled_study_ids)]\nprint(\"Series types:\")\ndisplay(meta.groupby([\"Anatomical_Plane\", \"Fluid_Sensitive\", \"Fat_Suppression\"]).size())\nprint(\"\\nMissing metadata:\")\nprint(meta[[\"Anatomical_Plane\", \"Fluid_Sensitive\", \"Fat_Suppression\"]].isna().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:52:28.643974Z","iopub.execute_input":"2026-09-04T04:52:28.644461Z","iopub.status.idle":"2026-09-04T04:52:28.673609Z","shell.execute_reply.started":"2026-09-04T04:52:28.644433Z","shell.execute_reply":"2026-09-04T04:52:28.672938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta = train_series[\n    train_series.StudyInstanceUID.isin(labelled_study_ids)\n].copy()\nmeta[\"Type\"] = (\n    meta[\"Anatomical_Plane\"].astype(str) + \"_\" +\n    meta[\"Fluid_Sensitive\"].astype(str) + \"_\" +\n    meta[\"Fat_Suppression\"].astype(str)\n)\ncheck = pd.crosstab(meta.StudyInstanceUID, meta.Type)\nprint(\"Studies:\", len(check))\ndisplay(check)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:52:33.576325Z","iopub.execute_input":"2026-09-04T04:52:33.576614Z","iopub.status.idle":"2026-09-04T04:52:33.621838Z","shell.execute_reply.started":"2026-09-04T04:52:33.576589Z","shell.execute_reply":"2026-09-04T04:52:33.620848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"good, bad = [], []\nfor i, sid in enumerate(labelled_study_ids, 1):\n    try:\n        data = load_study(sid)\n        n = sum(len(imgs) for imgs in data.values())\n        good.append(sid)\n        print(f\"{i}/58: {len(data)} series | {n} slices\")\n    except Exception as e:\n        bad.append((sid, str(e)))\n        print(f\"{i}/58: ERROR\")\nprint(f\"\\nSuccessful: {len(good)} | Failed: {len(bad)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:52:39.953281Z","iopub.execute_input":"2026-09-04T04:52:39.953977Z","iopub.status.idle":"2026-09-04T04:54:10.307414Z","shell.execute_reply.started":"2026-09-04T04:52:39.953944Z","shell.execute_reply":"2026-09-04T04:54:10.306721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nDATA_DIR = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\nTRAIN_DIR = os.path.join(DATA_DIR, \"train_series\")\nTEST_DIR = os.path.join(DATA_DIR, \"test_series\")\n\nprint(\"Train folders:\", len(os.listdir(TRAIN_DIR)))\nprint(\"Test folders:\", len(os.listdir(TEST_DIR)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:54:14.057979Z","iopub.execute_input":"2026-09-04T04:54:14.058662Z","iopub.status.idle":"2026-09-04T04:54:14.133689Z","shell.execute_reply.started":"2026-09-04T04:54:14.058615Z","shell.execute_reply":"2026-09-04T04:54:14.132717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ntrain = pd.read_csv(os.path.join(DATA_DIR, \"train.csv\"))\ntest = pd.read_csv(os.path.join(DATA_DIR, \"test.csv\"))\ntrain_series = pd.read_csv(os.path.join(DATA_DIR, \"train_series.csv\"))\ntest_series = pd.read_csv(os.path.join(DATA_DIR, \"test_series.csv\"))\n\ntarget_cols = [\n    \"ACL\",\"MCL\",\"Medial Meniscus\",\"Lateral Meniscus\",\n    \"Medial OA\",\"Lateral OA\",\"PF OA\",\"Effusion\",\n    \"Synovitis\",\"Baker's\",\"Contusion\",\"Fracture\"\n]\nlabelled_study_ids = train.loc[\n    train[target_cols].notna().all(axis=1), \"StudyInstanceUID\"\n]\nprint(\"Train studies:\", len(train))\nprint(\"Train MRI series:\", len(train_series))\nprint(\"Fully labelled studies:\", len(labelled_study_ids))\nprint(\"Test studies:\", len(test))\nprint(\"Test MRI series:\", len(test_series))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:54:18.521798Z","iopub.execute_input":"2026-09-04T04:54:18.522292Z","iopub.status.idle":"2026-09-04T04:54:18.676994Z","shell.execute_reply.started":"2026-09-04T04:54:18.52226Z","shell.execute_reply":"2026-09-04T04:54:18.675992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport torch\nfrom torch.utils.data import Dataset\n# Labelled studies + train/validation split\nlabelled_series = train_series[train_series.StudyInstanceUID.isin(labelled_study_ids)]\ncounts = labelled_series.groupby(\"StudyInstanceUID\").size()\ntrain_ids, val_ids = train_test_split(labelled_study_ids, test_size=0.2, random_state=42)\n\n# MRI dataset loader\nclass KneeMRI(Dataset):\n    def __init__(self, ids): self.ids = list(ids)\n    def __len__(self): return len(self.ids)\n    def __getitem__(self, i):\n        sid = self.ids[i]\n        data = load_study(sid)\n        series = [select_slices(preprocess(x)) for x in data.values()]\n        labels = train.loc[train.StudyInstanceUID == sid, target_cols].iloc[0].values.astype(\"float32\")\n        mask = ~np.isnan(labels)\n        return series, np.nan_to_num(labels), mask\ntrain_dataset, val_dataset = KneeMRI(train_ids), KneeMRI(val_ids)\nprint(f\"Labelled studies: {len(labelled_study_ids)} | Series: {len(labelled_series)}\")\nprint(f\"Series/study: {counts.min()}–{counts.max()} (avg {counts.mean():.2f})\")\nprint(f\"Train: {len(train_dataset)} | Validation: {len(val_dataset)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:54:22.423235Z","iopub.execute_input":"2026-09-04T04:54:22.423663Z","iopub.status.idle":"2026-09-04T04:54:22.441302Z","shell.execute_reply.started":"2026-09-04T04:54:22.42363Z","shell.execute_reply":"2026-09-04T04:54:22.440105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, pydicom, cv2\nimport numpy as np\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nIMG_SIZE, N_SLICES = 224, 16\ndef load_study(sid):\n    path = os.path.join(TRAIN_DIR, str(sid))\n    data = {}\n\n    for _, r in train_series[train_series.StudyInstanceUID == sid].iterrows():\n        sp = os.path.join(path, str(r.SeriesInstanceUID))\n        imgs = []\n        for f in os.listdir(sp):\n            try:\n                ds = pydicom.dcmread(os.path.join(sp, f))\n                if hasattr(ds, \"PixelData\"):\n                    img = cv2.resize(ds.pixel_array.astype(\"float32\"), (IMG_SIZE, IMG_SIZE))\n                    img /= max(img.max(), 1)\n                    imgs.append((getattr(ds, \"InstanceNumber\", 0), img))\n            except:\n                pass\n        imgs.sort()\n        imgs = np.array([x[1] for x in imgs])\n        idx = np.linspace(0, len(imgs)-1, N_SLICES).astype(int)\n        data[r.SeriesInstanceUID] = imgs[idx]\n    return data\n\nclass KneeMRI(Dataset):\n    def __init__(self, ids): self.ids = list(ids)\n    def __len__(self): return len(self.ids)\n    def __getitem__(self, i):\n        sid = self.ids[i]\n        data = load_study(sid)\n        series = list(data.values())\n        labels = train.loc[train.StudyInstanceUID == sid, target_cols].iloc[0].values.astype(\"float32\")\n        mask = ~np.isnan(labels)\n        return series, np.nan_to_num(labels), mask\ntrain_dataset, val_dataset = KneeMRI(train_ids), KneeMRI(val_ids)\n\n# Test one study\nseries, labels, mask = train_dataset[0]\nprint(\"Series:\", len(series))\nprint(\"Shapes:\", [x.shape for x in series])\nprint(\"Labels:\", labels)\nprint(\"Mask:\", mask)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:54:26.72041Z","iopub.execute_input":"2026-09-04T04:54:26.720787Z","iopub.status.idle":"2026-09-04T04:54:27.130753Z","shell.execute_reply.started":"2026-09-04T04:54:26.720756Z","shell.execute_reply":"2026-09-04T04:54:27.129833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader\n# 1. DataLoaders: batch=1 handles variable 3–14 series/study\ntrain_loader = DataLoader(train_dataset, batch_size=1, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=1, shuffle=False)\n\n# 2. Simple baseline CNN\nclass KneeCNN(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.cnn = nn.Sequential(\n            nn.Conv2d(1, 16, 3, padding=1), nn.ReLU(), nn.MaxPool2d(2),\n            nn.Conv2d(16, 32, 3, padding=1), nn.ReLU(), nn.AdaptiveAvgPool2d(1)\n        )\n        self.fc = nn.Linear(32, 12)\n    def forward(self, series):\n        features = []\n        for s in series[0]:\n            x = torch.tensor(s, dtype=torch.float32).unsqueeze(1)\n            f = self.cnn(x).squeeze(-1).squeeze(-1).mean(0)\n            features.append(f)\n        return self.fc(torch.stack(features).mean(0))\n\n# 3. Test model on one study\nmodel = KneeCNN()\nseries, labels, mask = train_dataset[0]\nwith torch.no_grad():\n    output = model([series])\n\n# 4. Convert outputs to probabilities + calculate loss\nprobs = torch.sigmoid(output)\nloss = nn.BCEWithLogitsLoss()(output, torch.tensor(labels))\nprint(\"Output shape:\", output.shape)\nprint(\"Probabilities:\", probs.numpy().round(3))\nprint(\"Loss:\", loss.item())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:54:33.072417Z","iopub.execute_input":"2026-09-04T04:54:33.072701Z","iopub.status.idle":"2026-09-04T04:54:34.430346Z","shell.execute_reply.started":"2026-09-04T04:54:33.072678Z","shell.execute_reply":"2026-09-04T04:54:34.429617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport numpy as np\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n\nclass KneeCNN(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.cnn = nn.Sequential(\n            nn.Conv2d(1, 16, 3, padding=1), nn.ReLU(),\n            nn.MaxPool2d(2),\n            nn.Conv2d(16, 32, 3, padding=1), nn.ReLU(),\n            nn.AdaptiveAvgPool2d(1)\n        )\n        self.fc = nn.Linear(32, 12)\n\n    def forward(self, series):\n        features = []\n\n        # Accept either series or [series]\n        if len(series) == 1 and isinstance(series[0], (list, tuple)):\n            series = series[0]\n\n        for s in series:\n            x = torch.tensor(s, dtype=torch.float32, device=device)\n\n            # (16,224,224) → (16,1,224,224)\n            if x.ndim == 3:\n                x = x.unsqueeze(1)\n\n            f = self.cnn(x).flatten(1).mean(0)\n            features.append(f)\n\n        return self.fc(torch.stack(features).mean(0))\n\n\n# Recreate and train model\nmodel = KneeCNN().to(device)\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-4)\ncriterion = nn.BCEWithLogitsLoss()\n\nfor epoch in range(3):\n    model.train()\n    total = 0\n\n    for series, labels, mask in train_dataset:\n        optimizer.zero_grad()\n\n        out = model(series)\n\n        y = torch.tensor(labels, dtype=torch.float32, device=device)\n        m = torch.tensor(mask, dtype=torch.bool, device=device)\n\n        loss = criterion(out[m], y[m])\n        loss.backward()\n        optimizer.step()\n\n        total += loss.item()\n\n    print(f\"Epoch {epoch+1}/3 | Loss: {total/len(train_dataset):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:54:40.88104Z","iopub.execute_input":"2026-09-04T04:54:40.881324Z","iopub.status.idle":"2026-09-04T04:55:59.542412Z","shell.execute_reply.started":"2026-09-04T04:54:40.881299Z","shell.execute_reply":"2026-09-04T04:55:59.541317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\nimport numpy as np\nimport torch\n\nmodel.eval()\npreds, true = [], []\n\nwith torch.no_grad():\n    for series, labels, mask in val_dataset:\n        p = torch.sigmoid(model(series)).cpu().numpy()\n        preds.append(p)\n        true.append(labels)\n\npreds, true = np.array(preds), np.array(true)\n\nprint(\"Validation shape:\", preds.shape)\n\nfor i, col in enumerate(target_cols):\n    m = ~np.isnan(true[:, i])\n    if len(np.unique(true[m, i])) == 2:\n        print(f\"{col:18s}: AUC = {roc_auc_score(true[m, i], preds[m, i]):.3f}\")\n    else:\n        print(f\"{col:18s}: AUC = N/A\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:56:49.623861Z","iopub.execute_input":"2026-09-04T04:56:49.625135Z","iopub.status.idle":"2026-09-04T04:56:56.005382Z","shell.execute_reply.started":"2026-09-04T04:56:49.625099Z","shell.execute_reply":"2026-09-04T04:56:56.004395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\nimport numpy as np\nimport torch\nimport torch.nn as nn\n\n# 1. Validation before improvement\ndef validate():\n    model.eval()\n    preds, true = [], []\n\n    with torch.no_grad():\n        for series, labels, mask in val_dataset:\n            preds.append(torch.sigmoid(model(series)).cpu().numpy())\n            true.append(labels)\n\n    preds, true = np.array(preds), np.array(true)\n\n    scores = []\n    for i, col in enumerate(target_cols):\n        m = ~np.isnan(true[:, i])\n        if len(np.unique(true[m, i])) == 2:\n            auc = roc_auc_score(true[m, i], preds[m, i])\n            scores.append(auc)\n            print(f\"{col:18s}: {auc:.3f}\")\n        else:\n            print(f\"{col:18s}: N/A\")\n\n    print(\"Mean AUC:\", round(np.mean(scores), 3))\n    return preds, true\n\nprint(\"=== BEFORE IMPROVEMENT ===\")\npreds, true = validate()\n\n\n# 2. Class-weighted training \npos = train.loc[train.StudyInstanceUID.isin(train_ids), target_cols].sum()\nneg = train.loc[train.StudyInstanceUID.isin(train_ids), target_cols].count() - pos\nweights = (neg / pos).replace([np.inf, -np.inf], 1).fillna(1)\n\ncriterion = nn.BCEWithLogitsLoss(\n    pos_weight=torch.tensor(weights.values, dtype=torch.float32, device=device)\n)\n\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-4)\n\nprint(\"\\n=== IMPROVED TRAINING ===\")\n\nfor epoch in range(5):\n    model.train()\n    total = 0\n\n    for series, labels, mask in train_dataset:\n        optimizer.zero_grad()\n        out = model(series)\n\n        y = torch.tensor(labels, dtype=torch.float32, device=device)\n        m = torch.tensor(mask, dtype=torch.bool, device=device)\n\n        loss = criterion(out[m], y[m])\n        loss.backward()\n        optimizer.step()\n        total += loss.item()\n\n    print(f\"Epoch {epoch+1}/5 | Loss: {total/len(train_dataset):.4f}\")\n\n\n# 3. Validation after improvement \nprint(\"\\n=== AFTER IMPROVEMENT ===\")\npreds, true = validate()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T04:59:29.109942Z","iopub.execute_input":"2026-09-04T04:59:29.110536Z","iopub.status.idle":"2026-09-04T05:01:43.5779Z","shell.execute_reply.started":"2026-09-04T04:59:29.110503Z","shell.execute_reply":"2026-09-04T05:01:43.577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save current improved model\n\ntorch.save(model.state_dict(), \"/kaggle/working/knee_best_model.pth\")\n\nprint(\"Best model saved.\")\nprint(\"Mean validation AUC:\", round(np.mean([\n    roc_auc_score(true[:, i], preds[:, i])\n    for i in range(12)\n    if len(np.unique(true[:, i])) == 2\n]), 3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T05:01:50.628893Z","iopub.execute_input":"2026-09-04T05:01:50.629316Z","iopub.status.idle":"2026-09-04T05:01:50.670384Z","shell.execute_reply.started":"2026-09-04T05:01:50.629286Z","shell.execute_reply":"2026-09-04T05:01:50.669512Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nMODEL_PATH = \"/kaggle/working/knee_best_model.pth\"\n\nprint(\"Model exists:\", os.path.exists(MODEL_PATH))\nprint(\"File size:\", os.path.getsize(MODEL_PATH) / 1024**2, \"MB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T05:01:54.472483Z","iopub.execute_input":"2026-09-04T05:01:54.472765Z","iopub.status.idle":"2026-09-04T05:01:54.478544Z","shell.execute_reply.started":"2026-09-04T05:01:54.47274Z","shell.execute_reply":"2026-09-04T05:01:54.477472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\nimport pandas as pd\nimport numpy as np\nimport torch\nimport pydicom\nfrom PIL import Image\n\nmodel.eval()\n# 1. TEST MRI LOADER\ndef load_test_study(\n    study_id,\n    num_slices=16,\n    image_size=224\n):\n\n    study_id = str(study_id)\n\n    # Get all series belonging to this study\n    rows = test_series[\n        test_series[\"StudyInstanceUID\"].astype(str)\n        == study_id\n    ]\n\n    all_series = []\n\n    for _, row in rows.iterrows():\n\n        series_id = str(\n            row[\"SeriesInstanceUID\"]\n        )\n\n        # Expected test series directory\n        series_path = os.path.join(\n            TEST_DIR,\n            study_id,\n            series_id\n        )\n\n        # If not found, search recursively\n        if not os.path.isdir(series_path):\n\n            matches = glob.glob(\n                f\"{TEST_DIR}/**/{series_id}\",\n                recursive=True\n            )\n\n            if len(matches) > 0:\n                series_path = matches[0]\n\n        if not os.path.isdir(series_path):\n            print(\n                \"Series directory not found:\",\n                series_id\n            )\n            continue\n\n        # Find DICOM files\n        dcm_files = glob.glob(\n            os.path.join(\n                series_path,\n                \"*.dcm\"\n            )\n        )\n\n        if len(dcm_files) == 0:\n\n            # Try all files if extension is different\n            dcm_files = glob.glob(\n                os.path.join(\n                    series_path,\n                    \"*\"\n                )\n            )\n\n        slices = []\n\n        for file in dcm_files:\n\n            try:\n\n                ds = pydicom.dcmread(\n                    file,\n                    force=True\n                )\n\n                image = ds.pixel_array.astype(\n                    np.float32\n                )\n\n                # Normalize\n                image = image - image.min()\n\n                if image.max() > 0:\n                    image = image / image.max()\n\n                instance = getattr(\n                    ds,\n                    \"InstanceNumber\",\n                    0\n                )\n\n                slices.append(\n                    (instance, image)\n                )\n\n            except Exception:\n                continue\n\n        if len(slices) == 0:\n            continue\n\n        # Sort slices\n        slices.sort(\n            key=lambda x: x[0]\n        )\n\n        images = [\n            x[1]\n            for x in slices\n        ]\n\n        # Select 16 representative slices\n        if len(images) >= num_slices:\n\n            indices = np.linspace(\n                0,\n                len(images) - 1,\n                num_slices\n            ).astype(int)\n\n            images = [\n                images[i]\n                for i in indices\n            ]\n\n        else:\n\n            # Repeat last slice\n            while len(images) < num_slices:\n                images.append(\n                    images[-1]\n                )\n\n        # Resize to 224 x 224\n        resized = []\n\n        for image in images:\n\n            image = Image.fromarray(\n                (image * 255).astype(\n                    np.uint8\n                )\n            )\n\n            image = image.resize(\n                (image_size, image_size)\n            )\n\n            image = np.asarray(\n                image,\n                dtype=np.float32\n            ) / 255.0\n\n            resized.append(image)\n\n        all_series.append(\n            np.stack(resized)\n        )\n\n    return all_series\n\n# 2. GENERATE TEST PREDICTIONS\ntest_preds = []\n\ntest_ids = (\n    test_series[\n        \"StudyInstanceUID\"\n    ]\n    .astype(str)\n    .unique()\n)\n\nprint(\n    \"Number of test studies:\",\n    len(test_ids)\n)\n\nfor sid in test_ids:\n\n    print(\n        \"\\nProcessing:\",\n        sid\n    )\n\n    # Load from TEST_DIR, not train loader\n    series = load_test_study(sid)\n\n    print(\n        \"Number of series loaded:\",\n        len(series)\n    )\n\n    outputs = []\n\n    with torch.no_grad():\n\n        for s in series:\n\n            # Model returns 12 predictions\n            output = model([s])\n\n            outputs.append(output)\n\n    if len(outputs) > 0:\n\n        # Average predictions across series\n        logits = torch.stack(\n            outputs\n        ).mean(0)\n\n        pred = torch.sigmoid(\n            logits\n        )\n\n    else:\n        print(\n            \"WARNING: No MRI series loaded!\"\n        )\n\n        pred = torch.full(\n            (12,),\n            0.5,\n            device=device\n        )\n\n    test_preds.append(\n        pred.cpu().numpy()\n    )\n\n\n# Convert to NumPy\ntest_preds = np.array(\n    test_preds\n)\n\n# 3. CHECK PREDICTIONS\nprint(\n    \"\\n================================\"\n)\n\nprint(\n    \"TEST PREDICTION RESULTS\"\n)\n\nprint(\n    \"================================\"\n)\n\nprint(\n    \"Prediction shape:\",\n    test_preds.shape\n)\n\nprint(\n    test_preds\n)\n\nprint(\n    \"\\nMinimum prediction:\",\n    test_preds.min()\n)\n\nprint(\n    \"Maximum prediction:\",\n    test_preds.max()\n)\n\n# 4. LOAD KAGGLE SAMPLE SUBMISSION\nsample_files = glob.glob(\n    \"/kaggle/input/**/sample_submission.csv\",\n    recursive=True\n)\n\nif len(sample_files) == 0:\n\n    raise FileNotFoundError(\n        \"sample_submission.csv not found in Kaggle Input.\"\n    )\n\nsample_path = sample_files[0]\n\nprint(\n    \"\\nSample submission:\",\n    sample_path\n)\n\nsample_submission = pd.read_csv(\n    sample_path\n)\n\n# 5. PUT PREDICTIONS INTO SAMPLE FILE\nfor i, sid in enumerate(\n    sample_submission[\n        \"StudyInstanceUID\"\n    ].astype(str)\n):\n\n    matches = np.where(\n        test_ids == sid\n    )[0]\n\n    if len(matches) > 0:\n\n        sample_submission.loc[\n            i,\n            target_cols\n        ] = test_preds[\n            matches[0]\n        ]\n\n# 6. SAVE FINAL SUBMISSION\noutput_path = (\n    \"/kaggle/working/\"\n    \"sample_submission.csv\"\n)\n\nsample_submission.to_csv(\n    output_path,\n    index=False\n)\n\n# 7. FINAL CHECK\nprint(\n    \"\\n================================\"\n)\n\nprint(\n    \"FINAL sample_submission.csv\"\n)\n\nprint(\n    \"================================\"\n)\n\nprint(\n    sample_submission\n)\n\nprint(\n    \"\\nShape:\",\n    sample_submission.shape\n)\n\nprint(\n    \"\\nSaved at:\"\n)\n\nprint(\n    output_path\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T05:13:54.296482Z","iopub.execute_input":"2026-09-04T05:13:54.297321Z","iopub.status.idle":"2026-09-04T05:16:55.023894Z","shell.execute_reply.started":"2026-09-04T05:13:54.29729Z","shell.execute_reply":"2026-09-04T05:16:55.022722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\npath = \"/kaggle/working/sample_submission.csv\"\nprint(\"Exists:\", os.path.exists(path))\nprint(\"Size:\", os.path.getsize(path), \"bytes\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T05:19:44.19067Z","iopub.execute_input":"2026-09-04T05:19:44.191527Z","iopub.status.idle":"2026-09-04T05:19:44.197384Z","shell.execute_reply.started":"2026-09-04T05:19:44.191494Z","shell.execute_reply":"2026-09-04T05:19:44.195948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"sample_submission columns:\")\nprint(sample_submission.columns.tolist())\n\nprint(\"\\nShape:\", sample_submission.shape)\nprint(\"\\nMissing values:\")\nprint(sample_submission.isna().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T05:19:48.047802Z","iopub.execute_input":"2026-09-04T05:19:48.048694Z","iopub.status.idle":"2026-09-04T05:19:48.055977Z","shell.execute_reply.started":"2026-09-04T05:19:48.04866Z","shell.execute_reply":"2026-09-04T05:19:48.055005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}