{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pydicom\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport glob\nimport os\n\n# 1. Dynamically find the competition root folder\nbase_path = \"\"\nfor root, dirs, files in os.walk('/kaggle/input'):\n    if 'train.csv' in files:\n        base_path = root\n        break\n\nprint(f\"Dataset dynamically found at: {base_path}\")\n\n# 2. Load the training data to get a valid Study ID\ntrain_df = pd.read_csv(f'{base_path}/train.csv')\nsample_study = train_df['StudyInstanceUID'].iloc[0]\n\n# 3. FOOLPROOF SEARCH: Recursively find any .dcm file inside the Study ID folder\nprint(f\"Searching for DICOMs belonging to Study: {sample_study}...\")\ndicom_path_pattern = f\"{base_path}/**/{sample_study}/**/*.dcm\"\ndicom_files = glob.glob(dicom_path_pattern, recursive=True)\n\nprint(f\"Found {len(dicom_files)} slices!\")\n\n# 4. Read the first DICOM file and Visualize\nif len(dicom_files) > 0:\n    sample_file = dicom_files[0]\n    dcm = pydicom.dcmread(sample_file)\n\n    # 5. Extract critical hidden metadata\n    print(\"\\n--- DICOM METADATA ---\")\n    print(f\"Patient ID: {getattr(dcm, 'PatientID', 'Not Found')}\")\n    print(f\"Study Description: {getattr(dcm, 'StudyDescription', 'Not Found')}\")\n    print(f\"Series Description: {getattr(dcm, 'SeriesDescription', 'Not Found')}\")\n    print(f\"Image Shape: {dcm.pixel_array.shape}\")\n\n    # 6. Visualize the raw MRI slice\n    plt.figure(figsize=(8, 8))\n    plt.imshow(dcm.pixel_array, cmap='bone')\n    plt.title(f\"Raw MRI Slice - Patient: {getattr(dcm, 'PatientID', 'Unknown')}\")\n    plt.axis('off')\n    plt.show()\nelse:\n    print(\"Error: Still could not find the image files. Kaggle might be linking the data differently.\")\n    print(\"Available folders in base path:\", os.listdir(base_path))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport glob\nimport cv2  # Added OpenCV for image resizing\n\n# 1. Grab all DICOM files for a specific series within our sample study\nseries_path_pattern = f\"{base_path}/**/{sample_study}/*/*.dcm\"\ndicom_files = glob.glob(series_path_pattern, recursive=True)\n\n# 2. Load all DICOMs in the series\nslices = [pydicom.dcmread(f) for f in dicom_files]\n\n# 3. Sort slices by their actual physical position in the scanner (InstanceNumber)\nslices.sort(key=lambda x: int(x.InstanceNumber))\n\nprint(f\"Loaded and sorted {len(slices)} slices for this series.\")\n\n# 4. Find the middle of the knee\nmid_idx = len(slices) // 2\n\n# 5. Define our Neural Network target resolution\nTARGET_SIZE = (256, 256)\n\n# 6. Extract slices AND resize them to identical dimensions\nslice_1 = cv2.resize(slices[mid_idx - 1].pixel_array.astype(float), TARGET_SIZE)\nslice_2 = cv2.resize(slices[mid_idx].pixel_array.astype(float), TARGET_SIZE)\nslice_3 = cv2.resize(slices[mid_idx + 1].pixel_array.astype(float), TARGET_SIZE)\n\n# 7. Normalize the pixel contrast\ndef normalize_image(img):\n    img_min = img.min()\n    img_max = img.max()\n    if img_max > img_min:\n        return (img - img_min) / (img_max - img_min)\n    return img\n\nslice_1 = normalize_image(slice_1)\nslice_2 = normalize_image(slice_2)\nslice_3 = normalize_image(slice_3)\n\n# 8. Stack them into a 3-Channel \"RGB\" format\nimage_25d = np.stack([slice_1, slice_2, slice_3], axis=-1)\n\nprint(f\"Successfully built 2.5D Array with uniform shape: {image_25d.shape}\")\n\n# 9. Visualize the 3 channels separately\nfig, axes = plt.subplots(1, 3, figsize=(15, 5))\naxes[0].imshow(image_25d[:, :, 0], cmap='gray')\naxes[0].set_title(f\"Slice {mid_idx - 1}\")\naxes[0].axis('off')\n\naxes[1].imshow(image_25d[:, :, 1], cmap='gray')\naxes[1].set_title(f\"Slice {mid_idx} (Center)\")\naxes[1].axis('off')\n\naxes[2].imshow(image_25d[:, :, 2], cmap='gray')\naxes[2].set_title(f\"Slice {mid_idx + 1}\")\naxes[2].axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install timm","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport pydicom\nimport cv2\nimport numpy as np\nimport glob\nimport timm\nimport pandas as pd\n\n\n# 1. The 2.5D PyTorch Dataset\n\nclass RSNAKnee25DDataset(Dataset):\n    def __init__(self, df, base_path, target_size=(256, 256)):\n        self.df = df\n        self.base_path = base_path\n        self.target_size = target_size\n        # Assuming the 12 targets are the remaining columns\n        self.target_cols = [c for c in df.columns if c not in ['StudyInstanceUID', 'Report', 'fold']]\n\n    def __len__(self):\n        return len(self.df)\n\n    def normalize_image(self, img):\n        img_min = img.min()\n        img_max = img.max()\n        if img_max > img_min:\n            return (img - img_min) / (img_max - img_min)\n        return img\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        study_id = row['StudyInstanceUID']\n        \n        \n        series_path_pattern = f\"{self.base_path}/**/{study_id}/*/*.dcm\"\n        dicom_files = glob.glob(series_path_pattern, recursive=True)\n        \n        \n        if len(dicom_files) < 3:\n            image_25d = np.zeros((3, self.target_size[0], self.target_size[1]), dtype=np.float32)\n        else:\n            slices = [pydicom.dcmread(f) for f in dicom_files]\n            slices.sort(key=lambda x: int(x.InstanceNumber))\n            \n            mid_idx = len(slices) // 2\n            \n            # Extract, resize, and normalize 3 slices\n            s1 = cv2.resize(slices[mid_idx - 1].pixel_array.astype(float), self.target_size)\n            s2 = cv2.resize(slices[mid_idx].pixel_array.astype(float), self.target_size)\n            s3 = cv2.resize(slices[mid_idx + 1].pixel_array.astype(float), self.target_size)\n            \n            s1 = self.normalize_image(s1)\n            s2 = self.normalize_image(s2)\n            s3 = self.normalize_image(s3)\n            \n            # PyTorch expects channels first: (Channels, Height, Width)\n            image_25d = np.stack([s1, s2, s3], axis=0).astype(np.float32)\n\n       \n        labels = row[self.target_cols].fillna(-1).values.astype(np.float32)\n        \n        return {\n            'image': torch.tensor(image_25d),\n            'labels': torch.tensor(labels)\n        }\n\n\n# 2. The Vision Model Architecture\n\nclass RSNAVisionModel(nn.Module):\n    def __init__(self, model_name='resnet18d', num_classes=12, pretrained=True):\n        super().__init__()\n        # Create the backbone using timm, stripping the original classification head (num_classes=0)\n        self.backbone = timm.create_model(model_name, pretrained=pretrained, num_classes=0)\n        \n        # Add our custom multi-label head\n        self.head = nn.Sequential(\n            nn.Dropout(0.2),\n            nn.Linear(self.backbone.num_features, num_classes)\n        )\n\n    def forward(self, x):\n        features = self.backbone(x)\n        return self.head(features)\n\n# --- Test the initialization to make sure everything links up ---\nprint(\"Initializing pipeline...\")\nmodel = RSNAVisionModel(model_name='resnet18d', num_classes=12)\nprint(f\"Model built successfully! Output features per image: {model.backbone.num_features}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T18:43:43.602265Z","iopub.execute_input":"2026-08-06T18:43:43.603192Z","iopub.status.idle":"2026-08-06T18:44:42.617539Z","shell.execute_reply.started":"2026-08-06T18:43:43.603154Z","shell.execute_reply":"2026-08-06T18:44:42.616775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\nimport warnings\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\nimport cv2\nimport pydicom\nimport numpy as np\n\nwarnings.filterwarnings('ignore')\n\n# 1. Dynamically locate base_path\nbase_path = \"\"\nfor root, dirs, files in os.walk('/kaggle/input'):\n    if 'train.csv' in files:\n        base_path = root\n        break\n\nprint(f\"Dataset path loaded: {base_path}\")\n\n# 2. Load Data and take a small subset (200 rows) for our speed test\ntrain_df = pd.read_csv(f'{base_path}/train.csv').head(200)\ntrain_data, val_data = train_test_split(train_df, test_size=0.2, random_state=42)\n\n# 3. UPGRADED DATASET CLASS (No recursive searching)\nclass RSNAKnee25DDataset(Dataset):\n    def __init__(self, df, base_path, target_size=(256, 256)):\n        self.df = df\n        self.base_path = base_path\n        self.target_size = target_size\n        self.target_cols = [c for c in df.columns if c not in ['StudyInstanceUID', 'Report', 'fold']]\n\n    def __len__(self):\n        return len(self.df)\n\n    def normalize_image(self, img):\n        img_min = img.min()\n        img_max = img.max()\n        if img_max > img_min:\n            return (img - img_min) / (img_max - img_min)\n        return img\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        study_id = row['StudyInstanceUID']\n        \n        # FIX 1: Direct wildcard path instead of recursive search. \n        # This is 1000x faster for the file system to locate.\n        series_path_pattern = f\"{self.base_path}/train_*/{study_id}/*/*.dcm\"\n        dicom_files = glob.glob(series_path_pattern)\n        \n        if len(dicom_files) < 3:\n            image_25d = np.zeros((3, self.target_size[0], self.target_size[1]), dtype=np.float32)\n        else:\n            slices = [pydicom.dcmread(f) for f in dicom_files]\n            slices.sort(key=lambda x: int(x.InstanceNumber))\n            \n            mid_idx = len(slices) // 2\n            \n            s1 = cv2.resize(slices[mid_idx - 1].pixel_array.astype(float), self.target_size)\n            s2 = cv2.resize(slices[mid_idx].pixel_array.astype(float), self.target_size)\n            s3 = cv2.resize(slices[mid_idx + 1].pixel_array.astype(float), self.target_size)\n            \n            s1 = self.normalize_image(s1)\n            s2 = self.normalize_image(s2)\n            s3 = self.normalize_image(s3)\n            \n            image_25d = np.stack([s1, s2, s3], axis=0).astype(np.float32)\n\n        labels = row[self.target_cols].fillna(-1).values.astype(np.float32)\n        \n        return {'image': torch.tensor(image_25d), 'labels': torch.tensor(labels)}\n\n# 4. Create DataLoaders\ntrain_dataset = RSNAKnee25DDataset(train_data, base_path)\nval_dataset = RSNAKnee25DDataset(val_data, base_path)\n\n# FIX 2: num_workers=0 prevents PyTorch from deadlocking on Kaggle's shared CPUs\ntrain_loader = DataLoader(train_dataset, batch_size=8, shuffle=True, num_workers=0)\nval_loader = DataLoader(val_dataset, batch_size=8, shuffle=False, num_workers=0)\n\n# 5. Setup Device, Model, Optimizer, and Loss\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nmodel = RSNAVisionModel(model_name='resnet18d', num_classes=12).to(device)\noptimizer = optim.AdamW(model.parameters(), lr=1e-4)\ncriterion = nn.BCEWithLogitsLoss(reduction='none') \n\n# 6. The Mini-Training Loop (1 Epoch)\nprint(f\"--- Starting Training on {device.type.upper()} ---\")\nmodel.train()\ntotal_loss = 0.0\n\nfor batch in tqdm(train_loader, desc=\"Training Epoch 1\"):\n    images = batch['image'].to(device)\n    labels = batch['labels'].to(device)\n    \n    optimizer.zero_grad()\n    outputs = model(images)\n    \n    valid_mask = (labels != -1).float()\n    loss = (criterion(outputs, labels) * valid_mask).sum() / torch.clamp(valid_mask.sum(), min=1.0)\n    \n    loss.backward()\n    optimizer.step()\n    total_loss += loss.item()\n\nprint(f\"\\nEpoch 1 Complete! | Average Training Loss: {total_loss/len(train_loader):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T18:53:53.847936Z","iopub.execute_input":"2026-08-06T18:53:53.848357Z","iopub.status.idle":"2026-08-06T18:58:39.330923Z","shell.execute_reply.started":"2026-08-06T18:53:53.848328Z","shell.execute_reply":"2026-08-06T18:58:39.329871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\nprint(\"Searching for CSV files in Kaggle input directory...\\n\")\n\n# This loops through all folders inside Kaggle's data directory\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if filename.endswith('.csv'):\n            filepath = os.path.join(dirname, filename)\n            print(f\"--- Found: {filepath} ---\")\n            \n            try:\n                # Read just the first row to get columns quickly\n                df = pd.read_csv(filepath, nrows=1)\n                print(\"Columns:\", df.columns.tolist())\n                print() # blank line for readability\n            except Exception as e:\n                print(f\"Could not read columns. Error: {e}\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T19:03:44.392523Z","iopub.execute_input":"2026-08-06T19:03:44.39351Z","iopub.status.idle":"2026-08-06T19:05:49.207208Z","shell.execute_reply.started":"2026-08-06T19:03:44.393474Z","shell.execute_reply":"2026-08-06T19:05:49.206456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pydicom\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport glob\n\n# 1. Setup Base Output Path\nBASE_DIR = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\nOUTPUT_DIR = \"/kaggle/working/train_npy_25d\"\nos.makedirs(OUTPUT_DIR, exist_ok=True)\n\n# 2. Load and Merge Data\ntrain_labels = pd.read_csv(f'{BASE_DIR}/train.csv')\ntrain_series = pd.read_csv(f'{BASE_DIR}/train_series.csv')\ndf_master = pd.merge(train_series, train_labels, on='StudyInstanceUID', how='left')\n\n\n# THE BLOODHOUND: Find where Kaggle actually hid the images\n\nfirst_study_id = str(df_master.iloc[0]['StudyInstanceUID'])\nprint(f\"Hunting for the actual location of Study ID: {first_study_id}...\")\n\nINPUT_DIR = None\n# Scan the input directory for the exact location of our first patient\nfor root, dirs, files in os.walk('/kaggle/input'):\n    if first_study_id in dirs:\n        INPUT_DIR = root\n        break\n\nif INPUT_DIR is None:\n    raise Exception(\"Could not find the image folders anywhere! Ensure the actual image dataset (not just the CSVs) is attached to your notebook environment.\")\n    \nprint(f\"Found them! The true image directory is: {INPUT_DIR}\\n\")\n\n\n\ndef extract_25d_stack(series_dir):\n    \"\"\"Robust extractor that handles missing paths and missing .dcm extensions.\"\"\"\n    if not os.path.exists(series_dir):\n        print(f\"  -> Missing Path: {series_dir}\")\n        return None\n        \n    # Check for standard .dcm files\n    dicom_files = glob.glob(os.path.join(series_dir, \"*.dcm\"))\n    \n    # Fallback: If no .dcm files, grab all files (Kaggle sometimes omits extensions)\n    if not dicom_files:\n        dicom_files = [os.path.join(series_dir, f) for f in os.listdir(series_dir) \n                       if os.path.isfile(os.path.join(series_dir, f))]\n        \n    if not dicom_files:\n        print(f\"  -> Empty Directory: {series_dir}\")\n        return None\n        \n    try:\n        # Read headers and sort by spatial location\n        slices = [pydicom.dcmread(f) for f in dicom_files]\n        slices.sort(key=lambda x: int(getattr(x, 'InstanceNumber', 0)))\n        \n        # 2.5D Logic: Bottom, Middle, Top\n        n = len(slices)\n        if n < 3:\n            idx = [0, n//2, n-1]\n        else:\n            idx = [n//4, n//2, 3*n//4]\n            \n        img_stack = np.stack([slices[i].pixel_array for i in idx])\n        return img_stack.astype(np.float32)\n    except Exception as e:\n        print(f\"  -> Error reading DICOMs in {series_dir}: {e}\")\n        return None\n\n# 3. Process the Data (Batch of 5)\nprocessed_records = []\nprint(\"Starting 2.5D conversion...\\n\")\n\nfor _, row in tqdm(df_master.head(100).iterrows(), total=min(100, len(df_master))):\n    \n    study_id = str(row['StudyInstanceUID'])\n    series_id = str(row['SeriesInstanceUID'])\n    plane = row['Anatomical_Plane']\n    \n    # Build the path using the true INPUT_DIR we hunted down\n    series_path = os.path.join(INPUT_DIR, study_id, series_id)\n    \n    # Extract 2.5D volume\n    img_array = extract_25d_stack(series_path)\n    if img_array is None:\n        continue\n        \n    # Save as fast .npy\n    save_name = f\"{study_id}_{series_id}_{plane}.npy\"\n    save_path = os.path.join(OUTPUT_DIR, save_name)\n    np.save(save_path, img_array)\n    \n    # Log successful conversion\n    row_dict = row.to_dict()\n    row_dict['npy_path'] = save_path\n    processed_records.append(row_dict)\n\n# 4. Save and Verify\ndf_processed = pd.DataFrame(processed_records)\n\n# Safe check before printing to avoid the IndexError\nif len(df_processed) == 0:\n    print(\"\\nFAILURE: No records were processed. Check the debug messages above.\")\nelse:\n    df_processed.to_csv('/kaggle/working/train_25d_metadata.csv', index=False)\n    print(\"\\nConversion complete! Metadata saved to 'train_25d_metadata.csv'\")\n    print(f\"Successfully processed {len(df_processed)} arrays.\")\n    print(f\"First array shape: {np.load(df_processed.iloc[0]['npy_path']).shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T19:24:44.40152Z","iopub.execute_input":"2026-08-06T19:24:44.402142Z","iopub.status.idle":"2026-08-06T19:24:55.965826Z","shell.execute_reply.started":"2026-08-06T19:24:44.402112Z","shell.execute_reply":"2026-08-06T19:24:55.964932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install iterative-stratification","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T19:17:27.871771Z","iopub.execute_input":"2026-08-06T19:17:27.872239Z","iopub.status.idle":"2026-08-06T19:17:31.651274Z","shell.execute_reply.started":"2026-08-06T19:17:27.872207Z","shell.execute_reply":"2026-08-06T19:17:31.650447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom iterstrat.ml_stratifiers import MultilabelStratifiedKFold\n\n# 1. Load the metadata we just created\ndf = pd.read_csv('/kaggle/working/train_25d_metadata.csv')\n\n# 2. Define our 12 target columns\nTARGETS = [\n    'ACL', 'MCL', 'Medial Meniscus', 'Lateral Meniscus', 'Medial OA', \n    'Lateral OA', 'PF OA', 'Effusion', 'Synovitis', \"Baker's\", \n    'Contusion', 'Fracture'\n]\n\n# CRITICAL FIX: Fill missing target values with 0. \n# The stratifier cannot perform distribution math on NaNs.\ndf[TARGETS] = df[TARGETS].fillna(0)\n\n# 3. Create a patient-level dataframe\n# Grouping by StudyInstanceUID ensures exactly one row per patient.\ndf_patients = df.groupby('StudyInstanceUID')[TARGETS].max().reset_index()\n\n# 4. Set up the Multilabel Stratified K-Fold (on the patients, not the individual scans)\n\nmskf = MultilabelStratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# 5. Initialize the fold column in the patient dataframe\ndf_patients['fold'] = -1\n\n# 6. Execute the split on the patient level\n# X = the patient dataframe (dummy data)\n# y = the multiple labels we need balanced\nfor fold, (train_idx, val_idx) in enumerate(mskf.split(X=df_patients, y=df_patients[TARGETS])):\n    df_patients.loc[val_idx, 'fold'] = fold\n\n# 7. Map the folds back to the main DataFrame\n# Because we merge on StudyInstanceUID, all scans from the same patient get the exact same fold\ndf = df.merge(df_patients[['StudyInstanceUID', 'fold']], on='StudyInstanceUID', how='left')\n\n# 8. Save the final training manifest\ndf.to_csv('/kaggle/working/train_25d_folds.csv', index=False)\n\nprint(\"Folds successfully assigned without leakage!\")\nprint(\"\\nScan counts per fold:\")\nprint(df['fold'].value_counts().sort_index())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T19:24:59.961345Z","iopub.execute_input":"2026-08-06T19:24:59.962245Z","iopub.status.idle":"2026-08-06T19:25:00.005077Z","shell.execute_reply.started":"2026-08-06T19:24:59.962207Z","shell.execute_reply":"2026-08-06T19:25:00.004037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import Dataset\nimport numpy as np\n\nclass RSNAKneeDataset25D(Dataset):\n    def __init__(self, df, targets, transforms=None):\n        self.df = df\n        self.targets = targets\n        self.transforms = transforms\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        \n        # 1. Load the ultra-fast .npy array (Shape: 3, Height, Width)\n        img = np.load(row['npy_path'])\n        \n        # 2. Apply Augmentations (if you have Albumentations/Torchvision later)\n        \n        if self.transforms:\n            pass \n        \n        # 3. Convert image and labels to PyTorch Tensors\n        img_tensor = torch.tensor(img, dtype=torch.float32)\n        label_tensor = torch.tensor(row[self.targets].values.astype(np.float32))\n        \n        return img_tensor, label_tensor\n\n\ntest_dataset = RSNAKneeDataset25D(df=df, targets=TARGETS)\nsample_img, sample_labels = test_dataset[0]\n\nprint(f\"Tensor Shape: {sample_img.shape}\")\nprint(f\"Labels Shape: {sample_labels.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T19:28:48.469314Z","iopub.execute_input":"2026-08-06T19:28:48.469846Z","iopub.status.idle":"2026-08-06T19:28:48.482231Z","shell.execute_reply.started":"2026-08-06T19:28:48.469815Z","shell.execute_reply":"2026-08-06T19:28:48.481227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.models as models\nimport torchvision.transforms as T\nimport numpy as np\n\n# 1. Define Dataset With Resizing\nclass RSNAKneeDataset25D(Dataset):\n    def __init__(self, df, targets, img_size=(512, 512)):\n        self.df = df\n        self.targets = targets\n        self.resize = T.Resize(img_size, antialias=True)\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img = np.load(row['npy_path'])\n        \n        img_tensor = torch.tensor(img, dtype=torch.float32)\n        img_tensor = self.resize(img_tensor)\n        \n        label_tensor = torch.tensor(row[self.targets].values.astype(np.float32))\n        \n        return img_tensor, label_tensor\n\n# 2. Define Targets\nTARGETS = [\n    'ACL', 'MCL', 'Medial Meniscus', 'Lateral Meniscus', 'Medial OA', \n    'Lateral OA', 'PF OA', 'Effusion', 'Synovitis', \"Baker's\", \n    'Contusion', 'Fracture'\n]\n\n# 3. Split Data\ntrain_df = df[df['fold'] != 0].reset_index(drop=True)\nvalid_df = df[df['fold'] == 0].reset_index(drop=True)\n\n# 4. Create DataLoaders\ntrain_dataset = RSNAKneeDataset25D(train_df, TARGETS)\nvalid_dataset = RSNAKneeDataset25D(valid_df, TARGETS)\n\ntrain_loader = DataLoader(train_dataset, batch_size=4, shuffle=True)\nvalid_loader = DataLoader(valid_dataset, batch_size=4, shuffle=False)\n\n# 5. Build Model\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f\"Using device: {device}\")\n\nmodel = models.resnet18(weights=models.ResNet18_Weights.DEFAULT)\nmodel.fc = nn.Linear(model.fc.in_features, len(TARGETS))\nmodel = model.to(device)\n\n# 6. Loss and Optimizer\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = optim.Adam(model.parameters(), lr=1e-4)\n\n# 7. Mini-Epoch Loop\nprint(\"\\nStarting Mini-Epoch Test...\")\nfor epoch in range(2):\n    model.train()\n    train_loss = 0.0\n    \n    for imgs, labels in train_loader:\n        imgs, labels = imgs.to(device), labels.to(device)\n        \n        optimizer.zero_grad()\n        outputs = model(imgs)\n        \n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        \n        train_loss += loss.item()\n        \n    avg_train_loss = train_loss / len(train_loader)\n    \n    model.eval()\n    val_loss = 0.0\n    with torch.no_grad():\n        for imgs, labels in valid_loader:\n            imgs, labels = imgs.to(device), labels.to(device)\n            outputs = model(imgs)\n            loss = criterion(outputs, labels)\n            val_loss += loss.item()\n            \n    avg_val_loss = val_loss / len(valid_loader)\n    \n    print(f\"Epoch {epoch+1}/2 | Train Loss: {avg_train_loss:.4f} | Val Loss: {avg_val_loss:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T19:33:18.585516Z","iopub.execute_input":"2026-08-06T19:33:18.586011Z","iopub.status.idle":"2026-08-06T19:33:22.522501Z","shell.execute_reply.started":"2026-08-06T19:33:18.585981Z","shell.execute_reply":"2026-08-06T19:33:22.521477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Clear Working Directory\nimport shutil\nimport os\n\nif os.path.exists('/kaggle/working/train_npy_25d'):\n    shutil.rmtree('/kaggle/working/train_npy_25d')\n    print(\"Cleared old files. Disk space restored!\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pydicom\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport glob\nimport cv2\n\n# Setup Paths\nBASE_DIR = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\nOUTPUT_DIR = \"/kaggle/working/train_npy_25d\"\nos.makedirs(OUTPUT_DIR, exist_ok=True)\n\n# Load Data\ntrain_labels = pd.read_csv(f'{BASE_DIR}/train.csv')\ntrain_series = pd.read_csv(f'{BASE_DIR}/train_series.csv')\ndf_master = pd.merge(train_series, train_labels, on='StudyInstanceUID', how='left')\n\n# Locate Image Directory\nfirst_study_id = str(df_master.iloc[0]['StudyInstanceUID'])\nprint(f\"Hunting for: {first_study_id}...\")\n\nINPUT_DIR = None\nfor root, dirs, files in os.walk('/kaggle/input'):\n    if first_study_id in dirs:\n        INPUT_DIR = root\n        break\n\nif INPUT_DIR is None:\n    raise Exception(\"Could not find the image folders anywhere!\")\n    \nprint(f\"Found true directory: {INPUT_DIR}\\n\")\n\n# Extraction and Compression Function\ndef extract_25d_stack(series_dir, img_size=(256, 256)):\n    if not os.path.exists(series_dir):\n        return None\n        \n    dicom_files = glob.glob(os.path.join(series_dir, \"*.dcm\"))\n    if not dicom_files:\n        dicom_files = [os.path.join(series_dir, f) for f in os.listdir(series_dir) \n                       if os.path.isfile(os.path.join(series_dir, f))]\n        \n    if not dicom_files:\n        return None\n        \n    try:\n        slices = [pydicom.dcmread(f) for f in dicom_files]\n        slices.sort(key=lambda x: int(getattr(x, 'InstanceNumber', 0)))\n        \n        n = len(slices)\n        if n < 3:\n            idx = [0, n//2, n-1]\n        else:\n            idx = [n//4, n//2, 3*n//4]\n            \n        processed_slices = []\n        for i in idx:\n            img = slices[i].pixel_array.astype(np.float32)\n            img = cv2.resize(img, img_size, interpolation=cv2.INTER_AREA)\n            processed_slices.append(img)\n            \n        img_stack = np.stack(processed_slices)\n        return img_stack.astype(np.float16)\n    except Exception:\n        return None\n\n# Process Dataset\nprocessed_records = []\nprint(\"Starting memory-optimized 2.5D dataset conversion...\\n\")\n\nfor _, row in tqdm(df_master.iterrows(), total=len(df_master)):\n    study_id = str(row['StudyInstanceUID'])\n    series_id = str(row['SeriesInstanceUID'])\n    plane = row['Anatomical_Plane']\n    \n    series_path = os.path.join(INPUT_DIR, study_id, series_id)\n    \n    img_array = extract_25d_stack(series_path)\n    if img_array is None:\n        continue\n        \n    save_name = f\"{study_id}_{series_id}_{plane}.npy\"\n    save_path = os.path.join(OUTPUT_DIR, save_name)\n    np.save(save_path, img_array)\n    \n    row_dict = row.to_dict()\n    row_dict['npy_path'] = save_path\n    processed_records.append(row_dict)\n\n# Save Metadata\ndf_processed = pd.DataFrame(processed_records)\n\nif len(df_processed) > 0:\n    df_processed.to_csv('/kaggle/working/train_25d_metadata.csv', index=False)\n    print(\"\\nFull conversion complete! Metadata saved to 'train_25d_metadata.csv'\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}