{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":13451,"datasetId":654585,"databundleVersionId":1188070}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-23T22:54:37.111368Z","iopub.execute_input":"2026-03-23T22:54:37.112004Z","iopub.status.idle":"2026-03-23T22:56:08.787367Z","shell.execute_reply.started":"2026-03-23T22:54:37.11197Z","shell.execute_reply":"2026-03-23T22:56:08.786281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\n\nprint(f\"PyTorch version: {torch.__version__}\")\nprint(f\"GPU available: {torch.cuda.is_available()}\")\nprint(f\"GPU name: {torch.cuda.get_device_name(0) if torch.cuda.is_available() else 'None'}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-25T10:51:32.158153Z","iopub.execute_input":"2026-03-25T10:51:32.158911Z","iopub.status.idle":"2026-03-25T10:51:36.644212Z","shell.execute_reply.started":"2026-03-25T10:51:32.158879Z","shell.execute_reply":"2026-03-25T10:51:36.643567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\npath = \"/kaggle/input/competitions/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection\"\nfiles = os.listdir(path)\nprint(f\"Files in dataset: {files}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-25T10:51:39.414558Z","iopub.execute_input":"2026-03-25T10:51:39.414916Z","iopub.status.idle":"2026-03-25T10:51:39.433993Z","shell.execute_reply.started":"2026-03-25T10:51:39.414891Z","shell.execute_reply":"2026-03-25T10:51:39.433172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nimport os\n\n# ── CHECK DAMAGE ─────────────────────────────────────────────────\ntotal_bytes = sum(\n    os.path.getsize(os.path.join(root, f))\n    for root, dirs, files in os.walk(OUT_DIR)\n    for f in files\n)\nprint(f\"Current disk usage: {total_bytes / 1024 / 1024 / 1024:.2f} GB\")\n\nfiles_written = len(os.listdir(f\"{OUT_DIR}/images/train\"))\nprint(f\"Files written so far: {files_written:,}\")\n\n# ── WIPE EVERYTHING AND START CLEAN ──────────────────────────────\nprint(\"\\nWiping output directory...\")\nshutil.rmtree(OUT_DIR)\n\nfor split in ['train', 'val']:\n    os.makedirs(f\"{OUT_DIR}/images/{split}\", exist_ok=True)\n    os.makedirs(f\"{OUT_DIR}/labels/{split}\", exist_ok=True)\n\nprint(\"Clean. Ready to restart with JPEG.\")\nprint(f\"Disk usage now: {sum(os.path.getsize(os.path.join(r,f)) for r,d,fs in os.walk(OUT_DIR) for f in fs) / 1024:.1f} KB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-25T12:10:48.680702Z","iopub.execute_input":"2026-03-25T12:10:48.681286Z","iopub.status.idle":"2026-03-25T12:10:49.977314Z","shell.execute_reply.started":"2026-03-25T12:10:48.681256Z","shell.execute_reply":"2026-03-25T12:10:49.976565Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_slice(row, split, class_map=CLASS_MAP):\n    slice_id = row['slice_id']\n    path     = os.path.join(DICOM_DIR, f\"{slice_id}.dcm\")\n\n    try:\n        hu  = load_hu(path)\n        rgb = dicom_to_rgb(hu)\n\n        # JPEG — 10x smaller than PNG, negligible quality loss for CT\n        img_path = f\"{OUT_DIR}/images/{split}/{slice_id}.jpg\"\n        Image.fromarray(rgb).save(img_path, format='JPEG', quality=85)\n\n        is_positive = any(row[t] == 1 for t in class_map.keys())\n        lbl_path    = f\"{OUT_DIR}/labels/{split}/{slice_id}.txt\"\n\n        if not is_positive:\n            open(lbl_path, 'w').close()\n            return 'ok'\n\n        brain_mask = strip_skull(hu)\n        lines      = []\n\n        for htype, class_id in class_map.items():\n            if row[htype] != 1:\n                continue\n            boxes = generate_bbox(hu, brain_mask)\n            for (xc, yc, bw, bh) in boxes:\n                lines.append(f\"{class_id} {xc:.6f} {yc:.6f} {bw:.6f} {bh:.6f}\")\n\n        if not lines:\n            os.remove(img_path)\n            return 'no_box'\n\n        with open(lbl_path, 'w') as f:\n            f.write('\\n'.join(lines))\n\n        return 'ok'\n\n    except Exception as e:\n        return 'error'\n\n# ── VERIFY JPEG SIZE ON 100 SLICES FIRST ─────────────────────────\nprint(\"Size check on 100 slices...\")\nprocess_split(train_df.head(100), 'train')\n\ntotal_bytes = sum(\n    os.path.getsize(os.path.join(root, f))\n    for root, dirs, files in os.walk(f\"{OUT_DIR}/images/train\")\n    for f in files\n)\navg_kb    = total_bytes / 100 / 1024\nprojected = avg_kb * 215866 / 1024 / 1024\n\nprint(f\"Average JPEG size: {avg_kb:.1f} KB\")\nprint(f\"Projected total:   {projected:.1f} GB\")\nprint(f\"Kaggle limit:      19.5 GB\")\nprint(f\"Safe: {'✅ YES' if projected < 15 else '⚠️ TIGHT' if projected < 19 else '❌ NO'}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-25T12:11:48.647515Z","iopub.execute_input":"2026-03-25T12:11:48.647781Z","iopub.status.idle":"2026-03-25T12:11:56.682883Z","shell.execute_reply.started":"2026-03-25T12:11:48.647757Z","shell.execute_reply":"2026-03-25T12:11:56.682279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── WIPE SAMPLE AND RUN FULL DATASET ─────────────────────────────\nimport shutil\n\nshutil.rmtree(OUT_DIR)\nfor split in ['train', 'val']:\n    os.makedirs(f\"{OUT_DIR}/images/{split}\", exist_ok=True)\n    os.makedirs(f\"{OUT_DIR}/labels/{split}\", exist_ok=True)\n\nprint(\"=\" * 50)\nprint(\"FULL RUN — 215,866 SLICES\")\nprint(\"=\" * 50)\nprint(f\"Estimated time: {215866 / 10.18 / 3600:.1f} hours\")\nprint(f\"Estimated size: ~5.2 GB\")\nprint()\n\ntrain_stats = process_split(train_df, 'train')\nval_stats   = process_split(val_df,   'val')\n\n# ── FINAL SUMMARY ────────────────────────────────────────────────\nprint(\"\\n\" + \"=\" * 50)\nprint(\"FINAL DATASET SUMMARY\")\nprint(\"=\" * 50)\n\ntrain_images = os.listdir(f\"{OUT_DIR}/images/train\")\nval_images   = os.listdir(f\"{OUT_DIR}/images/val\")\ntrain_labels = os.listdir(f\"{OUT_DIR}/labels/train\")\nval_labels   = os.listdir(f\"{OUT_DIR}/labels/val\")\n\ntrain_pos = sum(1 for l in train_labels\n                if os.path.getsize(f\"{OUT_DIR}/labels/train/{l}\") > 0)\nval_pos   = sum(1 for l in val_labels\n                if os.path.getsize(f\"{OUT_DIR}/labels/val/{l}\") > 0)\n\ntotal_bytes = sum(\n    os.path.getsize(os.path.join(root, f))\n    for root, dirs, files in os.walk(OUT_DIR)\n    for f in files\n)\n\nprint(f\"  train: {len(train_images):,} images  ({train_pos:,} pos + {len(train_labels)-train_pos:,} neg)\")\nprint(f\"  val:   {len(val_images):,} images  ({val_pos:,} pos + {len(val_labels)-val_pos:,} neg)\")\nprint(f\"  size:  {total_bytes/1024/1024/1024:.2f} GB\")\n\n# ── WRITE data.yaml ───────────────────────────────────────────────\nyaml_content = f\"\"\"path: {OUT_DIR}\ntrain: images/train\nval:   images/val\n\nnc: 5\nnames:\n  0: epidural\n  1: intraparenchymal\n  2: intraventricular\n  3: subarachnoid\n  4: subdural\n\"\"\"\n\nwith open(f\"{OUT_DIR}/data.yaml\", 'w') as f:\n    f.write(yaml_content)\n\nprint(f\"\\n  data.yaml written ✅\")\nprint(f\"  Dataset ready for YOLOv8 🧠🔥\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-25T12:12:57.002383Z","iopub.execute_input":"2026-03-25T12:12:57.002686Z","iopub.status.idle":"2026-03-25T18:09:30.873592Z","shell.execute_reply.started":"2026-03-25T12:12:57.002659Z","shell.execute_reply":"2026-03-25T18:09:30.872945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── INSTALL DEPENDENCIES ─────────────────────────────────────────\n!pip install ultralytics albumentations -q\n\nfrom ultralytics import YOLO\nimport torch\n\nprint(f\"GPU: {torch.cuda.get_device_name(0)}\")\nprint(f\"VRAM: {torch.cuda.get_device_properties(0).total_memory / 1e9:.1f} GB\")\n\n# ── LOAD MODEL ───────────────────────────────────────────────────\nmodel = YOLO('yolov8m.pt')\nprint(f\"\\nModel loaded ✅\")\nprint(f\"Parameters: {sum(p.numel() for p in model.model.parameters()):,}\")\n\n# ── ALBUMENTATIONS TRANSFORMS (GAUSSIAN NOISE) ──────────────────\n# Using dictionary format – works with any Albumentations version\ncustom_transforms = [\n    {\n        'name': 'GaussianNoise',\n        'p': 0.5,\n        'var_limit': (10.0, 50.0)      # adjust to your CT noise level\n    }\n    # Add more transforms as needed, e.g.:\n    # {\n    #     'name': 'RandomBrightnessContrast',\n    #     'p': 0.3,\n    #     'brightness_limit': 0.2,\n    #     'contrast_limit': 0.2\n    # }\n]\n\n# ── TRAIN ────────────────────────────────────────────────────────\nresults = model.train(\n    data    = f\"{OUT_DIR}/data.yaml\",    # make sure OUT_DIR is defined\n    epochs  = 50,\n    imgsz   = 512,\n    batch   = 16,\n    device  = 0,\n    workers = 2,\n\n    optimizer = 'AdamW',\n    lr0       = 0.001,\n    lrf       = 0.01,\n    weight_decay = 0.0005,\n    dropout      = 0.0,\n\n    # YOLO built‑in augmentations\n    hsv_h    = 0.0,\n    hsv_s    = 0.0,\n    hsv_v    = 0.3,\n    degrees  = 15.0,\n    translate= 0.1,\n    scale    = 0.3,\n    fliplr   = 0.5,\n    flipud   = 0.0,\n    mosaic   = 0.5,\n\n    cls      = 0.5,\n\n    project  = '/kaggle/working/neuroscan_runs',\n    name     = 'v1_yolov8m',\n    save     = True,\n    plots    = True,\n    verbose  = True,\n\n    # ── CORRECT PARAMETER NAME ───────────────────────────────────\n    augmentations = custom_transforms   # <-- was 'albumentations'\n)\n\nprint(\"\\nTraining complete 🧠🔥\")\nprint(f\"Best weights: {results.save_dir}/weights/best.pt\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-25T18:26:00.623396Z","iopub.execute_input":"2026-03-25T18:26:00.623924Z","execution_failed":"2026-03-25T22:49:12.377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── CELL 1: PATHS ────────────────────────────────────────────────\nimport os, shutil, time\nimport pydicom\nimport numpy as np\nimport pandas as pd\nfrom scipy import ndimage\nfrom PIL import Image\nfrom tqdm.notebook import tqdm\n\nDICOM_DIR = \"/kaggle/input/competitions/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train\"\nCSV_PATH  = \"/kaggle/input/competitions/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train.csv\"\nOUT_DIR   = \"/kaggle/working/cranioscan_dataset\"\nRUNS_DIR  = \"/kaggle/working/cranioscan_runs\"\n\nCLASS_MAP = {\n    'epidural':         0,\n    'intraparenchymal': 1,\n    'intraventricular': 2,\n    'subarachnoid':     3,\n    'subdural':         4,\n}\n\nfor split in ['train', 'val']:\n    os.makedirs(f\"{OUT_DIR}/images/{split}\", exist_ok=True)\n    os.makedirs(f\"{OUT_DIR}/labels/{split}\", exist_ok=True)\n\nprint(\"Paths and directories ready ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T01:17:51.754234Z","iopub.execute_input":"2026-03-26T01:17:51.754517Z","iopub.status.idle":"2026-03-26T01:17:54.317136Z","shell.execute_reply.started":"2026-03-26T01:17:51.754481Z","shell.execute_reply":"2026-03-26T01:17:54.316387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── CELL 2: ALL PIPELINE FUNCTIONS ───────────────────────────────\ndef load_hu(path):\n    dcm   = pydicom.dcmread(path)\n    raw   = dcm.pixel_array.astype(np.float32)\n    slope = float(getattr(dcm, 'RescaleSlope',     1))\n    inter = float(getattr(dcm, 'RescaleIntercept', 0))\n    return raw * slope + inter\n\ndef apply_window(hu, wc, ww):\n    low, high = wc - ww/2, wc + ww/2\n    return (np.clip(hu, low, high) - low) / (high - low)\n\ndef dicom_to_rgb(hu):\n    brain    = apply_window(hu, 40,   80)\n    subdural = apply_window(hu, 80,   200)\n    bone     = apply_window(hu, 600,  2800)\n    rgb = np.stack([brain, subdural, bone], axis=-1)\n    return (rgb * 255).astype(np.uint8)\n\ndef strip_skull(hu):\n    mask = hu > -200\n    labeled, num_features = ndimage.label(mask)\n    if num_features == 0:\n        return mask.astype(bool)\n    sizes = ndimage.sum(mask, labeled, range(1, num_features + 1))\n    largest = np.argmax(sizes) + 1\n    mask = labeled == largest\n    mask = ndimage.binary_fill_holes(mask)\n    mask = ndimage.binary_erosion(mask,  iterations=15)\n    mask = ndimage.binary_dilation(mask, iterations=6)\n    labeled, num_features = ndimage.label(mask)\n    if num_features == 0:\n        return mask.astype(bool)\n    sizes = ndimage.sum(mask, labeled, range(1, num_features + 1))\n    largest = np.argmax(sizes) + 1\n    return (labeled == largest).astype(bool)\n\ndef generate_bbox(hu, brain_mask,\n                  hemorrhage_hu_min=50,\n                  hemorrhage_hu_max=100,\n                  min_blob_size=300,\n                  max_blobs=3,\n                  edge_margin=0.05,\n                  max_distance_from_center=0.22):\n    H, W = hu.shape\n    center_y, center_x = H / 2, W / 2\n    hemorrhage_mask = (hu >= hemorrhage_hu_min) & (hu <= hemorrhage_hu_max)\n    hemorrhage_mask = hemorrhage_mask & brain_mask\n    labeled, num_blobs = ndimage.label(hemorrhage_mask)\n    if num_blobs == 0:\n        return []\n    margin_px_h = int(H * edge_margin)\n    margin_px_w = int(W * edge_margin)\n    candidates  = []\n    for blob_id in range(1, num_blobs + 1):\n        blob      = labeled == blob_id\n        blob_size = blob.sum()\n        if blob_size < min_blob_size:\n            continue\n        rows = np.where(blob.any(axis=1))[0]\n        cols = np.where(blob.any(axis=0))[0]\n        row_min, row_max = rows[0], rows[-1]\n        col_min, col_max = cols[0], cols[-1]\n        if (row_min < margin_px_h or row_max > H - margin_px_h or\n            col_min < margin_px_w or col_max > W - margin_px_w):\n            continue\n        blob_center_y = (row_min + row_max) / 2\n        blob_center_x = (col_min + col_max) / 2\n        dist = np.sqrt(\n            ((blob_center_y - center_y) / H) ** 2 +\n            ((blob_center_x - center_x) / W) ** 2\n        )\n        if dist > max_distance_from_center:\n            continue\n        x_center = blob_center_x / W\n        y_center = blob_center_y / H\n        width    = (col_max - col_min) / W\n        height   = (row_max - row_min) / H\n        candidates.append({'box': (x_center, y_center, width, height), 'size': blob_size})\n    candidates = sorted(candidates, key=lambda x: x['size'], reverse=True)[:max_blobs]\n    return [c['box'] for c in candidates]\n\ndef process_slice(row, split, class_map=CLASS_MAP):\n    slice_id = row['slice_id']\n    path     = os.path.join(DICOM_DIR, f\"{slice_id}.dcm\")\n    try:\n        hu  = load_hu(path)\n        rgb = dicom_to_rgb(hu)\n        img_path = f\"{OUT_DIR}/images/{split}/{slice_id}.jpg\"\n        Image.fromarray(rgb).save(img_path, format='JPEG', quality=85)\n        is_positive = any(row[t] == 1 for t in class_map.keys())\n        lbl_path    = f\"{OUT_DIR}/labels/{split}/{slice_id}.txt\"\n        if not is_positive:\n            open(lbl_path, 'w').close()\n            return 'ok'\n        brain_mask = strip_skull(hu)\n        lines      = []\n        for htype, class_id in class_map.items():\n            if row[htype] != 1:\n                continue\n            boxes = generate_bbox(hu, brain_mask)\n            for (xc, yc, bw, bh) in boxes:\n                lines.append(f\"{class_id} {xc:.6f} {yc:.6f} {bw:.6f} {bh:.6f}\")\n        if not lines:\n            os.remove(img_path)\n            return 'no_box'\n        with open(lbl_path, 'w') as f:\n            f.write('\\n'.join(lines))\n        return 'ok'\n    except Exception as e:\n        return 'error'\n\nprint(\"All functions loaded ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T01:18:00.466136Z","iopub.execute_input":"2026-03-26T01:18:00.466595Z","iopub.status.idle":"2026-03-26T01:18:00.483146Z","shell.execute_reply.started":"2026-03-26T01:18:00.466562Z","shell.execute_reply":"2026-03-26T01:18:00.482381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── CELL 3: LOAD LABELS + BUILD 5K SUBSET ────────────────────────\ndf = pd.read_csv(CSV_PATH)\ndf[['slice_id', 'hemorrhage_type']] = df['ID'].str.rsplit('_', n=1, expand=True)\ndf_clean = df.groupby(['slice_id', 'hemorrhage_type'])['Label'].max().reset_index()\ndf_pivot = df_clean.pivot(index='slice_id', columns='hemorrhage_type', values='Label').reset_index()\ndf_pivot.columns.name = None\n\nfiles_on_disk = set(f.replace('.dcm', '') for f in os.listdir(DICOM_DIR))\ndf_pivot = df_pivot[df_pivot['slice_id'].isin(files_on_disk)].reset_index(drop=True)\n\ntypes     = list(CLASS_MAP.keys())\npositives = df_pivot[df_pivot['any'] == 1].sample(frac=1, random_state=42).reset_index(drop=True)\nnegatives = df_pivot[df_pivot['any'] == 0].sample(frac=1, random_state=42).reset_index(drop=True)\n\n# 5k subset: 2500 pos + 2500 neg = balanced\nSUBSET = 2500\npos_train = positives[:int(SUBSET*0.8)]\npos_val   = positives[int(SUBSET*0.8):SUBSET]\nneg_train = negatives[:int(SUBSET*0.8)]\nneg_val   = negatives[int(SUBSET*0.8):SUBSET]\n\ntrain_df = pd.concat([pos_train, neg_train]).sample(frac=1, random_state=42).reset_index(drop=True)\nval_df   = pd.concat([pos_val,   neg_val  ]).sample(frac=1, random_state=42).reset_index(drop=True)\n\nprint(f\"Subset split:\")\nprint(f\"  Train: {len(train_df):,}  ({len(pos_train):,} pos + {len(neg_train):,} neg)\")\nprint(f\"  Val:   {len(val_df):,}  ({len(pos_val):,} pos + {len(neg_val):,} neg)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T01:18:08.100172Z","iopub.execute_input":"2026-03-26T01:18:08.101067Z","iopub.status.idle":"2026-03-26T01:18:33.91345Z","shell.execute_reply.started":"2026-03-26T01:18:08.101037Z","shell.execute_reply":"2026-03-26T01:18:33.912593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── CELL 4: GENERATE 5K DATASET ──────────────────────────────────\ndef process_split(df, split):\n    stats = {'ok': 0, 'no_box': 0, 'error': 0}\n    for _, row in tqdm(df.iterrows(), total=len(df), desc=f\"Processing {split}\"):\n        result = process_slice(row, split)\n        stats[result] += 1\n    print(f\"\\n── {split.upper()} COMPLETE ──\")\n    print(f\"  ok={stats['ok']:,}  no_box={stats['no_box']:,}  error={stats['error']:,}\")\n    return stats\n\nprocess_split(train_df, 'train')\nprocess_split(val_df,   'val')\n\n# write data.yaml\nyaml_content = f\"\"\"path: {OUT_DIR}\ntrain: images/train\nval:   images/val\n\nnc: 5\nnames:\n  0: epidural\n  1: intraparenchymal\n  2: intraventricular\n  3: subarachnoid\n  4: subdural\n\"\"\"\nwith open(f\"{OUT_DIR}/data.yaml\", 'w') as f:\n    f.write(yaml_content)\n\ntotal = sum(\n    os.path.getsize(os.path.join(r, f))\n    for r, d, files in os.walk(OUT_DIR)\n    for f in files\n)\nprint(f\"\\nDataset size: {total/1e6:.1f} MB ✅\")\nprint(f\"data.yaml written ✅\")\nprint(f\"Ready to train 🔥\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T01:18:57.670174Z","iopub.execute_input":"2026-03-26T01:18:57.670647Z","iopub.status.idle":"2026-03-26T01:26:23.75052Z","shell.execute_reply.started":"2026-03-26T01:18:57.670617Z","shell.execute_reply":"2026-03-26T01:26:23.749807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── INSTALL DEPENDENCIES ─────────────────────────────────────────\n!pip install ultralytics albumentations -q\n\nfrom ultralytics import YOLO\nimport torch\n\nprint(f\"GPU: {torch.cuda.get_device_name(0)}\")\nprint(f\"VRAM: {torch.cuda.get_device_properties(0).total_memory / 1e9:.1f} GB\")\n\n# ── LOAD MODEL ───────────────────────────────────────────────────\nmodel = YOLO('yolov8m.pt')\nprint(f\"\\nModel loaded ✅\")\nprint(f\"Parameters: {sum(p.numel() for p in model.model.parameters()):,}\")\n\n# ── ALBUMENTATIONS TRANSFORMS (GAUSSIAN NOISE) ──────────────────\n# Using dictionary format – works with any Albumentations version\ncustom_transforms = [\n    {\n        'name': 'GaussianNoise',\n        'p': 0.5,\n        'var_limit': (10.0, 50.0)      # adjust to your CT noise level\n    }\n    # Add more transforms as needed, e.g.:\n    # {\n    #     'name': 'RandomBrightnessContrast',\n    #     'p': 0.3,\n    #     'brightness_limit': 0.2,\n    #     'contrast_limit': 0.2\n    # }\n]\n\n# ── TRAIN ────────────────────────────────────────────────────────\nresults = model.train(\n    data    = f\"{OUT_DIR}/data.yaml\",    # make sure OUT_DIR is defined\n    epochs  = 20,\n    save_period = 1, \n    exist_ok = True,\n    imgsz   = 512,\n    batch   = 16,\n    device  = 0,\n    workers = 2,\n\n    optimizer = 'AdamW',\n    lr0       = 0.001,\n    lrf       = 0.01,\n    weight_decay = 0.0005,\n    dropout      = 0.0,\n\n    # YOLO built‑in augmentations\n    hsv_h    = 0.0,\n    hsv_s    = 0.0,\n    hsv_v    = 0.3,\n    degrees  = 15.0,\n    translate= 0.1,\n    scale    = 0.3,\n    fliplr   = 0.5,\n    flipud   = 0.0,\n    mosaic   = 0.5,\n\n    cls      = 0.5,\n\n    project  = '/kaggle/working/neuroscan_runs',\n    name     = 'v1_yolov8m',\n    save     = True,\n    plots    = True,\n    verbose  = True,\n\n    # ── CORRECT PARAMETER NAME ───────────────────────────────────\n    augmentations = custom_transforms   # <-- was 'albumentations'\n)\n\nprint(\"\\nTraining complete 🧠🔥\")\nprint(f\"Best weights: {results.save_dir}/weights/best.pt\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T01:29:29.58308Z","iopub.execute_input":"2026-03-26T01:29:29.583562Z","iopub.status.idle":"2026-03-26T02:01:27.283001Z","shell.execute_reply.started":"2026-03-26T01:29:29.583523Z","shell.execute_reply":"2026-03-26T02:01:27.282022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil, os\n\nos.makedirs('/kaggle/working/weights', exist_ok=True)\n\nshutil.copy(\n    '/kaggle/working/neuroscan_runs/v1_yolov8m/weights/best.pt',\n    '/kaggle/working/weights/best.pt'\n)\nshutil.copy(\n    '/kaggle/working/neuroscan_runs/v1_yolov8m/weights/last.pt',\n    '/kaggle/working/weights/last.pt'\n)\n\nprint(f\"best.pt: {os.path.getsize('/kaggle/working/weights/best.pt')/1e6:.1f} MB ✅\")\nprint(f\"last.pt: {os.path.getsize('/kaggle/working/weights/last.pt')/1e6:.1f} MB ✅\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T02:06:09.515645Z","iopub.execute_input":"2026-03-26T02:06:09.516329Z","iopub.status.idle":"2026-03-26T02:06:09.585094Z","shell.execute_reply.started":"2026-03-26T02:06:09.516286Z","shell.execute_reply":"2026-03-26T02:06:09.584413Z"}},"outputs":[],"execution_count":null}]}