{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":37333,"databundleVersionId":3949526,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\ndf = pd.read_csv(\"/kaggle/input/mayo-clinic-strip-ai/train.csv\")\nprint(df.head())","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-03T04:34:15.744351Z","iopub.execute_input":"2025-07-03T04:34:15.744737Z","iopub.status.idle":"2025-07-03T04:34:15.758489Z","shell.execute_reply.started":"2025-07-03T04:34:15.744714Z","shell.execute_reply":"2025-07-03T04:34:15.757396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Num rows:\", len(df))\nprint(\"Unique patients:\", df['patient_id'].nunique())\nprint(\"Images per patient (avg):\", df.groupby('patient_id').size().mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T04:34:15.760019Z","iopub.execute_input":"2025-07-03T04:34:15.760331Z","iopub.status.idle":"2025-07-03T04:34:15.784078Z","shell.execute_reply.started":"2025-07-03T04:34:15.760303Z","shell.execute_reply":"2025-07-03T04:34:15.782909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nsns.countplot(data=df, x='label')\nplt.title('Label Distribution (CE vs LAA)')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T04:34:15.784955Z","iopub.execute_input":"2025-07-03T04:34:15.785212Z","iopub.status.idle":"2025-07-03T04:34:15.97074Z","shell.execute_reply.started":"2025-07-03T04:34:15.785194Z","shell.execute_reply":"2025-07-03T04:34:15.969751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count occurrences of each label\nlabel_counts = df['label'].value_counts()\nprint(\"Label counts:\")\nprint(label_counts)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T04:34:15.971765Z","iopub.execute_input":"2025-07-03T04:34:15.972084Z","iopub.status.idle":"2025-07-03T04:34:15.979848Z","shell.execute_reply.started":"2025-07-03T04:34:15.972062Z","shell.execute_reply":"2025-07-03T04:34:15.978516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"center_counts = df['center_id'].value_counts()\nprint(\"Sample count per center:\")\nprint(center_counts)\ntrain_df = df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T04:34:15.982453Z","iopub.execute_input":"2025-07-03T04:34:15.982759Z","iopub.status.idle":"2025-07-03T04:34:16.002688Z","shell.execute_reply.started":"2025-07-03T04:34:15.982739Z","shell.execute_reply":"2025-07-03T04:34:16.001967Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.crosstab(train_df['center_id'], train_df['label'], normalize='index') * 100","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T04:34:16.004026Z","iopub.execute_input":"2025-07-03T04:34:16.004639Z","iopub.status.idle":"2025-07-03T04:34:16.037749Z","shell.execute_reply.started":"2025-07-03T04:34:16.004613Z","shell.execute_reply":"2025-07-03T04:34:16.036848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\nimport matplotlib.pyplot as plt\n\n# Path to your image file (example)\nimg_path = '/kaggle/input/mayo-clinic-strip-ai/train/008e5c_0.tif'\n\n# Load image with PIL\nimg = Image.open(img_path)\n\n# Display image with matplotlib\nplt.figure(figsize=(8, 8))\nplt.imshow(img)\nplt.axis('off')  # Hide axis for cleaner view\nplt.title('Sample CE Visualization')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T04:34:16.038622Z","iopub.execute_input":"2025-07-03T04:34:16.038886Z","iopub.status.idle":"2025-07-03T04:34:32.456919Z","shell.execute_reply.started":"2025-07-03T04:34:16.038867Z","shell.execute_reply":"2025-07-03T04:34:32.455882Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfrom PIL import Image\nImage.MAX_IMAGE_PIXELS = None        # 🔓 Disable the safeguard  ❗Use only if you trust the file\nimg = Image.open(\"/kaggle/input/mayo-clinic-strip-ai/train/008e5c_0.tif\")\n\n# Load image with PIL\nimg = Image.open(img_path)\n\n# Display image with matplotlib\nplt.figure(figsize=(8, 8))\nplt.imshow(img)\nplt.axis('off')  # Hide axis for cleaner view\nplt.title('Sample LAA Visualization')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T04:34:32.458581Z","iopub.execute_input":"2025-07-03T04:34:32.458923Z","iopub.status.idle":"2025-07-03T04:34:48.788841Z","shell.execute_reply.started":"2025-07-03T04:34:32.458898Z","shell.execute_reply":"2025-07-03T04:34:48.787889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_image_size(image_id):\n    path = f'/kaggle/input/mayo-clinic-strip-ai/train/{image_id}.tif'\n    with Image.open(path) as img:\n        return img.size  # returns (width, height)\n\n# Apply to all images (warning: slow for big datasets)\ndf['image_size'] = df['image_id'].apply(get_image_size)\n\n# Split width and height into separate columns\ndf['width'] = df['image_size'].apply(lambda x: x[0])\ndf['height'] = df['image_size'].apply(lambda x: x[1])\n\n# Plot distributions\nplt.figure(figsize=(12,5))\nplt.subplot(1,2,1)\ndf['width'].hist(bins=30)\nplt.title('Image Width Distribution')\nplt.xlabel('Width (pixels)')\nplt.ylabel('Count')\n\nplt.subplot(1,2,2)\ndf['height'].hist(bins=30)\nplt.title('Image Height Distribution')\nplt.xlabel('Height (pixels)')\nplt.ylabel('Count')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T04:34:48.789857Z","iopub.execute_input":"2025-07-03T04:34:48.790088Z","iopub.status.idle":"2025-07-03T04:34:50.517105Z","shell.execute_reply.started":"2025-07-03T04:34:48.790067Z","shell.execute_reply":"2025-07-03T04:34:50.516022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['area'] = df['width'] * df['height']\n\n# Find smallest and largest by area\nsmallest = df.loc[df['area'].idxmin()]\nlargest = df.loc[df['area'].idxmax()]\n\nprint(\"Smallest image:\")\nprint(f\"Image ID: {smallest['image_id']}\")\nprint(f\"Width: {smallest['width']} px, Height: {smallest['height']} px\")\nprint(f\"Area: {smallest['area']} pixels² ({smallest['area'] / 1_000_000:.2f} MP)\")\n\nprint(\"\\nLargest image:\")\nprint(f\"Image ID: {largest['image_id']}\")\nprint(f\"Width: {largest['width']} px, Height: {largest['height']} px\")\nprint(f\"Area: {largest['area']} pixels² ({largest['area'] / 1_000_000:.2f} MP)\")","metadata":{"execution":{"iopub.status.busy":"2025-07-03T04:34:50.518245Z","iopub.execute_input":"2025-07-03T04:34:50.519267Z","iopub.status.idle":"2025-07-03T04:34:50.52822Z","shell.execute_reply.started":"2025-07-03T04:34:50.519234Z","shell.execute_reply":"2025-07-03T04:34:50.526947Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, glob, cv2, random, math, json\nfrom pathlib import Path\nfrom collections import defaultdict\n\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport matplotlib.pyplot as plt\n\nimport torch, torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as T\nimport timm                              # for ResNet-50\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import log_loss","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:30:20.862009Z","iopub.execute_input":"2025-07-03T13:30:20.863176Z","iopub.status.idle":"2025-07-03T13:30:39.092838Z","shell.execute_reply.started":"2025-07-03T13:30:20.863133Z","shell.execute_reply":"2025-07-03T13:30:39.091583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport os, openslide, numpy as np\nfrom tqdm.auto import tqdm\n\ndef tile_wsi(wsi_path, out_dir, tile_size=224, stride=224, tissue_th=0.15):\n    slide = openslide.OpenSlide(str(wsi_path))\n    w, h = slide.dimensions\n    out_dir = Path(out_dir); out_dir.mkdir(parents=True, exist_ok=True)\n\n    for y in range(0, h - tile_size + 1, stride):           # ① include bottom edge\n        for x in range(0, w - tile_size + 1, stride):       # ① include right edge\n            tile = slide.read_region((x, y), 0, (tile_size, tile_size)).convert(\"RGB\")\n            if (np.array(tile) < 240).mean() > tissue_th:   # ② keep tissue tiles\n                tile.save(out_dir / f\"{Path(wsi_path).stem}_{x}_{y}.jpg\")\n\n# ③ --- tile every slide once ---\nsrc_dir  = Path('/kaggle/input/mayo-clinic-strip-ai/train')\ndest_dir = Path('/kaggle/working/tiles')\nfor tif in tqdm(src_dir.glob('*.tif')):\n    tile_wsi(tif, dest_dir)\nprint(\"Total tiles:\", len(list(dest_dir.glob('*.jpg'))))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T11:57:26.917735Z","iopub.execute_input":"2025-07-03T11:57:26.918333Z","iopub.status.idle":"2025-07-03T12:00:16.750285Z","shell.execute_reply.started":"2025-07-03T11:57:26.918301Z","shell.execute_reply":"2025-07-03T12:00:16.748675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"src_dir  = Path('/kaggle/input/mayo-clinic-strip-ai/train')\ndest_dir = Path('/kaggle/working/tiles')\nprint(\"Total tiles:\", len(list(dest_dir.glob('*.jpg'))))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:30:58.694096Z","iopub.execute_input":"2025-07-03T13:30:58.69451Z","iopub.status.idle":"2025-07-03T13:30:58.700908Z","shell.execute_reply.started":"2025-07-03T13:30:58.694485Z","shell.execute_reply":"2025-07-03T13:30:58.699735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TileDataset(Dataset):\n    \"\"\"\n    Dataset that returns individual tiles extracted from each WSI.\n\n    ┌────────────────────────────────────────────────────────────┐\n    │ Expected tile filenames: <image_id>_<x>_<y>.jpg            │\n    │   • <image_id> comes from the train.csv `image_id` column  │\n    │   • <x>, <y> are the top-left pixel of the tile in the WSI │\n    └────────────────────────────────────────────────────────────┘\n    \"\"\"\n\n    def __init__(\n        self,\n        df,                      # pandas DataFrame with cols: image_id, label, patient_id\n        tile_dir,                # directory that contains the *.jpg tiles\n        tform=None               # optional torchvision transform\n    ):\n        self.records = []        # list of (tile_path, label, patient_id, slide_id)\n        tile_dir = Path(tile_dir)\n\n        for _, row in df.iterrows():\n            slide_id  = Path(row[\"image_id\"]).stem          # drop \".tif\" if present\n            label     = row[\"label\"]\n            patient   = row[\"patient_id\"]\n\n            # gather every tile that belongs to this slide\n            for p in tile_dir.glob(f\"{slide_id}_*.jpg\"):\n                self.records.append((p, label, patient, slide_id))\n\n        # fallback to a simple transform if none supplied\n        self.tform = (\n            tform\n            or T.Compose([\n                    T.RandomHorizontalFlip(),\n                    T.RandomVerticalFlip(),\n                    T.ToTensor(),\n                    T.Normalize([0.485, 0.456, 0.406],\n                                [0.229, 0.224, 0.225]),\n               ])\n        )\n\n    def __len__(self):\n        return len(self.records)\n\n    def __getitem__(self, idx):\n        path, label, patient_id, slide_id = self.records[idx]\n\n        img = Image.open(path).convert(\"RGB\")\n        img = self.tform(img)\n\n        # BCEWithLogitsLoss expects float targets of shape [N, 1]\n        label = torch.tensor(label, dtype=torch.float32)\n\n        # return identifiers so you can aggregate tiles → slide prediction later\n        return img, label, patient_id, slide_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:30:39.102379Z","iopub.execute_input":"2025-07-03T13:30:39.102701Z","iopub.status.idle":"2025-07-03T13:30:39.122899Z","shell.execute_reply.started":"2025-07-03T13:30:39.102674Z","shell.execute_reply":"2025-07-03T13:30:39.121767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(\"Using\", device)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:30:39.124766Z","iopub.execute_input":"2025-07-03T13:30:39.125907Z","iopub.status.idle":"2025-07-03T13:30:39.146311Z","shell.execute_reply.started":"2025-07-03T13:30:39.125847Z","shell.execute_reply":"2025-07-03T13:30:39.145277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def make_model():\n    model = timm.create_model('resnet50', pretrained=False, num_classes=1)\n    return model.to(device)\n\n\n\nBCE = nn.BCEWithLogitsLoss()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:30:40.915436Z","iopub.execute_input":"2025-07-03T13:30:40.915763Z","iopub.status.idle":"2025-07-03T13:30:40.920974Z","shell.execute_reply.started":"2025-07-03T13:30:40.915742Z","shell.execute_reply":"2025-07-03T13:30:40.919858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_one_epoch(model, loader, optim):\n    model.train()\n    for imgs, labels, *_ in loader:\n        imgs   = imgs.to(device)\n        labels = labels.unsqueeze(1).to(device)      # shape [B,1] for BCE\n        optim.zero_grad()\n        loss = BCE(model(imgs), labels)\n        loss.backward(); optim.step()\n\n@torch.no_grad()\ndef eval_tile(model, loader):\n    model.eval()\n    tile_probs, slide_ids = [], []\n    for imgs, labels, *_ , sids in loader:\n        p = torch.sigmoid(model(imgs.to(device))).squeeze().cpu().numpy()\n        tile_probs.extend(p); slide_ids.extend(sids)\n    ...\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:30:45.35831Z","iopub.execute_input":"2025-07-03T13:30:45.358727Z","iopub.status.idle":"2025-07-03T13:30:45.365994Z","shell.execute_reply.started":"2025-07-03T13:30:45.3587Z","shell.execute_reply":"2025-07-03T13:30:45.364933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for fold,(tr_df,val_df) in enumerate(FOLDS):\n    tr_ds = TileDataset(tr_df, '/kaggle/working/tiles/')\n    val_ds = TileDataset(val_df, '/kaggle/working/tiles/',\n                         tform=T.Compose([T.ToTensor(), T.Normalize([0.485,0.456,0.406],\n                                                                     [0.229,0.224,0.225])]))\n    tr_loader = DataLoader(tr_ds, 32, shuffle=True, num_workers=4)\n    val_loader = DataLoader(val_ds, 32, shuffle=False, num_workers=4)\n\n    model = make_model()\n    opt = torch.optim.AdamW(model.parameters(), 1e-4)\n\n    train_one_epoch(model, tr_loader, opt)\n    ll = eval_tile(model, val_loader)\n    print(f\"Fold {fold}: slide-level LogLoss = {ll:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:30:47.423545Z","iopub.execute_input":"2025-07-03T13:30:47.424042Z","iopub.status.idle":"2025-07-03T13:30:47.510888Z","shell.execute_reply.started":"2025-07-03T13:30:47.424004Z","shell.execute_reply":"2025-07-03T13:30:47.509637Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RESIZE = 64\n\ndef load_vec(path):\n    arr = cv2.resize(cv2.imread(path), (RESIZE, RESIZE))\n    return arr.reshape(-1) / 255.0         # flatten & scale to [0,1]\n\nX, y, pids = [], [], []\nfor _, row in train_df.iterrows():\n    wsi = row['image_id']\n    label = row['label']\n    tile_path = f\"/kaggle/input/mayo-clinic-strip-ai/train/{wsi}.tif\"\n\n    # crude thumbnail read (anti-bomb)\n    Image.MAX_IMAGE_PIXELS = None\n    img = Image.open(tile_path).convert('RGB')\n    img = img.resize((RESIZE, RESIZE))\n    X.append(np.array(img).reshape(-1)/255.)\n    y.append(label); pids.append(row['patient_id'])\n\nX = np.stack(X); y = np.array(y)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T12:04:48.557509Z","iopub.execute_input":"2025-07-03T12:04:48.557845Z","execution_failed":"2025-07-03T13:19:59.703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scores = []\nfor fold, (tr, val) in enumerate(gkf.split(X, y, groups=pids)):\n    m = LogisticRegression(max_iter=1000, solver='saga')\n    m.fit(X[tr], y[tr])\n    proba = m.predict_proba(X[val])[:,1]\n    ll = log_loss(y[val], proba)\n    scores.append(ll)\n    print(f\"Fold {fold}: LogLoss = {ll:.4f}\")\n\nprint(\"Mean slide-level LogLoss:\", np.mean(scores))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:29:46.773233Z","iopub.execute_input":"2025-07-03T13:29:46.77356Z","iopub.status.idle":"2025-07-03T13:29:46.859052Z","shell.execute_reply.started":"2025-07-03T13:29:46.773533Z","shell.execute_reply":"2025-07-03T13:29:46.857887Z"}},"outputs":[],"execution_count":null}]}