{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":37333,"databundleVersionId":3949526}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Cell 1","metadata":{}},{"cell_type":"code","source":"!pip install -q zarr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T12:06:36.19273Z","iopub.execute_input":"2026-04-25T12:06:36.193099Z","iopub.status.idle":"2026-04-25T12:06:40.96288Z","shell.execute_reply.started":"2026-04-25T12:06:36.193065Z","shell.execute_reply":"2026-04-25T12:06:40.96157Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile worker.py\nimport os\nimport cv2\nimport numpy as np\nimport tifffile as tiff\nimport zarr\nimport gc\n\ndef process_single_wsi_worker(image_id, INPUT_DIR, OUTPUT_DIR, COMPOSITE_DIM, PATCH_SIZE, JPEG_QUALITY):\n    img_path = os.path.join(INPUT_DIR, f\"{image_id}.tif\")\n    save_path = os.path.join(OUTPUT_DIR, f\"{image_id}.jpg\")\n    if os.path.exists(save_path): return image_id, True, \"skipped\"\n\n    try:\n        # Pre-allocate buffer to fix memory footprint early\n        composite = np.full((COMPOSITE_DIM, COMPOSITE_DIM, 3), 255, dtype=np.uint8)\n        \n        with tiff.TiffFile(img_path) as tif:\n            series = tif.series[0]\n            # 1. Preprocessing: Fast Watershed on Thumbnail [cite: 74, 119]\n            thumb = series.levels[-1].asarray()\n            if thumb.ndim == 4: thumb = thumb[0]\n            if thumb.shape[-1] == 4: thumb = thumb[:, :, :3]\n            \n            gray = cv2.cvtColor(thumb, cv2.COLOR_RGB2GRAY)\n            # Sobel enhancement for boundaries [cite: 118]\n            mag = cv2.magnitude(cv2.Sobel(gray, cv2.CV_32F, 1, 0), cv2.Sobel(gray, cv2.CV_32F, 0, 1))\n            _, thresh = cv2.threshold(np.uint8(255*mag/(mag.max()+1e-7)), 0, 255, cv2.THRESH_BINARY+cv2.THRESH_OTSU)\n            \n            # Watershed markers via distance transform [cite: 32, 120]\n            dist = cv2.distanceTransform(thresh, cv2.DIST_L2, 5)\n            _, markers = cv2.connectedComponents(np.uint8(dist > 0.2 * dist.max()))\n            markers = cv2.watershed(thumb, markers + 1)\n            \n            # Identify salient regions [cite: 14, 128]\n            mask = np.zeros(gray.shape, dtype=\"uint8\")\n            mask[markers > 1] = 255\n            contours, _ = cv2.findContours(mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n            contours = sorted(contours, key=cv2.contourArea, reverse=True)[:16]\n            \n            del gray, mag, thresh, dist, markers, mask, thumb\n            \n            # 2. Optimized Extraction from Level 0 [cite: 75, 129]\n            full_res = series.levels[0]\n            sX = full_res.shape[1]/series.levels[-1].shape[1]\n            sY = full_res.shape[0]/series.levels[-1].shape[0]\n            \n            # Zarr demand-paging is faster than full-level access [cite: 15, 87]\n            with zarr.open(full_res.aszarr(), mode='r') as zarr_img:\n                patch_count = 0\n                for c in contours:\n                    if patch_count >= 16: break\n                    M = cv2.moments(c)\n                    if M[\"m00\"] == 0: continue\n                    \n                    cX, cY = int((M[\"m10\"]/M[\"m00\"])*sX), int((M[\"m01\"]/M[\"m00\"])*sY)\n                    x1 = max(0, min(cX - PATCH_SIZE//2, full_res.shape[1]-PATCH_SIZE))\n                    y1 = max(0, min(cY - PATCH_SIZE//2, full_res.shape[0]-PATCH_SIZE))\n                    \n                    # Direct buffer write to avoid RAM drawing\n                    p = zarr_img[y1:y1+PATCH_SIZE, x1:x1+PATCH_SIZE]\n                    if p.shape[-1] == 4: p = p[:,:,:3]\n                    \n                    if p.mean() < 240 and p.std() > 5:\n                        r, col = (patch_count // 4) * PATCH_SIZE, (patch_count % 4) * PATCH_SIZE\n                        composite[r:r+PATCH_SIZE, col:col+PATCH_SIZE] = p\n                        patch_count += 1\n            del contours\n            \n        # Morphology-preserving composite save [cite: 18, 76]\n        cv2.imwrite(save_path, cv2.cvtColor(composite, cv2.COLOR_RGB2BGR), [int(cv2.IMWRITE_JPEG_QUALITY), JPEG_QUALITY])\n        del composite\n        gc.collect()\n        return image_id, True, \"ok\"\n    except Exception as e:\n        return image_id, False, str(e)[:50]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T12:06:40.965212Z","iopub.execute_input":"2026-04-25T12:06:40.965622Z","iopub.status.idle":"2026-04-25T12:06:40.973797Z","shell.execute_reply.started":"2026-04-25T12:06:40.965588Z","shell.execute_reply":"2026-04-25T12:06:40.972921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport multiprocessing\nimport importlib\nfrom tqdm import tqdm\nfrom functools import partial\nimport worker \n\n# Force-load the Watershed worker engine\nimportlib.reload(worker)\n\n# --- Configuration (Directly from Manuscript) ---\nINPUT_DIR      = \"/kaggle/input/competitions/mayo-clinic-strip-ai/train\"\nCSV_PATH       = \"/kaggle/input/competitions/mayo-clinic-strip-ai/train.csv\"\nOUTPUT_DIR     = \"/kaggle/working/roi_composites/\"\nCOMPOSITE_DIM  = 512 # [cite: 149]\nPATCH_SIZE     = 128\nJPEG_QUALITY   = 70\n\n# To guarantee RAM never exceeds 25GB, we use a single worker with hard recycling\nNUM_WORKERS    = 2 \nMAXTASKS_CHILD = 1 # Wipes RAM completely after every slide to prevent leaks\n\nos.makedirs(OUTPUT_DIR, exist_ok=True)\n\nif __name__ == '__main__':\n    try:\n        multiprocessing.set_start_method(\"spawn\", force=True)\n    except RuntimeError:\n        pass\n\n    # Ensure path robustness for Kaggle environment\n    actual_csv = CSV_PATH if os.path.exists(CSV_PATH) else CSV_PATH.replace(\"competitions/\", \"\")\n    \n    if os.path.exists(actual_csv):\n        df = pd.read_csv(actual_csv)\n        image_ids = df['image_id'].unique()\n        # Strictly process the 754 verified WSIs [cite: 42]\n        pending = [i for i in image_ids if not os.path.exists(os.path.join(OUTPUT_DIR, f\"{i}.jpg\"))]\n\n        if pending:\n            actual_input = INPUT_DIR if os.path.isdir(INPUT_DIR) else INPUT_DIR.replace(\"competitions/\", \"\")\n            \n            worker_func = partial(\n                worker.process_single_wsi_worker, \n                INPUT_DIR=actual_input, \n                OUTPUT_DIR=OUTPUT_DIR, \n                COMPOSITE_DIM=COMPOSITE_DIM, \n                PATCH_SIZE=PATCH_SIZE, \n                JPEG_QUALITY=JPEG_QUALITY\n            )\n\n            print(f\"Executing Paper-Strict Watershed ROI Extraction: {len(pending)} slides.\")\n            \n            with multiprocessing.Pool(processes=NUM_WORKERS, maxtasksperchild=MAXTASKS_CHILD) as pool:\n                results = pool.imap_unordered(worker_func, pending, chunksize=1)\n                with tqdm(total=len(pending), desc=\"Watershed Extraction\", unit=\"slide\") as pbar:\n                    for _ in results: pbar.update(1)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-25T12:06:40.975392Z","iopub.execute_input":"2026-04-25T12:06:40.975691Z","execution_failed":"2026-04-25T12:15:54.78Z"}},"outputs":[],"execution_count":null}]}