{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:04.08729Z","iopub.execute_input":"2026-09-26T01:55:04.087881Z","iopub.status.idle":"2026-09-26T01:55:10.131846Z","shell.execute_reply.started":"2026-09-26T01:55:04.087851Z","shell.execute_reply":"2026-09-26T01:55:10.131181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\n\nprint(\"PyTorch version:\", torch.__version__)\nprint(\"CUDA available:\", torch.cuda.is_available())\n\nif torch.cuda.is_available():\n    print(\"GPU:\", torch.cuda.get_device_name(0))\n    print(\"GPU count:\", torch.cuda.device_count())\nelse:\n    print(\"❌ GPU NOT AVAILABLE\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:10.133255Z","iopub.execute_input":"2026-09-26T01:55:10.133598Z","iopub.status.idle":"2026-09-26T01:55:10.13856Z","shell.execute_reply.started":"2026-09-26T01:55:10.133578Z","shell.execute_reply":"2026-09-26T01:55:10.13774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(os.listdir(\"/kaggle/input\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:10.139458Z","iopub.execute_input":"2026-09-26T01:55:10.13997Z","iopub.status.idle":"2026-09-26T01:55:10.161004Z","shell.execute_reply.started":"2026-09-26T01:55:10.139946Z","shell.execute_reply":"2026-09-26T01:55:10.160281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls -lah /kaggle/input/competitions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:10.161823Z","iopub.execute_input":"2026-09-26T01:55:10.162002Z","iopub.status.idle":"2026-09-26T01:55:10.306625Z","shell.execute_reply.started":"2026-09-26T01:55:10.161984Z","shell.execute_reply":"2026-09-26T01:55:10.305941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!find /kaggle/input/competitions -maxdepth 2 -type d | head -30","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:10.309003Z","iopub.execute_input":"2026-09-26T01:55:10.309298Z","iopub.status.idle":"2026-09-26T01:55:10.547627Z","shell.execute_reply.started":"2026-09-26T01:55:10.309268Z","shell.execute_reply":"2026-09-26T01:55:10.546991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_root = \"/kaggle/input/competitions/UBC-OCEAN\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:10.548828Z","iopub.execute_input":"2026-09-26T01:55:10.549151Z","iopub.status.idle":"2026-09-26T01:55:10.5534Z","shell.execute_reply.started":"2026-09-26T01:55:10.549092Z","shell.execute_reply":"2026-09-26T01:55:10.552535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ndataset_root = \"/kaggle/input/competitions/UBC-OCEAN\"\n\nprint(\"Dataset exists:\", os.path.exists(dataset_root))\nprint(\"\\nContents:\")\nprint(os.listdir(dataset_root))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:10.554384Z","iopub.execute_input":"2026-09-26T01:55:10.554672Z","iopub.status.idle":"2026-09-26T01:55:10.569482Z","shell.execute_reply.started":"2026-09-26T01:55:10.55465Z","shell.execute_reply":"2026-09-26T01:55:10.568636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ntrain_csv = os.path.join(dataset_root, \"train.csv\")\n\ntrain_df = pd.read_csv(train_csv)\n\nprint(\"Train shape:\", train_df.shape)\nprint(\"\\nColumns:\")\nprint(train_df.columns.tolist())\n\nprint(\"\\nFirst 5 rows:\")\ndisplay(train_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:10.570416Z","iopub.execute_input":"2026-09-26T01:55:10.570685Z","iopub.status.idle":"2026-09-26T01:55:10.619189Z","shell.execute_reply.started":"2026-09-26T01:55:10.570661Z","shell.execute_reply":"2026-09-26T01:55:10.618541Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\nimport pandas as pd\n\ndataset_root = \"/kaggle/input/competitions/UBC-OCEAN\"\n\n# Load training metadata\ntrain_df = pd.read_csv(os.path.join(dataset_root, \"train.csv\"))\n\n# Select the first actual WSI (not TMA)\nwsi_row = train_df[train_df[\"is_tma\"] == False].iloc[0]\n\nimage_id = wsi_row[\"image_id\"]\n\nprint(\"Selected image ID:\", image_id)\nprint(\"Label:\", wsi_row[\"label\"])\nprint(\"Is TMA:\", wsi_row[\"is_tma\"])\nprint(\"Dimensions:\", wsi_row[\"image_width\"], \"x\", wsi_row[\"image_height\"])\n\n# Find the actual image file\npattern = os.path.join(\n    dataset_root,\n    \"train_images\",\n    f\"{image_id}.*\"\n)\n\nmatches = glob.glob(pattern)\n\nprint(\"\\nMatching files:\")\nprint(matches)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:10.620162Z","iopub.execute_input":"2026-09-26T01:55:10.620458Z","iopub.status.idle":"2026-09-26T01:55:10.631534Z","shell.execute_reply.started":"2026-09-26T01:55:10.620437Z","shell.execute_reply":"2026-09-26T01:55:10.630576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nthumbnail_dir = \"/kaggle/input/competitions/UBC-OCEAN/train_thumbnails\"\n\nfiles = os.listdir(thumbnail_dir)\n\nprint(\"Number of thumbnails:\", len(files))\nprint(\"First 20 files:\")\nprint(files[:20])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:10.63255Z","iopub.execute_input":"2026-09-26T01:55:10.632924Z","iopub.status.idle":"2026-09-26T01:55:10.648989Z","shell.execute_reply.started":"2026-09-26T01:55:10.632885Z","shell.execute_reply":"2026-09-26T01:55:10.64829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"matches = [f for f in files if f.startswith(\"4\")]\n\nprint(\"Files starting with 4:\")\nprint(matches)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:10.649868Z","iopub.execute_input":"2026-09-26T01:55:10.650284Z","iopub.status.idle":"2026-09-26T01:55:10.663316Z","shell.execute_reply.started":"2026-09-26T01:55:10.650261Z","shell.execute_reply":"2026-09-26T01:55:10.662555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nthumbnail_path = os.path.join(\n    dataset_root,\n    \"train_thumbnails\",\n    f\"{image_id}_thumbnail.png\"\n)\n\nprint(\"Thumbnail path:\", thumbnail_path)\nprint(\"File exists:\", os.path.exists(thumbnail_path))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:10.66425Z","iopub.execute_input":"2026-09-26T01:55:10.664536Z","iopub.status.idle":"2026-09-26T01:55:10.680925Z","shell.execute_reply.started":"2026-09-26T01:55:10.664504Z","shell.execute_reply":"2026-09-26T01:55:10.680278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\nimport matplotlib.pyplot as plt\n\nthumbnail = Image.open(thumbnail_path)\n\nprint(\"Thumbnail size:\", thumbnail.size)\n\nplt.figure(figsize=(12, 8))\nplt.imshow(thumbnail)\nplt.axis(\"off\")\nplt.title(f\"WSI Thumbnail - Image ID {image_id} - {wsi_row['label']}\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:10.68178Z","iopub.execute_input":"2026-09-26T01:55:10.682045Z","iopub.status.idle":"2026-09-26T01:55:11.690339Z","shell.execute_reply.started":"2026-09-26T01:55:10.682014Z","shell.execute_reply":"2026-09-26T01:55:11.689457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\n\n# Temporarily disable Pillow's pixel-count safety check\n# ONLY for opening/inspecting this trusted competition image.\nImage.MAX_IMAGE_PIXELS = None\n\nwith Image.open(wsi_path) as wsi:\n    print(\"WSI opened successfully!\")\n    print(\"Format:\", wsi.format)\n    print(\"Size:\", wsi.size)\n    print(\"Mode:\", wsi.mode)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.69306Z","iopub.execute_input":"2026-09-26T01:55:11.69383Z","iopub.status.idle":"2026-09-26T01:55:11.704028Z","shell.execute_reply.started":"2026-09-26T01:55:11.693803Z","shell.execute_reply":"2026-09-26T01:55:11.699936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport pandas as pd\n\n\nclass UBCOceanMetadata:\n    \"\"\"\n    Handles metadata and file paths for the UBC-OCEAN dataset.\n    \"\"\"\n\n    def __init__(self, dataset_root: str):\n        self.dataset_root = Path(dataset_root)\n\n        self.train_csv = self.dataset_root / \"train.csv\"\n        self.train_images_dir = self.dataset_root / \"train_images\"\n        self.train_thumbnails_dir = self.dataset_root / \"train_thumbnails\"\n\n        if not self.dataset_root.exists():\n            raise FileNotFoundError(\n                f\"Dataset root not found: {self.dataset_root}\"\n            )\n\n        if not self.train_csv.exists():\n            raise FileNotFoundError(\n                f\"train.csv not found: {self.train_csv}\"\n            )\n\n        self.df = pd.read_csv(self.train_csv)\n\n        self._validate_columns()\n\n    def _validate_columns(self):\n        required_columns = {\n            \"image_id\",\n            \"label\",\n            \"image_width\",\n            \"image_height\",\n            \"is_tma\",\n        }\n\n        missing = required_columns - set(self.df.columns)\n\n        if missing:\n            raise ValueError(\n                f\"Missing required columns: {sorted(missing)}\"\n            )\n\n    def get_record(self, image_id):\n        \"\"\"Return metadata for one image.\"\"\"\n\n        rows = self.df[self.df[\"image_id\"] == image_id]\n\n        if rows.empty:\n            raise ValueError(\n                f\"Image ID {image_id} not found in train.csv\"\n            )\n\n        row = rows.iloc[0]\n\n        return {\n            \"image_id\": int(row[\"image_id\"]),\n            \"label\": str(row[\"label\"]),\n            \"width\": int(row[\"image_width\"]),\n            \"height\": int(row[\"image_height\"]),\n            \"is_tma\": bool(row[\"is_tma\"]),\n        }\n\n    def get_image_path(self, image_id):\n        \"\"\"Return path to the original training image.\"\"\"\n\n        return self.train_images_dir / f\"{image_id}.png\"\n\n    def get_thumbnail_path(self, image_id):\n        \"\"\"\n        Return thumbnail path for a WSI.\n\n        TMAs do not have thumbnails.\n        \"\"\"\n\n        record = self.get_record(image_id)\n\n        if record[\"is_tma\"]:\n            return None\n\n        return (\n            self.train_thumbnails_dir\n            / f\"{image_id}_thumbnail.png\"\n        )\n\n    def get_all_records(self):\n        \"\"\"Return metadata for all training images.\"\"\"\n\n        records = []\n\n        for _, row in self.df.iterrows():\n            image_id = int(row[\"image_id\"])\n\n            records.append(\n                {\n                    \"image_id\": image_id,\n                    \"label\": str(row[\"label\"]),\n                    \"width\": int(row[\"image_width\"]),\n                    \"height\": int(row[\"image_height\"]),\n                    \"is_tma\": bool(row[\"is_tma\"]),\n                }\n            )\n\n        return records","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.704728Z","iopub.status.idle":"2026-09-26T01:55:11.705035Z","shell.execute_reply.started":"2026-09-26T01:55:11.704911Z","shell.execute_reply":"2026-09-26T01:55:11.704926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(\"Current directory:\")\nprint(os.getcwd())\n\nprint(\"\\nContents of /kaggle/working:\")\nprint(os.listdir(\"/kaggle/working\"))\n\nprint(\"\\nSearching for metadata.py:\")\nfor root, dirs, files in os.walk(\"/kaggle/working\"):\n    if \"metadata.py\" in files:\n        print(os.path.join(root, \"metadata.py\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.706194Z","iopub.status.idle":"2026-09-26T01:55:11.706608Z","shell.execute_reply.started":"2026-09-26T01:55:11.706408Z","shell.execute_reply":"2026-09-26T01:55:11.706434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nproject_root = Path(\"/kaggle/working/ovarian-trustworthy-ai\")\n\nfolders = [\n    \"config\",\n    \"src\",\n    \"src/preprocessing\",\n    \"src/quality\",\n    \"src/features\",\n    \"src/classification\",\n    \"src/calibration\",\n    \"src/explainability\",\n    \"src/routing\",\n    \"src/evaluation\",\n    \"cache/tiles\",\n    \"cache/embeddings\",\n    \"outputs/logs\",\n    \"outputs/metrics\",\n    \"outputs/visualizations\",\n    \"tests\",\n    \"notebooks\",\n]\n\nfor folder in folders:\n    (project_root / folder).mkdir(parents=True, exist_ok=True)\n\n# Create Python package marker files\npackage_files = [\n    \"src/__init__.py\",\n    \"src/preprocessing/__init__.py\",\n    \"src/quality/__init__.py\",\n    \"src/features/__init__.py\",\n    \"src/classification/__init__.py\",\n    \"src/calibration/__init__.py\",\n    \"src/explainability/__init__.py\",\n    \"src/routing/__init__.py\",\n    \"src/evaluation/__init__.py\",\n]\n\nfor file in package_files:\n    (project_root / file).touch()\n\nprint(\"Project structure created!\")\nprint(project_root)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.707658Z","iopub.status.idle":"2026-09-26T01:55:11.707892Z","shell.execute_reply.started":"2026-09-26T01:55:11.707778Z","shell.execute_reply":"2026-09-26T01:55:11.707792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nfor root, dirs, files in os.walk(\"/kaggle/working/ovarian-trustworthy-ai\"):\n    level = root.replace(\"/kaggle/working/ovarian-trustworthy-ai\", \"\").count(os.sep)\n    indent = \"    \" * level\n    print(f\"{indent}{os.path.basename(root)}/\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.709606Z","iopub.status.idle":"2026-09-26T01:55:11.710047Z","shell.execute_reply.started":"2026-09-26T01:55:11.7098Z","shell.execute_reply":"2026-09-26T01:55:11.709821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nmetadata_code = r'''\nfrom pathlib import Path\nimport pandas as pd\n\n\nclass UBCOceanMetadata:\n    \"\"\"\n    Handles metadata and file paths for the UBC-OCEAN dataset.\n    \"\"\"\n\n    def __init__(self, dataset_root: str):\n        self.dataset_root = Path(dataset_root)\n\n        self.train_csv = self.dataset_root / \"train.csv\"\n        self.train_images_dir = self.dataset_root / \"train_images\"\n        self.train_thumbnails_dir = self.dataset_root / \"train_thumbnails\"\n\n        if not self.dataset_root.exists():\n            raise FileNotFoundError(\n                f\"Dataset root not found: {self.dataset_root}\"\n            )\n\n        if not self.train_csv.exists():\n            raise FileNotFoundError(\n                f\"train.csv not found: {self.train_csv}\"\n            )\n\n        self.df = pd.read_csv(self.train_csv)\n\n        self._validate_columns()\n\n    def _validate_columns(self):\n        required_columns = {\n            \"image_id\",\n            \"label\",\n            \"image_width\",\n            \"image_height\",\n            \"is_tma\",\n        }\n\n        missing = required_columns - set(self.df.columns)\n\n        if missing:\n            raise ValueError(\n                f\"Missing required columns: {sorted(missing)}\"\n            )\n\n    def get_record(self, image_id):\n        \"\"\"Return metadata for one image.\"\"\"\n\n        rows = self.df[self.df[\"image_id\"] == image_id]\n\n        if rows.empty:\n            raise ValueError(\n                f\"Image ID {image_id} not found in train.csv\"\n            )\n\n        row = rows.iloc[0]\n\n        return {\n            \"image_id\": int(row[\"image_id\"]),\n            \"label\": str(row[\"label\"]),\n            \"width\": int(row[\"image_width\"]),\n            \"height\": int(row[\"image_height\"]),\n            \"is_tma\": bool(row[\"is_tma\"]),\n        }\n\n    def get_image_path(self, image_id):\n        \"\"\"Return path to the original training image.\"\"\"\n\n        return self.train_images_dir / f\"{image_id}.png\"\n\n    def get_thumbnail_path(self, image_id):\n        \"\"\"\n        Return thumbnail path for a WSI.\n\n        TMAs do not have thumbnails.\n        \"\"\"\n\n        record = self.get_record(image_id)\n\n        if record[\"is_tma\"]:\n            return None\n\n        return (\n            self.train_thumbnails_dir\n            / f\"{image_id}_thumbnail.png\"\n        )\n\n    def get_all_records(self):\n        \"\"\"Return metadata for all training images.\"\"\"\n\n        records = []\n\n        for _, row in self.df.iterrows():\n            image_id = int(row[\"image_id\"])\n\n            records.append(\n                {\n                    \"image_id\": image_id,\n                    \"label\": str(row[\"label\"]),\n                    \"width\": int(row[\"image_width\"]),\n                    \"height\": int(row[\"image_height\"]),\n                    \"is_tma\": bool(row[\"is_tma\"]),\n                }\n            )\n\n        return records\n'''\n\nmetadata_path = (\n    Path(\"/kaggle/working/ovarian-trustworthy-ai\")\n    / \"src\"\n    / \"preprocessing\"\n    / \"metadata.py\"\n)\n\nmetadata_path.write_text(metadata_code)\n\nprint(\"Created:\")\nprint(metadata_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.711555Z","iopub.status.idle":"2026-09-26T01:55:11.712004Z","shell.execute_reply.started":"2026-09-26T01:55:11.711817Z","shell.execute_reply":"2026-09-26T01:55:11.711841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport os\n\nproject_root = Path(\"/kaggle/working/ovarian-trustworthy-ai\")\n\nprint(\"Project exists:\", project_root.exists())\nprint(\"Project path:\", project_root)\n\nprint(\"\\nProject contents:\")\nif project_root.exists():\n    for path in project_root.rglob(\"*\"):\n        print(path)\nelse:\n    print(\"PROJECT DOES NOT EXIST\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.713576Z","iopub.status.idle":"2026-09-26T01:55:11.713879Z","shell.execute_reply.started":"2026-09-26T01:55:11.71374Z","shell.execute_reply":"2026-09-26T01:55:11.713761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nfrom pathlib import Path\n\nproject_root = Path(\"/kaggle/working/ovarian-trustworthy-ai\")\n\nprint(\"Project exists:\", project_root.exists())\nprint(\"src exists:\", (project_root / \"src\").exists())\nprint(\"metadata.py exists:\", (project_root / \"src/preprocessing/metadata.py\").exists())\n\n# Add the project root to Python's import path\nsys.path.insert(0, str(project_root))\n\nprint(\"\\nProject root added to sys.path:\")\nprint(str(project_root) in sys.path)\n\nprint(\"\\nFirst entries in sys.path:\")\nprint(sys.path[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.714916Z","iopub.status.idle":"2026-09-26T01:55:11.715309Z","shell.execute_reply.started":"2026-09-26T01:55:11.715094Z","shell.execute_reply":"2026-09-26T01:55:11.715139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import importlib.util\n\nspec = importlib.util.find_spec(\"src\")\n\nprint(\"\\nPython can find src:\", spec is not None)\n\nif spec is not None:\n    print(\"src location:\", spec.origin)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.716592Z","iopub.status.idle":"2026-09-26T01:55:11.716952Z","shell.execute_reply.started":"2026-09-26T01:55:11.716788Z","shell.execute_reply":"2026-09-26T01:55:11.716817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nfrom pathlib import Path\nimport importlib\n\nproject_root = \"/kaggle/working/ovarian-trustworthy-ai\"\n\nprint(\"1. Project root exists:\")\nprint(Path(project_root).exists())\n\nprint(\"\\n2. src exists:\")\nprint(Path(project_root, \"src\").exists())\n\nprint(\"\\n3. src __init__.py exists:\")\nprint(Path(project_root, \"src\", \"__init__.py\").exists())\n\nprint(\"\\n4. metadata.py exists:\")\nprint(Path(project_root, \"src\", \"preprocessing\", \"metadata.py\").exists())\n\nprint(\"\\n5. Project root in sys.path BEFORE:\")\nprint(project_root in sys.path)\n\nsys.path.insert(0, project_root)\n\nprint(\"\\n6. Project root in sys.path AFTER:\")\nprint(project_root in sys.path)\n\nprint(\"\\n7. sys.path first 10 entries:\")\nfor p in sys.path[:10]:\n    print(repr(p))\n\nprint(\"\\n8. Importlib caches invalidated\")\nimportlib.invalidate_caches()\n\nprint(\"\\n9. Find src:\")\nprint(importlib.util.find_spec(\"src\"))\n\nprint(\"\\n10. Find metadata directly:\")\nmetadata_file = Path(project_root) / \"src\" / \"preprocessing\" / \"metadata.py\"\nprint(metadata_file.exists())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.718047Z","iopub.status.idle":"2026-09-26T01:55:11.718542Z","shell.execute_reply.started":"2026-09-26T01:55:11.718309Z","shell.execute_reply":"2026-09-26T01:55:11.718331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from src.preprocessing.metadata import UBCOceanMetadata\n\nprint(\"Import successful! ✅\")\nprint(UBCOceanMetadata)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.71967Z","iopub.status.idle":"2026-09-26T01:55:11.720506Z","shell.execute_reply.started":"2026-09-26T01:55:11.720326Z","shell.execute_reply":"2026-09-26T01:55:11.720346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_root = \"/kaggle/input/competitions/UBC-OCEAN\"\n\nmetadata = UBCOceanMetadata(dataset_root)\n\nprint(\"Metadata loaded successfully!\")\nprint(\"Number of training records:\", len(metadata.df))\n\nrecord = metadata.get_record(4)\n\nprint(\"\\nImage 4 metadata:\")\nfor key, value in record.items():\n    print(f\"{key}: {value}\")\n\nprint(\"\\nOriginal image:\")\nprint(metadata.get_image_path(4))\n\nprint(\"\\nThumbnail:\")\nprint(metadata.get_thumbnail_path(4))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.72236Z","iopub.status.idle":"2026-09-26T01:55:11.722613Z","shell.execute_reply.started":"2026-09-26T01:55:11.7225Z","shell.execute_reply":"2026-09-26T01:55:11.722514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pyvips\n\nprint(\"pyvips imported successfully! ✅\")\nprint(\"libvips version:\", pyvips.version(0), \n      pyvips.version(1), \n      pyvips.version(2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.724292Z","iopub.status.idle":"2026-09-26T01:55:11.724664Z","shell.execute_reply.started":"2026-09-26T01:55:11.724467Z","shell.execute_reply":"2026-09-26T01:55:11.724488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q \"pyvips[binary]\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.725544Z","iopub.status.idle":"2026-09-26T01:55:11.7259Z","shell.execute_reply.started":"2026-09-26T01:55:11.725722Z","shell.execute_reply":"2026-09-26T01:55:11.725743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pyvips\n\nprint(\"pyvips imported successfully! ✅\")\nprint(\n    \"libvips version:\",\n    pyvips.version(0),\n    pyvips.version(1),\n    pyvips.version(2)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.727413Z","iopub.status.idle":"2026-09-26T01:55:11.727759Z","shell.execute_reply.started":"2026-09-26T01:55:11.72759Z","shell.execute_reply":"2026-09-26T01:55:11.727612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nproject_root = Path(\"/kaggle/working/ovarian-trustworthy-ai\")\nfile_path = project_root / \"src/preprocessing/image_reader.py\"\n\ncode = '''\nfrom pathlib import Path\nimport numpy as np\nimport pyvips\n\nfrom .metadata import UBCOceanMetadata\n\n\nclass UBCOceanImageReader:\n    \"\"\"\n    Safely opens UBC-OCEAN images and reads small image regions.\n\n    Large WSI files are never loaded completely into NumPy.\n    \"\"\"\n\n    def __init__(self, dataset_root: str):\n        self.metadata = UBCOceanMetadata(dataset_root)\n\n    def open_image(self, image_id: int):\n        \"\"\"\n        Open an image using pyvips.\n\n        Only image metadata/header information is accessed initially.\n        \"\"\"\n        image_path = self.metadata.get_image_path(image_id)\n\n        if not image_path.exists():\n            raise FileNotFoundError(\n                f\"Image file not found: {image_path}\"\n            )\n\n        image = pyvips.Image.new_from_file(\n            str(image_path),\n            access=\"random\"\n        )\n\n        return image\n\n    def get_region(\n        self,\n        image_id: int,\n        x: int,\n        y: int,\n        width: int,\n        height: int\n    ) -> np.ndarray:\n        \"\"\"\n        Extract a small region from an image and return it as NumPy RGB.\n\n        The full WSI is never converted to NumPy.\n        \"\"\"\n\n        image = self.open_image(image_id)\n\n        if x < 0 or y < 0:\n            raise ValueError(\"x and y must be non-negative.\")\n\n        if width <= 0 or height <= 0:\n            raise ValueError(\"width and height must be positive.\")\n\n        if x >= image.width or y >= image.height:\n            raise ValueError(\n                f\"Region origin ({x}, {y}) is outside \"\n                f\"image dimensions ({image.width}, {image.height}).\"\n            )\n\n        width = min(width, image.width - x)\n        height = min(height, image.height - y)\n\n        region = image.crop(x, y, width, height)\n\n        array = np.ndarray(\n            buffer=region.write_to_memory(),\n            dtype=np.uint8,\n            shape=(region.height, region.width, region.bands)\n        )\n\n        if array.shape[2] >= 3:\n            array = array[:, :, :3]\n\n        return array\n\n    def get_image_dimensions(self, image_id: int):\n        \"\"\"\n        Return actual image dimensions from pyvips.\n        \"\"\"\n        image = self.open_image(image_id)\n        return image.width, image.height\n'''\n\nfile_path.write_text(code)\n\nprint(f\"Created: {file_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.728852Z","iopub.status.idle":"2026-09-26T01:55:11.729231Z","shell.execute_reply.started":"2026-09-26T01:55:11.729022Z","shell.execute_reply":"2026-09-26T01:55:11.729053Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nimport importlib\n\nproject_root = \"/kaggle/working/ovarian-trustworthy-ai\"\n\nif project_root not in sys.path:\n    sys.path.insert(0, project_root)\n\nimportlib.invalidate_caches()\n\nfrom src.preprocessing.image_reader import UBCOceanImageReader\n\ndataset_root = \"/kaggle/input/competitions/UBC-OCEAN\"\n\nreader = UBCOceanImageReader(dataset_root)\n\nprint(\"Image reader imported successfully! ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.730328Z","iopub.status.idle":"2026-09-26T01:55:11.730621Z","shell.execute_reply.started":"2026-09-26T01:55:11.730499Z","shell.execute_reply":"2026-09-26T01:55:11.730513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image = reader.open_image(4)\n\nprint(\"Image opened successfully! ✅\")\nprint(\"Width :\", image.width)\nprint(\"Height:\", image.height)\nprint(\"Bands :\", image.bands)\nprint(\"Format:\", image.format)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.732238Z","iopub.status.idle":"2026-09-26T01:55:11.732589Z","shell.execute_reply.started":"2026-09-26T01:55:11.732409Z","shell.execute_reply":"2026-09-26T01:55:11.732432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tile = reader.get_region(\n    image_id=4,\n    x=5000,\n    y=5000,\n    width=256,\n    height=256\n)\n\nprint(\"Region extracted successfully! ✅\")\nprint(\"Shape:\", tile.shape)\nprint(\"Data type:\", tile.dtype)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.734185Z","iopub.status.idle":"2026-09-26T01:55:11.734496Z","shell.execute_reply.started":"2026-09-26T01:55:11.734336Z","shell.execute_reply":"2026-09-26T01:55:11.734356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(5, 5))\nplt.imshow(tile)\nplt.axis(\"off\")\nplt.title(\"WSI 4 — 256 × 256 Region\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.735804Z","iopub.status.idle":"2026-09-26T01:55:11.736155Z","shell.execute_reply.started":"2026-09-26T01:55:11.735955Z","shell.execute_reply":"2026-09-26T01:55:11.735975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nproject_root = Path(\"/kaggle/working/ovarian-trustworthy-ai\")\nfile_path = project_root / \"src/preprocessing/slide_type.py\"\n\ncode = '''\nfrom dataclasses import dataclass\n\n\n@dataclass\nclass SlideInfo:\n    image_id: int\n    slide_type: str\n    width: int\n    height: int\n    magnification: str\n\n\nclass SlideTypeHandler:\n    \"\"\"\n    Determines how a UBC-OCEAN image should be handled\n    based on the official train.csv metadata.\n    \"\"\"\n\n    def __init__(self, metadata):\n        self.metadata = metadata\n\n    def get_slide_info(self, image_id: int) -> SlideInfo:\n\n        record = self.metadata.get_record(image_id)\n\n        if record[\"is_tma\"]:\n            slide_type = \"TMA\"\n            magnification = \"40x\"\n        else:\n            slide_type = \"WSI\"\n            magnification = \"20x\"\n\n        return SlideInfo(\n            image_id=record[\"image_id\"],\n            slide_type=slide_type,\n            width=record[\"width\"],\n            height=record[\"height\"],\n            magnification=magnification,\n        )\n\n    def is_wsi(self, image_id: int) -> bool:\n        return not self.metadata.get_record(image_id)[\"is_tma\"]\n\n    def is_tma(self, image_id: int) -> bool:\n        return self.metadata.get_record(image_id)[\"is_tma\"]\n'''\n\nfile_path.write_text(code)\n\nprint(f\"Created: {file_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.739013Z","iopub.status.idle":"2026-09-26T01:55:11.739472Z","shell.execute_reply.started":"2026-09-26T01:55:11.739331Z","shell.execute_reply":"2026-09-26T01:55:11.739353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nimport importlib\n\nproject_root = \"/kaggle/working/ovarian-trustworthy-ai\"\n\nif project_root not in sys.path:\n    sys.path.insert(0, project_root)\n\nimportlib.invalidate_caches()\n\nfrom src.preprocessing.metadata import UBCOceanMetadata\nfrom src.preprocessing.slide_type import SlideTypeHandler\n\ndataset_root = \"/kaggle/input/competitions/UBC-OCEAN\"\n\nmetadata = UBCOceanMetadata(dataset_root)\nslide_handler = SlideTypeHandler(metadata)\n\ninfo = slide_handler.get_slide_info(4)\n\nprint(info)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.740803Z","iopub.status.idle":"2026-09-26T01:55:11.741218Z","shell.execute_reply.started":"2026-09-26T01:55:11.741032Z","shell.execute_reply":"2026-09-26T01:55:11.741053Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"info_tma = slide_handler.get_slide_info(91)\n\nprint(info_tma)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.742449Z","iopub.status.idle":"2026-09-26T01:55:11.742823Z","shell.execute_reply.started":"2026-09-26T01:55:11.742675Z","shell.execute_reply":"2026-09-26T01:55:11.7427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nproject_root = Path(\"/kaggle/working/ovarian-trustworthy-ai\")\nfile_path = project_root / \"src/preprocessing/tile_extractor.py\"\n\ncode = '''\nfrom dataclasses import dataclass\nfrom typing import List\n\nimport cv2\nimport numpy as np\n\nfrom .image_reader import UBCOceanImageReader\nfrom .slide_type import SlideTypeHandler\n\n\n@dataclass\nclass TileCoordinate:\n    image_id: int\n    x: int\n    y: int\n    source_size: int\n    standardized_size: int\n    slide_type: str\n\n\nclass TileExtractor:\n    \"\"\"\n    Extracts standardized tiles from UBC-OCEAN WSI and TMA images.\n\n    WSI:\n        256 x 256 native tiles at approximately 20x.\n\n    TMA:\n        512 x 512 native tiles at approximately 40x,\n        resized to 256 x 256 for standardization.\n    \"\"\"\n\n    def __init__(self, image_reader, slide_handler):\n        self.image_reader = image_reader\n        self.slide_handler = slide_handler\n\n    def get_tile_size(self, image_id: int):\n        \"\"\"\n        Return the native tile size according to slide type.\n        \"\"\"\n\n        if self.slide_handler.is_tma(image_id):\n            return 512\n\n        return 256\n\n    def generate_grid(\n        self,\n        image_id: int,\n        stride: int = None\n    ) -> List[TileCoordinate]:\n        \"\"\"\n        Generate candidate tile coordinates.\n\n        This function only generates coordinates.\n        It does NOT read the image.\n        \"\"\"\n\n        info = self.slide_handler.get_slide_info(image_id)\n\n        tile_size = self.get_tile_size(image_id)\n\n        if stride is None:\n            stride = tile_size\n\n        coordinates = []\n\n        for y in range(0, info.height - tile_size + 1, stride):\n            for x in range(0, info.width - tile_size + 1, stride):\n\n                coordinates.append(\n                    TileCoordinate(\n                        image_id=image_id,\n                        x=x,\n                        y=y,\n                        source_size=tile_size,\n                        standardized_size=256,\n                        slide_type=info.slide_type,\n                    )\n                )\n\n        return coordinates\n\n    def extract_tile(\n        self,\n        coordinate: TileCoordinate\n    ) -> np.ndarray:\n        \"\"\"\n        Extract one tile and convert it to the standardized\n        256 x 256 RGB representation.\n        \"\"\"\n\n        tile = self.image_reader.get_region(\n            image_id=coordinate.image_id,\n            x=coordinate.x,\n            y=coordinate.y,\n            width=coordinate.source_size,\n            height=coordinate.source_size,\n        )\n\n        if coordinate.source_size != coordinate.standardized_size:\n\n            tile = cv2.resize(\n                tile,\n                (\n                    coordinate.standardized_size,\n                    coordinate.standardized_size\n                ),\n                interpolation=cv2.INTER_AREA,\n            )\n\n        return tile\n\n    def extract_tiles(\n        self,\n        image_id: int,\n        max_tiles: int = None\n    ) -> List[np.ndarray]:\n        \"\"\"\n        Extract tiles from a regular grid.\n\n        max_tiles can be used to prevent excessive memory usage.\n        \"\"\"\n\n        coordinates = self.generate_grid(image_id)\n\n        if max_tiles is not None:\n            coordinates = coordinates[:max_tiles]\n\n        tiles = []\n\n        for coordinate in coordinates:\n\n            tile = self.extract_tile(coordinate)\n\n            tiles.append(tile)\n\n        return tiles\n'''\n\nfile_path.write_text(code)\n\nprint(f\"Created: {file_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.744286Z","iopub.status.idle":"2026-09-26T01:55:11.744622Z","shell.execute_reply.started":"2026-09-26T01:55:11.74451Z","shell.execute_reply":"2026-09-26T01:55:11.744524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nimport importlib\n\nproject_root = \"/kaggle/working/ovarian-trustworthy-ai\"\n\nif project_root not in sys.path:\n    sys.path.insert(0, project_root)\n\nimportlib.invalidate_caches()\n\nfrom src.preprocessing.tile_extractor import TileExtractor\n\nprint(\"Tile extractor imported successfully! ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.746174Z","iopub.status.idle":"2026-09-26T01:55:11.746524Z","shell.execute_reply.started":"2026-09-26T01:55:11.746336Z","shell.execute_reply":"2026-09-26T01:55:11.746358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from src.preprocessing.metadata import UBCOceanMetadata\nfrom src.preprocessing.image_reader import UBCOceanImageReader\nfrom src.preprocessing.slide_type import SlideTypeHandler\nfrom src.preprocessing.tile_extractor import TileExtractor\n\ndataset_root = \"/kaggle/input/competitions/UBC-OCEAN\"\n\nmetadata = UBCOceanMetadata(dataset_root)\n\nreader = UBCOceanImageReader(dataset_root)\n\nslide_handler = SlideTypeHandler(metadata)\n\ntile_extractor = TileExtractor(\n    image_reader=reader,\n    slide_handler=slide_handler\n)\n\nprint(\"Tile extraction pipeline initialized! ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.747372Z","iopub.status.idle":"2026-09-26T01:55:11.747617Z","shell.execute_reply.started":"2026-09-26T01:55:11.74751Z","shell.execute_reply":"2026-09-26T01:55:11.747524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"wsi_coords = tile_extractor.generate_grid(\n    image_id=4,\n    stride=256\n)\n\nprint(\"Number of candidate WSI tiles:\", len(wsi_coords))\nprint(\"First 5 coordinates:\")\n\nfor coord in wsi_coords[:5]:\n    print(coord)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.748539Z","iopub.status.idle":"2026-09-26T01:55:11.748756Z","shell.execute_reply.started":"2026-09-26T01:55:11.748654Z","shell.execute_reply":"2026-09-26T01:55:11.748667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"wsi_tiles = []\n\nfor coord in wsi_coords[:3]:\n    tile = tile_extractor.extract_tile(coord)\n    wsi_tiles.append(tile)\n\nprint(\"Extracted:\", len(wsi_tiles), \"tiles\")\n\nfor i, tile in enumerate(wsi_tiles):\n    print(f\"Tile {i}: shape={tile.shape}, dtype={tile.dtype}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.749944Z","iopub.status.idle":"2026-09-26T01:55:11.750226Z","shell.execute_reply.started":"2026-09-26T01:55:11.750057Z","shell.execute_reply":"2026-09-26T01:55:11.75007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"wsi_tiles = []\n\nfor coord in wsi_coords[:3]:\n    tile = tile_extractor.extract_tile(coord)\n    wsi_tiles.append(tile)\n\nprint(\"Extracted:\", len(wsi_tiles), \"tiles\")\n\nfor i, tile in enumerate(wsi_tiles):\n    print(f\"Tile {i}: shape={tile.shape}, dtype={tile.dtype}\")\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.751777Z","iopub.status.idle":"2026-09-26T01:55:11.75205Z","shell.execute_reply.started":"2026-09-26T01:55:11.751945Z","shell.execute_reply":"2026-09-26T01:55:11.751959Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tma_coords = tile_extractor.generate_grid(\n    image_id=91,\n    stride=512\n)\n\nprint(\"Number of candidate TMA tiles:\", len(tma_coords))\n\nprint(\"First 3 coordinates:\")\n\nfor coord in tma_coords[:3]:\n    print(coord)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.753506Z","iopub.status.idle":"2026-09-26T01:55:11.753846Z","shell.execute_reply.started":"2026-09-26T01:55:11.7537Z","shell.execute_reply":"2026-09-26T01:55:11.753714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tma_tile = tile_extractor.extract_tile(tma_coords[0])\n\nprint(\"TMA standardized tile:\")\nprint(\"Shape:\", tma_tile.shape)\nprint(\"Dtype:\", tma_tile.dtype)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.75478Z","iopub.status.idle":"2026-09-26T01:55:11.755076Z","shell.execute_reply.started":"2026-09-26T01:55:11.754964Z","shell.execute_reply":"2026-09-26T01:55:11.754978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nproject_root = Path(\"/kaggle/working/ovarian-trustworthy-ai\")\nfile_path = project_root / \"src/preprocessing/thumbnail_selector.py\"\n\ncode = '''\nfrom dataclasses import replace\n\nimport cv2\nimport numpy as np\nimport pyvips\n\n\nclass ThumbnailCandidateSelector:\n    \"\"\"\n    Uses a WSI thumbnail to identify candidate tissue regions.\n\n    This is a coarse candidate-selection step.\n    Detailed adaptive quality assessment is performed later in M2.\n    \"\"\"\n\n    def __init__(self, metadata):\n        self.metadata = metadata\n\n    def load_thumbnail(self, image_id: int) -> np.ndarray:\n        \"\"\"\n        Load the WSI thumbnail as an RGB NumPy array.\n\n        TMAs do not have official thumbnails in UBC-OCEAN,\n        so this method is intended for WSIs only.\n        \"\"\"\n\n        thumbnail_path = self.metadata.get_thumbnail_path(image_id)\n\n        if thumbnail_path is None:\n            raise ValueError(\n                f\"Image {image_id} is a TMA and has no official thumbnail.\"\n            )\n\n        if not thumbnail_path.exists():\n            raise FileNotFoundError(\n                f\"Thumbnail not found: {thumbnail_path}\"\n            )\n\n        thumbnail = pyvips.Image.new_from_file(\n            str(thumbnail_path),\n            access=\"sequential\"\n        )\n\n        array = np.ndarray(\n            buffer=thumbnail.write_to_memory(),\n            dtype=np.uint8,\n            shape=(thumbnail.height, thumbnail.width, thumbnail.bands)\n        )\n\n        if array.shape[2] >= 3:\n            array = array[:, :, :3]\n\n        return array\n\n    def create_tissue_mask(self, image_id: int) -> np.ndarray:\n        \"\"\"\n        Create a coarse tissue mask using HSV saturation\n        and Otsu thresholding.\n\n        Higher-saturation regions are treated as likely tissue.\n        \"\"\"\n\n        thumbnail = self.load_thumbnail(image_id)\n\n        hsv = cv2.cvtColor(thumbnail, cv2.COLOR_RGB2HSV)\n\n        saturation = hsv[:, :, 1]\n\n        _, mask = cv2.threshold(\n            saturation,\n            0,\n            255,\n            cv2.THRESH_BINARY + cv2.THRESH_OTSU\n        )\n\n        # Remove tiny isolated regions.\n        kernel = np.ones((3, 3), np.uint8)\n\n        mask = cv2.morphologyEx(\n            mask,\n            cv2.MORPH_OPEN,\n            kernel\n        )\n\n        mask = cv2.morphologyEx(\n            mask,\n            cv2.MORPH_CLOSE,\n            kernel\n        )\n\n        return mask > 0\n\n    def get_thumbnail_tissue_ratio(\n        self,\n        image_id: int,\n        x: int,\n        y: int,\n        width: int,\n        height: int\n    ) -> float:\n        \"\"\"\n        Estimate the tissue fraction corresponding to one\n        full-resolution WSI tile.\n        \"\"\"\n\n        record = self.metadata.get_record(image_id)\n\n        if record[\"is_tma\"]:\n            raise ValueError(\n                \"Thumbnail-guided selection is only for WSIs.\"\n            )\n\n        mask = self.create_tissue_mask(image_id)\n\n        thumb_height, thumb_width = mask.shape\n\n        scale_x = thumb_width / record[\"width\"]\n        scale_y = thumb_height / record[\"height\"]\n\n        tx0 = max(0, int(np.floor(x * scale_x)))\n        ty0 = max(0, int(np.floor(y * scale_y)))\n\n        tx1 = min(\n            thumb_width,\n            int(np.ceil((x + width) * scale_x))\n        )\n\n        ty1 = min(\n            thumb_height,\n            int(np.ceil((y + height) * scale_y))\n        )\n\n        if tx1 <= tx0 or ty1 <= ty0:\n            return 0.0\n\n        region = mask[ty0:ty1, tx0:tx1]\n\n        return float(region.mean())\n\n    def select_candidates(\n        self,\n        coordinates,\n        tissue_ratio_threshold: float = 0.20\n    ):\n        \"\"\"\n        Keep candidate WSI coordinates whose corresponding\n        thumbnail region contains enough likely tissue.\n\n        TMA coordinates are retained because TMAs do not have\n        official thumbnails.\n        \"\"\"\n\n        if not coordinates:\n            return []\n\n        image_id = coordinates[0].image_id\n\n        record = self.metadata.get_record(image_id)\n\n        if record[\"is_tma\"]:\n            return list(coordinates)\n\n        mask = self.create_tissue_mask(image_id)\n\n        thumb_height, thumb_width = mask.shape\n\n        scale_x = thumb_width / record[\"width\"]\n        scale_y = thumb_height / record[\"height\"]\n\n        selected = []\n\n        for coordinate in coordinates:\n\n            tx0 = max(\n                0,\n                int(np.floor(coordinate.x * scale_x))\n            )\n\n            ty0 = max(\n                0,\n                int(np.floor(coordinate.y * scale_y))\n            )\n\n            tx1 = min(\n                thumb_width,\n                int(np.ceil(\n                    (coordinate.x + coordinate.source_size) * scale_x\n                ))\n            )\n\n            ty1 = min(\n                thumb_height,\n                int(np.ceil(\n                    (coordinate.y + coordinate.source_size) * scale_y\n                ))\n            )\n\n            if tx1 <= tx0 or ty1 <= ty0:\n                continue\n\n            region = mask[ty0:ty1, tx0:tx1]\n\n            tissue_ratio = float(region.mean())\n\n            if tissue_ratio >= tissue_ratio_threshold:\n                selected.append(coordinate)\n\n        return selected\n'''\n\nfile_path.write_text(code)\n\nprint(f\"Created: {file_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.756093Z","iopub.status.idle":"2026-09-26T01:55:11.756358Z","shell.execute_reply.started":"2026-09-26T01:55:11.756247Z","shell.execute_reply":"2026-09-26T01:55:11.756262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nimport importlib\n\nproject_root = \"/kaggle/working/ovarian-trustworthy-ai\"\n\nif project_root not in sys.path:\n    sys.path.insert(0, project_root)\n\nimportlib.invalidate_caches()\n\nfrom src.preprocessing.thumbnail_selector import ThumbnailCandidateSelector\n\nprint(\"Thumbnail selector imported successfully! ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.756829Z","iopub.status.idle":"2026-09-26T01:55:11.757062Z","shell.execute_reply.started":"2026-09-26T01:55:11.756956Z","shell.execute_reply":"2026-09-26T01:55:11.756969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selector = ThumbnailCandidateSelector(metadata)\n\nprint(\"Thumbnail candidate selector initialized! ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.757569Z","iopub.status.idle":"2026-09-26T01:55:11.757798Z","shell.execute_reply.started":"2026-09-26T01:55:11.757672Z","shell.execute_reply":"2026-09-26T01:55:11.757685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"wsi_mask = selector.create_tissue_mask(4)\n\nprint(\"WSI tissue mask created! ✅\")\nprint(\"Mask shape:\", wsi_mask.shape)\nprint(\"Tissue fraction:\", wsi_mask.mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.758342Z","iopub.status.idle":"2026-09-26T01:55:11.758582Z","shell.execute_reply.started":"2026-09-26T01:55:11.758474Z","shell.execute_reply":"2026-09-26T01:55:11.758486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(8, 7))\nplt.imshow(wsi_mask, cmap=\"gray\")\nplt.title(\"WSI 4 — Thumbnail Tissue Mask\")\nplt.axis(\"off\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.759922Z","iopub.status.idle":"2026-09-26T01:55:11.760301Z","shell.execute_reply.started":"2026-09-26T01:55:11.760101Z","shell.execute_reply":"2026-09-26T01:55:11.760142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_wsi_coords = selector.select_candidates(\n    wsi_coords,\n    tissue_ratio_threshold=0.20\n)\n\nprint(\"Original WSI candidates:\", len(wsi_coords))\nprint(\"Thumbnail-selected candidates:\", len(selected_wsi_coords))\nprint(\n    \"Candidates removed:\",\n    len(wsi_coords) - len(selected_wsi_coords)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.761507Z","iopub.status.idle":"2026-09-26T01:55:11.761744Z","shell.execute_reply.started":"2026-09-26T01:55:11.761627Z","shell.execute_reply":"2026-09-26T01:55:11.76164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"First 5 selected coordinates:\")\n\nfor coord in selected_wsi_coords[:5]:\n    print(coord)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.762958Z","iopub.status.idle":"2026-09-26T01:55:11.763292Z","shell.execute_reply.started":"2026-09-26T01:55:11.763143Z","shell.execute_reply":"2026-09-26T01:55:11.763171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_tile = tile_extractor.extract_tile(\n    selected_wsi_coords[0]\n)\n\nprint(\"Selected tile extracted successfully! ✅\")\nprint(\"Shape:\", selected_tile.shape)\nprint(\"Dtype:\", selected_tile.dtype)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.764352Z","iopub.status.idle":"2026-09-26T01:55:11.764686Z","shell.execute_reply.started":"2026-09-26T01:55:11.764504Z","shell.execute_reply":"2026-09-26T01:55:11.764525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(5, 5))\nplt.imshow(selected_tile)\nplt.title(\"Thumbnail-Selected WSI Tile\")\nplt.axis(\"off\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.766154Z","iopub.status.idle":"2026-09-26T01:55:11.766467Z","shell.execute_reply.started":"2026-09-26T01:55:11.766282Z","shell.execute_reply":"2026-09-26T01:55:11.766295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_tma_coords = selector.select_candidates(\n    tma_coords,\n    tissue_ratio_threshold=0.20\n)\n\nprint(\"Original TMA candidates:\", len(tma_coords))\nprint(\"Selected TMA candidates:\", len(selected_tma_coords))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.767753Z","iopub.status.idle":"2026-09-26T01:55:11.768167Z","shell.execute_reply.started":"2026-09-26T01:55:11.767938Z","shell.execute_reply":"2026-09-26T01:55:11.767962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile /kaggle/working/ovarian-trustworthy-ai/src/preprocessing/stain_normalization.py\n\nfrom dataclasses import dataclass\n\nimport numpy as np\n\n\n@dataclass\nclass StainReference:\n    stain_matrix: np.ndarray\n    max_concentration: np.ndarray\n    io: float = 255.0\n\n\nclass MacenkoNormalizer:\n    \"\"\"\n    Macenko stain normalization for H&E histopathology images.\n\n    Pipeline:\n        RGB\n          -> Optical Density\n          -> foreground filtering\n          -> covariance/eigenvector analysis\n          -> angular percentile stain vectors\n          -> stain concentration estimation\n          -> concentration normalization\n          -> RGB reconstruction\n    \"\"\"\n\n    def __init__(\n        self,\n        beta: float = 0.15,\n        alpha: float = 1.0,\n        max_samples: int = 10000,\n        io: float = 255.0,\n    ):\n        self.beta = beta\n        self.alpha = alpha\n        self.max_samples = max_samples\n        self.io = io\n\n        self.reference = None\n\n    # =========================================================\n    # RGB -> Optical Density\n    # =========================================================\n    def rgb_to_od(self, image: np.ndarray) -> np.ndarray:\n\n        image = image.astype(np.float32)\n\n        image = np.clip(image, 0, 255)\n\n        od = -np.log(\n            (image + 1.0) / self.io\n        )\n\n        return od\n\n    # =========================================================\n    # Optical Density -> RGB\n    # =========================================================\n    def od_to_rgb(self, od: np.ndarray) -> np.ndarray:\n\n        rgb = self.io * np.exp(-od) - 1.0\n\n        rgb = np.clip(rgb, 0, 255)\n\n        return rgb.astype(np.uint8)\n\n    # =========================================================\n    # Pixel sampling\n    # =========================================================\n    def _sample_pixels(\n        self,\n        pixels: np.ndarray\n    ) -> np.ndarray:\n\n        if len(pixels) <= self.max_samples:\n            return pixels\n\n        rng = np.random.default_rng(42)\n\n        indices = rng.choice(\n            len(pixels),\n            size=self.max_samples,\n            replace=False,\n        )\n\n        return pixels[indices]\n\n    # =========================================================\n    # Estimate Macenko stain matrix\n    # =========================================================\n    def _estimate_stain_matrix(\n        self,\n        image: np.ndarray,\n    ) -> np.ndarray:\n\n        if image.ndim != 3 or image.shape[2] != 3:\n            raise ValueError(\n                \"Image must have shape H x W x 3.\"\n            )\n\n        # -----------------------------------------------------\n        # Step 1: RGB -> OD\n        # -----------------------------------------------------\n        od = self.rgb_to_od(image)\n\n        pixels = od.reshape(-1, 3)\n\n        # -----------------------------------------------------\n        # Step 2: Remove low-OD/background pixels\n        # -----------------------------------------------------\n        mask = np.min(pixels, axis=1) >= self.beta\n\n        odhat = pixels[mask]\n\n        if len(odhat) < 100:\n            raise ValueError(\n                \"Not enough foreground pixels for \"\n                \"Macenko stain estimation.\"\n            )\n\n        # -----------------------------------------------------\n        # Step 3: Sample pixels for efficiency\n        # -----------------------------------------------------\n        odhat = self._sample_pixels(odhat)\n\n        # -----------------------------------------------------\n        # Step 4: Covariance matrix\n        # -----------------------------------------------------\n        covariance = np.cov(\n            odhat.T\n        )\n\n        # -----------------------------------------------------\n        # Step 5: Eigen decomposition\n        # -----------------------------------------------------\n        eigvals, eigvecs = np.linalg.eigh(\n            covariance\n        )\n\n        # Eigenvalues are returned in ascending order.\n        # Select the two largest eigenvectors.\n        V = eigvecs[:, [2, 1]]\n\n        # -----------------------------------------------------\n        # Step 6: Consistent eigenvector orientation\n        # -----------------------------------------------------\n        if V[0, 0] < 0:\n            V[:, 0] *= -1\n\n        if V[0, 1] < 0:\n            V[:, 1] *= -1\n\n        # -----------------------------------------------------\n        # Step 7: Project OD data onto stain plane\n        # -----------------------------------------------------\n        projected = odhat @ V\n\n        # -----------------------------------------------------\n        # Step 8: Calculate angular coordinates\n        # -----------------------------------------------------\n        phi = np.arctan2(\n            projected[:, 1],\n            projected[:, 0],\n        )\n\n        # -----------------------------------------------------\n        # Step 9: Robust angular extremes\n        # -----------------------------------------------------\n        min_phi = np.percentile(\n            phi,\n            self.alpha,\n        )\n\n        max_phi = np.percentile(\n            phi,\n            100.0 - self.alpha,\n        )\n\n        # -----------------------------------------------------\n        # Step 10: Convert extremes back to OD space\n        # -----------------------------------------------------\n        v_min = V @ np.array([\n            np.cos(min_phi),\n            np.sin(min_phi),\n        ])\n\n        v_max = V @ np.array([\n            np.cos(max_phi),\n            np.sin(max_phi),\n        ])\n\n        # Normalize vectors\n        v_min = v_min / (\n            np.linalg.norm(v_min) + 1e-8\n        )\n\n        v_max = v_max / (\n            np.linalg.norm(v_max) + 1e-8\n        )\n\n        # -----------------------------------------------------\n        # Step 11: Consistent H&E ordering\n        # -----------------------------------------------------\n        if v_min[0] > v_max[0]:\n            stain_matrix = np.array([\n                v_min,\n                v_max,\n            ])\n        else:\n            stain_matrix = np.array([\n                v_max,\n                v_min,\n            ])\n\n        # Normalize rows\n        stain_matrix = (\n            stain_matrix\n            / (\n                np.linalg.norm(\n                    stain_matrix,\n                    axis=1,\n                    keepdims=True,\n                )\n                + 1e-8\n            )\n        )\n\n        return stain_matrix\n\n    # =========================================================\n    # Estimate stain concentrations\n    # =========================================================\n    def _get_concentrations(\n        self,\n        image: np.ndarray,\n        stain_matrix: np.ndarray,\n    ):\n\n        od = self.rgb_to_od(image)\n\n        height, width, _ = od.shape\n\n        pixels = od.reshape(-1, 3)\n\n        # Foreground mask\n        mask = np.min(\n            pixels,\n            axis=1\n        ) >= self.beta\n\n        odhat = pixels[mask]\n\n        if len(odhat) == 0:\n            raise ValueError(\n                \"No foreground pixels found.\"\n            )\n\n        # Solve:\n        #\n        # OD = C @ stain_matrix\n        #\n        concentrations = np.linalg.lstsq(\n            stain_matrix.T,\n            odhat.T,\n            rcond=None,\n        )[0].T\n\n        return (\n            concentrations,\n            mask,\n            (height, width),\n        )\n\n    # =========================================================\n    # Fit reference\n    # =========================================================\n    def fit(\n        self,\n        reference_images,\n    ):\n\n        if not reference_images:\n            raise ValueError(\n                \"At least one reference image is required.\"\n            )\n\n        stain_matrices = []\n        concentration_samples = []\n\n        for image in reference_images:\n\n            stain_matrix = (\n                self._estimate_stain_matrix(image)\n            )\n\n            concentrations, _, _ = (\n                self._get_concentrations(\n                    image,\n                    stain_matrix,\n                )\n            )\n\n            stain_matrices.append(\n                stain_matrix\n            )\n\n            concentration_samples.append(\n                concentrations\n            )\n\n        # First reference tile determines\n        # the target stain appearance.\n        target_stain_matrix = (\n            stain_matrices[0]\n        )\n\n        all_concentrations = np.concatenate(\n            concentration_samples,\n            axis=0,\n        )\n\n        target_max_concentration = (\n            np.percentile(\n                all_concentrations,\n                99,\n                axis=0,\n            )\n        )\n\n        self.reference = StainReference(\n            stain_matrix=target_stain_matrix,\n            max_concentration=target_max_concentration,\n            io=self.io,\n        )\n\n        return self\n\n    # =========================================================\n    # Transform / normalize image\n    # =========================================================\n    def transform(\n        self,\n        image: np.ndarray,\n    ) -> np.ndarray:\n\n        if self.reference is None:\n            raise RuntimeError(\n                \"Normalizer has not been fitted. \"\n                \"Call fit() first.\"\n            )\n\n        source_stain_matrix = (\n            self._estimate_stain_matrix(image)\n        )\n\n        concentrations, mask, shape = (\n            self._get_concentrations(\n                image,\n                source_stain_matrix,\n            )\n        )\n\n        # -----------------------------------------------------\n        # Source stain quantity\n        # -----------------------------------------------------\n        source_max = np.percentile(\n            concentrations,\n            99,\n            axis=0,\n        )\n\n        source_max = np.maximum(\n            source_max,\n            1e-8,\n        )\n\n        # -----------------------------------------------------\n        # Match source concentration to reference\n        # -----------------------------------------------------\n        normalized_concentrations = (\n            concentrations\n            / source_max\n        )\n\n        normalized_concentrations *= (\n            self.reference.max_concentration\n        )\n\n        height, width = shape\n\n        # -----------------------------------------------------\n        # Reconstruct normalized OD\n        # -----------------------------------------------------\n        normalized_od = np.zeros(\n            (height * width, 3),\n            dtype=np.float32,\n        )\n\n        normalized_od[mask] = (\n            normalized_concentrations\n            @ self.reference.stain_matrix\n        )\n\n        normalized_od = normalized_od.reshape(\n            height,\n            width,\n            3,\n        )\n\n        # -----------------------------------------------------\n        # OD -> RGB\n        # -----------------------------------------------------\n        normalized_image = (\n            self.od_to_rgb(\n                normalized_od\n            )\n        )\n\n        return normalized_image\n\n    # =========================================================\n    # Fit + Transform\n    # =========================================================\n    def fit_transform(\n        self,\n        reference_images,\n        image: np.ndarray,\n    ) -> np.ndarray:\n\n        self.fit(reference_images)\n\n        return self.transform(image)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.769274Z","iopub.status.idle":"2026-09-26T01:55:11.769604Z","shell.execute_reply.started":"2026-09-26T01:55:11.769475Z","shell.execute_reply":"2026-09-26T01:55:11.769498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nimport importlib\n\nproject_root = \"/kaggle/working/ovarian-trustworthy-ai\"\n\nif project_root not in sys.path:\n    sys.path.insert(0, project_root)\n\nimport src.preprocessing.stain_normalization as stain_module\n\nimportlib.reload(stain_module)\n\nMacenkoNormalizer = stain_module.MacenkoNormalizer\n\nprint(\"Corrected Macenko implementation loaded! ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.771408Z","iopub.status.idle":"2026-09-26T01:55:11.771758Z","shell.execute_reply.started":"2026-09-26T01:55:11.77157Z","shell.execute_reply":"2026-09-26T01:55:11.77159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"normalizer = MacenkoNormalizer()\n\nnormalizer.fit([reference_tile])\n\nprint(\"Macenko fitted successfully! ✅\")\n\nprint(\"\\nCorrected stain matrix:\")\nprint(normalizer.reference.stain_matrix)\n\nprint(\"\\nMaximum concentrations:\")\nprint(normalizer.reference.max_concentration)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.772649Z","iopub.status.idle":"2026-09-26T01:55:11.772975Z","shell.execute_reply.started":"2026-09-26T01:55:11.772795Z","shell.execute_reply":"2026-09-26T01:55:11.77283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nproject_root = Path(\"/kaggle/working/ovarian-trustworthy-ai\")\n\nprint(\"Project exists:\", project_root.exists())\n\npreprocessing_dir = project_root / \"src\" / \"preprocessing\"\n\nprint(\"\\nPreprocessing folder:\")\nfor item in preprocessing_dir.iterdir():\n    print(\" -\", item.name)\n\nprint(\"\\nStain normalization file exists:\",\n      (preprocessing_dir / \"stain_normalization.py\").exists())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.774542Z","iopub.status.idle":"2026-09-26T01:55:11.774871Z","shell.execute_reply.started":"2026-09-26T01:55:11.774703Z","shell.execute_reply":"2026-09-26T01:55:11.774724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nfile_path = Path(\n    \"/kaggle/working/ovarian-trustworthy-ai/\"\n    \"src/preprocessing/stain_normalization.py\"\n)\n\nfile_path.parent.mkdir(parents=True, exist_ok=True)\n\nfile_path.touch()\n\nprint(\"File created:\", file_path)\nprint(\"Exists:\", file_path.exists())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.776008Z","iopub.status.idle":"2026-09-26T01:55:11.776357Z","shell.execute_reply.started":"2026-09-26T01:55:11.776185Z","shell.execute_reply":"2026-09-26T01:55:11.776207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nproject_root = \"/kaggle/working/ovarian-trustworthy-ai\"\n\nos.makedirs(\n    f\"{project_root}/src/preprocessing\",\n    exist_ok=True\n)\n\nprint(\"Folder created/check passed ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.78392Z","iopub.status.idle":"2026-09-26T01:55:11.784261Z","shell.execute_reply.started":"2026-09-26T01:55:11.784099Z","shell.execute_reply":"2026-09-26T01:55:11.78414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile /kaggle/working/ovarian-trustworthy-ai/src/preprocessing/stain_normalization.py\n\nfrom dataclasses import dataclass\n\nimport numpy as np\n\n\n@dataclass\nclass StainReference:\n    stain_matrix: np.ndarray\n    max_concentration: np.ndarray\n    io: float = 255.0\n\n\nclass MacenkoNormalizer:\n\n    def __init__(\n        self,\n        beta: float = 0.15,\n        alpha: float = 1.0,\n        max_samples: int = 10000,\n        io: float = 255.0,\n        black_threshold: float = 30.0,\n    ):\n        self.beta = beta\n        self.alpha = alpha\n        self.max_samples = max_samples\n        self.io = io\n        self.black_threshold = black_threshold\n\n        self.reference = None\n\n    # ---------------------------------------------------------\n    # RGB -> Optical Density\n    # ---------------------------------------------------------\n    def rgb_to_od(self, image: np.ndarray) -> np.ndarray:\n\n        image = image.astype(np.float32)\n        image = np.clip(image, 0, 255)\n\n        od = -np.log10(\n            (image + 1.0) / self.io\n        )\n\n        return od\n\n    # ---------------------------------------------------------\n    # Optical Density -> RGB\n    # ---------------------------------------------------------\n    def od_to_rgb(self, od: np.ndarray) -> np.ndarray:\n\n        rgb = (\n            self.io * np.power(10.0, -od)\n            - 1.0\n        )\n\n        rgb = np.clip(rgb, 0, 255)\n\n        return rgb.astype(np.uint8)\n\n    # ---------------------------------------------------------\n    # Randomly sample pixels for efficiency\n    # ---------------------------------------------------------\n    def _sample_pixels(\n        self,\n        pixels: np.ndarray,\n    ) -> np.ndarray:\n\n        if len(pixels) <= self.max_samples:\n            return pixels\n\n        rng = np.random.default_rng(42)\n\n        indices = rng.choice(\n            len(pixels),\n            size=self.max_samples,\n            replace=False,\n        )\n\n        return pixels[indices]\n\n    # ---------------------------------------------------------\n    # Create stain/foreground mask\n    # ---------------------------------------------------------\n    def _get_stain_mask(\n        self,\n        image: np.ndarray,\n        od: np.ndarray,\n    ) -> np.ndarray:\n\n        rgb_pixels = image.reshape(-1, 3).astype(\n            np.float32\n        )\n\n        od_pixels = od.reshape(-1, 3)\n\n        # Remove near-black scanner / border pixels\n        not_black = (\n            np.max(rgb_pixels, axis=1)\n            >= self.black_threshold\n        )\n\n        # Standard Macenko-style OD foreground filtering\n        sufficient_od = (\n            np.min(od_pixels, axis=1)\n            >= self.beta\n        )\n\n        mask = not_black & sufficient_od\n\n        return mask\n\n    # ---------------------------------------------------------\n    # Estimate H&E stain matrix\n    # ---------------------------------------------------------\n    def _estimate_stain_matrix(\n        self,\n        image: np.ndarray,\n    ):\n\n        if image.ndim != 3 or image.shape[2] != 3:\n            raise ValueError(\n                \"Image must have shape H x W x 3.\"\n            )\n\n        od = self.rgb_to_od(image)\n\n        pixels = od.reshape(-1, 3)\n\n        # Create foreground/stain mask\n        mask = self._get_stain_mask(\n            image,\n            od,\n        )\n\n        od_pixels = pixels[mask]\n\n        if len(od_pixels) < 100:\n            raise ValueError(\n                \"Not enough stained pixels \"\n                \"to estimate stain matrix.\"\n            )\n\n        # Limit computation for large tiles\n        od_pixels = self._sample_pixels(\n            od_pixels\n        )\n\n        # -----------------------------------------------------\n        # Singular Value Decomposition\n        # -----------------------------------------------------\n        _, _, vh = np.linalg.svd(\n            od_pixels,\n            full_matrices=False,\n        )\n\n        # First two principal directions\n        v = vh[:2].T\n\n        # Project OD values onto SVD plane\n        projected = od_pixels @ v\n\n        # Calculate angular coordinates\n        phi = np.arctan2(\n            projected[:, 1],\n            projected[:, 0],\n        )\n\n        # Robust percentile limits\n        min_phi = np.percentile(\n            phi,\n            self.alpha,\n        )\n\n        max_phi = np.percentile(\n            phi,\n            100.0 - self.alpha,\n        )\n\n        vector_1 = (\n            v[:, 0] * np.cos(min_phi)\n            + v[:, 1] * np.sin(min_phi)\n        )\n\n        vector_2 = (\n            v[:, 0] * np.cos(max_phi)\n            + v[:, 1] * np.sin(max_phi)\n        )\n\n        # Normalize stain vectors\n        vector_1 = vector_1 / (\n            np.linalg.norm(vector_1)\n            + 1e-8\n        )\n\n        vector_2 = vector_2 / (\n            np.linalg.norm(vector_2)\n            + 1e-8\n        )\n\n        # Consistent H&E ordering\n        if vector_1[0] > vector_2[0]:\n            stain_matrix = np.stack(\n                [vector_1, vector_2],\n                axis=0,\n            )\n        else:\n            stain_matrix = np.stack(\n                [vector_2, vector_1],\n                axis=0,\n            )\n\n        return stain_matrix\n\n    # ---------------------------------------------------------\n    # Estimate stain concentrations\n    # ---------------------------------------------------------\n    def _get_concentrations(\n        self,\n        image: np.ndarray,\n        stain_matrix: np.ndarray,\n    ):\n\n        od = self.rgb_to_od(image)\n\n        height, width, _ = od.shape\n\n        pixels = od.reshape(-1, 3)\n\n        # Same foreground mask used during\n        # stain matrix estimation\n        mask = self._get_stain_mask(\n            image,\n            od,\n        )\n\n        od_stained = pixels[mask]\n\n        if len(od_stained) == 0:\n            raise ValueError(\n                \"No stained pixels found in image.\"\n            )\n\n        # Solve:\n        # OD = stain_matrix.T @ concentrations\n        concentrations = np.linalg.lstsq(\n            stain_matrix.T,\n            od_stained.T,\n            rcond=None,\n        )[0]\n\n        # Prevent small numerical negative values\n        concentrations = np.maximum(\n            concentrations,\n            0.0,\n        )\n\n        return (\n            concentrations,\n            mask,\n            (height, width),\n        )\n\n    # ---------------------------------------------------------\n    # Fit reference stain appearance\n    # ---------------------------------------------------------\n    def fit(\n        self,\n        reference_images,\n    ):\n\n        if not reference_images:\n            raise ValueError(\n                \"At least one reference image \"\n                \"is required.\"\n            )\n\n        stain_matrices = []\n        concentration_samples = []\n\n        for image in reference_images:\n\n            if image.ndim != 3 or image.shape[2] != 3:\n                raise ValueError(\n                    \"Reference image must have \"\n                    \"shape H x W x 3.\"\n                )\n\n            stain_matrix = (\n                self._estimate_stain_matrix(\n                    image\n                )\n            )\n\n            concentrations, _, _ = (\n                self._get_concentrations(\n                    image,\n                    stain_matrix,\n                )\n            )\n\n            stain_matrices.append(\n                stain_matrix\n            )\n\n            concentration_samples.append(\n                concentrations\n            )\n\n        # First reference image defines\n        # target stain color appearance\n        target_stain_matrix = (\n            stain_matrices[0]\n        )\n\n        # Combine concentration samples\n        all_concentrations = np.concatenate(\n            concentration_samples,\n            axis=1,\n        )\n\n        # 99th percentile defines target\n        # stain intensity\n        target_max_concentration = (\n            np.percentile(\n                all_concentrations,\n                99,\n                axis=1,\n            )\n        )\n\n        self.reference = StainReference(\n            stain_matrix=target_stain_matrix,\n            max_concentration=(\n                target_max_concentration\n            ),\n            io=self.io,\n        )\n\n        return self\n\n    # ---------------------------------------------------------\n    # Transform one image\n    # ---------------------------------------------------------\n    def transform(\n        self,\n        image: np.ndarray,\n    ) -> np.ndarray:\n\n        if self.reference is None:\n            raise RuntimeError(\n                \"Normalizer has not been fitted. \"\n                \"Call fit() first.\"\n            )\n\n        if image.ndim != 3 or image.shape[2] != 3:\n            raise ValueError(\n                \"Image must have shape H x W x 3.\"\n            )\n\n        # Estimate source stain matrix\n        source_stain_matrix = (\n            self._estimate_stain_matrix(\n                image\n            )\n        )\n\n        # Get source concentrations\n        concentrations, mask, shape = (\n            self._get_concentrations(\n                image,\n                source_stain_matrix,\n            )\n        )\n\n        # Source 99th percentile\n        source_max_concentration = (\n            np.percentile(\n                concentrations,\n                99,\n                axis=1,\n            )\n        )\n\n        source_max_concentration = (\n            np.maximum(\n                source_max_concentration,\n                1e-8,\n            )\n        )\n\n        # Normalize stain intensity\n        normalized_concentrations = (\n            concentrations\n            / source_max_concentration[:, None]\n        )\n\n        # Match target stain intensity\n        normalized_concentrations *= (\n            self.reference.max_concentration[\n                :, None\n            ]\n        )\n\n        height, width = shape\n\n        # -----------------------------------------------------\n        # IMPORTANT:\n        # Start with ORIGINAL image.\n        #\n        # This preserves:\n        # - black borders\n        # - white background\n        # - non-stain pixels\n        # -----------------------------------------------------\n        normalized_image = image.copy()\n\n        # Reconstruct only the pixels that\n        # were identified as stained tissue\n        normalized_od_stained = (\n            self.reference.stain_matrix.T\n            @ normalized_concentrations\n        ).T\n\n        normalized_rgb_stained = (\n            self.od_to_rgb(\n                normalized_od_stained\n            )\n        )\n\n        # Put normalized pixels back into\n        # their original locations\n        normalized_image[\n            mask.reshape(height, width)\n        ] = normalized_rgb_stained\n\n        return normalized_image\n\n    # ---------------------------------------------------------\n    # Fit + Transform\n    # ---------------------------------------------------------\n    def fit_transform(\n        self,\n        reference_images,\n        image: np.ndarray,\n    ):\n\n        self.fit(reference_images)\n\n        return self.transform(image)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.785824Z","iopub.status.idle":"2026-09-26T01:55:11.786185Z","shell.execute_reply.started":"2026-09-26T01:55:11.785977Z","shell.execute_reply":"2026-09-26T01:55:11.785991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\npath = \"/kaggle/working/ovarian-trustworthy-ai/src/preprocessing/stain_normalization.py\"\n\nprint(\"Exists:\", os.path.exists(path))\nprint(\"Size:\", os.path.getsize(path), \"bytes\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.787294Z","iopub.status.idle":"2026-09-26T01:55:11.787665Z","shell.execute_reply.started":"2026-09-26T01:55:11.787505Z","shell.execute_reply":"2026-09-26T01:55:11.787528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nfile_path = Path(\n    \"/kaggle/working/ovarian-trustworthy-ai/\"\n    \"src/preprocessing/stain_normalization.py\"\n)\n\nprint(\"Exists:\", file_path.exists())\nprint(\"Size:\", file_path.stat().st_size, \"bytes\")\n\nprint(\"\\nFirst 5 lines:\")\nprint(\"\\n\".join(file_path.read_text().splitlines()[:5]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.789202Z","iopub.status.idle":"2026-09-26T01:55:11.789588Z","shell.execute_reply.started":"2026-09-26T01:55:11.789343Z","shell.execute_reply":"2026-09-26T01:55:11.789383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nimport importlib\n\nproject_root = \"/kaggle/working/ovarian-trustworthy-ai\"\n\nif project_root not in sys.path:\n    sys.path.insert(0, project_root)\n\nimportlib.invalidate_caches()\n\nfrom src.preprocessing.stain_normalization import MacenkoNormalizer\n\nprint(\"Macenko normalizer imported successfully! ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.79084Z","iopub.status.idle":"2026-09-26T01:55:11.791137Z","shell.execute_reply.started":"2026-09-26T01:55:11.790992Z","shell.execute_reply":"2026-09-26T01:55:11.791006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# M1.5.2 — Get representative candidate tiles\n\nimport sys\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nproject_root = \"/kaggle/working/ovarian-trustworthy-ai\"\n\nif project_root not in sys.path:\n    sys.path.insert(0, project_root)\n\nfrom src.preprocessing.metadata import UBCOceanMetadata\nfrom src.preprocessing.image_reader import UBCOceanImageReader\nfrom src.preprocessing.slide_type import SlideTypeHandler\nfrom src.preprocessing.tile_extractor import TileExtractor\nfrom src.preprocessing.thumbnail_selector import ThumbnailCandidateSelector\n\n\ndataset_root = \"/kaggle/input/competitions/UBC-OCEAN\"\n\nmetadata = UBCOceanMetadata(dataset_root)\nimage_reader = UBCOceanImageReader(dataset_root)\nslide_handler = SlideTypeHandler(metadata)\n\ntile_extractor = TileExtractor(\n    image_reader,\n    slide_handler\n)\n\nselector = ThumbnailCandidateSelector(metadata)\n\n\n# Generate WSI candidate grid\ncoordinates = tile_extractor.generate_grid(4)\n\n# Apply thumbnail tissue filtering\nselected_coordinates = selector.select_candidates(\n    coordinates,\n    tissue_ratio_threshold=0.20\n)\n\nprint(\"Total candidates:\", len(coordinates))\nprint(\"Selected candidates:\", len(selected_coordinates))\n\n\n# Select 6 spread-out candidates\nindices = np.linspace(\n    0,\n    len(selected_coordinates) - 1,\n    6,\n    dtype=int\n)\n\nreference_tiles = []\n\nfor idx in indices:\n    coordinate = selected_coordinates[idx]\n\n    tile = tile_extractor.extract_tile(coordinate)\n\n    reference_tiles.append(tile)\n\n    print(\n        f\"Tile {len(reference_tiles)}: \"\n        f\"x={coordinate.x}, \"\n        f\"y={coordinate.y}, \"\n        f\"shape={tile.shape}\"\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.791991Z","iopub.status.idle":"2026-09-26T01:55:11.792335Z","shell.execute_reply.started":"2026-09-26T01:55:11.792147Z","shell.execute_reply":"2026-09-26T01:55:11.792171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# M1.5.2 — Visualize candidate reference tiles\n\nimport matplotlib.pyplot as plt\n\nfig, axes = plt.subplots(2, 3, figsize=(12, 8))\n\nfor i, (ax, tile) in enumerate(zip(axes.ravel(), reference_tiles)):\n    ax.imshow(tile)\n    ax.set_title(\n        f\"Reference Candidate {i + 1}\\n\"\n        f\"x={selected_coordinates[indices[i]].x}, \"\n        f\"y={selected_coordinates[indices[i]].y}\"\n    )\n    ax.axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.793727Z","iopub.status.idle":"2026-09-26T01:55:11.79396Z","shell.execute_reply.started":"2026-09-26T01:55:11.793849Z","shell.execute_reply":"2026-09-26T01:55:11.793862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# M1.5.3 — Fit Macenko using Candidate 2\n\nfrom src.preprocessing.stain_normalization import MacenkoNormalizer\n\n# Candidate 2 is index 1 because Python uses zero-based indexing\nreference_tile = reference_tiles[1]\n\nprint(\"Reference tile shape:\", reference_tile.shape)\nprint(\"Reference tile dtype:\", reference_tile.dtype)\n\nnormalizer = MacenkoNormalizer()\n\nnormalizer.fit([reference_tile])\n\nprint(\"Macenko fitted successfully! ✅\")\nprint(\"\\nReference stain matrix:\")\nprint(normalizer.reference.stain_matrix)\n\nprint(\"\\nReference maximum concentrations:\")\nprint(normalizer.reference.max_concentration)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.795641Z","iopub.status.idle":"2026-09-26T01:55:11.795917Z","shell.execute_reply.started":"2026-09-26T01:55:11.795806Z","shell.execute_reply":"2026-09-26T01:55:11.795819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# M1.5.4 — Normalize a visually different tissue tile\n\ntest_tile = reference_tiles[4]   # Candidate 5\n\nnormalized_tile = normalizer.transform(test_tile)\n\nprint(\"Original tile:\")\nprint(\"  Shape:\", test_tile.shape)\nprint(\"  Dtype:\", test_tile.dtype)\n\nprint(\"\\nNormalized tile:\")\nprint(\"  Shape:\", normalized_tile.shape)\nprint(\"  Dtype:\", normalized_tile.dtype)\n\nprint(\"\\nNormalization completed successfully! ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.796758Z","iopub.status.idle":"2026-09-26T01:55:11.797073Z","shell.execute_reply.started":"2026-09-26T01:55:11.796953Z","shell.execute_reply":"2026-09-26T01:55:11.796969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nfig, axes = plt.subplots(1, 2, figsize=(10, 5))\n\naxes[0].imshow(test_tile)\naxes[0].set_title(\"Original — Candidate 5\")\naxes[0].axis(\"off\")\n\naxes[1].imshow(normalized_tile)\naxes[1].set_title(\"Macenko Normalized\")\naxes[1].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.798283Z","iopub.status.idle":"2026-09-26T01:55:11.798627Z","shell.execute_reply.started":"2026-09-26T01:55:11.798456Z","shell.execute_reply":"2026-09-26T01:55:11.798472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# M1.5.5 — Normalize multiple candidate tiles\n\ncomparison_results = []\n\nfor i, tile in enumerate(reference_tiles):\n\n    if i == 1:\n        # Candidate 2 is our reference.\n        # Don't normalize it against itself.\n        continue\n\n    normalized = normalizer.transform(tile)\n\n    comparison_results.append(\n        {\n            \"candidate\": i + 1,\n            \"original\": tile,\n            \"normalized\": normalized,\n        }\n    )\n\nprint(\n    \"Successfully normalized\",\n    len(comparison_results),\n    \"test tiles. ✅\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.799712Z","iopub.status.idle":"2026-09-26T01:55:11.799983Z","shell.execute_reply.started":"2026-09-26T01:55:11.799826Z","shell.execute_reply":"2026-09-26T01:55:11.799839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visual comparison\n\nfig, axes = plt.subplots(\n    5,\n    2,\n    figsize=(10, 20)\n)\n\nfor row, result in enumerate(comparison_results):\n\n    axes[row, 0].imshow(result[\"original\"])\n    axes[row, 0].set_title(\n        f\"Candidate {result['candidate']} — Original\"\n    )\n    axes[row, 0].axis(\"off\")\n\n    axes[row, 1].imshow(result[\"normalized\"])\n    axes[row, 1].set_title(\n        f\"Candidate {result['candidate']} — Normalized\"\n    )\n    axes[row, 1].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.801858Z","iopub.status.idle":"2026-09-26T01:55:11.802208Z","shell.execute_reply.started":"2026-09-26T01:55:11.802025Z","shell.execute_reply":"2026-09-26T01:55:11.802044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nproject_root = \"/kaggle/working/ovarian-trustworthy-ai\"\n\nprint(\"Project exists:\", os.path.exists(project_root))\nprint(\"\\nProject contents:\")\nprint(os.listdir(project_root))\n\nprint(\"\\nSRC contents:\")\nprint(os.listdir(os.path.join(project_root, \"src\")))\n\nprint(\"\\nPREPROCESSING contents:\")\nprint(os.listdir(os.path.join(project_root, \"src\", \"preprocessing\")))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.803807Z","iopub.status.idle":"2026-09-26T01:55:11.804225Z","shell.execute_reply.started":"2026-09-26T01:55:11.803999Z","shell.execute_reply":"2026-09-26T01:55:11.804023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nproject_root = \"/kaggle/working/ovarian-trustworthy-ai\"\n\nfolders = [\n    \"src\",\n    \"src/preprocessing\",\n    \"src/quality\",\n    \"src/features\",\n    \"src/classification\",\n    \"src/calibration\",\n    \"src/explainability\",\n    \"src/routing\",\n    \"src/evaluation\",\n    \"config\",\n    \"data\",\n    \"cache/tiles\",\n    \"cache/embeddings\",\n    \"outputs/logs\",\n    \"outputs/metrics\",\n    \"outputs/visualizations\",\n    \"tests\",\n    \"notebooks\",\n]\n\nfor folder in folders:\n    os.makedirs(os.path.join(project_root, folder), exist_ok=True)\n\nprint(\"Project structure recreated ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.806222Z","iopub.status.idle":"2026-09-26T01:55:11.806583Z","shell.execute_reply.started":"2026-09-26T01:55:11.806399Z","shell.execute_reply":"2026-09-26T01:55:11.806422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nproject_root = Path(\"/kaggle/working/ovarian-trustworthy-ai\")\n\ninit_files = [\n    project_root / \"src\" / \"__init__.py\",\n    project_root / \"src\" / \"preprocessing\" / \"__init__.py\",\n    project_root / \"src\" / \"quality\" / \"__init__.py\",\n    project_root / \"src\" / \"features\" / \"__init__.py\",\n    project_root / \"src\" / \"classification\" / \"__init__.py\",\n    project_root / \"src\" / \"calibration\" / \"__init__.py\",\n    project_root / \"src\" / \"explainability\" / \"__init__.py\",\n    project_root / \"src\" / \"routing\" / \"__init__.py\",\n    project_root / \"src\" / \"evaluation\" / \"__init__.py\",\n    project_root / \"config\" / \"__init__.py\",\n]\n\nfor path in init_files:\n    path.touch(exist_ok=True)\n\nprint(\"Package files recreated ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.807471Z","iopub.status.idle":"2026-09-26T01:55:11.80783Z","shell.execute_reply.started":"2026-09-26T01:55:11.807647Z","shell.execute_reply":"2026-09-26T01:55:11.807669Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile /kaggle/working/ovarian-trustworthy-ai/src/preprocessing/metadata.py\n\nfrom pathlib import Path\nimport pandas as pd\n\n\nclass UBCOceanMetadata:\n\n    def __init__(self, dataset_root: str):\n\n        self.dataset_root = Path(dataset_root)\n\n        self.train_csv = self.dataset_root / \"train.csv\"\n        self.train_images_dir = self.dataset_root / \"train_images\"\n        self.train_thumbnails_dir = self.dataset_root / \"train_thumbnails\"\n\n        if not self.dataset_root.exists():\n            raise FileNotFoundError(\n                f\"Dataset root not found: {self.dataset_root}\"\n            )\n\n        if not self.train_csv.exists():\n            raise FileNotFoundError(\n                f\"train.csv not found: {self.train_csv}\"\n            )\n\n        self.df = pd.read_csv(self.train_csv)\n\n        self._validate_columns()\n\n\n    def _validate_columns(self):\n\n        required_columns = {\n            \"image_id\",\n            \"label\",\n            \"image_width\",\n            \"image_height\",\n            \"is_tma\"\n        }\n\n        missing = required_columns - set(self.df.columns)\n\n        if missing:\n            raise ValueError(\n                f\"Missing required columns: {sorted(missing)}\"\n            )\n\n\n    def get_record(self, image_id):\n\n        rows = self.df[self.df[\"image_id\"] == image_id]\n\n        if rows.empty:\n            raise ValueError(\n                f\"Image ID {image_id} not found in train.csv\"\n            )\n\n        row = rows.iloc[0]\n\n        return {\n            \"image_id\": int(row[\"image_id\"]),\n            \"label\": str(row[\"label\"]),\n            \"width\": int(row[\"image_width\"]),\n            \"height\": int(row[\"image_height\"]),\n            \"is_tma\": bool(row[\"is_tma\"])\n        }\n\n\n    def get_image_path(self, image_id):\n\n        return self.train_images_dir / f\"{image_id}.png\"\n\n\n    def get_thumbnail_path(self, image_id):\n\n        record = self.get_record(image_id)\n\n        # TMA slides do not have official thumbnails\n        if record[\"is_tma\"]:\n            return None\n\n        return self.train_thumbnails_dir / f\"{image_id}_thumbnail.png\"\n\n\n    def get_all_records(self):\n\n        records = []\n\n        for _, row in self.df.iterrows():\n\n            records.append({\n                \"image_id\": int(row[\"image_id\"]),\n                \"label\": str(row[\"label\"]),\n                \"width\": int(row[\"image_width\"]),\n                \"height\": int(row[\"image_height\"]),\n                \"is_tma\": bool(row[\"is_tma\"])\n            })\n\n        return records","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.809544Z","iopub.status.idle":"2026-09-26T01:55:11.809902Z","shell.execute_reply.started":"2026-09-26T01:55:11.809722Z","shell.execute_reply":"2026-09-26T01:55:11.809744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nimport importlib\n\nproject_root = \"/kaggle/working/ovarian-trustworthy-ai\"\n\nif project_root not in sys.path:\n    sys.path.insert(0, project_root)\n\nimportlib.invalidate_caches()\n\nfrom src.preprocessing.metadata import UBCOceanMetadata\n\ndataset_root = \"/kaggle/input/competitions/UBC-OCEAN\"\n\nmetadata = UBCOceanMetadata(dataset_root)\n\nprint(\"Number of records:\", len(metadata.df))\n\nrecord = metadata.get_record(4)\n\nprint(\"\\nImage 4:\")\nprint(record)\n\nprint(\"\\nImage path:\")\nprint(metadata.get_image_path(4))\n\nprint(\"\\nThumbnail path:\")\nprint(metadata.get_thumbnail_path(4))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.811005Z","iopub.status.idle":"2026-09-26T01:55:11.811394Z","shell.execute_reply.started":"2026-09-26T01:55:11.8112Z","shell.execute_reply":"2026-09-26T01:55:11.811226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile /kaggle/working/ovarian-trustworthy-ai/src/preprocessing/image_reader.py\n\nfrom pathlib import Path\nimport numpy as np\nimport pyvips\n\nfrom .metadata import UBCOceanMetadata\n\n\nclass UBCOceanImageReader:\n\n    def __init__(self, dataset_root: str):\n\n        self.metadata = UBCOceanMetadata(dataset_root)\n\n\n    def open_image(self, image_id):\n\n        image_path = self.metadata.get_image_path(image_id)\n\n        if not image_path.exists():\n            raise FileNotFoundError(\n                f\"Image not found: {image_path}\"\n            )\n\n        return pyvips.Image.new_from_file(\n            str(image_path),\n            access=\"random\"\n        )\n\n\n    def get_region(\n        self,\n        image_id,\n        x,\n        y,\n        width,\n        height\n    ):\n\n        image = self.open_image(image_id)\n\n        image_width = image.width\n        image_height = image.height\n\n        if x < 0 or y < 0:\n            raise ValueError(\"x and y must be >= 0\")\n\n        if width <= 0 or height <= 0:\n            raise ValueError(\"width and height must be > 0\")\n\n        if x + width > image_width:\n            raise ValueError(\n                f\"Requested region exceeds image width. \"\n                f\"x={x}, width={width}, image_width={image_width}\"\n            )\n\n        if y + height > image_height:\n            raise ValueError(\n                f\"Requested region exceeds image height. \"\n                f\"y={y}, height={height}, image_height={image_height}\"\n            )\n\n        region = image.crop(\n            x,\n            y,\n            width,\n            height\n        )\n\n        array = np.ndarray(\n            buffer=region.write_to_memory(),\n            dtype=np.uint8,\n            shape=(region.height, region.width, region.bands)\n        )\n\n        # Keep RGB channels only\n        if array.shape[2] > 3:\n            array = array[:, :, :3]\n\n        return array\n\n\n    def get_image_dimensions(self, image_id):\n\n        image = self.open_image(image_id)\n\n        return image.width, image.height","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.812574Z","iopub.status.idle":"2026-09-26T01:55:11.812913Z","shell.execute_reply.started":"2026-09-26T01:55:11.812776Z","shell.execute_reply":"2026-09-26T01:55:11.812802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q \"pyvips[binary]\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.814795Z","iopub.status.idle":"2026-09-26T01:55:11.815149Z","shell.execute_reply.started":"2026-09-26T01:55:11.814976Z","shell.execute_reply":"2026-09-26T01:55:11.815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pyvips\n\nprint(\"pyvips version:\", pyvips.__version__)\nprint(\"libvips version:\", pyvips.version(0))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.8162Z","iopub.status.idle":"2026-09-26T01:55:11.816487Z","shell.execute_reply.started":"2026-09-26T01:55:11.81635Z","shell.execute_reply":"2026-09-26T01:55:11.81637Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nimport importlib\n\nproject_root = \"/kaggle/working/ovarian-trustworthy-ai\"\n\nif project_root not in sys.path:\n    sys.path.insert(0, project_root)\n\nimportlib.invalidate_caches()\n\nfrom src.preprocessing.image_reader import UBCOceanImageReader\n\ndataset_root = \"/kaggle/input/competitions/UBC-OCEAN\"\n\nreader = UBCOceanImageReader(dataset_root)\n\nwidth, height = reader.get_image_dimensions(4)\n\nprint(\"Image 4 dimensions:\")\nprint(\"Width :\", width)\nprint(\"Height:\", height)\n\ntile = reader.get_region(\n    image_id=4,\n    x=8960,\n    y=4608,\n    width=256,\n    height=256\n)\n\nprint(\"\\nRegion:\")\nprint(\"Shape :\", tile.shape)\nprint(\"Dtype :\", tile.dtype)\nprint(\"Range :\", tile.min(), \"to\", tile.max())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.817741Z","iopub.status.idle":"2026-09-26T01:55:11.818099Z","shell.execute_reply.started":"2026-09-26T01:55:11.817956Z","shell.execute_reply":"2026-09-26T01:55:11.817978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile /kaggle/working/ovarian-trustworthy-ai/src/preprocessing/slide_type.py\n\nfrom dataclasses import dataclass\n\n\n@dataclass\nclass SlideInfo:\n    image_id: int\n    slide_type: str\n    width: int\n    height: int\n    magnification: str\n\n\nclass SlideTypeHandler:\n\n    def __init__(self, metadata):\n        self.metadata = metadata\n\n\n    def get_slide_info(self, image_id):\n\n        record = self.metadata.get_record(image_id)\n\n        if record[\"is_tma\"]:\n            slide_type = \"TMA\"\n            magnification = \"40x\"\n        else:\n            slide_type = \"WSI\"\n            magnification = \"20x\"\n\n        return SlideInfo(\n            image_id=record[\"image_id\"],\n            slide_type=slide_type,\n            width=record[\"width\"],\n            height=record[\"height\"],\n            magnification=magnification\n        )\n\n\n    def is_wsi(self, image_id):\n\n        return not self.metadata.get_record(image_id)[\"is_tma\"]\n\n\n    def is_tma(self, image_id):\n\n        return self.metadata.get_record(image_id)[\"is_tma\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.81969Z","iopub.status.idle":"2026-09-26T01:55:11.820039Z","shell.execute_reply.started":"2026-09-26T01:55:11.819891Z","shell.execute_reply":"2026-09-26T01:55:11.819921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nimport importlib\n\nproject_root = \"/kaggle/working/ovarian-trustworthy-ai\"\n\nif project_root not in sys.path:\n    sys.path.insert(0, project_root)\n\nimportlib.invalidate_caches()\n\nfrom src.preprocessing.metadata import UBCOceanMetadata\nfrom src.preprocessing.slide_type import SlideTypeHandler\n\ndataset_root = \"/kaggle/input/competitions/UBC-OCEAN\"\n\nmetadata = UBCOceanMetadata(dataset_root)\nslide_handler = SlideTypeHandler(metadata)\n\n# WSI test\nwsi_info = slide_handler.get_slide_info(4)\n\nprint(\"Image 4:\")\nprint(wsi_info)\n\n# TMA test\ntma_info = slide_handler.get_slide_info(91)\n\nprint(\"\\nImage 91:\")\nprint(tma_info)\n\nprint(\"\\nType checks:\")\nprint(\"Image 4 is WSI:\", slide_handler.is_wsi(4))\nprint(\"Image 91 is TMA:\", slide_handler.is_tma(91))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.821457Z","iopub.status.idle":"2026-09-26T01:55:11.821821Z","shell.execute_reply.started":"2026-09-26T01:55:11.821624Z","shell.execute_reply":"2026-09-26T01:55:11.821645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile /kaggle/working/ovarian-trustworthy-ai/src/preprocessing/tile_extractor.py\n\nfrom dataclasses import dataclass\nfrom typing import List\n\nimport cv2\nimport numpy as np\n\nfrom .image_reader import UBCOceanImageReader\nfrom .slide_type import SlideTypeHandler\n\n\n@dataclass\nclass TileCoordinate:\n    image_id: int\n    x: int\n    y: int\n    source_size: int\n    standardized_size: int\n    slide_type: str\n\n\nclass TileExtractor:\n\n    def __init__(self, reader, slide_handler):\n\n        self.reader = reader\n        self.slide_handler = slide_handler\n\n        self.standardized_size = 256\n\n\n    def get_tile_size(self, image_id):\n\n        if self.slide_handler.is_tma(image_id):\n            return 512\n\n        return 256\n\n\n    def generate_grid(self, image_id, stride=None):\n\n        slide_info = self.slide_handler.get_slide_info(image_id)\n\n        tile_size = self.get_tile_size(image_id)\n\n        if stride is None:\n            stride = tile_size\n\n        coordinates = []\n\n        for y in range(\n            0,\n            slide_info.height - tile_size + 1,\n            stride\n        ):\n\n            for x in range(\n                0,\n                slide_info.width - tile_size + 1,\n                stride\n            ):\n\n                coordinates.append(\n                    TileCoordinate(\n                        image_id=image_id,\n                        x=x,\n                        y=y,\n                        source_size=tile_size,\n                        standardized_size=self.standardized_size,\n                        slide_type=slide_info.slide_type\n                    )\n                )\n\n        return coordinates\n\n\n    def extract_tile(self, coordinate):\n\n        tile = self.reader.get_region(\n            image_id=coordinate.image_id,\n            x=coordinate.x,\n            y=coordinate.y,\n            width=coordinate.source_size,\n            height=coordinate.source_size\n        )\n\n        if coordinate.source_size != coordinate.standardized_size:\n\n            tile = cv2.resize(\n                tile,\n                (\n                    coordinate.standardized_size,\n                    coordinate.standardized_size\n                ),\n                interpolation=cv2.INTER_AREA\n            )\n\n        return tile\n\n\n    def extract_tiles(self, image_id, max_tiles=None):\n\n        coordinates = self.generate_grid(image_id)\n\n        if max_tiles is not None:\n            coordinates = coordinates[:max_tiles]\n\n        tiles = []\n\n        for coordinate in coordinates:\n\n            tile = self.extract_tile(coordinate)\n\n            tiles.append(\n                (coordinate, tile)\n            )\n\n        return tiles","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.822897Z","iopub.status.idle":"2026-09-26T01:55:11.823241Z","shell.execute_reply.started":"2026-09-26T01:55:11.823044Z","shell.execute_reply":"2026-09-26T01:55:11.823058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nimport importlib\n\nproject_root = \"/kaggle/working/ovarian-trustworthy-ai\"\n\nif project_root not in sys.path:\n    sys.path.insert(0, project_root)\n\nimportlib.invalidate_caches()\n\nfrom src.preprocessing.metadata import UBCOceanMetadata\nfrom src.preprocessing.image_reader import UBCOceanImageReader\nfrom src.preprocessing.slide_type import SlideTypeHandler\nfrom src.preprocessing.tile_extractor import TileExtractor\n\n\ndataset_root = \"/kaggle/input/competitions/UBC-OCEAN\"\n\nmetadata = UBCOceanMetadata(dataset_root)\n\nreader = UBCOceanImageReader(dataset_root)\n\nslide_handler = SlideTypeHandler(metadata)\n\ntile_extractor = TileExtractor(\n    reader,\n    slide_handler\n)\n\n\n# Generate WSI tile coordinates\ncoordinates = tile_extractor.generate_grid(4)\n\nprint(\"Number of candidate tiles:\", len(coordinates))\n\nprint(\"\\nFirst 5 coordinates:\")\n\nfor coordinate in coordinates[:5]:\n    print(coordinate)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.824417Z","iopub.status.idle":"2026-09-26T01:55:11.824666Z","shell.execute_reply.started":"2026-09-26T01:55:11.82455Z","shell.execute_reply":"2026-09-26T01:55:11.824563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_tiles = tile_extractor.extract_tiles(\n    image_id=4,\n    max_tiles=3\n)\n\nprint(\"Extracted tiles:\", len(test_tiles))\n\nfor i, (coordinate, tile) in enumerate(test_tiles):\n\n    print(\n        f\"Tile {i+1}:\",\n        \"coordinate=\", (coordinate.x, coordinate.y),\n        \"shape=\", tile.shape,\n        \"dtype=\", tile.dtype\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.82572Z","iopub.status.idle":"2026-09-26T01:55:11.826078Z","shell.execute_reply.started":"2026-09-26T01:55:11.825942Z","shell.execute_reply":"2026-09-26T01:55:11.825959Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tma_coordinates = tile_extractor.generate_grid(91)\n\nprint(\"Image 91 candidate tiles:\", len(tma_coordinates))\n\nprint(\"\\nFirst coordinate:\")\nprint(tma_coordinates[0])\n\ntma_tile = tile_extractor.extract_tile(\n    tma_coordinates[0]\n)\n\nprint(\"\\nExtracted TMA tile:\")\nprint(\"Shape:\", tma_tile.shape)\nprint(\"Dtype:\", tma_tile.dtype)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.827309Z","iopub.status.idle":"2026-09-26T01:55:11.827542Z","shell.execute_reply.started":"2026-09-26T01:55:11.827433Z","shell.execute_reply":"2026-09-26T01:55:11.827447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile /kaggle/working/ovarian-trustworthy-ai/src/preprocessing/thumbnail_selector.py\n\nfrom pathlib import Path\n\nimport cv2\nimport numpy as np\nimport pyvips\n\nfrom .metadata import UBCOceanMetadata\nfrom .tile_extractor import TileCoordinate\n\n\nclass ThumbnailSelector:\n\n    def __init__(\n        self,\n        metadata,\n        tissue_threshold=0.20\n    ):\n\n        self.metadata = metadata\n        self.tissue_threshold = tissue_threshold\n\n\n    def _load_thumbnail(self, image_id):\n\n        thumbnail_path = self.metadata.get_thumbnail_path(image_id)\n\n        if thumbnail_path is None:\n            return None\n\n        if not thumbnail_path.exists():\n            raise FileNotFoundError(\n                f\"Thumbnail not found: {thumbnail_path}\"\n            )\n\n        image = pyvips.Image.new_from_file(\n            str(thumbnail_path),\n            access=\"random\"\n        )\n\n        array = np.ndarray(\n            buffer=image.write_to_memory(),\n            dtype=np.uint8,\n            shape=(image.height, image.width, image.bands)\n        )\n\n        if array.shape[2] > 3:\n            array = array[:, :, :3]\n\n        return array\n\n\n    def _create_tissue_mask(self, thumbnail):\n\n        hsv = cv2.cvtColor(\n            thumbnail,\n            cv2.COLOR_RGB2HSV\n        )\n\n        saturation = hsv[:, :, 1]\n\n        # Otsu automatically chooses a saturation threshold\n        _, mask = cv2.threshold(\n            saturation,\n            0,\n            255,\n            cv2.THRESH_BINARY + cv2.THRESH_OTSU\n        )\n\n        # Remove small isolated regions\n        kernel = np.ones(\n            (5, 5),\n            dtype=np.uint8\n        )\n\n        mask = cv2.morphologyEx(\n            mask,\n            cv2.MORPH_OPEN,\n            kernel\n        )\n\n        mask = cv2.morphologyEx(\n            mask,\n            cv2.MORPH_CLOSE,\n            kernel\n        )\n\n        return mask > 0\n\n\n    def get_thumbnail_tissue_ratio(\n        self,\n        image_id,\n        x,\n        y,\n        tile_size\n    ):\n\n        thumbnail = self._load_thumbnail(image_id)\n\n        if thumbnail is None:\n            # TMA has no official thumbnail\n            return 1.0\n\n        tissue_mask = self._create_tissue_mask(\n            thumbnail\n        )\n\n        record = self.metadata.get_record(image_id)\n\n        scale_x = thumbnail.shape[1] / record[\"width\"]\n        scale_y = thumbnail.shape[0] / record[\"height\"]\n\n        x1 = int(x * scale_x)\n        y1 = int(y * scale_y)\n\n        x2 = int((x + tile_size) * scale_x)\n        y2 = int((y + tile_size) * scale_y)\n\n        x1 = max(0, min(x1, thumbnail.shape[1]))\n        x2 = max(0, min(x2, thumbnail.shape[1]))\n\n        y1 = max(0, min(y1, thumbnail.shape[0]))\n        y2 = max(0, min(y2, thumbnail.shape[0]))\n\n        if x2 <= x1 or y2 <= y1:\n            return 0.0\n\n        region = tissue_mask[y1:y2, x1:x2]\n\n        if region.size == 0:\n            return 0.0\n\n        return float(region.mean())\n\n\n    def select_candidates(\n        self,\n        image_id,\n        coordinates\n    ):\n\n        record = self.metadata.get_record(image_id)\n\n        # TMA images do not have official thumbnails.\n        # Keep their generated coordinates for M2 QA.\n        if record[\"is_tma\"]:\n            return list(coordinates)\n\n        selected = []\n\n        for coordinate in coordinates:\n\n            ratio = self.get_thumbnail_tissue_ratio(\n                image_id=image_id,\n                x=coordinate.x,\n                y=coordinate.y,\n                tile_size=coordinate.source_size\n            )\n\n            if ratio >= self.tissue_threshold:\n                selected.append(coordinate)\n\n        return selected","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.828874Z","iopub.status.idle":"2026-09-26T01:55:11.829267Z","shell.execute_reply.started":"2026-09-26T01:55:11.829039Z","shell.execute_reply":"2026-09-26T01:55:11.829055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nimport importlib\n\nproject_root = \"/kaggle/working/ovarian-trustworthy-ai\"\n\nif project_root not in sys.path:\n    sys.path.insert(0, project_root)\n\nimportlib.invalidate_caches()\n\nfrom src.preprocessing.metadata import UBCOceanMetadata\nfrom src.preprocessing.thumbnail_selector import ThumbnailSelector\n\n\ndataset_root = \"/kaggle/input/competitions/UBC-OCEAN\"\n\nmetadata = UBCOceanMetadata(dataset_root)\n\nselector = ThumbnailSelector(\n    metadata=metadata,\n    tissue_threshold=0.20\n)\n\nratio = selector.get_thumbnail_tissue_ratio(\n    image_id=4,\n    x=8960,\n    y=4608,\n    tile_size=256\n)\n\nprint(\"Tissue ratio:\", ratio)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.830743Z","iopub.status.idle":"2026-09-26T01:55:11.831141Z","shell.execute_reply.started":"2026-09-26T01:55:11.830919Z","shell.execute_reply":"2026-09-26T01:55:11.830934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from src.preprocessing.image_reader import UBCOceanImageReader\nfrom src.preprocessing.slide_type import SlideTypeHandler\nfrom src.preprocessing.tile_extractor import TileExtractor\n\nreader = UBCOceanImageReader(dataset_root)\n\nslide_handler = SlideTypeHandler(metadata)\n\ntile_extractor = TileExtractor(\n    reader,\n    slide_handler\n)\n\ncoordinates = tile_extractor.generate_grid(4)\n\nprint(\"Total candidate tiles:\", len(coordinates))\n\nselected_coordinates = selector.select_candidates(\n    image_id=4,\n    coordinates=coordinates\n)\n\nprint(\"Selected tissue candidates:\", len(selected_coordinates))\n\nremoved = len(coordinates) - len(selected_coordinates)\n\nprint(\"Removed background candidates:\", removed)\n\nprint(\n    \"Percentage removed:\",\n    round(100 * removed / len(coordinates), 2),\n    \"%\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.832261Z","iopub.status.idle":"2026-09-26T01:55:11.83257Z","shell.execute_reply.started":"2026-09-26T01:55:11.832419Z","shell.execute_reply":"2026-09-26T01:55:11.832464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tma_coordinates = tile_extractor.generate_grid(91)\n\ntma_selected = selector.select_candidates(\n    image_id=91,\n    coordinates=tma_coordinates\n)\n\nprint(\"TMA total:\", len(tma_coordinates))\nprint(\"TMA selected:\", len(tma_selected))\n\nprint(\n    \"All TMA candidates preserved:\",\n    len(tma_coordinates) == len(tma_selected)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.833772Z","iopub.status.idle":"2026-09-26T01:55:11.834101Z","shell.execute_reply.started":"2026-09-26T01:55:11.833945Z","shell.execute_reply":"2026-09-26T01:55:11.833968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nimport importlib\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nproject_root = \"/kaggle/working/ovarian-trustworthy-ai\"\n\nif project_root not in sys.path:\n    sys.path.insert(0, project_root)\n\nimportlib.invalidate_caches()\n\nfrom src.preprocessing.image_reader import UBCOceanImageReader\nfrom src.preprocessing.stain_normalization import MacenkoNormalizer\n\ndataset_root = \"/kaggle/input/competitions/UBC-OCEAN\"\n\nreader = UBCOceanImageReader(dataset_root)\n\nimage_id = 4\n\ntile = reader.get_region(\n    image_id=image_id,\n    x=8960,\n    y=4608,\n    width=256,\n    height=256\n)\n\nprint(\"Tile shape:\", tile.shape)\nprint(\"Tile dtype:\", tile.dtype)\nprint(\"Pixel range:\", tile.min(), \"to\", tile.max())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.835478Z","iopub.status.idle":"2026-09-26T01:55:11.835758Z","shell.execute_reply.started":"2026-09-26T01:55:11.835639Z","shell.execute_reply":"2026-09-26T01:55:11.835655Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"normalizer = MacenkoNormalizer(\n    beta=0.15,\n    alpha=1.0,\n    max_samples=10000,\n    io=255.0,\n    black_threshold=30.0\n)\n\nnormalizer.fit([tile])\n\nprint(\"Reference stain matrix:\")\nprint(normalizer.reference.stain_matrix)\n\nprint(\"\\nReference maximum concentration:\")\nprint(normalizer.reference.max_concentration)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.83686Z","iopub.status.idle":"2026-09-26T01:55:11.837225Z","shell.execute_reply.started":"2026-09-26T01:55:11.837043Z","shell.execute_reply":"2026-09-26T01:55:11.837068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get several tissue tiles from image 4\nreference_coordinates = [\n    (8960, 4608),\n    (9216, 4608),\n    (9472, 4608),\n    (9728, 4608),\n    (8960, 4864),\n]\n\nreference_tiles = []\n\nfor x, y in reference_coordinates:\n    tile = reader.get_region(\n        image_id=4,\n        x=x,\n        y=y,\n        width=256,\n        height=256\n    )\n    reference_tiles.append(tile)\n\nprint(\"Number of reference tiles:\", len(reference_tiles))\nprint(\"Tile shapes:\", [t.shape for t in reference_tiles])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.838568Z","iopub.status.idle":"2026-09-26T01:55:11.838818Z","shell.execute_reply.started":"2026-09-26T01:55:11.838707Z","shell.execute_reply":"2026-09-26T01:55:11.838721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(normalizer.__dict__.keys())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.840033Z","iopub.status.idle":"2026-09-26T01:55:11.840427Z","shell.execute_reply.started":"2026-09-26T01:55:11.840218Z","shell.execute_reply":"2026-09-26T01:55:11.840245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for key, value in normalizer.__dict__.items():\n    print(key, type(value), getattr(value, \"shape\", None))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.841454Z","iopub.status.idle":"2026-09-26T01:55:11.841669Z","shell.execute_reply.started":"2026-09-26T01:55:11.841561Z","shell.execute_reply":"2026-09-26T01:55:11.841573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(normalizer.reference)\nprint(normalizer.reference.__dict__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.842542Z","iopub.status.idle":"2026-09-26T01:55:11.84277Z","shell.execute_reply.started":"2026-09-26T01:55:11.842663Z","shell.execute_reply":"2026-09-26T01:55:11.842677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(normalizer.reference.__dict__.keys())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.843678Z","iopub.status.idle":"2026-09-26T01:55:11.843946Z","shell.execute_reply.started":"2026-09-26T01:55:11.843829Z","shell.execute_reply":"2026-09-26T01:55:11.843844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"normalizer.fit(reference_tiles)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.845673Z","iopub.status.idle":"2026-09-26T01:55:11.846006Z","shell.execute_reply.started":"2026-09-26T01:55:11.84583Z","shell.execute_reply":"2026-09-26T01:55:11.845852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\ntile = reference_tiles[0]\n\nod = normalizer.rgb_to_od(tile)\n\nmask = normalizer._get_stain_mask(tile, od)\n\nprint(\"Total pixels:\", tile.shape[0] * tile.shape[1])\nprint(\"Foreground pixels:\", mask.sum())\nprint(\"Foreground ratio:\", mask.mean())\n\nstain_matrix = normalizer._estimate_stain_matrix(tile)\n\nprint(\"\\nEstimated stain matrix:\")\nprint(stain_matrix)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.847213Z","iopub.status.idle":"2026-09-26T01:55:11.847568Z","shell.execute_reply.started":"2026-09-26T01:55:11.847395Z","shell.execute_reply":"2026-09-26T01:55:11.847417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\nAngle between stain vectors:\")\n\nv1 = stain_matrix[0]\nv2 = stain_matrix[1]\n\ncosine = np.dot(v1, v2) / (\n    np.linalg.norm(v1) * np.linalg.norm(v2)\n)\n\nangle = np.degrees(np.arccos(np.clip(cosine, -1, 1)))\n\nprint(angle, \"degrees\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.849367Z","iopub.status.idle":"2026-09-26T01:55:11.849663Z","shell.execute_reply.started":"2026-09-26T01:55:11.849496Z","shell.execute_reply":"2026-09-26T01:55:11.84951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import inspect\n\nprint(inspect.getsource(normalizer._estimate_stain_matrix))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.850487Z","iopub.status.idle":"2026-09-26T01:55:11.850851Z","shell.execute_reply.started":"2026-09-26T01:55:11.850665Z","shell.execute_reply":"2026-09-26T01:55:11.85069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(inspect.getsource(normalizer._get_stain_mask))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.852257Z","iopub.status.idle":"2026-09-26T01:55:11.852614Z","shell.execute_reply.started":"2026-09-26T01:55:11.852429Z","shell.execute_reply":"2026-09-26T01:55:11.852453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"od = normalizer.rgb_to_od(tile)\n\nmask = normalizer._get_stain_mask(tile, od)\nod_pixels = od.reshape(-1, 3)[mask]\n\nod_pixels = normalizer._sample_pixels(od_pixels)\n\n# SVD\n_, singular_values, vh = np.linalg.svd(\n    od_pixels,\n    full_matrices=False\n)\n\nv = vh[:2].T\n\nprojected = od_pixels @ v\n\nphi = np.arctan2(\n    projected[:, 1],\n    projected[:, 0]\n)\n\nprint(\"Number of OD pixels:\", len(od_pixels))\n\nprint(\"\\nSingular values:\")\nprint(singular_values)\n\nprint(\"\\nOD min:\")\nprint(od_pixels.min(axis=0))\n\nprint(\"\\nOD max:\")\nprint(od_pixels.max(axis=0))\n\nprint(\"\\nOD mean:\")\nprint(od_pixels.mean(axis=0))\n\nprint(\"\\nPhi min/max:\")\nprint(phi.min(), phi.max())\n\nprint(\"\\nPhi percentiles:\")\nprint(np.percentile(phi, [0, 1, 5, 25, 50, 75, 95, 99, 100]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.853725Z","iopub.status.idle":"2026-09-26T01:55:11.854156Z","shell.execute_reply.started":"2026-09-26T01:55:11.85395Z","shell.execute_reply":"2026-09-26T01:55:11.853978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check whether the angle distribution is wrapping around -pi / +pi\n\nphi_shifted = np.mod(phi, 2 * np.pi)\n\nprint(\"Original phi percentiles:\")\nprint(np.percentile(phi, [1, 5, 50, 95, 99]))\n\nprint(\"\\nShifted phi percentiles:\")\nprint(np.percentile(phi_shifted, [1, 5, 50, 95, 99]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:11.855588Z","iopub.status.idle":"2026-09-26T01:55:11.856097Z","shell.execute_reply.started":"2026-09-26T01:55:11.85592Z","shell.execute_reply":"2026-09-26T01:55:11.85595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"SVD basis V:\")\nprint(v)\n\nprint(\"\\nProjected coordinate ranges:\")\nprint(\"X:\", projected[:, 0].min(), \"to\", projected[:, 0].max())\nprint(\"Y:\", projected[:, 1].min(), \"to\", projected[:, 1].max())\n\nprint(\"\\nProjected coordinate means:\")\nprint(projected.mean(axis=0))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-26T01:55:32.177324Z","iopub.execute_input":"2026-09-26T01:55:32.177581Z","iopub.status.idle":"2026-09-26T01:55:32.184323Z","shell.execute_reply.started":"2026-09-26T01:55:32.177558Z","shell.execute_reply":"2026-09-26T01:55:32.183384Z"}},"outputs":[],"execution_count":null}]}