{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":18647,"databundleVersionId":1126921,"sourceType":"competition"}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Master Imports and Path Setup**","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport openslide\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix, cohen_kappa_score\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom tensorflow.keras.applications import DenseNet121, EfficientNetB0, ResNet50\nfrom tensorflow.keras.layers import Input, GlobalAveragePooling2D, Concatenate, Dense, Dropout, Rescaling\nfrom tensorflow.keras.models import Model\n\n# 1. Clear Session\ntf.keras.backend.clear_session()\n\n# 2. Setup GPU Strategy\nstrategy = tf.distribute.MirroredStrategy()\nprint(f'✅ GPUs Detected: {strategy.num_replicas_in_sync}')\n\n# 3. CRITICAL CONSTANTS FOR TILING\nBASE_PATH = '/kaggle/input/prostate-cancer-grade-assessment/'\nIMAGES_DIR = os.path.join(BASE_PATH, 'train_images')\nTILE_SIZE = 128\nN_TILES = 16\nSIZE = int(math.sqrt(N_TILES)) * TILE_SIZE  # 4 * 128 = 512\nBATCH_SIZE = 8 * strategy.num_replicas_in_sync # 16 total (Safe for 512x512)\n\n# 4. Load & Limit Data (Safety Config)\ntrain_df = pd.read_csv(os.path.join(BASE_PATH, 'train.csv'))\n# Limit to 6000 to ensure <9 hour runtime\ntrain_df = train_df.sample(n=6000, random_state=42).reset_index(drop=True)\nprint(f\"✅ Configuration: {N_TILES} tiles @ {TILE_SIZE}px = {SIZE}x{SIZE} input.\")\nprint(f\"✅ Dataset: {len(train_df)} slides selected for training.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"2. **Image Processing Functions**","metadata":{}},{"cell_type":"code","source":"def get_tiles(img, mode=0):\n    \"\"\"\n    1. Pads image to be divisible by TILE_SIZE\n    2. Cuts into tiles\n    3. Selects the N_TILES with the most tissue (info)\n    \"\"\"\n    h, w, c = img.shape\n    pad_h = (TILE_SIZE - h % TILE_SIZE) % TILE_SIZE + ((TILE_SIZE * mode) // 2)\n    pad_w = (TILE_SIZE - w % TILE_SIZE) % TILE_SIZE + ((TILE_SIZE * mode) // 2)\n    \n    img = np.pad(img, [[pad_h // 2, pad_h - pad_h // 2], \n                      [pad_w // 2, pad_w - pad_w // 2], \n                      [0, 0]], constant_values=255)\n    \n    img = img.reshape(img.shape[0] // TILE_SIZE, TILE_SIZE,\n                      img.shape[1] // TILE_SIZE, TILE_SIZE, 3)\n    img = img.transpose(0, 2, 1, 3, 4).reshape(-1, TILE_SIZE, TILE_SIZE, 3)\n    \n    # Sort by \"tissue content\" (sum of pixels; lower sum = more tissue/darker)\n    if len(img) < N_TILES:\n        img = np.pad(img, [[0, N_TILES - len(img)], [0, 0], [0, 0], [0, 0]], constant_values=255)\n        \n    idxs = np.argsort(img.reshape(img.shape[0], -1).sum(-1))[:N_TILES]\n    img = img[idxs]\n    return img","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"3. **Data Generator and Splitting**","metadata":{}},{"cell_type":"code","source":"class TiledGenerator(tf.keras.utils.Sequence):\n    def __init__(self, df, batch_size=16, shuffle=True, augment=False):\n        self.df = df.reset_index(drop=True)\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.augment = augment\n        self.indices = np.arange(len(self.df))\n        self.on_epoch_end()\n\n    def __len__(self):\n        return int(np.ceil(len(self.df) / self.batch_size))\n\n    def on_epoch_end(self):\n        if self.shuffle: np.random.shuffle(self.indices)\n\n    def __getitem__(self, index):\n        idxs = self.indices[index*self.batch_size : (index+1)*self.batch_size]\n        \n        # 4x4 Grid = 16 Tiles\n        X = np.zeros((self.batch_size, SIZE, SIZE, 3), dtype=np.float32)\n        y = np.zeros((self.batch_size, 6), dtype=np.float32)\n        \n        valid_count = 0\n        for i, ID in enumerate(idxs):\n            row = self.df.iloc[ID]\n            path = os.path.join(IMAGES_DIR, f\"{row['image_id']}.tiff\")\n            try:\n                slide = openslide.OpenSlide(path)\n                # Level 1 is usually 4x downsample (good balance of speed/detail)\n                img = np.array(slide.read_region((0,0), 1, slide.level_dimensions[1]).convert('RGB'))\n                \n                # Extract tiles and stitch\n                tiles = get_tiles(img)\n                \n                # Stitch 16 tiles into 4x4 grid\n                stitched = np.zeros((SIZE, SIZE, 3), dtype=np.uint8)\n                for j in range(N_TILES):\n                    r, c = divmod(j, 4) # 4 is sqrt(16)\n                    stitched[r*TILE_SIZE:(r+1)*TILE_SIZE, \n                             c*TILE_SIZE:(c+1)*TILE_SIZE] = tiles[j]\n                \n                if self.augment and np.random.rand() > 0.5:\n                    stitched = np.fliplr(stitched)\n                    \n                # Normalize 0-1\n                X[i] = stitched.astype('float32') / 255.0\n                y[i, row['isup_grade']] = 1.0\n                valid_count += 1\n            except:\n                continue # Skip corrupt\n                \n        # Fill Batch if errors occurred\n        if valid_count < self.batch_size and valid_count > 0:\n             for i in range(valid_count, self.batch_size):\n                X[i] = X[0]\n                y[i] = y[0]\n                \n        return X, y","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"4. **Build the Hybrid Ensemble Model**","metadata":{}},{"cell_type":"code","source":"def build_tiled_model():\n    with strategy.scope():\n        # Input is now 512x512 (High Res)\n        inputs = Input(shape=(SIZE, SIZE, 3))\n        \n        # Rescale inputs for ImageNet models (they expect 0-255 range usually)\n        x_in = Rescaling(255.0)(inputs)\n        \n        # 1. Backbones (Unfrozen)\n        d121 = DenseNet121(weights='imagenet', include_top=False, input_shape=(SIZE, SIZE, 3))\n        eff0 = EfficientNetB0(weights='imagenet', include_top=False, input_shape=(SIZE, SIZE, 3))\n        res50 = ResNet50(weights='imagenet', include_top=False, input_shape=(SIZE, SIZE, 3))\n        \n        d121.trainable = True\n        eff0.trainable = True\n        res50.trainable = True\n        \n        # 2. Features\n        out_d = GlobalAveragePooling2D(name='pool_d')(d121(x_in))\n        out_e = GlobalAveragePooling2D(name='pool_e')(eff0(x_in))\n        out_r = GlobalAveragePooling2D(name='pool_r')(res50(x_in))\n        \n        # 3. Concatenate\n        merged = Concatenate()([out_d, out_e, out_r])\n        \n        # 4. Classification Head\n        x = Dense(512, activation='relu')(merged)\n        x = Dropout(0.4)(x)\n        outputs = Dense(6, activation='softmax', dtype='float32')(x)\n        \n        model = Model(inputs=inputs, outputs=outputs)\n        \n        # Low LR for unfrozen training\n        opt = tf.keras.optimizers.Adam(learning_rate=1e-5, clipnorm=1.0)\n        model.compile(optimizer=opt, loss='categorical_crossentropy', metrics=['accuracy'])\n        return model\n\nmodel = build_tiled_model()\nprint(f\"✅ Tiled Model Built. Input Shape: {model.input_shape}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"5. **Data Splitting and Initialization**","metadata":{}},{"cell_type":"code","source":"# Split\ntrain_sub, val_sub = train_test_split(\n    train_df, test_size=0.15, stratify=train_df['isup_grade'], random_state=42\n)\n\n# Generators\ntrain_gen = TiledGenerator(train_sub, batch_size=BATCH_SIZE, augment=True)\nval_gen = TiledGenerator(val_sub, batch_size=BATCH_SIZE, shuffle=False)\n\n# Weights\nweights = compute_class_weight('balanced', classes=np.unique(train_sub['isup_grade']), y=train_sub['isup_grade'])\nclass_weight_dict = {i: w for i, w in enumerate(weights)}\n\nprint(f\"🚀 Ready to train on {len(train_sub)} samples (512x512 resolution)\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"6. **Training Loop and Native Save**","metadata":{}},{"cell_type":"code","source":"callbacks = [\n    tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=2, verbose=1),\n    tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=4, restore_best_weights=True),\n    tf.keras.callbacks.ModelCheckpoint('panda_tiled_final.keras', save_best_only=True)\n]\n\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=10, \n    class_weight=class_weight_dict,\n    callbacks=callbacks,\n    verbose=1\n)\n\nmodel.save('panda_tiled_final_finished.keras')\nprint(\"✅ Training Complete.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"7. **Performance Evaluation (Confusion Matrix)**","metadata":{}},{"cell_type":"code","source":"# Run Inference\nval_gen.shuffle = False\nval_gen.on_epoch_end()\n\nprint(\"🔍 Calculating Final Metrics...\")\npreds = model.predict(val_gen, verbose=1)\ny_pred = np.argmax(preds[:len(val_sub)], axis=1)\ny_true = val_sub['isup_grade'].values\n\n# Metrics\nqwk = cohen_kappa_score(y_true, y_pred, weights='quadratic')\nacc = np.mean(y_pred == y_true)\n\n# Plot\nplt.figure(figsize=(10, 8))\nsns.heatmap(confusion_matrix(y_true, y_pred), annot=True, fmt='d', cmap='Greens')\nplt.title(f'Tiled Model Results - Acc: {acc:.2%} | QWK: {qwk:.4f}')\nplt.show()\n\nprint(classification_report(y_true, y_pred))","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}