{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\n\nfrom PIL import Image\n\nprint(\"TensorFlow Version:\", tf.__version__)\nprint(\"GPUs Available:\", tf.config.list_physical_devices('GPU'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T15:40:37.100929Z","iopub.execute_input":"2026-09-29T15:40:37.101864Z","iopub.status.idle":"2026-09-29T15:40:37.107547Z","shell.execute_reply.started":"2026-09-29T15:40:37.101824Z","shell.execute_reply":"2026-09-29T15:40:37.106919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_path = \"/kaggle/input/datasets/felipekitamura/head-ct-hemorrhage\"\nlabels_path = os.path.join(base_path, \"labels.csv\")\nimage_folder = os.path.join(base_path, \"head_ct\", \"head_ct\")\n\ndf = pd.read_csv(labels_path)\n\ndf.columns = df.columns.str.strip()\n\nprint(df.head())\nprint(\"\\nColumns:\", df.columns.tolist())\nprint(\"\\nShape:\", df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T15:40:52.094845Z","iopub.execute_input":"2026-09-29T15:40:52.095121Z","iopub.status.idle":"2026-09-29T15:40:52.143051Z","shell.execute_reply.started":"2026-09-29T15:40:52.095098Z","shell.execute_reply":"2026-09-29T15:40:52.142339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['image_path'] = df['id'].apply(\n    lambda x: os.path.join(\n        image_folder,\n        str(int(x)).zfill(3) + \".png\"\n    )\n)\n\nprint(\"Class Distribution:\")\nprint(df['hemorrhage'].value_counts())\n\nprint(\"\\nFirst 5 Rows:\")\nprint(df[['id', 'hemorrhage', 'image_path']].head())\n\nprint(\"\\nChecking Files:\")\n\nfor path in df['image_path'].head():\n    print(os.path.exists(path), path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T15:41:04.9783Z","iopub.execute_input":"2026-09-29T15:41:04.978798Z","iopub.status.idle":"2026-09-29T15:41:05.06268Z","shell.execute_reply.started":"2026-09-29T15:41:04.978767Z","shell.execute_reply":"2026-09-29T15:41:05.06201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_data = df.sample(10, random_state=42)\n\nfig, axes = plt.subplots(2, 5, figsize=(15, 6))\n\nfor ax, (_, row) in zip(axes.flatten(), sample_data.iterrows()):\n\n    img = Image.open(row['image_path'])\n\n    ax.imshow(img, cmap='gray')\n\n    label = \"Hemorrhage\" if row['hemorrhage'] == 1 else \"Normal\"\n\n    ax.set_title(label)\n    ax.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T15:42:51.394662Z","iopub.execute_input":"2026-09-29T15:42:51.39508Z","iopub.status.idle":"2026-09-29T15:42:52.48453Z","shell.execute_reply.started":"2026-09-29T15:42:51.395046Z","shell.execute_reply":"2026-09-29T15:42:52.483554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_df, temp_df = train_test_split(\n    df,\n    test_size=0.30,\n    random_state=42,\n    stratify=df['hemorrhage']\n)\n\nval_df, test_df = train_test_split(\n    temp_df,\n    test_size=0.50,\n    random_state=42,\n    stratify=temp_df['hemorrhage']\n)\n\ntrain_df = train_df.reset_index(drop=True)\nval_df = val_df.reset_index(drop=True)\ntest_df = test_df.reset_index(drop=True)\n\nprint(\"Total Images:\", len(df))\nprint(\"Training Images:\", len(train_df))\nprint(\"Validation Images:\", len(val_df))\nprint(\"Testing Images:\", len(test_df))\n\nprint(\"\\nTraining Distribution:\")\nprint(train_df['hemorrhage'].value_counts())\n\nprint(\"\\nValidation Distribution:\")\nprint(val_df['hemorrhage'].value_counts())\n\nprint(\"\\nTesting Distribution:\")\nprint(test_df['hemorrhage'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T15:43:03.037914Z","iopub.execute_input":"2026-09-29T15:43:03.038325Z","iopub.status.idle":"2026-09-29T15:43:03.193548Z","shell.execute_reply.started":"2026-09-29T15:43:03.038296Z","shell.execute_reply":"2026-09-29T15:43:03.19276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = (224, 224)\nBATCH_SIZE = 8\n\ndef load_image(path, label):\n\n    image = tf.io.read_file(path)\n\n    image = tf.image.decode_png(\n        image,\n        channels=3\n    )\n\n    image = tf.image.resize(\n        image,\n        IMG_SIZE\n    )\n\n    image = tf.cast(image, tf.float32)\n\n    label = tf.cast(label, tf.float32)\n\n    return image, label\n\n\ndef create_dataset(dataframe, shuffle=False):\n\n    paths = dataframe['image_path'].values\n    labels = dataframe['hemorrhage'].values\n\n    dataset = tf.data.Dataset.from_tensor_slices(\n        (paths, labels)\n    )\n\n    dataset = dataset.map(\n        load_image,\n        num_parallel_calls=tf.data.AUTOTUNE\n    )\n\n    if shuffle:\n        dataset = dataset.shuffle(\n            buffer_size=len(dataframe),\n            seed=42\n        )\n\n    dataset = dataset.batch(BATCH_SIZE)\n\n    dataset = dataset.prefetch(\n        tf.data.AUTOTUNE\n    )\n\n    return dataset\n\n\ntrain_data = create_dataset(train_df, shuffle=True)\nval_data = create_dataset(val_df)\ntest_data = create_dataset(test_df)\n\nfor images, labels in train_data.take(1):\n    print(\"Image Batch Shape:\", images.shape)\n    print(\"Label Batch Shape:\", labels.shape)\n    print(\"Minimum Pixel Value:\", tf.reduce_min(images).numpy())\n    print(\"Maximum Pixel Value:\", tf.reduce_max(images).numpy())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T15:45:17.594959Z","iopub.execute_input":"2026-09-29T15:45:17.595742Z","iopub.status.idle":"2026-09-29T15:45:18.08146Z","shell.execute_reply.started":"2026-09-29T15:45:17.595716Z","shell.execute_reply":"2026-09-29T15:45:18.080648Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\nfrom tensorflow.keras.applications import DenseNet121\nfrom tensorflow.keras.applications.densenet import preprocess_input as densenet_preprocess\n\nfrom tensorflow.keras.layers import (\n    Input,\n    Lambda,\n    GlobalAveragePooling2D,\n    Dense,\n    Dropout\n)\n\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n# -----------------------------\n# Load pretrained DenseNet121\n# -----------------------------\nbase_model1 = DenseNet121(\n    weights='imagenet',\n    include_top=False,\n    input_shape=(224, 224, 3)\n)\n\n# Freeze pretrained layers\nbase_model1.trainable = False\n\n# -----------------------------\n# Build Model\n# -----------------------------\ninputs = Input(shape=(224, 224, 3))\n\n# DenseNet-specific preprocessing\nx = Lambda(densenet_preprocess)(inputs)\n\nx = base_model1(\n    x,\n    training=False\n)\n\nx = GlobalAveragePooling2D()(x)\n\nx = Dense(\n    128,\n    activation='relu'\n)(x)\n\nx = Dropout(0.5)(x)\n\noutputs = Dense(\n    1,\n    activation='sigmoid'\n)(x)\n\nmodel1 = Model(\n    inputs=inputs,\n    outputs=outputs\n)\n\n# -----------------------------\n# Compile\n# -----------------------------\nmodel1.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['accuracy']\n)\n\n# -----------------------------\n# Early Stopping\n# -----------------------------\nearly_stop1 = EarlyStopping(\n    monitor='val_loss',\n    patience=5,\n    restore_best_weights=True,\n    verbose=1\n)\n\n# Model architecture\nmodel1.summary()\n\n# -----------------------------\n# Training\n# -----------------------------\nhistory1 = model1.fit(\n    train_data,\n    validation_data=val_data,\n    epochs=50,\n    callbacks=[early_stop1]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T15:48:30.685091Z","iopub.execute_input":"2026-09-29T15:48:30.685816Z","iopub.status.idle":"2026-09-29T15:49:38.210638Z","shell.execute_reply.started":"2026-09-29T15:48:30.685787Z","shell.execute_reply":"2026-09-29T15:49:38.209985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.layers import (\n    Input,\n    GlobalAveragePooling2D,\n    Dense,\n    Dropout\n)\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n# -----------------------------\n# Load pretrained EfficientNetB0\n# -----------------------------\nbase_model2 = EfficientNetB0(\n    weights='imagenet',\n    include_top=False,\n    input_shape=(224, 224, 3)\n)\n\n# Freeze pretrained layers\nbase_model2.trainable = False\n\n# -----------------------------\n# Build Model\n# -----------------------------\ninputs = Input(shape=(224, 224, 3))\n\n# EfficientNetB0 in TensorFlow includes its own input rescaling\nx = base_model2(\n    inputs,\n    training=False\n)\n\nx = GlobalAveragePooling2D()(x)\n\nx = Dense(\n    128,\n    activation='relu'\n)(x)\n\nx = Dropout(0.5)(x)\n\noutputs = Dense(\n    1,\n    activation='sigmoid'\n)(x)\n\nmodel2 = Model(\n    inputs=inputs,\n    outputs=outputs\n)\n\n# -----------------------------\n# Compile\n# -----------------------------\nmodel2.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['accuracy']\n)\n\n# -----------------------------\n# Early Stopping\n# -----------------------------\nearly_stop2 = EarlyStopping(\n    monitor='val_loss',\n    patience=5,\n    restore_best_weights=True,\n    verbose=1\n)\n\n# Show architecture\nmodel2.summary()\n\n# -----------------------------\n# Training\n# -----------------------------\nhistory2 = model2.fit(\n    train_data,\n    validation_data=val_data,\n    epochs=50,\n    callbacks=[early_stop2]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T15:57:23.554149Z","iopub.execute_input":"2026-09-29T15:57:23.554567Z","iopub.status.idle":"2026-09-29T15:58:13.287841Z","shell.execute_reply.started":"2026-09-29T15:57:23.554538Z","shell.execute_reply":"2026-09-29T15:58:13.287167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.applications import Xception\nfrom tensorflow.keras.applications.xception import preprocess_input as xception_preprocess\n\nfrom tensorflow.keras.layers import Input, Lambda, GlobalAveragePooling2D, Dense, Dropout\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n# -----------------------------\n# Load pretrained Xception\n# -----------------------------\nbase_model3 = Xception(\n    weights='imagenet',\n    include_top=False,\n    input_shape=(224, 224, 3)\n)\n\nbase_model3.trainable = False\n\n# -----------------------------\n# Build Model\n# -----------------------------\ninputs = Input(shape=(224, 224, 3))\n\nx = Lambda(xception_preprocess)(inputs)\n\nx = base_model3(\n    x,\n    training=False\n)\n\nx = GlobalAveragePooling2D()(x)\n\nx = Dense(\n    128,\n    activation='relu'\n)(x)\n\nx = Dropout(0.5)(x)\n\noutputs = Dense(\n    1,\n    activation='sigmoid'\n)(x)\n\nmodel3 = Model(\n    inputs=inputs,\n    outputs=outputs\n)\n\n# -----------------------------\n# Compile\n# -----------------------------\nmodel3.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['accuracy']\n)\n\n# -----------------------------\n# Early Stopping\n# -----------------------------\nearly_stop3 = EarlyStopping(\n    monitor='val_loss',\n    patience=5,\n    restore_best_weights=True,\n    verbose=1\n)\n\nmodel3.summary()\n\n# -----------------------------\n# Train\n# -----------------------------\nhistory3 = model3.fit(\n    train_data,\n    validation_data=val_data,\n    epochs=50,\n    callbacks=[early_stop3]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T16:04:40.057717Z","iopub.execute_input":"2026-09-29T16:04:40.058119Z","iopub.status.idle":"2026-09-29T16:05:18.38888Z","shell.execute_reply.started":"2026-09-29T16:04:40.058091Z","shell.execute_reply":"2026-09-29T16:05:18.388007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom sklearn.metrics import (\n    accuracy_score,\n    precision_score,\n    recall_score,\n    f1_score\n)\n\ndef evaluate_model(model, dataset, model_name):\n\n    y_true = []\n    y_pred = []\n\n    for images, labels in dataset:\n\n        predictions = model.predict(\n            images,\n            verbose=0\n        )\n\n        predicted_labels = (\n            predictions > 0.5\n        ).astype(int).flatten()\n\n        y_true.extend(\n            labels.numpy().astype(int)\n        )\n\n        y_pred.extend(\n            predicted_labels\n        )\n\n    return {\n        \"Model\": model_name,\n        \"Accuracy\": accuracy_score(y_true, y_pred),\n        \"Precision\": precision_score(y_true, y_pred, zero_division=0),\n        \"Recall\": recall_score(y_true, y_pred, zero_division=0),\n        \"F1 Score\": f1_score(y_true, y_pred, zero_division=0)\n    }\n\n\nresults = pd.DataFrame([\n    evaluate_model(model1, test_data, \"DenseNet121\"),\n    evaluate_model(model2, test_data, \"EfficientNetB0\"),\n    evaluate_model(model3, test_data, \"Xception\")\n])\n\nfor column in [\n    \"Accuracy\",\n    \"Precision\",\n    \"Recall\",\n    \"F1 Score\"\n]:\n    results[column] = (\n        results[column] * 100\n    ).round(2)\n\nprint(\"Final Model Comparison:\")\ndisplay(results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T16:08:18.586424Z","iopub.execute_input":"2026-09-29T16:08:18.587311Z","iopub.status.idle":"2026-09-29T16:08:20.146574Z","shell.execute_reply.started":"2026-09-29T16:08:18.587283Z","shell.execute_reply":"2026-09-29T16:08:20.146007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nfrom sklearn.metrics import (\n    confusion_matrix,\n    ConfusionMatrixDisplay\n)\n\ndef get_predictions(model, dataset):\n\n    y_true = []\n    y_pred = []\n\n    for images, labels in dataset:\n\n        predictions = model.predict(\n            images,\n            verbose=0\n        )\n\n        predicted_labels = (\n            predictions > 0.5\n        ).astype(int).flatten()\n\n        y_true.extend(\n            labels.numpy().astype(int)\n        )\n\n        y_pred.extend(\n            predicted_labels\n        )\n\n    return np.array(y_true), np.array(y_pred)\n\n\nmodels = [\n    (\"DenseNet121\", model1),\n    (\"EfficientNetB0\", model2),\n    (\"Xception\", model3)\n]\n\nfor name, model in models:\n\n    y_true, y_pred = get_predictions(\n        model,\n        test_data\n    )\n\n    cm = confusion_matrix(\n        y_true,\n        y_pred\n    )\n\n    disp = ConfusionMatrixDisplay(\n        confusion_matrix=cm,\n        display_labels=[\n            \"Normal\",\n            \"Hemorrhage\"\n        ]\n    )\n\n    disp.plot()\n\n    plt.title(\n        name + \" - Confusion Matrix\"\n    )\n\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T16:08:26.95222Z","iopub.execute_input":"2026-09-29T16:08:26.952773Z","iopub.status.idle":"2026-09-29T16:08:28.848054Z","shell.execute_reply.started":"2026-09-29T16:08:26.952737Z","shell.execute_reply":"2026-09-29T16:08:28.847404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_history(history, model_name):\n\n    # Accuracy\n    plt.figure(figsize=(8, 5))\n\n    plt.plot(\n        history.history['accuracy'],\n        label='Training Accuracy'\n    )\n\n    plt.plot(\n        history.history['val_accuracy'],\n        label='Validation Accuracy'\n    )\n\n    plt.title(\n        model_name + \" - Accuracy\"\n    )\n\n    plt.xlabel(\"Epoch\")\n    plt.ylabel(\"Accuracy\")\n    plt.legend()\n    plt.show()\n\n    # Loss\n    plt.figure(figsize=(8, 5))\n\n    plt.plot(\n        history.history['loss'],\n        label='Training Loss'\n    )\n\n    plt.plot(\n        history.history['val_loss'],\n        label='Validation Loss'\n    )\n\n    plt.title(\n        model_name + \" - Loss\"\n    )\n\n    plt.xlabel(\"Epoch\")\n    plt.ylabel(\"Loss\")\n    plt.legend()\n    plt.show()\n\n\nplot_history(history1, \"DenseNet121\")\nplot_history(history2, \"EfficientNetB0\")\nplot_history(history3, \"Xception\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T16:08:47.35522Z","iopub.execute_input":"2026-09-29T16:08:47.35586Z","iopub.status.idle":"2026-09-29T16:08:48.227954Z","shell.execute_reply.started":"2026-09-29T16:08:47.35583Z","shell.execute_reply":"2026-09-29T16:08:48.227307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\nmodels = [\n    (\"DenseNet121\", model1),\n    (\"EfficientNetB0\", model2),\n    (\"Xception\", model3)\n]\n\nfor name, model in models:\n\n    y_true, y_pred = get_predictions(\n        model,\n        test_data\n    )\n\n    print(\"\\n================================\")\n    print(name)\n    print(\"================================\")\n\n    print(\n        classification_report(\n            y_true,\n            y_pred,\n            target_names=[\n                \"Normal\",\n                \"Hemorrhage\"\n            ],\n            digits=4,\n            zero_division=0\n        )\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T16:09:04.167999Z","iopub.execute_input":"2026-09-29T16:09:04.168849Z","iopub.status.idle":"2026-09-29T16:09:05.707063Z","shell.execute_reply.started":"2026-09-29T16:09:04.168817Z","shell.execute_reply":"2026-09-29T16:09:05.706246Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\n\ndef get_probabilities(model, dataset):\n\n    y_true = []\n    y_prob = []\n\n    for images, labels in dataset:\n\n        predictions = model.predict(\n            images,\n            verbose=0\n        ).flatten()\n\n        y_true.extend(\n            labels.numpy().astype(int)\n        )\n\n        y_prob.extend(predictions)\n\n    return np.array(y_true), np.array(y_prob)\n\n\nmodels = [\n    (\"DenseNet121\", model1),\n    (\"EfficientNetB0\", model2),\n    (\"Xception\", model3)\n]\n\nplt.figure(figsize=(8, 6))\n\nfor name, model in models:\n\n    y_true, y_prob = get_probabilities(\n        model,\n        test_data\n    )\n\n    fpr, tpr, _ = roc_curve(\n        y_true,\n        y_prob\n    )\n\n    roc_auc = auc(\n        fpr,\n        tpr\n    )\n\n    plt.plot(\n        fpr,\n        tpr,\n        label=f\"{name} AUC = {roc_auc:.3f}\"\n    )\n\nplt.plot(\n    [0, 1],\n    [0, 1],\n    linestyle='--'\n)\n\nplt.xlabel(\"False Positive Rate\")\nplt.ylabel(\"True Positive Rate\")\nplt.title(\"ROC Curve Comparison\")\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T16:09:21.364989Z","iopub.execute_input":"2026-09-29T16:09:21.365805Z","iopub.status.idle":"2026-09-29T16:09:23.081714Z","shell.execute_reply.started":"2026-09-29T16:09:21.365776Z","shell.execute_reply":"2026-09-29T16:09:23.080916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model1.save(\"/kaggle/working/densenet121_brain_hemorrhage.keras\")\n\nmodel2.save(\"/kaggle/working/efficientnetb0_brain_hemorrhage.keras\")\n\nmodel3.save(\"/kaggle/working/xception_brain_hemorrhage.keras\")\n\nprint(\"All 3 models saved successfully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T16:09:57.294438Z","iopub.execute_input":"2026-09-29T16:09:57.295267Z","iopub.status.idle":"2026-09-29T16:09:59.603943Z","shell.execute_reply.started":"2026-09-29T16:09:57.295221Z","shell.execute_reply":"2026-09-29T16:09:59.603127Z"}},"outputs":[],"execution_count":null}]}