{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#import necessary libs\nimport os\n\nimport keras_cv\nimport keras_core as keras\nfrom keras_core import layers\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom matplotlib import pyplot as plt\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:41:43.05535Z","iopub.execute_input":"2023-09-03T04:41:43.05572Z","iopub.status.idle":"2023-09-03T04:41:49.607219Z","shell.execute_reply.started":"2023-09-03T04:41:43.055682Z","shell.execute_reply":"2023-09-03T04:41:49.606222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Configuration","metadata":{}},{"cell_type":"markdown","source":"refer this for EDA - https://www.kaggle.com/code/aritrag/eda-train-csv","metadata":{}},{"cell_type":"code","source":"class Config:\n    SEED = 69\n    IMAGE_SIZE = [256,256]\n    BATCH_SIZE=32\n    EPOCHS = 5\n    TARGET_COLS = [\n        \"bowel_injury\", \"extravasation_injury\",\n        \"kidney_healthy\", \"kidney_low\", \"kidney_high\",\n        \"liver_healthy\", \"liver_low\", \"liver_high\",\n        \"spleen_healthy\", \"spleen_low\", \"spleen_high\"\n    ]\n    AUTOTUNE = tf.data.AUTOTUNE\n    \nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:41:49.60962Z","iopub.execute_input":"2023-09-03T04:41:49.61031Z","iopub.status.idle":"2023-09-03T04:41:49.615953Z","shell.execute_reply.started":"2023-09-03T04:41:49.610271Z","shell.execute_reply":"2023-09-03T04:41:49.615047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Reproducibility\n\nWe would want this notebook to have reproducible results. Here we set the seed for all the random algorithms so that we can reproduce the experiments each time exactly the same way.\n\n","metadata":{}},{"cell_type":"code","source":"keras.utils.set_random_seed(seed=config.SEED)","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:41:49.617418Z","iopub.execute_input":"2023-09-03T04:41:49.618236Z","iopub.status.idle":"2023-09-03T04:41:49.63453Z","shell.execute_reply.started":"2023-09-03T04:41:49.618198Z","shell.execute_reply":"2023-09-03T04:41:49.633502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset provided in the competition consists of DICOM images. We will not be training on the DICOM images, rather would work on PNG image which are extracted from the DICOM format.\n\nA helpful resource on the conversion of DICOM to PNG - https://www.kaggle.com/code/radek1/how-to-process-dicom-images-to-pngs","metadata":{}},{"cell_type":"code","source":"BASE_PATH = f'/kaggle/input/rsna-atd-512x512-png-v2-dataset'","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:41:49.637456Z","iopub.execute_input":"2023-09-03T04:41:49.637847Z","iopub.status.idle":"2023-09-03T04:41:49.645664Z","shell.execute_reply.started":"2023-09-03T04:41:49.637814Z","shell.execute_reply":"2023-09-03T04:41:49.644705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"train.csv contains metadata","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(f'{BASE_PATH}/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:41:49.646948Z","iopub.execute_input":"2023-09-03T04:41:49.647816Z","iopub.status.idle":"2023-09-03T04:41:49.724488Z","shell.execute_reply.started":"2023-09-03T04:41:49.647782Z","shell.execute_reply":"2023-09-03T04:41:49.723198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"image_path\"] = f\"{BASE_PATH}/train_images\"\\\n                    + \"/\" + df.patient_id.astype(str)\\\n                    + \"/\" + df.series_id.astype(str)\\\n                    + \"/\" + df.instance_number.astype(str) +\".png\"\n\ndf = df.drop_duplicates()","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:41:49.72631Z","iopub.execute_input":"2023-09-03T04:41:49.726687Z","iopub.status.idle":"2023-09-03T04:41:49.786064Z","shell.execute_reply.started":"2023-09-03T04:41:49.726633Z","shell.execute_reply":"2023-09-03T04:41:49.784937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:41:49.787356Z","iopub.execute_input":"2023-09-03T04:41:49.787806Z","iopub.status.idle":"2023-09-03T04:41:49.816202Z","shell.execute_reply.started":"2023-09-03T04:41:49.787772Z","shell.execute_reply":"2023-09-03T04:41:49.815331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We split the training dataset into train and validation. This is a common practise in the Machine Learning pipelines. We not only want to train our model, but also want to validate it's training.\n\nA small catch here is that the training and validation data should have an aligned data distribution. Here we handle that by grouping the lables and then splitting the dataset. This ensures an aligned data distribution between the training and the validation splits.","metadata":{}},{"cell_type":"code","source":"#Split_group function \"\"\ndef split_group(group,test_size=0.2):\n    if len(group)==1:\n        return (group, pd.DataFrame()) if np.random.rand() < test_size else (pd.DataFrame()\n                                                                             ,group)\n    else:\n        return train_test_split(group,test_size=test_size,random_state=69)\n    \n","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:41:49.8175Z","iopub.execute_input":"2023-09-03T04:41:49.817985Z","iopub.status.idle":"2023-09-03T04:41:49.823946Z","shell.execute_reply.started":"2023-09-03T04:41:49.817947Z","shell.execute_reply":"2023-09-03T04:41:49.823014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#initialize 2 empty dfs for train and val\n\ntrain_data = pd.DataFrame()\nval_data = pd.DataFrame()","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:41:49.825267Z","iopub.execute_input":"2023-09-03T04:41:49.825849Z","iopub.status.idle":"2023-09-03T04:41:49.838145Z","shell.execute_reply.started":"2023-09-03T04:41:49.825812Z","shell.execute_reply":"2023-09-03T04:41:49.837193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate through the groups and split them, handling single-sample groups\nfor _, group in df.groupby(config.TARGET_COLS):\n    train_group, val_group = split_group(group)\n    train_data = pd.concat([train_data,train_group],ignore_index=True)\n    val_data = pd.concat([val_data,val_group],ignore_index=True)\n    \nprint(train_data.shape)\nprint(val_data.shape)","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:41:49.842753Z","iopub.execute_input":"2023-09-03T04:41:49.84301Z","iopub.status.idle":"2023-09-03T04:41:49.939081Z","shell.execute_reply.started":"2023-09-03T04:41:49.842987Z","shell.execute_reply":"2023-09-03T04:41:49.938088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data Pipeline /w tf.data\n'\n\nHere we build the data pipeline using tf.data. Using tf.data we can map out data to an augmentation pipeline simple by using the map API.\n\nAdding augmentations to the data pipeline is as simple as adding a layer into the list of layers that the Augmenter processes.\n\nReference: https://keras.io/api/keras_cv/layers/augmentation/","metadata":{}},{"cell_type":"code","source":"#from keras_cv.augment.preprocess.augment import RandomFlip\nfrom tensorflow.keras.layers.experimental import preprocessing\n\ndef decode_image_and_label(image_path, label):\n    file_bytes = tf.io.read_file(image_path)\n    image = tf.io.decode_png(file_bytes, channels=3, dtype=tf.uint8)\n    image = tf.image.resize(image, config.IMAGE_SIZE, method=\"bilinear\")\n    image = tf.cast(image, tf.float32) / 255.0\n    \n    label = tf.cast(label, tf.float32)\n    #         bowel       fluid       kidney      liver       spleen\n    labels = (label[0:1], label[1:2], label[2:5], label[5:8], label[8:11])\n    \n    return (image, labels)\n\n\ndef apply_augmentation(images, labels):\n    augmenter = keras.Sequential(\n        \n        layers=[\n            keras_cv.layers.RandomFlip(mode=\"horizontal_and_vertical\"),\n            keras_cv.layers.RandomCutout(height_factor=0.2, width_factor=0.2),\n            \n        ]\n    )\n    aug = augmenter(images)\n    return (aug, labels)\n\ndef aug(image,labels):\n    #image = preprocessing.Rescaling(1.0 / 255)(image)  # Rescale pixel values\n    image = preprocessing.RandomFlip(mode=\"horizontal_and_vertical\")(image)\n    image = preprocessing.RandomRotation(factor=0.2)(image)\n    #image = preprocessing.RandomZoom(height_factor=(0.8, 1.2), width_factor=(0.8, 1.2))(image)\n    #image = preprocessing.RandomContrast(factor=0.2)(image)\n    image = preprocessing.RandomTranslation(height_factor=0.2, width_factor=0.2)(image)\n    return (image, labels)","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:41:49.940382Z","iopub.execute_input":"2023-09-03T04:41:49.940907Z","iopub.status.idle":"2023-09-03T04:41:49.951487Z","shell.execute_reply.started":"2023-09-03T04:41:49.940876Z","shell.execute_reply":"2023-09-03T04:41:49.950649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_dataset(image_paths, labels):\n    ds = (\n        tf.data.Dataset.from_tensor_slices((image_paths, labels))\n        .map(decode_image_and_label, num_parallel_calls=config.AUTOTUNE)\n        .shuffle(config.BATCH_SIZE * 10)\n        .batch(config.BATCH_SIZE)\n        .map(aug, num_parallel_calls=config.AUTOTUNE)\n        .prefetch(config.AUTOTUNE)\n    )\n    return ds","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:41:49.952592Z","iopub.execute_input":"2023-09-03T04:41:49.953449Z","iopub.status.idle":"2023-09-03T04:41:49.96644Z","shell.execute_reply.started":"2023-09-03T04:41:49.953415Z","shell.execute_reply":"2023-09-03T04:41:49.96547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths  = train_data.image_path.tolist()\nlabels = train_data[config.TARGET_COLS].values\n\nds = build_dataset(image_paths=paths, labels=labels)\nimages, labels = next(iter(ds))\nimages.shape, [label.shape for label in labels]","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:41:49.967762Z","iopub.execute_input":"2023-09-03T04:41:49.968292Z","iopub.status.idle":"2023-09-03T04:41:55.395597Z","shell.execute_reply.started":"2023-09-03T04:41:49.968255Z","shell.execute_reply":"2023-09-03T04:41:55.394646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras_cv.visualization.plot_image_gallery(\n    images=images,\n    value_range=(0,1),\n    rows=2,\n    cols=2\n)","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:41:55.397483Z","iopub.execute_input":"2023-09-03T04:41:55.398603Z","iopub.status.idle":"2023-09-03T04:41:55.74388Z","shell.execute_reply.started":"2023-09-03T04:41:55.398561Z","shell.execute_reply":"2023-09-03T04:41:55.742682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"BUILD MODEL","metadata":{}},{"cell_type":"code","source":"def build_model(warmup_steps, decay_steps):\n    # Define Input\n    inputs = keras.Input(shape=config.IMAGE_SIZE + [3,], batch_size=config.BATCH_SIZE)\n    \n    # Define Backbone\n    backbone = keras_cv.models.YOLOV8Backbone.from_preset('yolo_v8_xl_backbone_coco')\n    include_rescaling = False\n    x = inputs\n    \n    # GAP to get the activation maps\n    gap = keras.layers.GlobalAveragePooling2D()\n    x = gap(x)\n\n    # Define 'necks' for each head\n    x_bowel = keras.layers.Dense(32, activation='silu')(x)\n    x_extra = keras.layers.Dense(32, activation='silu')(x)\n    x_liver = keras.layers.Dense(32, activation='silu')(x)\n    x_kidney = keras.layers.Dense(32, activation='silu')(x)\n    x_spleen = keras.layers.Dense(32, activation='silu')(x)\n\n    # Define heads\n    out_bowel = keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel) # use sigmoid to convert predictions to [0-1]\n    out_extra = keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra) # use sigmoid to convert predictions to [0-1]\n    out_liver = keras.layers.Dense(3, name='liver', activation='softmax')(x_liver) # use softmax for the liver head\n    out_kidney = keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney) # use softmax for the kidney head\n    out_spleen = keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen) # use softmax for the spleen head\n    \n    # Concatenate the outputs\n    outputs = [out_bowel, out_extra, out_liver, out_kidney, out_spleen]\n\n    # Create model\n    print(\"[INFO] Building the model...\")\n    model = keras.Model(inputs=inputs, outputs=outputs)\n    \n    # Cosine Decay\n    cosine_decay = keras.optimizers.schedules.CosineDecay(\n        initial_learning_rate=1e-4,\n        decay_steps=decay_steps,\n        alpha=0.0,\n        warmup_target=1e-3,\n        warmup_steps=warmup_steps,\n    )\n\n    # Compile the model\n    optimizer = keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = {\n        \"bowel\":keras.losses.BinaryCrossentropy(),\n        \"extra\":keras.losses.BinaryCrossentropy(),\n        \"liver\":keras.losses.CategoricalCrossentropy(),\n        \"kidney\":keras.losses.CategoricalCrossentropy(),\n        \"spleen\":keras.losses.CategoricalCrossentropy(),\n    }\n    metrics = {\n        \"bowel\":[\"accuracy\"],\n        \"extra\":[\"accuracy\"],\n        \"liver\":[\"accuracy\"],\n        \"kidney\":[\"accuracy\"],\n        \"spleen\":[\"accuracy\"],\n    }\n    print(\"[INFO] Compiling the model...\")\n    model.compile(\n        optimizer=optimizer,\n      loss=loss,\n      metrics=metrics\n    )\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:41:55.745391Z","iopub.execute_input":"2023-09-03T04:41:55.745792Z","iopub.status.idle":"2023-09-03T04:41:55.76151Z","shell.execute_reply.started":"2023-09-03T04:41:55.745757Z","shell.execute_reply":"2023-09-03T04:41:55.760623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get image_paths and labels\nprint(\"[INFO] Building the dataset...\")\ntrain_paths = train_data.image_path.values; train_labels = train_data[config.TARGET_COLS].values.astype(np.float32)\nvalid_paths = val_data.image_path.values; valid_labels = val_data[config.TARGET_COLS].values.astype(np.float32)\n\n# train and valid dataset\ntrain_ds = build_dataset(image_paths=train_paths, labels=train_labels)\nval_ds = build_dataset(image_paths=valid_paths, labels=valid_labels)\n\ntotal_train_steps = train_ds.cardinality().numpy() * config.BATCH_SIZE * config.EPOCHS\nwarmup_steps = int(total_train_steps * 0.10)\ndecay_steps = total_train_steps - warmup_steps\n\nprint(f\"{total_train_steps=}\")\nprint(f\"{warmup_steps=}\")\nprint(f\"{decay_steps=}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:41:55.763611Z","iopub.execute_input":"2023-09-03T04:41:55.764377Z","iopub.status.idle":"2023-09-03T04:41:56.310859Z","shell.execute_reply.started":"2023-09-03T04:41:55.76433Z","shell.execute_reply":"2023-09-03T04:41:56.309812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build the model\nprint(\"[INFO] Building the model...\")\nmodel = build_model(warmup_steps, decay_steps)\n\n# train\nprint(\"[INFO] Training...\")\nhistory = model.fit(\n    train_ds,\n    epochs=config.EPOCHS,\n    validation_data=val_ds,\n)","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:42:14.441234Z","iopub.execute_input":"2023-09-03T04:42:14.441637Z","iopub.status.idle":"2023-09-03T04:56:38.678895Z","shell.execute_reply.started":"2023-09-03T04:42:14.441595Z","shell.execute_reply":"2023-09-03T04:56:38.677884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a 3x2 grid for the subplots\nfig, axes = plt.subplots(5, 1, figsize=(5, 15))\n\n# Flatten axes to iterate through them\naxes = axes.flatten()\n\n# Iterate through the metrics and plot them\nfor i, name in enumerate([\"bowel\", \"extra\", \"kidney\", \"liver\", \"spleen\"]):\n    # Plot training accuracy\n    axes[i].plot(history.history[name + '_accuracy'], label='Training ' + name)\n    # Plot validation accuracy\n    axes[i].plot(history.history['val_' + name + '_accuracy'], label='Validation ' + name)\n    axes[i].set_title(name)\n    axes[i].set_xlabel('Epoch')\n    axes[i].set_ylabel('Accuracy')\n    axes[i].legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:56:38.681749Z","iopub.execute_input":"2023-09-03T04:56:38.682108Z","iopub.status.idle":"2023-09-03T04:56:40.025571Z","shell.execute_reply.started":"2023-09-03T04:56:38.682072Z","shell.execute_reply":"2023-09-03T04:56:40.024676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history[\"loss\"], label=\"loss\")\nplt.plot(history.history[\"val_loss\"], label=\"val loss\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:56:40.02717Z","iopub.execute_input":"2023-09-03T04:56:40.027867Z","iopub.status.idle":"2023-09-03T04:56:40.299392Z","shell.execute_reply.started":"2023-09-03T04:56:40.02783Z","shell.execute_reply":"2023-09-03T04:56:40.29834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# store best results\nbest_epoch = np.argmin(history.history['val_loss'])\nbest_loss = history.history['val_loss'][best_epoch]\nbest_acc_bowel = history.history['val_bowel_accuracy'][best_epoch]\nbest_acc_extra = history.history['val_extra_accuracy'][best_epoch]\nbest_acc_liver = history.history['val_liver_accuracy'][best_epoch]\nbest_acc_kidney = history.history['val_kidney_accuracy'][best_epoch]\nbest_acc_spleen = history.history['val_spleen_accuracy'][best_epoch]\n\n# Find mean accuracy\nbest_acc = np.mean(\n    [best_acc_bowel,\n     best_acc_extra,\n     best_acc_liver,\n     best_acc_kidney,\n     best_acc_spleen\n])\n\n\nprint(f'>>>> BEST Loss  : {best_loss:.3f}\\n>>>> BEST Acc   : {best_acc:.3f}\\n>>>> BEST Epoch : {best_epoch}\\n')\nprint('ORGAN Acc:')\nprint(f'  >>>> {\"Bowel\".ljust(15)} : {best_acc_bowel:.3f}')\nprint(f'  >>>> {\"Extravasation\".ljust(15)} : {best_acc_extra:.3f}')\nprint(f'  >>>> {\"Liver\".ljust(15)} : {best_acc_liver:.3f}')\nprint(f'  >>>> {\"Kidney\".ljust(15)} : {best_acc_kidney:.3f}')\nprint(f'  >>>> {\"Spleen\".ljust(15)} : {best_acc_spleen:.3f}')","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:56:40.301775Z","iopub.execute_input":"2023-09-03T04:56:40.302112Z","iopub.status.idle":"2023-09-03T04:56:40.311685Z","shell.execute_reply.started":"2023-09-03T04:56:40.302079Z","shell.execute_reply":"2023-09-03T04:56:40.310491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:56:40.313475Z","iopub.execute_input":"2023-09-03T04:56:40.314001Z","iopub.status.idle":"2023-09-03T04:56:40.350429Z","shell.execute_reply.started":"2023-09-03T04:56:40.31396Z","shell.execute_reply":"2023-09-03T04:56:40.349533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the model\n#import keras.models\nmodel.save(\"yolov8_xl_atd.keras\")","metadata":{"execution":{"iopub.status.busy":"2023-09-03T04:56:40.351752Z","iopub.execute_input":"2023-09-03T04:56:40.352303Z","iopub.status.idle":"2023-09-03T04:56:40.424474Z","shell.execute_reply.started":"2023-09-03T04:56:40.35227Z","shell.execute_reply":"2023-09-03T04:56:40.423366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}