{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom matplotlib import pyplot as plt\nfrom sklearn.model_selection import train_test_split\n\nclass Config:\n    SEED = 42\n    IMAGE_SIZE = [256, 256]\n    BATCH_SIZE = 64\n    EPOCHS = 5  # Increase the number of epochs\n    TARGET_COLS = [\n        \"bowel_injury\", \"extravasation_injury\",\n        \"kidney_healthy\", \"kidney_low\", \"kidney_high\",\n        \"liver_healthy\", \"liver_low\", \"liver_high\",\n        \"spleen_healthy\", \"spleen_low\", \"spleen_high\",\n    ]\n    AUTOTUNE = tf.data.AUTOTUNE\n\nconfig = Config()\n\nBASE_PATH = f\"/kaggle/input/rsna-atd-512x512-png-v2-dataset\"\n\ndataframe = pd.read_csv(f\"{BASE_PATH}/train.csv\")\ndataframe[\"image_path\"] = f\"{BASE_PATH}/train_images\"\\\n                    + \"/\" + dataframe.patient_id.astype(str)\\\n                    + \"/\" + dataframe.series_id.astype(str)\\\n                    + \"/\" + dataframe.instance_number.astype(str) +\".png\"\ndataframe = dataframe.drop_duplicates()\n\ndef split_group(group, test_size=0.2):\n    if len(group) == 1:\n        return (group, pd.DataFrame()) if np.random.rand() < test_size else (pd.DataFrame(), group)\n    else:\n        return train_test_split(group, test_size=test_size, random_state=42)\n\ntrain_data = pd.DataFrame()\nval_data = pd.DataFrame()\n\nfor _, group in dataframe.groupby(config.TARGET_COLS):\n    train_group, val_group = split_group(group)\n    train_data = pd.concat([train_data, train_group], ignore_index=True)\n    val_data = pd.concat([val_data, val_group], ignore_index=True)\n\ndef decode_image_and_label(image_path, label):\n    file_bytes = tf.io.read_file(image_path)\n    image = tf.io.decode_png(file_bytes, channels=3, dtype=tf.uint8)\n    image = tf.image.resize(image, config.IMAGE_SIZE, method=\"bilinear\")\n    image = tf.cast(image, tf.float32) / 255.0\n    \n    label = tf.cast(label, tf.float32)\n    labels = (label[0:1], label[1:2], label[2:5], label[5:8], label[8:11])\n    \n    return (image, labels)\n\ndef build_dataset(image_paths, labels, is_training=True):\n    ds = (\n        tf.data.Dataset.from_tensor_slices((image_paths, labels))\n        .map(decode_image_and_label, num_parallel_calls=config.AUTOTUNE)\n        .shuffle(config.BATCH_SIZE * 10)\n        .batch(config.BATCH_SIZE)\n        .prefetch(config.AUTOTUNE)\n    )\n    \n    if is_training:\n        # Apply data augmentation using tf.image\n        def apply_augmentation(image, labels):\n            image = tf.image.random_flip_left_right(image)\n            image = tf.image.random_flip_up_down(image)\n            image = tf.image.random_brightness(image, max_delta=0.2)\n            image = tf.image.random_contrast(image, lower=0.8, upper=1.2)\n            image = tf.image.random_saturation(image, lower=0.8, upper=1.2)\n            # Add more augmentations as needed\n            return image, labels\n        \n        ds = ds.map(apply_augmentation, num_parallel_calls=config.AUTOTUNE)\n    \n    return ds\n\npaths = train_data.image_path.tolist()\nlabels = train_data[config.TARGET_COLS].values\n\nds = build_dataset(image_paths=paths, labels=labels, is_training=True)\nimages, labels = next(iter(ds))\n\n# Build InceptionV3 model\ndef build_inception_model():\n    base_model = tf.keras.applications.InceptionV3(weights=None, include_top=False, input_shape=(256, 256, 3))\n    \n    # Load weights from a local file path\n    weights_path = '/kaggle/input/inceptionv3/inception_v3_weights_tf_dim_ordering_tf_kernels_notop.h5'\n    base_model.load_weights(weights_path)\n    \n    for layer in base_model.layers:\n        layer.trainable = False\n\n    x = tf.keras.layers.GlobalAveragePooling2D()(base_model.output)\n    x = tf.keras.layers.Dense(1024, activation='relu')(x)\n    x = tf.keras.layers.Dropout(0.5)(x)\n\n    x_bowel = tf.keras.layers.Dense(16, activation='relu')(x)\n    x_extra = tf.keras.layers.Dense(16, activation='relu')(x)\n    x_liver = tf.keras.layers.Dense(16, activation='relu')(x)\n    x_kidney = tf.keras.layers.Dense(16, activation='relu')(x)\n    x_spleen = tf.keras.layers.Dense(16, activation='relu')(x)\n\n    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel)\n    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra)\n    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver)\n    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney)\n    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen)\n\n    outputs = [out_bowel, out_extra, out_liver, out_kidney, out_spleen]\n\n    model = tf.keras.Model(inputs=base_model.input, outputs=outputs)\n\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4),\n        loss=['binary_crossentropy', 'binary_crossentropy', 'categorical_crossentropy', 'categorical_crossentropy', 'categorical_crossentropy'],\n        metrics=['accuracy']\n    )\n\n    return model\n\n# Train the model\ntrain_paths = train_data.image_path.values\ntrain_labels = train_data[config.TARGET_COLS].values.astype(np.float32)\nvalid_paths = val_data.image_path.values\nvalid_labels = val_data[config.TARGET_COLS].values.astype(np.float32)\n\ntrain_ds = build_dataset(image_paths=train_paths, labels=train_labels, is_training=True)\nval_ds = build_dataset(image_paths=valid_paths, labels=valid_labels, is_training=False)\n\nmodel = build_inception_model()\n\nhistory = model.fit(\n    train_ds,\n    epochs=config.EPOCHS,\n    validation_data=val_ds,\n)\nfig, axes = plt.subplots(5, 1, figsize=(5, 15))\naxes = axes.flatten()\n\nfor i, name in enumerate([\"bowel\", \"extra\", \"kidney\", \"liver\", \"spleen\"]):\n    axes[i].plot(history.history[name + '_accuracy'], label='Training ' + name)\n    axes[i].plot(history.history['val_' + name + '_accuracy'], label='Validation ' + name)\n    axes[i].set_title(name)\n    axes[i].set_xlabel('Epoch')\n    axes[i].set_ylabel('Accuracy')\n    axes[i].legend()\n\nplt.tight_layout()\nplt.show()\n\nplt.plot(history.history[\"loss\"], label=\"loss\")\nplt.plot(history.history[\"val_loss\"], label=\"val loss\")\nplt.legend()\nplt.show()\n\nbest_epoch = np.argmin(history.history['val_loss'])\nbest_loss = history.history['val_loss'][best_epoch]\nbest_acc_bowel = history.history['val_bowel_accuracy'][best_epoch]\nbest_acc_extra = history.history['val_extra_accuracy'][best_epoch]\nbest_acc_liver = history.history['val_liver_accuracy'][best_epoch]\nbest_acc_kidney = history.history['val_kidney_accuracy'][best_epoch]\nbest_acc_spleen = history.history['val_spleen_accuracy'][best_epoch]\n\nbest_acc = np.mean(\n    [best_acc_bowel,\n     best_acc_extra,\n     best_acc_liver,\n     best_acc_kidney,\n     best_acc_spleen\n])\n\nprint(f'>>>> BEST Loss  : {best_loss:.3f}\\n>>>> BEST Acc   : {best_acc:.3f}\\n>>>> BEST Epoch : {best_epoch}\\n')\nprint('ORGAN Acc:')\nprint(f'  >>>> {\"Bowel\".ljust(15)} : {best_acc_bowel:.3f}')\nprint(f'  >>>> {\"Extravasation\".ljust(15)} : {best_acc_extra:.3f}')\nprint(f'  >>>> {\"Liver\".ljust(15)} : {best_acc_liver:.3f}')\nprint(f'  >>>> {\"Kidney\".ljust(15)} : {best_acc_kidney:.3f}')\nprint(f'  >>>> {\"Spleen\".ljust(15)} : {best_acc_spleen:.3f}')\n\nmodel.save(\"rsna-atd_cnn.h5\")\n","metadata":{"execution":{"iopub.status.busy":"2023-09-08T08:16:18.038879Z","iopub.execute_input":"2023-09-08T08:16:18.039254Z","iopub.status.idle":"2023-09-08T08:27:18.835787Z","shell.execute_reply.started":"2023-09-08T08:16:18.039221Z","shell.execute_reply":"2023-09-08T08:27:18.834792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\n\n\n\n# Load the test image paths\nBASE_PATH = \"/kaggle/input/rsna-atd-512x512-png-v2-dataset\"\ntest_image_info = [\n    (\"48843\", \"62825\", \"30.png\"),\n    (\"50046\", \"24574\", \"30.png\"),\n    (\"63706\", \"39279\", \"30.png\"),\n]\n\ntest_paths = []\n\nfor patient_id, series_id, image_name in test_image_info:\n    image_path = os.path.join(\n        BASE_PATH, \"test_images\", patient_id, series_id, image_name\n    )\n    test_paths.append(image_path)\n\n# Load and preprocess the test data\ndef decode_image(image_path):\n    file_bytes = tf.io.read_file(image_path)\n    image = tf.io.decode_png(file_bytes, channels=3, dtype=tf.uint8)\n    image = tf.image.resize(image, config.IMAGE_SIZE, method=\"bilinear\")\n    image = tf.cast(image, tf.float32) / 255.0\n    return image\n\ntest_images = [decode_image(image_path) for image_path in test_paths]\ntest_images = tf.stack(test_images)\n\n# Create a DataFrame to store the predictions\nsubmission_df = pd.DataFrame()\n\n# Load your saved model\nmodel = tf.keras.models.load_model(\"/kaggle/working/rsna-atd_cnn.h5\")\n\n# Make predictions on the test data\npredictions = model.predict(test_images)\n\n# Organize predictions and save them to the submission DataFrame\nsubmission_df[\"patient_id\"] = [int(patient_id) for patient_id, _, _ in test_image_info]\nsubmission_df[\"bowel_healthy\"] = 1 - predictions[0]\nsubmission_df[\"bowel_injury\"] = predictions[0]\nsubmission_df[\"extravasation_healthy\"] = 1 - predictions[1]\nsubmission_df[\"extravasation_injury\"] = predictions[1]\nsubmission_df[\"kidney_healthy\"] = predictions[2][:, 0]\nsubmission_df[\"kidney_low\"] = predictions[2][:, 1]\nsubmission_df[\"kidney_high\"] = predictions[2][:, 2]\nsubmission_df[\"liver_healthy\"] = predictions[3][:, 0]\nsubmission_df[\"liver_low\"] = predictions[3][:, 1]\nsubmission_df[\"liver_high\"] = predictions[3][:, 2]\nsubmission_df[\"spleen_healthy\"] = predictions[4][:, 0]\nsubmission_df[\"spleen_low\"] = predictions[4][:, 1]\nsubmission_df[\"spleen_high\"] = predictions[4][:, 2]\n\n# Save the submission DataFrame to a CSV file\nsubmission_df.to_csv(\"submission.csv\", index=False)\n\nprint(\"Submission CSV file saved successfully.\")\n","metadata":{"execution":{"iopub.status.busy":"2023-09-08T08:27:38.507429Z","iopub.execute_input":"2023-09-08T08:27:38.507828Z","iopub.status.idle":"2023-09-08T08:27:44.946296Z","shell.execute_reply.started":"2023-09-08T08:27:38.507798Z","shell.execute_reply":"2023-09-08T08:27:44.94522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.read_csv(\"/kaggle/working/submission.csv\")\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2023-09-08T08:28:24.552402Z","iopub.execute_input":"2023-09-08T08:28:24.553398Z","iopub.status.idle":"2023-09-08T08:28:24.584339Z","shell.execute_reply.started":"2023-09-08T08:28:24.553355Z","shell.execute_reply":"2023-09-08T08:28:24.583306Z"},"trusted":true},"execution_count":null,"outputs":[]}]}