{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfrom glob import glob\nimport pydicom\nimport tensorflow as tf\nimport tqdm as tqdm\nimport tensorflow_io as tfio\nimport pathlib\nimport datetime\n# tensorboard\n%load_ext tensorboard\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\n\n\"\"\"\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\"\"\"\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":5.477112,"end_time":"2022-08-01T23:24:50.07098","exception":false,"start_time":"2022-08-01T23:24:44.593868","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-22T23:15:48.245752Z","iopub.execute_input":"2022-08-22T23:15:48.247171Z","iopub.status.idle":"2022-08-22T23:15:54.273944Z","shell.execute_reply.started":"2022-08-22T23:15:48.247114Z","shell.execute_reply":"2022-08-22T23:15:54.272905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q ../input/for-pydicom/pylibjpeg-1.4.0-py3-none-any.whl\n!pip install -q ../input/for-pydicom/python_gdcm-3.0.14-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n!pip install -q ../input/for-pydicom/pylibjpeg_libjpeg-1.3.1-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl","metadata":{"execution":{"iopub.status.busy":"2022-08-22T23:14:41.158269Z","iopub.execute_input":"2022-08-22T23:14:41.15869Z","iopub.status.idle":"2022-08-22T23:15:12.993993Z","shell.execute_reply.started":"2022-08-22T23:14:41.158607Z","shell.execute_reply":"2022-08-22T23:15:12.992666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating the Dataset tf.Dataset.from_tensor_slices #","metadata":{"papermill":{"duration":0.003234,"end_time":"2022-08-01T23:24:50.076828","exception":false,"start_time":"2022-08-01T23:24:50.073594","status":"completed"},"tags":[]}},{"cell_type":"code","source":"class cl_CreatingDataset:\n    def __init__(self, imageHeight, imageWidth, batch_size):\n        self.imageHeight = imageHeight\n        self.imageWidth = imageWidth\n        self.batch_size = batch_size\n\n    \n    def creatingPathList(self, trainImagesPath, trainDfPath):\n        trainDf = pd.read_csv(trainDfPath)\n        trainImagesPathList = []\n        labels = []\n        # reading the tain CSV\n        trainDf = pd.read_csv(trainDfPath)\n        for i in tqdm.tqdm(range(len(trainDf))):\n            folderName = trainDf[\"StudyInstanceUID\"].iloc[i]\n            imageFolderPath = os.path.join(trainImagesPath, folderName)\n            for file in glob(os.path.join(imageFolderPath, \"*.dcm\")):\n                # taking the imageName\n                trainImagesPathList.append(file)\n                # creating the labels\n                label = np.array([trainDf[\"C1\"].iloc[i],trainDf[\"C2\"].iloc[i], trainDf[\"C3\"].iloc[i], trainDf[\"C4\"].iloc[i], trainDf[\"C5\"].iloc[i], \n                                 trainDf[\"C6\"].iloc[i], trainDf[\"C7\"].iloc[i], trainDf[\"patient_overall\"].iloc[i]])\n                labels.append(label)\n        \n        return trainImagesPathList, labels\n    \n    def parse_function(self, filename, label):\n        image_bytes = tf.io.read_file(filename)\n        image = tfio.image.decode_dicom_image(image_bytes, dtype=tf.float32)\n        #image = tf.image.convert_image_dtype(image, tf.float32)\n        resized_image = tf.image.resize(image, [self.imageHeight, self.imageWidth])\n        compressedImage = tf.squeeze(resized_image, axis = 0)\n        finalImage = tf.repeat(compressedImage, repeats=[3], axis = 2)\n        return finalImage, label\n        \n    def train_preprocess(self, image, label):\n        return image, label\n\n\n    def creatingDataset(self, filenames, labels):\n        dataset = tf.data.Dataset.from_tensor_slices((filenames, labels))\n        dataset = dataset.shuffle(len(filenames))\n        dataset = dataset.map(self.parse_function, num_parallel_calls=4)\n        dataset = dataset.map(self.train_preprocess, num_parallel_calls=4)\n        dataset = dataset.batch(self.batch_size)\n        dataset = dataset.prefetch(1)\n\n        return dataset","metadata":{"papermill":{"duration":0.01753,"end_time":"2022-08-01T23:24:50.096873","exception":false,"start_time":"2022-08-01T23:24:50.079343","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating Dataset using Generator Concept (tf.data.Dataset.fromgenerator) #","metadata":{}},{"cell_type":"code","source":"DATA_DIR = \"../input/rsna-2022-cervical-spine-fracture-detection/\"\ntrainCsv = \"../input/rsna-2022-cervical-spine-fracture-detection/train.csv\"\ntrain_df = pd.read_csv(trainCsv)\nprint(train_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-22T23:16:00.832305Z","iopub.execute_input":"2022-08-22T23:16:00.832942Z","iopub.status.idle":"2022-08-22T23:16:00.859797Z","shell.execute_reply.started":"2022-08-22T23:16:00.832903Z","shell.execute_reply":"2022-08-22T23:16:00.858876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dicom(path):\n    \"\"\"\n    reads a dicom file and loads the image array inside it\n    inputs:\n        path: the path of the required dicom file\n    returns:\n        data: image pixel arrays\n    \"\"\"\n    img=pydicom.dcmread(path)\n    data=img.pixel_array\n    data=data-np.min(data)\n    if np.max(data) != 0:\n        data=data/np.max(data)\n    data=(data*255).astype(np.uint8)\n    return data","metadata":{"execution":{"iopub.status.busy":"2022-08-22T23:16:02.764297Z","iopub.execute_input":"2022-08-22T23:16:02.76502Z","iopub.status.idle":"2022-08-22T23:16:02.771013Z","shell.execute_reply.started":"2022-08-22T23:16:02.764985Z","shell.execute_reply":"2022-08-22T23:16:02.769775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_generator():\n    for i, study_instance in enumerate(train_df.StudyInstanceUID[:]):\n        for dcm in os.listdir(DATA_DIR + f\"train_images/{study_instance}\"):\n            train_labels = []\n            path = DATA_DIR + f\"train_images/{study_instance}/{dcm}\"\n            img = load_dicom(path)\n            \n            # resize each image into a shape of (512, 512)\n            img = np.resize(img, (512, 512))\n            #  normalize image\n            img = img / 255.0\n            # convert from gray scale to rgb, this will be helpful incase we want to use pretrained models\n            img = tf.expand_dims(img, axis=-1)\n            img = tf.image.grayscale_to_rgb(img)\n            \n            train_labels.extend([\n                train_df.loc[i, \"C1\"],\n                train_df.loc[i, \"C2\"],\n                train_df.loc[i, \"C3\"],\n                train_df.loc[i, \"C4\"],\n                train_df.loc[i, \"C5\"],\n                train_df.loc[i, \"C6\"],\n                train_df.loc[i, \"C7\"],\n                train_df.loc[i, \"patient_overall\"] # end with patient overall\n            ])\n            yield img, train_labels","metadata":{"execution":{"iopub.status.busy":"2022-08-22T23:16:05.070249Z","iopub.execute_input":"2022-08-22T23:16:05.070641Z","iopub.status.idle":"2022-08-22T23:16:05.07956Z","shell.execute_reply.started":"2022-08-22T23:16:05.070608Z","shell.execute_reply":"2022-08-22T23:16:05.0785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Splitting the dataset ##","metadata":{}},{"cell_type":"code","source":"def splitDataset(dataset, trainFactor, img_count): # here it refers to tf.dataset\n    train_dataset = dataset.take(int(trainFactor * img_count))\n    validation_dataset = dataset.take(int((1 - trainFactor)* img_count))\n    \n    return train_dataset, validation_dataset\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-22T23:16:07.085274Z","iopub.execute_input":"2022-08-22T23:16:07.086199Z","iopub.status.idle":"2022-08-22T23:16:07.092156Z","shell.execute_reply.started":"2022-08-22T23:16:07.086155Z","shell.execute_reply":"2022-08-22T23:16:07.091068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def configure_for_performance(data):\n    data = data.cache()\n    data = data.batch(16)\n    data = data.prefetch(buffer_size=tf.data.AUTOTUNE)\n    return data","metadata":{"execution":{"iopub.status.busy":"2022-08-22T23:16:08.994457Z","iopub.execute_input":"2022-08-22T23:16:08.994976Z","iopub.status.idle":"2022-08-22T23:16:09.00552Z","shell.execute_reply.started":"2022-08-22T23:16:08.99493Z","shell.execute_reply":"2022-08-22T23:16:09.004155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating the model ##","metadata":{}},{"cell_type":"code","source":"def creating_model(opt):\n    IMG_SHAPE = (512, 512, 3)\n    base_model = tf.keras.applications.EfficientNetB5(input_shape=IMG_SHAPE,\n                                               include_top=False,\n                                               weights='imagenet')\n    \n    base_model.trainable = False\n    inputs = tf.keras.Input(shape=IMG_SHAPE)\n    x = base_model(inputs)\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    x = tf.keras.layers.Dropout(0.2)(x)\n    outputs = tf.keras.layers.Dense(8, activation = 'sigmoid')(x)\n    model = tf.keras.Model(inputs, outputs)\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-08-22T23:16:10.509748Z","iopub.execute_input":"2022-08-22T23:16:10.510722Z","iopub.status.idle":"2022-08-22T23:16:10.517753Z","shell.execute_reply.started":"2022-08-22T23:16:10.510606Z","shell.execute_reply":"2022-08-22T23:16:10.516752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"trainCsv = \"../input/rsna-2022-cervical-spine-fracture-detection/train.csv\"\ntrainImagePath = \"../input/rsna-2022-cervical-spine-fracture-detection/train_images\"\ncreDataset = cl_CreatingDataset(224, 224, 4)\ntrainImagesPathList, labels = creDataset.creatingPathList(trainImagePath, trainCsv)\nprint(\"train Image list :\", len(trainImagePath), \".....\", \"train label list : \", len(labels))\n","metadata":{"execution":{"iopub.status.busy":"2022-08-14T21:02:18.871133Z","iopub.execute_input":"2022-08-14T21:02:18.871793Z","iopub.status.idle":"2022-08-14T21:05:46.011647Z","shell.execute_reply.started":"2022-08-14T21:02:18.871754Z","shell.execute_reply":"2022-08-14T21:05:46.010225Z"}}},{"cell_type":"code","source":"# this is required because of length of dataset is not valid for generators\ndef getImageCount():\n    img_count = 0\n    for _, study_instance in enumerate(train_df.StudyInstanceUID[:5]):\n        for _ in os.listdir(DATA_DIR + f\"train_images/{study_instance}\"):\n            img_count += 1\n            \n    return img_count","metadata":{"execution":{"iopub.status.busy":"2022-08-22T23:16:13.618328Z","iopub.execute_input":"2022-08-22T23:16:13.618677Z","iopub.status.idle":"2022-08-22T23:16:13.625736Z","shell.execute_reply.started":"2022-08-22T23:16:13.618647Z","shell.execute_reply":"2022-08-22T23:16:13.624577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Callbacks ##","metadata":{}},{"cell_type":"code","source":"def createTensorboardCallback(logdir):\n    return tf.keras.callbacks.TensorBoard(log_dir=logdir, histogram_freq=1)\n\n    \ndef saveModelCallback(filePath):\n    checkpoint = tf.keras.callbacks.ModelCheckpoint(filepath=filePath, monitor='val_loss',verbose=1, save_best_only=True, mode='min')\n    \n    return checkpoint\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-22T23:18:02.100304Z","iopub.execute_input":"2022-08-22T23:18:02.101028Z","iopub.status.idle":"2022-08-22T23:18:02.107363Z","shell.execute_reply.started":"2022-08-22T23:18:02.100975Z","shell.execute_reply":"2022-08-22T23:18:02.106139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Displaying the training graphs ##","metadata":{}},{"cell_type":"code","source":"def display(history):\n    plt.plot(history.history['binary_accuracy'])\n    plt.plot(history.history['val_binary_accuracy'])\n    plt.title('model accuracy')\n    plt.ylabel('accuracy')\n    plt.xlabel('epoch')\n    plt.legend(['train', 'test'], loc='upper left')\n    plt.show()\n    # summarize history for loss\n    plt.plot(history.history['loss'])\n    plt.plot(history.history['val_loss'])\n    plt.title('model loss')\n    plt.ylabel('loss')\n    plt.xlabel('epoch')\n    plt.legend(['train', 'test'], loc='upper left')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-22T23:30:48.239966Z","iopub.execute_input":"2022-08-22T23:30:48.240473Z","iopub.status.idle":"2022-08-22T23:30:48.250612Z","shell.execute_reply.started":"2022-08-22T23:30:48.240439Z","shell.execute_reply":"2022-08-22T23:30:48.249539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.compat.v1.logging.set_verbosity(tf.compat.v1.logging.ERROR)\n# creating the dataset\ndataset = tf.data.Dataset.from_generator(data_generator, (tf.float32, tf.int8))\n\n# printing a sample data for checking\nfor img, label in dataset.take(1):\n    print(img.shape)\n    print(label.shape)\n    print(label)\n\n# splitting the dataset\ntrainFactor = 0.8\nimg_count = getImageCount()\nprint(\"[***] Images for training and validation : \", img_count)\ntrain_data, validation_data = splitDataset(dataset, trainFactor, img_count)\n#\ntrain_dataset = configure_for_performance(train_data)\nvalidation_dataset = configure_for_performance(validation_data)\n\n# creating and compiling the model\nINIT_LR = 1e-3\nopt = tf.keras.optimizers.Adam(learning_rate=INIT_LR)\nmodel = creating_model(opt)\n#model = alex_net()\nmodel.summary()\nmodel.compile(optimizer=tf.keras.optimizers.Adam(), \n              loss=tf.keras.losses.BinaryCrossentropy(),\n              metrics=[tf.keras.metrics.BinaryAccuracy()])\n# training\nEPOCHS = 10\nBATCH_SIZE = 4\n# tensorboard logging dir\nfoldername = \"/kaggle/working/tensorboardRecord\"\nos.makedirs(foldername, exist_ok=True)\nlogdir = os.path.join(foldername, datetime.datetime.now().strftime(\"%Y%m%d-%H%M%S\"))\n\n\n# modelsave checkpoint\nmodelDir = \"/kaggle/working/modelDir\"\nos.makedirs(modelDir, exist_ok=True)\nfilepath = 'my_best_model.epoch{epoch:02d}-loss{val_loss:.2f}.hdf5'\nmodelSavePath = os.path.join(modelDir, filepath)\n\ntensorboardCallback = createTensorboardCallback(logdir)\nmodeSaveCallback = saveModelCallback(modelSavePath)\nhistory = model.fit(train_dataset, batch_size = BATCH_SIZE, epochs = EPOCHS, validation_data = validation_dataset, callbacks=[modeSaveCallback])\n","metadata":{"execution":{"iopub.status.busy":"2022-08-22T23:19:10.978221Z","iopub.execute_input":"2022-08-22T23:19:10.978856Z","iopub.status.idle":"2022-08-22T23:26:56.169117Z","shell.execute_reply.started":"2022-08-22T23:19:10.978803Z","shell.execute_reply":"2022-08-22T23:26:56.168045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ndisplay(history)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T23:31:28.478686Z","iopub.execute_input":"2022-08-22T23:31:28.479147Z","iopub.status.idle":"2022-08-22T23:31:29.037207Z","shell.execute_reply.started":"2022-08-22T23:31:28.479104Z","shell.execute_reply":"2022-08-22T23:31:29.036034Z"},"trusted":true},"execution_count":null,"outputs":[]}]}