{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\nIn this notebook, we tested a training procedure without converting all the DICOM image files into png or jpeg.\nFor this we used the tensorflow conversion method and then used a tensorflow dataset creation method to create the training and validation data.\nWe applied the method to a sample of 1000 images.\nUsing an image size of 128 x 128, the training time on 1000 images is about 13min.","metadata":{}},{"cell_type":"markdown","source":"# Importing fundamental libraries and selected all dicom files\n","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nfrom os import path\nimport itertools","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:51:53.954837Z","iopub.execute_input":"2023-01-23T22:51:53.955302Z","iopub.status.idle":"2023-01-23T22:51:53.960697Z","shell.execute_reply.started":"2023-01-23T22:51:53.955267Z","shell.execute_reply":"2023-01-23T22:51:53.959495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To naturally sort all files in alphanumeric order\n!pip install natsort","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:51:57.023849Z","iopub.execute_input":"2023-01-23T22:51:57.024299Z","iopub.status.idle":"2023-01-23T22:52:10.615897Z","shell.execute_reply.started":"2023-01-23T22:51:57.024252Z","shell.execute_reply":"2023-01-23T22:52:10.614777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob \nfrom natsort import natsorted\nfiles_all = glob.glob('/kaggle/input/rsna-breast-cancer-detection/train_images/**/*.dcm', \n                   recursive = True)\nfiles_all_sort= natsorted(files_all) # sorted dicom files (.dcm extension) according to numeric number in each files\nprint(\"number of dicom files in directory:\",len(files_all_sort))","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:52:16.534654Z","iopub.execute_input":"2023-01-23T22:52:16.535118Z","iopub.status.idle":"2023-01-23T22:53:12.134802Z","shell.execute_reply.started":"2023-01-23T22:52:16.535078Z","shell.execute_reply":"2023-01-23T22:53:12.133495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# First 5 files\nfiles_all_sort[0:5]","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:53:12.136499Z","iopub.execute_input":"2023-01-23T22:53:12.13684Z","iopub.status.idle":"2023-01-23T22:53:12.147481Z","shell.execute_reply.started":"2023-01-23T22:53:12.136808Z","shell.execute_reply":"2023-01-23T22:53:12.146038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Formating csv files of train\nThis will be used to sort train dataset in the same order as files in the directory","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\n\ntrain['slach']= train.apply(lambda x: '/', axis = 1)\ntrain['patient_id']= train['patient_id'].astype(str)\ntrain['image_id']= train['image_id'].astype(str)\n# creating another id (final_id) for correspondance in directory (for example: 10006/1459541791 )\ntrain['final_id'] = train[['patient_id', 'slach', 'image_id']].agg(' '.join, axis=1)\n# sorting train.csv\nfrom natsort import index_natsorted\ntrain_sort = train.sort_values(\n    by=\"final_id\",\n    key=lambda x: np.argsort(index_natsorted(train[\"final_id\"])))","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:53:21.399375Z","iopub.execute_input":"2023-01-23T22:53:21.39984Z","iopub.status.idle":"2023-01-23T22:53:24.470945Z","shell.execute_reply.started":"2023-01-23T22:53:21.399802Z","shell.execute_reply":"2023-01-23T22:53:24.46974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Verify correspondance with files in sorted directrory. Sems cery good.\ntrain_sort.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:53:24.473771Z","iopub.execute_input":"2023-01-23T22:53:24.47428Z","iopub.status.idle":"2023-01-23T22:53:24.499455Z","shell.execute_reply.started":"2023-01-23T22:53:24.474213Z","shell.execute_reply":"2023-01-23T22:53:24.498287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing \n","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nimport tensorflow_io as tfio\nfrom tensorflow.keras.models import Sequential \nfrom keras.callbacks import EarlyStopping","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:53:47.316913Z","iopub.execute_input":"2023-01-23T22:53:47.317346Z","iopub.status.idle":"2023-01-23T22:53:54.399039Z","shell.execute_reply.started":"2023-01-23T22:53:47.317312Z","shell.execute_reply":"2023-01-23T22:53:54.397803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training and validation datasets","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split  # used to split data as training and test sets\n# We used a set of 10000 images for the demonstration.\ny = train_sort['cancer'].values[:1000]\nX = files_all_sort[:1000]\n# We will keep 20% (test size) of data in validation set \nX_train, X_val, y_train, y_val = train_test_split(X, y,test_size=0.2, random_state=0 )\n\n# Creating tensorflow like dataset with images directory and corresponding label\ntrain_ds = tf.data.Dataset.from_tensor_slices((X_train,y_train))\nval_ds = tf.data.Dataset.from_tensor_slices((X_val,y_val))\n\nprint(\"X_train size:\" , len(X_train))\nprint(\"X_val size:\" , len(X_val))\nprint(\"train_ds:\" , train_ds)\nprint(\"val_ds:\" , val_ds)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:53:54.401083Z","iopub.execute_input":"2023-01-23T22:53:54.401764Z","iopub.status.idle":"2023-01-23T22:53:54.462388Z","shell.execute_reply.started":"2023-01-23T22:53:54.401725Z","shell.execute_reply":"2023-01-23T22:53:54.461074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Verification of counts : 0=no bresat cancer, 1= breast cancer\npd.DataFrame(y).value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:54:01.279927Z","iopub.execute_input":"2023-01-23T22:54:01.280389Z","iopub.status.idle":"2023-01-23T22:54:01.293011Z","shell.execute_reply.started":"2023-01-23T22:54:01.280352Z","shell.execute_reply":"2023-01-23T22:54:01.292059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Decode DICOM images \n","metadata":{}},{"cell_type":"code","source":"# We used methods developped by Marcelo Lerendegui and Ouwen Huang: \n# https://www.g.org/io/api_docs/python/tfio/image/decode_dicom_image\ndef get_image_train(X_train):\n    image_bytes = tf.io.read_file(X_train)\n    image = tfio.image.decode_dicom_image(image_bytes, scale='auto', dtype=tf.uint16)\n    image= tf.image.resize(image, (128, 128))\n    #image= tf.image.per_image_standardization(image)\n    return image\n\ndef get_image_val(X_val):\n    image_bytes = tf.io.read_file(X_val)\n    image = tfio.image.decode_dicom_image(image_bytes, scale='auto', dtype=tf.uint16)\n    image= tf.image.resize(image, (128, 128))\n    #image= tf.image.per_image_standardization(image)\n    return image","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:54:08.166629Z","iopub.execute_input":"2023-01-23T22:54:08.16722Z","iopub.status.idle":"2023-01-23T22:54:08.179714Z","shell.execute_reply.started":"2023-01-23T22:54:08.167157Z","shell.execute_reply":"2023-01-23T22:54:08.177121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Define functions to obtain Labels","metadata":{}},{"cell_type":"code","source":"def get_label_train(y_train):\n    label=y_train\n    return label\n\ndef get_label_val(y_val):\n    label=y_val\n    return label","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:54:30.015934Z","iopub.execute_input":"2023-01-23T22:54:30.017297Z","iopub.status.idle":"2023-01-23T22:54:30.023112Z","shell.execute_reply.started":"2023-01-23T22:54:30.017231Z","shell.execute_reply":"2023-01-23T22:54:30.021534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Final functions combining image files and labels","metadata":{}},{"cell_type":"code","source":"def dataset_process_train(X_train, y_train):\n    image=get_image_train(X_train)\n    label=get_label_train(y_train)\n    return image,label\n\ndef dataset_process_val(X_val, y_val):\n    image=get_image_val(X_val)\n    label=get_label_val(y_val)\n    return image,label","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:54:39.685589Z","iopub.execute_input":"2023-01-23T22:54:39.685995Z","iopub.status.idle":"2023-01-23T22:54:39.69243Z","shell.execute_reply.started":"2023-01-23T22:54:39.685964Z","shell.execute_reply":"2023-01-23T22:54:39.691203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Mapping images and labels to train and validation sets ","metadata":{}},{"cell_type":"code","source":"train_ds = train_ds.map(dataset_process_train,  num_parallel_calls=tf.data.AUTOTUNE)\nval_ds = val_ds.map(dataset_process_val,  num_parallel_calls=tf.data.AUTOTUNE)\ntrain_ds, val_ds","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:54:45.28528Z","iopub.execute_input":"2023-01-23T22:54:45.285842Z","iopub.status.idle":"2023-01-23T22:54:45.909625Z","shell.execute_reply.started":"2023-01-23T22:54:45.285794Z","shell.execute_reply":"2023-01-23T22:54:45.908333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Scale images\nIn tensorflow dicom decoder, we used tf.uint16 with pixels value varing from 0 to 65535","metadata":{}},{"cell_type":"code","source":"train_ds= train_ds.map(lambda x,y: (x/65535, y),num_parallel_calls=tf.data.AUTOTUNE)\nval_ds= val_ds.map(lambda x,y: (x/65535, y),num_parallel_calls=tf.data.AUTOTUNE)  \ntrain_ds, val_ds","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:54:52.177617Z","iopub.execute_input":"2023-01-23T22:54:52.178149Z","iopub.status.idle":"2023-01-23T22:54:52.256205Z","shell.execute_reply.started":"2023-01-23T22:54:52.178103Z","shell.execute_reply":"2023-01-23T22:54:52.254775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Samples of images and tensors","metadata":{}},{"cell_type":"code","source":"train_ds_iterator=train_ds.as_numpy_iterator()\nimport matplotlib.pyplot as plt \nfor image, label in train_ds_iterator:\n    image= np.reshape(image[0], (128,128,1))\n    print(\"minimal pixel of image:\",np.min(image))\n    print(\"maximal pixel of image:\",np.max(image))\n    print('minimal  pixel tf:', tf.math.reduce_min(image))\n    print('maximal  pixel tf:', tf.math.reduce_max(image))\n    plt.imshow(image, cmap=\"gray\")\n    plt.title(print(label)) \n    break","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:54:57.688272Z","iopub.execute_input":"2023-01-23T22:54:57.688669Z","iopub.status.idle":"2023-01-23T22:55:00.203775Z","shell.execute_reply.started":"2023-01-23T22:54:57.688636Z","shell.execute_reply":"2023-01-23T22:55:00.202471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i,l in train_ds.take(1):\n    print(i)\n    i= np.reshape(i[0], (128,128,1))\n    plt.imshow(i, cmap=\"gray\")\n    print(\"minimal pixel of i:\",np.min(i))\n    print(\"maximal pixel of i:\",np.max(i))\n    print('minimal  pixel tf:', tf.math.reduce_min(i))\n    print('maximal  pixel tf:', tf.math.reduce_max(i))\n    plt.title(\"label\")\n    #break\n","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:55:07.828282Z","iopub.execute_input":"2023-01-23T22:55:07.829632Z","iopub.status.idle":"2023-01-23T22:55:12.122732Z","shell.execute_reply.started":"2023-01-23T22:55:07.829578Z","shell.execute_reply":"2023-01-23T22:55:12.121569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tensorflow preprocessing for fast training\ncaching, batching and prefetch. We have already shuffle datasets during split\n","metadata":{}},{"cell_type":"code","source":"# To preserve train and validation datasets\ntraining_dataset= train_ds \nvalidation_dataset= val_ds ","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:55:30.295823Z","iopub.execute_input":"2023-01-23T22:55:30.296267Z","iopub.status.idle":"2023-01-23T22:55:30.301421Z","shell.execute_reply.started":"2023-01-23T22:55:30.296219Z","shell.execute_reply":"2023-01-23T22:55:30.300253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_dataset = training_dataset.cache()\n#training_dataset = training_dataset.shuffle(len(training_dataset))\ntraining_dataset = training_dataset.batch(64).prefetch(tf.data.AUTOTUNE)\n#training_dataset = training_dataset.batch(64)\n#training_dataset = training_dataset.prefetch(tf.data.AUTOTUNE)\n#shape_file =training_dataset.map(shape_file)\nlen(training_dataset), training_dataset\n","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:55:36.987343Z","iopub.execute_input":"2023-01-23T22:55:36.987758Z","iopub.status.idle":"2023-01-23T22:55:36.999862Z","shell.execute_reply.started":"2023-01-23T22:55:36.987725Z","shell.execute_reply":"2023-01-23T22:55:36.998621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_dataset = validation_dataset.batch(64)\nvalidation_dataset = validation_dataset.cache()\nvalidationdataset = validation_dataset.prefetch(tf.data.AUTOTUNE)\nlen(validation_dataset), validation_dataset","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:55:43.769761Z","iopub.execute_input":"2023-01-23T22:55:43.770189Z","iopub.status.idle":"2023-01-23T22:55:43.782131Z","shell.execute_reply.started":"2023-01-23T22:55:43.770156Z","shell.execute_reply":"2023-01-23T22:55:43.780873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define simple model for training and save the model","metadata":{}},{"cell_type":"code","source":"# Define a simple sequential model\ndef create_model():\n    model = Sequential()\n    model.add(layers.InputLayer(input_shape=(128,128)))\n    model.add(layers.Reshape((128,128,1)))\n    model.add(layers.Conv2D(filters=64, kernel_size=3, activation='relu', padding='same'))\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Dense(32, activation='relu'))\n    model.add(layers.Dense(32, activation='relu'))\n    model.add(layers.Dense(1, activation='sigmoid'))\n\n\n    model.compile(\n    optimizer=tf.keras.optimizers.Adam(epsilon=0.01),\n    loss = 'binary_crossentropy',\n    metrics=['binary_accuracy']\n)\n    return model\n\nes = tf.keras.callbacks.EarlyStopping(patience=2)\n\n# Create a basic model instance\nmodel = create_model()\n\n# Display the model's architecture\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:55:50.143402Z","iopub.execute_input":"2023-01-23T22:55:50.144774Z","iopub.status.idle":"2023-01-23T22:55:50.257979Z","shell.execute_reply.started":"2023-01-23T22:55:50.144716Z","shell.execute_reply":"2023-01-23T22:55:50.256836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create and train a new model instance.\nfrom tqdm import tqdm\nfor i in tqdm(range(10)):\n    model = create_model()\n    model.fit(\n    training_dataset,\n    validation_data= validation_dataset,\n    epochs=10,\n    callbacks =[es]\n)\n\n# Save the entire model as a SavedModel.\n!mkdir -p saved_model\nmodel.save('saved_model/my_model')","metadata":{"execution":{"iopub.status.busy":"2023-01-23T22:56:00.769518Z","iopub.execute_input":"2023-01-23T22:56:00.769942Z"},"trusted":true},"execution_count":null,"outputs":[]}]}