{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%capture\n# Source: https://www.kaggle.com/code/remekkinas/fast-dicom-processing-1-6-2x-faster?scriptVersionId=113360473\n\n!pip install /kaggle/input/rsnamodules/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl \n\ntry:\n    import pylibjpeg\nexcept:\n   !pip install /kaggle/input/rsna-2022-whl/{pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom matplotlib import pyplot as plt\nimport pathlib\nimport os\n\n# Look at the pipeline module for ROI cropping\n# https://www.kaggle.com/code/achoxic/pipeline\nimport pipeline as pip\n\nfrom typing import Tuple","metadata":{"execution":{"iopub.status.busy":"2023-01-09T19:38:01.615652Z","iopub.execute_input":"2023-01-09T19:38:01.616567Z","iopub.status.idle":"2023-01-09T19:38:04.136072Z","shell.execute_reply.started":"2023-01-09T19:38:01.616442Z","shell.execute_reply":"2023-01-09T19:38:04.134789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tensorflow is very verbose.\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '1'","metadata":{"execution":{"iopub.status.busy":"2023-01-09T19:38:04.137955Z","iopub.execute_input":"2023-01-09T19:38:04.138598Z","iopub.status.idle":"2023-01-09T19:38:04.143718Z","shell.execute_reply.started":"2023-01-09T19:38:04.138562Z","shell.execute_reply":"2023-01-09T19:38:04.142425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_DIR = pathlib.Path('/kaggle/input/rsna-breast-cancer-detection')","metadata":{"execution":{"iopub.status.busy":"2023-01-09T19:38:04.145167Z","iopub.execute_input":"2023-01-09T19:38:04.145504Z","iopub.status.idle":"2023-01-09T19:38:04.158453Z","shell.execute_reply.started":"2023-01-09T19:38:04.145473Z","shell.execute_reply":"2023-01-09T19:38:04.157133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv_path = INPUT_DIR / 'train.csv'\ntrain_df = pd.read_csv(train_csv_path)","metadata":{"execution":{"iopub.status.busy":"2023-01-09T19:38:04.161708Z","iopub.execute_input":"2023-01-09T19:38:04.162206Z","iopub.status.idle":"2023-01-09T19:38:04.256565Z","shell.execute_reply.started":"2023-01-09T19:38:04.162169Z","shell.execute_reply":"2023-01-09T19:38:04.255068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv_path = INPUT_DIR / 'train.csv'\ntrain_images_dir = INPUT_DIR / 'train_images'\noutput_path = pathlib.Path('/kaggle/working/')\n\nopts = pip.Options(\n  patients_csv_path=train_csv_path,\n  root_dir=train_images_dir,\n  output_path=output_path,\n  output_shape=(384, 768),\n  num_examples_per_record=50,\n)\n\npip.make_dataset(opts)","metadata":{"execution":{"iopub.status.busy":"2023-01-09T19:41:23.915653Z","iopub.execute_input":"2023-01-09T19:41:23.917367Z","iopub.status.idle":"2023-01-09T19:51:03.606919Z","shell.execute_reply.started":"2023-01-09T19:41:23.917299Z","shell.execute_reply":"2023-01-09T19:51:03.604941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decodepng(encoded_img: tf.string) -> tf.Tensor:\n  img = tf.image.decode_png(encoded_img, dtype=tf.uint8, channels=1)\n  return tf.image.convert_image_dtype(img, tf.float32) # Between 0 and 1.\n\ndef decode_tfrecord(record_bytes):\n    features = tf.io.parse_single_example(record_bytes, {\n        'image/L_CC': tf.io.FixedLenFeature([], tf.string),\n        'image/L_MLO': tf.io.FixedLenFeature([], tf.string),\n        'image/R_CC': tf.io.FixedLenFeature([], tf.string),\n        'image/R_MLO': tf.io.FixedLenFeature([], tf.string),\n        'cancer/L': tf.io.FixedLenFeature([], tf.int64),\n        'cancer/R': tf.io.FixedLenFeature([], tf.int64),\n        'cancer/case': tf.io.FixedLenFeature([], tf.int64),\n        'patient_id': tf.io.FixedLenFeature([], tf.int64),\n    })\n    \n    image_lcc = decodepng(features['image/L_CC'])\n    image_lmlo = decodepng(features['image/L_MLO'])\n    image_rcc = decodepng(features['image/R_CC'])\n    image_rmlo = decodepng(features['image/R_MLO'])\n    lcancer = features['cancer/L']\n    rcancer = features['cancer/R']\n    patient_id = features['patient_id']\n       \n    return image_lcc, image_rcc, lcancer, rcancer, patient_id","metadata":{"execution":{"iopub.status.busy":"2023-01-09T19:51:07.329785Z","iopub.execute_input":"2023-01-09T19:51:07.331119Z","iopub.status.idle":"2023-01-09T19:51:07.343827Z","shell.execute_reply.started":"2023-01-09T19:51:07.331064Z","shell.execute_reply":"2023-01-09T19:51:07.34204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2023-01-09T19:51:08.144715Z","iopub.execute_input":"2023-01-09T19:51:08.14528Z","iopub.status.idle":"2023-01-09T19:51:09.277936Z","shell.execute_reply.started":"2023-01-09T19:51:08.14523Z","shell.execute_reply":"2023-01-09T19:51:09.276052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_dataset():\n    # Read all TFRecord file paths\n    patterns = ['/kaggle/working/cancer/*.tfrecords', '/kaggle/working/noncancer/*.tfrecords']\n    files = tf.io.gfile.glob(patterns)\n    # initialize TFRecord dataset\n    train_dataset = tf.data.TFRecordDataset(files, num_parallel_reads=1, compression_type='GZIP')\n    # Decode samples by mapping with decode function\n    train_dataset = train_dataset.map(decode_tfrecord)\n    # Batch samples\n    train_dataset = train_dataset.batch(32)\n    \n    return train_dataset","metadata":{"execution":{"iopub.status.busy":"2023-01-09T19:51:09.281545Z","iopub.execute_input":"2023-01-09T19:51:09.282081Z","iopub.status.idle":"2023-01-09T19:51:09.290798Z","shell.execute_reply.started":"2023-01-09T19:51:09.282043Z","shell.execute_reply":"2023-01-09T19:51:09.289497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds = get_train_dataset()","metadata":{"execution":{"iopub.status.busy":"2023-01-09T19:51:10.709857Z","iopub.execute_input":"2023-01-09T19:51:10.71139Z","iopub.status.idle":"2023-01-09T19:51:10.980613Z","shell.execute_reply.started":"2023-01-09T19:51:10.711342Z","shell.execute_reply":"2023-01-09T19:51:10.978987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shows a batch of images\ndef show_batch(dataset, rows=16, cols=2):\n  limgs, rimgs, lys, rys, pids = next(iter(dataset))\n\n  fig, axes = plt.subplots(nrows=rows, ncols=cols, figsize=(cols*6, rows*6))\n  for r in range(rows):\n    axes[r, 0].imshow(limgs[r], cmap='gray')\n    axes[r, 0].set_title(f'L, Target: {lys[r]} (patient_id: {pids[r]}', size=16)\n        \n    axes[r, 1].imshow(rimgs[r], cmap='gray')\n    axes[r, 1].set_title(f'R Target: {rys[r]} (patient_id: {pids[r]})', size=16)\n        \n  plt.show()\n    \nshow_batch(ds)","metadata":{"execution":{"iopub.status.busy":"2023-01-09T19:51:11.278909Z","iopub.execute_input":"2023-01-09T19:51:11.279342Z","iopub.status.idle":"2023-01-09T19:51:17.181067Z","shell.execute_reply.started":"2023-01-09T19:51:11.279307Z","shell.execute_reply":"2023-01-09T19:51:17.17896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}