{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# To Do Notes  \n\n1. Additional images (external)\n1. changing image background\n1. ROI detection https://www.kaggle.com/code/remekkinas/breast-cancer-roi-brest-extractor\n\n### Image Augmentation \n* Mild rotations (up to 6 degrees)\n* Shift, scale, shear\n* horizontal flip\n* for some images random level of blur, noise and gamma changes\n* limited the amount of brightness / gamma augmentations\n\n#### Other \n* Imbalanced Dataset\n* Wide and deep model \n* Model pretrained on RSNA data?\n\n### References \n* [Pneumonia Detection RSNA](https://paperswithcode.com/paper/deep-learning-for-automatic-pneumonia) \n* [Pydicom vs dicomsdl by REMEK KINAS](https://www.kaggle.com/code/remekkinas/fast-dicom-processing-1-6-2x-faster)","metadata":{}},{"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nimport matplotlib.pyplot as plt\nimport os\nimport glob\nimport gc\nimport time \nimport shutil\nfrom sklearn.model_selection import train_test_split\nfrom kaggle_datasets import KaggleDatasets\n\nimport tensorflow as tf\nfrom tensorflow.keras.applications.resnet50 import ResNet50\nfrom tensorflow.python.client import device_lib\nfrom tensorflow.keras import Model, layers, Input, Sequential\nfrom sklearn.metrics import log_loss\n\nfrom joblib import Parallel, delayed\nimport cv2","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_tpu():\n    try: \n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect() \n        tf.keras.mixed_precision.set_global_policy(\"mixed_bfloat16\")\n        tf.config.set_soft_device_placement(True)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        physical_devices = tf.config.list_logical_devices('TPU')\n        return strategy, physical_devices\n    except:\n        False\n        \ndef set_cpu_gpus():\n    try: \n        # printed out the detected devices\n        list_ld = device_lib.list_local_devices()\n        for dev in list_ld:\n            print(dev.name,dev.memory_limit)\n        \n        physical_devices = tf.config.list_physical_devices(\n            'GPU' if len(list_ld) - 1 else 'CPU'\n        )\n        tf.config.optimizer.set_jit(True)\n        tf.keras.mixed_precision.set_global_policy(\"mixed_float16\")\n        \n        # For GPU devices, set growth memory constraint\n        if 'GPU' in physical_devices[-1]:\n            for pd in physical_devices:\n                tf.config.experimental.set_memory_growth(pd, True)\n                pass\n        \n        strategy = tf.distribute.MirroredStrategy()\n        return strategy, physical_devices\n    except: \n        print('No Device Detected!')\n        \nstrategy, physical_devices = set_tpu() or set_cpu_gpus()\ntf.get_logger().setLevel('ERROR')\nphysical_devices, tf.__version__","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTOTUNE = tf.data.AUTOTUNE # allows for parallel loading \n\nDEBUG = True\nRESIZE = 512\nEPOCHS = 50\nBATCH_SIZE = 16  * strategy.num_replicas_in_sync","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'TPU' in str(strategy):\n    # TPU requires the data to be in a GDS bucket we therefore use the GDS path links \n    Train_In_Path = KaggleDatasets().get_gcs_path('rsna-breast-cancer-detection-poi-images')\n    Train_In_Path = Train_In_Path+ '/bc_768_roi/bc_768_roi/train/'\n\n    Test_In_Path = KaggleDatasets().get_gcs_path('rsna-breast-cancer-detection-poi-images')\n    Test_In_Path = Test_In_Path+'/bc_768_roi/bc_768_roi/test/'\nelse:\n    Train_In_Path =\"/kaggle/input/rsna-breast-cancer-detection-poi-images/bc_768_roi/bc_768_roi/train/\"\n    Test_In_Path = \"/kaggle/input/rsna-breast-cancer-detection-poi-images/bc_768_roi/bc_768_roi/test/\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data \n* Includes DCM images of breast mammographs . The dcm format needs to be read using pydicom. \n* These can then be viewed or loaded when converted to numpy arrays ","metadata":{}},{"cell_type":"code","source":"train_files = glob.glob ('/kaggle/input/rsna-breast-cancer-detection-poi-images/bc_768_roi/bc_768_roi/train/'+\"*.png\")\ntest_files = glob.glob ('/kaggle/input/rsna-breast-cancer-detection-poi-images/bc_768_roi/bc_768_roi/test/'+\"*.png\")\nprint(\"Number of training images:\",len(train_files))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\nsub_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv')\nsub_df.to_csv('submission.csv',index=False)\ny = df[\"cancer\"]\n\nif DEBUG:\n    y = y.sample(frac= 0.8)\n    df= df.iloc[y.index]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Same image path to dataframe\ndf[\"image_path\"] = Train_In_Path+df[\"patient_id\"].astype(str) +\"_\"+ df [\"image_id\"].astype(str)+\".png\"\ntest_df[\"image_path\"] = Test_In_Path+test_df[\"patient_id\"].astype(str) +\"_\"+ test_df [\"image_id\"].astype(str)+\".png\"\ntest_df[\"image_path\"][0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Helper Functions","metadata":{}},{"cell_type":"code","source":"# Custom Metric \n# Probabilistic F Score taken from competition \n\ndef Fscore_metric(labels, preds, beta=1):\n    preds = tf.clip_by_value(preds, 0, 1)\n    y_true_count = tf.reduce_sum(labels)\n    ctp = tf.reduce_sum(preds[labels==1])\n    cfp = tf.reduce_sum(preds[labels==0])\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp)\n    c_recall = ctp / y_true_count\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n        return result\n    else:\n        return 0.0\n    \ndef Fscore_numpy(labels, preds, beta=1):\n    preds = preds.clip(0, 1)\n    y_true_count = labels.sum()\n    ctp = preds[labels==1].sum()\n    cfp = preds[labels==0].sum()\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp)\n    c_recall = ctp / y_true_count\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n        return result\n    else:\n        return 0.0","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualise Data","metadata":{}},{"cell_type":"code","source":"fig,ax = plt.subplots(1,5,figsize = (20,8))\nax= np.ravel(ax)\n\nfor i in range(5):\n    img = cv2.imread(train_files[i])\n    ax[i].imshow(img)\n    \nfig.suptitle(\"Train files examples\")\nplt.tight_layout(pad = 0) \nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df[\"image_path\"]\ny = df[\"cancer\"]\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.33, random_state=42)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Loader \n* https://financial-engineering.medium.com/tensorflow-2-0-load-images-to-tensorflow-897b8b067fc2\n* https://notebook.community/tensorflow/docs/site/en/r1/tutorials/load_data/images\n\n\n1. Get list of image paths (tf.data.Dataset.list_files)\n1. Read images with tensorflow (tf.io.read_file)\n1. Decode images into dataset \n1. Cache, shuffle, prefetch  ","metadata":{}},{"cell_type":"code","source":"#decode file back to image (note the 3 channels for colored images)\ndef decode_read(file_path,label=None):\n    file_bytes = tf.io.read_file(file_path) #reads the image file path as a string\n    img = tf.image.decode_png(file_bytes, channels=3) #color images\n    img = tf.image.convert_image_dtype(img, tf.float32) #convert unit8 tensor to floats in the [0,1]range\n    img = tf.image.resize(img, [RESIZE, RESIZE]) \n    return img, label if label!=None else img\n\n# save date in cache for faster loading, buffered prefetching will load images from disk without having I/O become blocking\ndef prepare(ds,shuffle_buffer_size=1000):\n    ds = ds.cache()\n    ds = ds.shuffle(8 * BATCH_SIZE, reshuffle_each_iteration = True)\n    #ds = ds.repeat() #re-initialise dataset (i.e start from beginning of dataset not the end)\n    ds = ds.batch(BATCH_SIZE)\n    ds = ds.prefetch(buffer_size=AUTOTUNE)\n    return ds\n\ndef augment_image(img,label):\n    img = tf.image.random_flip_left_right(img)\n    #img = tf.keras.layers.experimental.preprocessing.RandomRotation(0.2)(img)\n    #img = layers.experimental.preprocessing.Rescaling(1./255)(img)\n    return img,label\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = tf.data.Dataset.from_tensor_slices((X_train.values,y_train.values))\nval_ds = tf.data.Dataset.from_tensor_slices((X_test.values,y_test.values))\ntest_ds = tf.data.Dataset.from_tensor_slices((test_df[\"image_path\"].values))\n\ntrain_ds = train_ds.map(decode_read, num_parallel_calls=AUTOTUNE)\nval_ds= val_ds.map(decode_read, num_parallel_calls=AUTOTUNE)\ntest_ds= test_ds.map(decode_read, num_parallel_calls=AUTOTUNE)\n\n#train images only\ntrain_ds= train_ds.map(augment_image, num_parallel_calls=AUTOTUNE)\n\n# Cache, shuffle, batch \ntrain_ds = prepare(train_ds)\nval_ds = prepare(val_ds)\ntest_ds = prepare(test_ds)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_batch, label_batch = next(iter(train_ds))\n\nfig, ax = plt.subplots(2,4,figsize=(20,8), sharey = True, sharex = True)\nax = np.ravel(ax)\nfor i in range(8):\n    img = tf.image.convert_image_dtype(image_batch[i].numpy(), tf.float32) \n    ax[i].imshow(img)\n    ax[i].set_title(f\"label:{label_batch[i].numpy()}\")\n    ax[i].axis('off') \n                   \nplt.tight_layout()\nplt.show()  ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modelling \n","metadata":{}},{"cell_type":"code","source":"tf.keras.backend.clear_session()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# issue with different tensorflow versions (in GPU and TPU accelerators in Kaggle)\nif 'TPU' in str(strategy):\n        Rescaling = tf.keras.layers.experimental.preprocessing.Rescaling(1./255,name = \"Rescaling0_1\")\n        RandomFlip = tf.keras.layers.experimental.preprocessing.RandomFlip('horizontal')\n        RandomRotation = tf.keras.layers.experimental.preprocessing.RandomRotation(0.2)\nelse:\n    Rescaling = layers.Rescaling(1./255,name = \"Rescaling0_1\")\n    RandomFlip = layers.RandomFlip('horizontal')\n    RandomRotation = layers.RandomRotation(0.2) \n    \ndata_augmentation  = Sequential([\n    #layers.InputLayer(input_shape=(RESIZE,RESIZE,3)), \n    RandomFlip,\n    RandomRotation])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    \n    ResNet = ResNet50(weights='/kaggle/input/resnet-preweights-imagenet/resnet50.h5',\n                  include_top=False,\n                  input_shape=(RESIZE,RESIZE,3),\n                  pooling = 'avg')\n    ResNet.trainable = False\n    \n    model = Sequential(\n        [layers.InputLayer(input_shape=(RESIZE,RESIZE,3)),         \n         layers.experimental.preprocessing.Rescaling(1./255,name = \"Rescaling0_1\"),\n         ResNet,\n         layers.Flatten(),\n         layers.Dense(32, activation = \"relu\"),\n            layers.Dense(1,activation =\"sigmoid\",dtype='float32')\n        ]\n    )\n    \nmodel.compile(optimizer=\"adam\",loss= tf.keras.losses.BinaryCrossentropy(),metrics = [Fscore_metric])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n  train_ds,\n  validation_data=val_ds,\n  epochs=EPOCHS\n)","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fscore = history.history['Fscore_metric']\nval_fscore = history.history['val_Fscore_metric']\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs_range = range(EPOCHS)\n\nplt.figure(figsize=(20, 8))\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, fscore, label='Training F Score')\nplt.plot(epochs_range, val_fscore, label='Validation F Score')\nplt.legend(loc='lower right')\nplt.title('Training and Validation F Score')\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Training Loss')\nplt.plot(epochs_range, val_loss, label='Validation Loss')\nplt.legend(loc='upper right')\nplt.title('Training and Validation Loss')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission and Prediction","metadata":{}},{"cell_type":"code","source":"test_preds = model.predict(test_ds)\ntest_preds.reshape(-1)","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\npred_df = pd.DataFrame({'prediction_id':test_df.prediction_id,\n                        'cancer':test_preds.reshape(-1)})\ndel sub_df['cancer']\n\nsub_df = sub_df.merge(pred_df, on='prediction_id', how='left')\nsub_df = sub_df.groupby('prediction_id')['cancer'].max().reset_index()  # merge duplicate prediction_id\nsub_df.to_csv('submission.csv',index=False)\nsub_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_val = np.concatenate([y for x, y in val_ds], axis=0)\nval_preds= model.predict(val_ds)","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"log loss score:\",log_loss(y_val, val_preds))\nprint(\"F probability score:\",Fscore_numpy(y_val, val_preds) )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"log loss if only no cancer pred:\",log_loss(y_val, np.zeros(len(y_val))))\nprint(\"F probability score if only no cancer pred:\",Fscore_numpy(y_val, np.zeros(len(y_val))) )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(y_val -val_preds.reshape(-1))\nplt.title(\"Residuals\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}