{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport warnings\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '2'\nwarnings.filterwarnings('ignore')\n\nimport numpy as np\nimport pandas as pd\n# from sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom tensorflow.keras import layers\n# import tensorflow_io as tfio\nimport matplotlib.pyplot as plt\nfrom kaggle_datasets import KaggleDatasets\n# import pydicom\n\n\nprint(\"Tensorflow version \" + tf.__version__)\n# print(\"pydicom version \" + pydicom.__version__)\n\npd.set_option('display.max_colwidth', 200)\n# tf.get_logger().setLevel('ERROR')\n# tf.autograph.set_verbosity(0)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-29T04:17:04.358301Z","iopub.execute_input":"2023-03-29T04:17:04.359375Z","iopub.status.idle":"2023-03-29T04:17:15.265274Z","shell.execute_reply.started":"2023-03-29T04:17:04.359315Z","shell.execute_reply":"2023-03-29T04:17:15.263627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()  # TPU detection. No parameters necessary if TPU_NAME environment variable is set. On Kaggle this is always the case.\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n#     strategy = tf.distribute.get_strategy() # default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.MirroredStrategy() # пока так, но надо проверять множественность GPU\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-03-29T04:17:15.267475Z","iopub.execute_input":"2023-03-29T04:17:15.268947Z","iopub.status.idle":"2023-03-29T04:17:20.586973Z","shell.execute_reply.started":"2023-03-29T04:17:15.268903Z","shell.execute_reply":"2023-03-29T04:17:20.585544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_SIZE = [768, 512]\nGCS_TFREC_PATH = KaggleDatasets().get_gcs_path('rsna-bcd-tfrecords-512x512-10k')\nMAIN_PATH = '/kaggle/input/rsna-breast-cancer-detection/'\nTRAIN_PATH = MAIN_PATH + 'train_images/'\nTEST_PATH = MAIN_PATH + 'test_images/'","metadata":{"execution":{"iopub.status.busy":"2023-03-29T04:27:48.809164Z","iopub.execute_input":"2023-03-29T04:27:48.809733Z","iopub.status.idle":"2023-03-29T04:27:49.384348Z","shell.execute_reply.started":"2023-03-29T04:27:48.809686Z","shell.execute_reply":"2023-03-29T04:27:49.383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Analize metadata**","metadata":{}},{"cell_type":"code","source":"meta_df = pd.read_csv(MAIN_PATH + 'train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-03-29T04:28:01.660261Z","iopub.execute_input":"2023-03-29T04:28:01.660883Z","iopub.status.idle":"2023-03-29T04:28:01.803958Z","shell.execute_reply.started":"2023-03-29T04:28:01.660831Z","shell.execute_reply":"2023-03-29T04:28:01.802448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for model.fit\nclass_weight = {0: 1 - meta_df.cancer.mean(),\n               1: meta_df.cancer.mean()}\nprint(class_weight)","metadata":{"execution":{"iopub.status.busy":"2023-03-29T07:31:13.327361Z","iopub.execute_input":"2023-03-29T07:31:13.328708Z","iopub.status.idle":"2023-03-29T07:31:13.33671Z","shell.execute_reply.started":"2023-03-29T07:31:13.328638Z","shell.execute_reply":"2023-03-29T07:31:13.335486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Preparing datasets**","metadata":{}},{"cell_type":"code","source":"# EPOCHS = 25\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nNUM_TRAINING_IMAGES = len(meta_df)*.95 # train is 95% of whole dataset\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-03-29T07:47:44.688128Z","iopub.execute_input":"2023-03-29T07:47:44.688669Z","iopub.status.idle":"2023-03-29T07:47:44.695572Z","shell.execute_reply.started":"2023-03-29T07:47:44.688626Z","shell.execute_reply":"2023-03-29T07:47:44.694092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEPS_PER_EPOCH","metadata":{"execution":{"iopub.status.busy":"2023-03-29T07:40:29.735766Z","iopub.execute_input":"2023-03-29T07:40:29.73664Z","iopub.status.idle":"2023-03-29T07:40:29.742256Z","shell.execute_reply.started":"2023-03-29T07:40:29.736596Z","shell.execute_reply":"2023-03-29T07:40:29.741341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-03-29T07:40:38.127701Z","iopub.execute_input":"2023-03-29T07:40:38.128147Z","iopub.status.idle":"2023-03-29T07:40:38.134922Z","shell.execute_reply.started":"2023-03-29T07:40:38.128115Z","shell.execute_reply":"2023-03-29T07:40:38.133945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), # tensor encoded as bytestring\n        \"label\": tf.io.FixedLenFeature([], tf.int64),  # shape [] means single element\n        \"meta_data\": tf.io.RaggedFeature(tf.string),\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = tf.io.parse_tensor(example['image'], tf.string)\n#     image = tf.image.decode_jpeg(image, channels=3)\n#     image = tf.reshape(image, [*IMAGE_SIZE, 3]) # explicite size needed for TPU\n    image = tf.image.decode_jpeg(image)\n    image = tf.reshape(image, [*IMAGE_SIZE, 1]) # explicite size needed for TPU\n#     image = tf.image.decode_jpeg(image)\n    label = tf.cast(example['label'], tf.uint8)\n    meta_data = example['meta_data']\n    return image, label#, meta_data\n\ndef load_dataset(filenames, labeled=True, ordered=False, augmentation=False):\n    # Read from TFRecords. For optimal performance, reading from multiple files at once and\n    # disregarding data order. Order does not matter since we will be shuffling the data anyway.\n\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False # disable order, increase speed\n\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.AUTOTUNE) # automatically interleaves reads from multiple files\n    dataset = dataset.with_options(ignore_order) # uses data as soon as it streams in, rather than in its original order\n    dataset = dataset.map(read_unlabeled_tfrecord if not labeled \n                          else read_labeled_tfrecord_wa if augmentation else read_labeled_tfrecord)\n#     dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord)\n    # returns a dataset of (image, label) pairs if labeled=True or (image, id) pairs if labeled=False\n    return dataset\n\ndef get_training_dataset(augmentation=False):\n    dataset = load_dataset(tf.io.gfile.glob(GCS_TFREC_PATH + '/train/*.tfrecords'), \n                           labeled=True, augmentation=augmentation)\n    dataset = dataset.repeat() # the training dataset must repeat for several epochs\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(2)\n    return dataset\n\ndef get_validation_dataset():\n    dataset = load_dataset(tf.io.gfile.glob(GCS_TFREC_PATH + '/train/*.tfrecords'), \n                           labeled=True)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2023-03-29T07:40:54.349952Z","iopub.execute_input":"2023-03-29T07:40:54.35046Z","iopub.status.idle":"2023-03-29T07:40:54.369019Z","shell.execute_reply.started":"2023-03-29T07:40:54.350412Z","shell.execute_reply":"2023-03-29T07:40:54.367575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_dataset = get_training_dataset()\nvalidation_dataset = get_validation_dataset()","metadata":{"execution":{"iopub.status.busy":"2023-03-29T07:47:55.169054Z","iopub.execute_input":"2023-03-29T07:47:55.169576Z","iopub.status.idle":"2023-03-29T07:47:55.320558Z","shell.execute_reply.started":"2023-03-29T07:47:55.169513Z","shell.execute_reply":"2023-03-29T07:47:55.319313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(training_dataset.element_spec)\nprint(validation_dataset.element_spec)","metadata":{"execution":{"iopub.status.busy":"2023-03-29T04:29:21.944495Z","iopub.execute_input":"2023-03-29T04:29:21.945055Z","iopub.status.idle":"2023-03-29T04:29:21.953303Z","shell.execute_reply.started":"2023-03-29T04:29:21.945011Z","shell.execute_reply":"2023-03-29T04:29:21.951292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model","metadata":{}},{"cell_type":"code","source":"def pf1_score(labels, predictions):\n\n    pTP = tf.math.reduce_sum(labels * predictions)\n    pFP = tf.math.reduce_sum((1-labels) * predictions)\n\n    pPrecision = pTP/(pTP+pFP)\n    pRecall = pTP/tf.math.reduce_sum(labels)\n    \n\n    if (pPrecision > 0 and pRecall > 0):\n        pF1 = 2 * pPrecision * pRecall/(pPrecision + pRecall)\n        return pF1\n    else:\n        return 0.0","metadata":{"execution":{"iopub.status.busy":"2023-03-29T04:29:41.885076Z","iopub.execute_input":"2023-03-29T04:29:41.885845Z","iopub.status.idle":"2023-03-29T04:29:41.895093Z","shell.execute_reply.started":"2023-03-29T04:29:41.885791Z","shell.execute_reply":"2023-03-29T04:29:41.893591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():  \n    pretrained_model = tf.keras.applications.DenseNet201(weights='imagenet', include_top=False ,input_shape=[*IMAGE_SIZE, 3])\n#     pretrained_model = tf.keras.applications.DenseNet121(weights='imagenet', include_top=False ,input_shape=[*IMAGE_SIZE, 3])\n#     pretrained_model = tf.keras.applications.Xception(weights='imagenet', include_top=False ,input_shape=[*IMAGE_SIZE, 3])\n    pretrained_model.trainable = False\n    \n    inputs = tf.keras.Input(shape=[*IMAGE_SIZE, 1])\n    x = tf.keras.layers.Conv2D(3, 3, activation='relu', padding='same')(inputs)\n    x = pretrained_model(x)\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    x = tf.keras.layers.Dropout(0.2)(x)\n    x = tf.keras.layers.Dense(1024, activation='relu')(x)\n    x = tf.keras.layers.Dropout(0.2)(x)\n    x = tf.keras.layers.Dense(512, activation='relu')(x)\n    x = tf.keras.layers.Dropout(0.2)(x)\n    x = tf.keras.layers.Dense(256, activation='relu')(x)\n    x = tf.keras.layers.Dropout(0.2)(x)\n    \n    outputs = tf.keras.layers.Dense(1, activation='sigmoid')(x)\n    \n    model = tf.keras.Model(inputs, outputs)","metadata":{"execution":{"iopub.status.busy":"2023-03-29T07:41:04.299459Z","iopub.execute_input":"2023-03-29T07:41:04.300551Z","iopub.status.idle":"2023-03-29T07:42:09.257978Z","shell.execute_reply.started":"2023-03-29T07:41:04.30048Z","shell.execute_reply":"2023-03-29T07:42:09.256363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-29T07:42:30.950941Z","iopub.execute_input":"2023-03-29T07:42:30.952588Z","iopub.status.idle":"2023-03-29T07:42:31.045776Z","shell.execute_reply.started":"2023-03-29T07:42:30.952498Z","shell.execute_reply":"2023-03-29T07:42:31.044582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EPOCHS = 5\nmodel.compile(\n    optimizer='adam',\n#     optimizer=tf.keras.optimizers.Adamax(),\n    loss = 'binary_crossentropy',\n    metrics = pf1_score,\n)\n\nhistorical = model.fit(training_dataset,\n                       steps_per_epoch=STEPS_PER_EPOCH, \n                       epochs=EPOCHS, \n                       validation_data=validation_dataset,\n                       class_weight = class_weight\n                      )","metadata":{"execution":{"iopub.status.busy":"2023-03-29T07:48:06.211037Z","iopub.execute_input":"2023-03-29T07:48:06.211686Z","iopub.status.idle":"2023-03-29T08:10:55.526974Z","shell.execute_reply.started":"2023-03-29T07:48:06.211623Z","shell.execute_reply":"2023-03-29T08:10:55.525449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EPOCHS = 40\nwith strategy.scope():    \n    pretrained_model.trainable = True \n        \nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=3, verbose=1, min_lr=1e-8)\nearly_stop = tf.keras.callbacks.EarlyStopping(monitor = 'val_pf1_score', patience=3, start_from_epoch=30)\n        \nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-5),\n#     optimizer=tf.keras.optimizers.Adamax(),\n    loss = 'binary_crossentropy',\n    metrics = pf1_score,\n)\n\nhistorical = model.fit(training_dataset,\n                       steps_per_epoch=STEPS_PER_EPOCH,\n                       epochs=EPOCHS,\n                       callbacks=[reduce_lr, early_stop],\n                       validation_data=validation_dataset,\n                       class_weight = class_weight\n                      )","metadata":{"execution":{"iopub.status.busy":"2023-03-29T08:33:59.461704Z","iopub.execute_input":"2023-03-29T08:33:59.462204Z","iopub.status.idle":"2023-03-29T12:06:01.911095Z","shell.execute_reply.started":"2023-03-29T08:33:59.46216Z","shell.execute_reply":"2023-03-29T12:06:01.908812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_learning_curves(history):\n    acc = history.history['pf1_score']\n    #val_acc = history.history['val_accuracy']\n    loss = history.history['loss']\n    #val_loss = history.history['val_loss']\n\n    epochs = range(len(acc))\n\n#     plt.plot(epochs, acc, 'r', label='Training accuracy')\n    plt.plot(epochs, loss, 'b', label='Traning loss')\n    plt.title('Training and validation accuracy')\n    plt.legend(loc=0)\n    plt.figure();\n\n\n    plt.show();\n","metadata":{"execution":{"iopub.status.busy":"2023-03-29T12:09:13.49914Z","iopub.execute_input":"2023-03-29T12:09:13.499849Z","iopub.status.idle":"2023-03-29T12:09:13.509604Z","shell.execute_reply.started":"2023-03-29T12:09:13.499801Z","shell.execute_reply":"2023-03-29T12:09:13.508461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_learning_curves(historical)","metadata":{"execution":{"iopub.status.busy":"2023-03-29T12:09:16.173843Z","iopub.execute_input":"2023-03-29T12:09:16.174355Z","iopub.status.idle":"2023-03-29T12:09:16.507845Z","shell.execute_reply.started":"2023-03-29T12:09:16.174301Z","shell.execute_reply":"2023-03-29T12:09:16.506612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('dn121-768x512.h5')","metadata":{"execution":{"iopub.status.busy":"2023-03-29T12:09:29.918685Z","iopub.execute_input":"2023-03-29T12:09:29.919499Z","iopub.status.idle":"2023-03-29T12:09:39.541764Z","shell.execute_reply.started":"2023-03-29T12:09:29.91945Z","shell.execute_reply":"2023-03-29T12:09:39.54006Z"},"trusted":true},"execution_count":null,"outputs":[]}]}