{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip -q install tensorflow==2.3.0","metadata":{"editable":false,"execution":{"iopub.status.busy":"2022-03-21T19:21:29.043422Z","iopub.execute_input":"2022-03-21T19:21:29.043763Z","iopub.status.idle":"2022-03-21T19:21:36.34257Z","shell.execute_reply.started":"2022-03-21T19:21:29.043727Z","shell.execute_reply":"2022-03-21T19:21:36.341304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Basics / Data manipulation\nimport numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nimport zipfile\nimport os\nimport glob\nimport shutil\n\n# Visualization\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport cv2\nimport skimage.io\n\n# ML\nfrom sklearn import model_selection\nfrom sklearn.model_selection import train_test_split\n\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import optimizers\nfrom tensorflow.keras import models\nfrom tensorflow.keras.models import Model, Sequential\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import NASNetMobile\nfrom tensorflow.keras.applications.imagenet_utils import preprocess_input\n\n#Use this to check if the GPU is configured correctly\nfrom tensorflow.python.client import device_lib\nprint(device_lib.list_local_devices())\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-03-21T19:21:36.345256Z","iopub.execute_input":"2022-03-21T19:21:36.345604Z","iopub.status.idle":"2022-03-21T19:21:36.372286Z","shell.execute_reply.started":"2022-03-21T19:21:36.345562Z","shell.execute_reply":"2022-03-21T19:21:36.371297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data\n10k+ of .tiff images\n*    **80%** for training \n*    **20%** for internal testing\n            *  10% Validation\n            *  10% Testing","metadata":{"editable":false}},{"cell_type":"markdown","source":"# Checking if GPU is being used","metadata":{"editable":false}},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()  # TPU detection\n    print(\"Running on TPU \", tpu.cluster_spec().as_dict()[\"worker\"])\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept ValueError:\n    print(\"Not connected to a TPU runtime. Using CPU/GPU strategy\")\n    strategy = tf.distribute.MirroredStrategy()","metadata":{"editable":false,"execution":{"iopub.status.busy":"2022-03-21T19:21:36.373895Z","iopub.execute_input":"2022-03-21T19:21:36.374213Z","iopub.status.idle":"2022-03-21T19:21:36.386038Z","shell.execute_reply.started":"2022-03-21T19:21:36.374176Z","shell.execute_reply":"2022-03-21T19:21:36.38516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model \nThe model will have the follow configuration:\n______________\n1st layer: NASNetMobile (224, 224, 3) input images\n______________\n2nd layer: GlobalMaxPooling2D\n______________\n3rd layer: Dropout with learning rate = 2e-5\n______________\n4th layer: Denser layer x 6 that will classify the image","metadata":{"editable":false}},{"cell_type":"code","source":"base_model = NASNetMobile(weights=\"imagenet\", include_top=False, input_shape=(224, 224, 3))\n#print(base_model.summary())","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-03-21T19:21:36.388356Z","iopub.execute_input":"2022-03-21T19:21:36.38883Z","iopub.status.idle":"2022-03-21T19:21:45.998325Z","shell.execute_reply.started":"2022-03-21T19:21:36.388791Z","shell.execute_reply":"2022-03-21T19:21:45.997534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = models.Sequential()\nmodel.add(base_model)\nbase_model.trainable = True\nmodel.add(layers.GlobalMaxPooling2D(name=\"gap\"))\n# Avoid overfitting\nmodel.add(layers.Dropout(rate=0.5))\nmodel.add(layers.Dense(2, activation=\"softmax\", name=\"fc_out\"))\n\nmodel.compile(\n    loss=\"categorical_crossentropy\",\n    optimizer=optimizers.RMSprop(lr=2e-5),\n    metrics=[\"acc\"])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-03-21T19:21:46.004137Z","iopub.execute_input":"2022-03-21T19:21:46.004358Z","iopub.status.idle":"2022-03-21T19:21:49.420311Z","shell.execute_reply.started":"2022-03-21T19:21:46.004332Z","shell.execute_reply":"2022-03-21T19:21:49.419557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Unzipping The Files\nThe original images have been transformed into tiled mosaics. Each image_id has 8 mosaic variations; the variations have been grouped into their seperate own zips\nto work within a Kaggle restriction of  max 5 GB in ./kaggle/working & and max 20 GB in ./kaggle/tmp while training.csv + validation.csv + testing.csv are seperate & global. \n\nThe dataset has been split 90% Training, 7.5% Validation, and 2.5% Internal Testing. If you want to use all 10% of the data for Validation, just merge the appropriate dataframes & zipfile contents. \n\nOnly unzip a single variation at a time! The notebook will fail if you use up all the space in ./kaggle/working and your model will not be saved! If you need to move on to a different variation, delete the old files! You can unzip them again later, no problem.\n","metadata":{}},{"cell_type":"code","source":"NUMBER_OF_TRAINING_IMAGES = len(pd.read_csv('../input/8-fold-pc-dataset-gen-0-8/training.csv'))\nNUMBER_OF_VALIDATION_IMAGES = len(pd.read_csv('../input/8-fold-pc-dataset-gen-0-8/validation.csv'))\nNUMBER_OF_TESTING_IMAGES = len(pd.read_csv('../input/8-fold-pc-dataset-gen-0-8/testing.csv'))","metadata":{"execution":{"iopub.status.busy":"2022-03-21T19:21:49.421995Z","iopub.execute_input":"2022-03-21T19:21:49.422207Z","iopub.status.idle":"2022-03-21T19:21:49.480245Z","shell.execute_reply.started":"2022-03-21T19:21:49.422184Z","shell.execute_reply":"2022-03-21T19:21:49.479479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#TVT = [\"./train\", \"./validation\", \"./test\"]\nTVT = [\"./train\", \"./validation\"]\nOUT = [0, 1]\nOut = [\"/Positive\",\"/Negative\"]\nGLE = [\"/GLEASON_SCORE_[!0]+[!0]\", \"/GLEASON_SCORE_0+0\"]\n\ndef binarize():\n    for grouping in TVT:\n        for outcomes in OUT:\n            if not os.path.exists(grouping + Out[outcomes]):\n                os.makedirs(grouping + Out[outcomes])\n            for file in glob.iglob(grouping + GLE[outcomes] + \"/*\"):\n                os.replace(file, grouping + Out[outcomes] + \"/\" + file.split(\"/\")[3])\n            for folder in glob.iglob(grouping + GLE[outcomes]): \n                os.rmdir(folder)  \n","metadata":{"execution":{"iopub.status.busy":"2022-03-21T19:21:49.481752Z","iopub.execute_input":"2022-03-21T19:21:49.482038Z","iopub.status.idle":"2022-03-21T19:21:49.492457Z","shell.execute_reply.started":"2022-03-21T19:21:49.482002Z","shell.execute_reply":"2022-03-21T19:21:49.490294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### CAUTION ###\n\n#variations = [\"A\", \"B\", \"C\", \"D\", \"E\", \"F\", \"G\", \"H\"]\nvariations = [\"A\"]\n\ndef zippity(variant):\n    print(f'Variation {variant}')\n    # Train\n    with zipfile.ZipFile(f'../input/8-fold-pc-dataset-gen-{variations.index(variant) + 1}-8-{variant.lower()}/train{variant}.zip','r') as z:\n        z.extractall(\".\")\n                    \n    # Valid\n    with zipfile.ZipFile(f'../input/8-fold-pc-dataset-gen-{variations.index(variant) + 1}-8-{variant.lower()}/validation{variant}.zip','r') as z:\n        z.extractall(\".\")\n                    \n    # Test\n#     with zipfile.ZipFile(\"../input/pc-data-dataset-gen/test.zip\",\"r\") as z:\n#         z.extractall(\".\")\n    binarize()\n    \n    for path in glob.glob(\"./*/GLEASON_SCORE_?+?/\"):\n        os.rmdir(path)","metadata":{"execution":{"iopub.status.busy":"2022-03-21T19:21:49.49401Z","iopub.execute_input":"2022-03-21T19:21:49.494308Z","iopub.status.idle":"2022-03-21T19:21:49.502418Z","shell.execute_reply.started":"2022-03-21T19:21:49.494271Z","shell.execute_reply":"2022-03-21T19:21:49.501646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def zappity():\n    # Deleting image folders to avoid over-saturate the output\n    !rm -r train\n    !rm -r validation\n#     !rm -r test","metadata":{"execution":{"iopub.status.busy":"2022-03-21T19:21:49.50373Z","iopub.execute_input":"2022-03-21T19:21:49.504273Z","iopub.status.idle":"2022-03-21T19:21:49.511519Z","shell.execute_reply.started":"2022-03-21T19:21:49.504236Z","shell.execute_reply":"2022-03-21T19:21:49.510661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Additional Data Augmentation","metadata":{}},{"cell_type":"code","source":"image_gen = ImageDataGenerator(\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    rescale=1/255,\n    shear_range=0.2,\n    zoom_range=0.2,\n    fill_mode=\"nearest\",\n    preprocessing_function=tf.keras.applications.nasnet.preprocess_input)","metadata":{"execution":{"iopub.status.busy":"2022-03-21T19:21:49.513015Z","iopub.execute_input":"2022-03-21T19:21:49.513493Z","iopub.status.idle":"2022-03-21T19:21:49.521441Z","shell.execute_reply.started":"2022-03-21T19:21:49.513459Z","shell.execute_reply":"2022-03-21T19:21:49.520618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sample = plt.imread(\"../input/panda2/train_images/0005f7aaab2800f6170c399693a96917.png\")\n\n#plt.imshow(image_gen.random_transform(sample))","metadata":{"execution":{"iopub.status.busy":"2022-03-21T19:21:49.523366Z","iopub.execute_input":"2022-03-21T19:21:49.523665Z","iopub.status.idle":"2022-03-21T19:21:49.533495Z","shell.execute_reply.started":"2022-03-21T19:21:49.52363Z","shell.execute_reply":"2022-03-21T19:21:49.532931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 32\n\ndef which_image_gen(which):\n    if(which == \"train\"):\n        which_gen = image_gen.flow_from_directory(\"./train\",\n                                                  target_size=(224, 224),\n                                                  batch_size=batch_size,\n                                                  class_mode=\"categorical\")\n        \n    \n    elif(which == \"valid\"):\n        which_gen = image_gen.flow_from_directory(\"./validation\",\n                                                  target_size=(224, 224),\n                                                  batch_size=batch_size,\n                                                  class_mode=\"categorical\")\n    \n#     elif(which == \"test\"):\n#         which_gen = image_gen.flow_from_directory(\"./test\",\n#                                                   target_size=(224, 224),\n#                                                   batch_size=batch_size,\n#                                                   class_mode=\"categorical\")\n    return which_gen\n","metadata":{"execution":{"iopub.status.busy":"2022-03-21T19:21:49.536224Z","iopub.execute_input":"2022-03-21T19:21:49.536795Z","iopub.status.idle":"2022-03-21T19:21:49.544987Z","shell.execute_reply.started":"2022-03-21T19:21:49.536757Z","shell.execute_reply":"2022-03-21T19:21:49.544179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for variety in variations:\n    zippity(variety)\n    \n    train_image_gen = which_image_gen(\"train\")\n    validation_image_gen = which_image_gen(\"valid\")\n#     test_image_gen = which_image_gen(\"test\")\n\n#     Flowing through directories to see the classes and the number of images\n#     print(image_gen.flow_from_directory(\"./train\"))\n#     print(image_gen.flow_from_directory(\"./validation\"))\n#     print(image_gen.flow_from_directory(\"./test\"))\n\n#     train_image_gen.class_indices\n#     validation_image_gen.class_indices\n#     test_image_gen.class_indices\n\n    results = model.fit(\n        train_image_gen,\n        steps_per_epoch=NUMBER_OF_TRAINING_IMAGES // batch_size,\n        epochs=50,\n        validation_data=validation_image_gen,\n        validation_steps=NUMBER_OF_VALIDATION_IMAGES // batch_size,\n        verbose=20,\n        use_multiprocessing=True,\n        workers=4)\n    \n    # Saving the synaptic weights of the model\n    model.save(\"./NASNetMobile-model.h5\")\n    \n    zappity()\n","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-03-21T19:21:49.546492Z","iopub.execute_input":"2022-03-21T19:21:49.546934Z","iopub.status.idle":"2022-03-21T19:29:19.473596Z","shell.execute_reply.started":"2022-03-21T19:21:49.546898Z","shell.execute_reply":"2022-03-21T19:29:19.471622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_hist_acc(hist):\n    plt.plot(hist.history[\"acc\"])\n    plt.plot(hist.history[\"val_acc\"])\n    plt.title(\"Model Accuracy\")\n    plt.ylabel(\"Accuracy\")\n    plt.xlabel(\"Epoch\")\n    plt.legend([\"Accuracy\", \"Validation Accuracy\"], loc=\"upper left\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-21T19:29:19.47819Z","iopub.execute_input":"2022-03-21T19:29:19.478545Z","iopub.status.idle":"2022-03-21T19:29:19.485349Z","shell.execute_reply.started":"2022-03-21T19:29:19.478502Z","shell.execute_reply":"2022-03-21T19:29:19.48441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_hist_loss(hist):\n    plt.plot(hist.history[\"loss\"])\n    plt.plot(hist.history[\"val_loss\"])\n    plt.title(\"Model Loss\")\n    plt.ylabel(\"Errors\")\n    plt.xlabel(\"Epoch\")\n    plt.legend([\"Loss\", \"Validation Loss\"], loc=\"upper left\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-21T19:29:19.486621Z","iopub.execute_input":"2022-03-21T19:29:19.487037Z","iopub.status.idle":"2022-03-21T19:29:19.506231Z","shell.execute_reply.started":"2022-03-21T19:29:19.487004Z","shell.execute_reply":"2022-03-21T19:29:19.505473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving the synaptic weights of the model\nmodel.save(\"./NASNetMobile-model.h5\")","metadata":{"editable":false,"execution":{"iopub.status.busy":"2022-03-21T19:29:19.509598Z","iopub.execute_input":"2022-03-21T19:29:19.50984Z","iopub.status.idle":"2022-03-21T19:29:21.15393Z","shell.execute_reply.started":"2022-03-21T19:29:19.509817Z","shell.execute_reply":"2022-03-21T19:29:21.1532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results_df = pd.DataFrame({\"epoch\":[i + 1 for i in range(len(results.history[\"acc\"]))], \"acc\":results.history[\"acc\"], \"val_acc\":results.history[\"val_acc\"], \"loss\":results.history[\"loss\"], \"val_loss\":results.history[\"val_loss\"]})\nresults_df","metadata":{"editable":false,"execution":{"iopub.status.busy":"2022-03-21T19:29:21.156415Z","iopub.execute_input":"2022-03-21T19:29:21.156881Z","iopub.status.idle":"2022-03-21T19:29:21.208075Z","shell.execute_reply.started":"2022-03-21T19:29:21.156841Z","shell.execute_reply":"2022-03-21T19:29:21.207485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_hist_acc(results)","metadata":{"execution":{"iopub.status.busy":"2022-03-21T19:29:21.209493Z","iopub.execute_input":"2022-03-21T19:29:21.209896Z","iopub.status.idle":"2022-03-21T19:29:21.502415Z","shell.execute_reply.started":"2022-03-21T19:29:21.209859Z","shell.execute_reply":"2022-03-21T19:29:21.501566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_hist_loss(results)","metadata":{"execution":{"iopub.status.busy":"2022-03-21T19:29:21.504212Z","iopub.execute_input":"2022-03-21T19:29:21.504647Z","iopub.status.idle":"2022-03-21T19:29:21.748623Z","shell.execute_reply.started":"2022-03-21T19:29:21.504604Z","shell.execute_reply":"2022-03-21T19:29:21.747801Z"},"trusted":true},"execution_count":null,"outputs":[]}]}