{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip -q install tensorflow==2.3.0","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nimport zipfile\nimport os\n\n# Visualization\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport cv2\nimport skimage.io\n\n# ML\nfrom sklearn import model_selection\nfrom sklearn.model_selection import train_test_split\n\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import optimizers\nfrom tensorflow.keras import models\nfrom tensorflow.keras.models import Model, Sequential\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.applications.imagenet_utils import preprocess_input\n\n#Use this to check if the GPU is configured correctly\nfrom tensorflow.python.client import device_lib\nprint(device_lib.list_local_devices())\n\n%matplotlib inline","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()  # TPU detection\n    print(\"Running on TPU \", tpu.cluster_spec().as_dict()[\"worker\"])\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept ValueError:\n    print(\"Not connected to a TPU runtime. Using CPU/GPU strategy\")\n    strategy = tf.distribute.MirroredStrategy()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conv_base = ResNet50(weights=\"imagenet\", include_top=False, input_shape=(224, 224, 3))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = models.Sequential()\nmodel.add(conv_base)\nconv_base.trainable = False\nmodel.add(layers.GlobalMaxPooling2D(name=\"gap\"))\n# Avoid overfitting\nmodel.add(layers.Dropout(rate=0.5))\nmodel.add(layers.Dense(10, activation=\"softmax\", name=\"fc_out\"))\n\nmodel.compile(\n    loss=\"categorical_crossentropy\",\n    optimizer=optimizers.RMSprop(lr=2e-5),\n    metrics=[\"acc\"])\n\nmodel.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### CAUTION ###\n\nvariations = [\"A\", \"B\", \"C\", \"D\", \"E\", \"F\", \"G\", \"H\"]\n#variations = [\"A\", \"B\"]\n\ndef zippity(variant):\n    print(f'Variation {variant}')\n\n    with zipfile.ZipFile(f'../input/8-fold-pc-dataset-gen-{variations.index(variant) + 1}-8-{variant.lower()}/train{variant}.zip','r') as z:\n        z.extractall(\".\")\n    \n    with zipfile.ZipFile(f'../input/8-fold-pc-dataset-gen-{variations.index(variant) + 1}-8-{variant.lower()}/validation{variant}.zip','r') as z:\n        z.extractall(\".\")\n    \n#     with zipfile.ZipFile(\"../input/pc-data-dataset-gen/test.zip\",\"r\") as z:\n#         z.extractall(\".\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def zappity():\n    # Deleting image folders to avoid over-saturate the output\n    !rm -r train\n    !rm -r validation\n#     !rm -r test","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_gen = ImageDataGenerator(\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    rescale=1/255,\n    shear_range=0.2,\n    zoom_range=0.2,\n    fill_mode=\"nearest\",\n    preprocessing_function=tf.keras.applications.nasnet.preprocess_input)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 32\n\ndef which_image_gen(which):\n    if(which == \"train\"):\n        which_gen = image_gen.flow_from_directory(\"./train\",\n                                                  target_size=(224, 224),\n                                                  batch_size=batch_size,\n                                                  class_mode=\"categorical\")\n        \n    \n    elif(which == \"valid\"):\n        which_gen = image_gen.flow_from_directory(\"./validation\",\n                                                  target_size=(224, 224),\n                                                  batch_size=batch_size,\n                                                  class_mode=\"categorical\")\n    \n#     elif(which == \"test\"):\n#         which_gen = image_gen.flow_from_directory(\"./test\",\n#                                                   target_size=(224, 224),\n#                                                   batch_size=batch_size,\n#                                                   class_mode=\"categorical\")\n    return which_gen","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUMBER_OF_TRAINING_IMAGES = len(pd.read_csv('../input/8-fold-pc-dataset-gen-0-8/training.csv'))\nNUMBER_OF_VALIDATION_IMAGES = len(pd.read_csv('../input/8-fold-pc-dataset-gen-0-8/validation.csv'))\nNUMBER_OF_TESTING_IMAGES = len(pd.read_csv('../input/8-fold-pc-dataset-gen-0-8/testing.csv'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for variety in variations:\n    zippity(variety)\n    \n    train_image_gen = which_image_gen(\"train\")\n    validation_image_gen = which_image_gen(\"valid\")\n#     test_image_gen = which_image_gen(\"test\")\n\n#     Flowing through directories to see the classes and the number of images\n#     print(image_gen.flow_from_directory(\"./train\"))\n#     print(image_gen.flow_from_directory(\"./validation\"))\n#     print(image_gen.flow_from_directory(\"./test\"))\n\n#     train_image_gen.class_indices\n#     validation_image_gen.class_indices\n#     test_image_gen.class_indices\n\n    results = model.fit(\n        train_image_gen,\n        steps_per_epoch=NUMBER_OF_TRAINING_IMAGES // batch_size,\n        epochs=10,\n        validation_data=validation_image_gen,\n        validation_steps=NUMBER_OF_VALIDATION_IMAGES // batch_size,\n        verbose=1,\n        use_multiprocessing=True,\n        workers=4,\n    )\n    \n    # Saving the synaptic weights of the model\n    model.save(\"./ResNet50-model.h5\")\n    \n    zappity()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_hist_acc(hist):\n    plt.plot(hist.history[\"acc\"])\n    plt.plot(hist.history[\"val_acc\"])\n    plt.title(\"Model Accuracy\")\n    plt.ylabel(\"Accuracy\")\n    plt.xlabel(\"Epoch\")\n    plt.legend([\"Accuracy\", \"Validation Accuracy\"], loc=\"upper left\")\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_hist_loss(hist):\n    plt.plot(hist.history[\"loss\"])\n    plt.plot(hist.history[\"val_loss\"])\n    plt.title(\"Model Loss\")\n    plt.ylabel(\"Accuracy\")\n    plt.xlabel(\"Epoch\")\n    plt.legend([\"Loss\", \"Validation Loss\"], loc=\"upper left\")\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results_df = pd.DataFrame({\"epoch\":[i + 1 for i in range(len(results.history[\"acc\"]))], \"acc\":results.history[\"acc\"], \"val_acc\":results.history[\"val_acc\"], \"loss\":results.history[\"loss\"], \"val_loss\":results.history[\"val_loss\"]})\nresults_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_hist_acc(results)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_hist_loss(results)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}