{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n#import numpy as np # linear algebra\n#import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n#import os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n    #for filename in filenames:\n        #print(os.path.join(dirname, filename))\n        #break\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-19T14:15:49.960084Z","iopub.execute_input":"2023-10-19T14:15:49.960741Z","iopub.status.idle":"2023-10-19T14:15:49.964893Z","shell.execute_reply.started":"2023-10-19T14:15:49.960713Z","shell.execute_reply":"2023-10-19T14:15:49.963882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import All Libraries","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\nimport os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nfrom sklearn.metrics import classification_report , confusion_matrix , accuracy_score , auc\nfrom sklearn.model_selection import train_test_split\n\nimport cv2\nfrom PIL import Image \nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras import Sequential\nfrom keras.layers import Input, Dense,Conv2D , MaxPooling2D, Flatten,BatchNormalization,Dropout\nfrom tensorflow.keras.preprocessing import image_dataset_from_directory\nimport tensorflow_hub as hub ","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:52:15.315309Z","iopub.execute_input":"2023-10-19T14:52:15.315677Z","iopub.status.idle":"2023-10-19T14:52:27.5582Z","shell.execute_reply.started":"2023-10-19T14:52:15.315646Z","shell.execute_reply":"2023-10-19T14:52:27.55696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:52:27.560648Z","iopub.execute_input":"2023-10-19T14:52:27.561658Z","iopub.status.idle":"2023-10-19T14:52:27.567954Z","shell.execute_reply.started":"2023-10-19T14:52:27.561605Z","shell.execute_reply":"2023-10-19T14:52:27.566002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test Image Data Preprocessing ","metadata":{}},{"cell_type":"code","source":"path = \"/kaggle/input/UBC-OCEAN/test.csv\"\ntest_df = pd.read_csv(path)\ntest_df","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:52:27.569645Z","iopub.execute_input":"2023-10-19T14:52:27.571051Z","iopub.status.idle":"2023-10-19T14:52:27.617248Z","shell.execute_reply.started":"2023-10-19T14:52:27.571004Z","shell.execute_reply":"2023-10-19T14:52:27.616171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['is_tma']=((test_df['image_width'] < 5000) & (test_df['image_height'] < 5000))\ntest_df","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:52:32.397182Z","iopub.execute_input":"2023-10-19T14:52:32.398256Z","iopub.status.idle":"2023-10-19T14:52:32.414289Z","shell.execute_reply.started":"2023-10-19T14:52:32.398222Z","shell.execute_reply":"2023-10-19T14:52:32.412579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['is_tma']=test_df['is_tma'].astype(\"int8\")\ntest_df","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:52:33.458392Z","iopub.execute_input":"2023-10-19T14:52:33.459437Z","iopub.status.idle":"2023-10-19T14:52:33.470273Z","shell.execute_reply.started":"2023-10-19T14:52:33.459396Z","shell.execute_reply":"2023-10-19T14:52:33.468998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Image.MAX_IMAGE_PIXELS = 10000000000\n# Define patch size and overlap (if needed)\npatch_size = (128,128)  # Adjust this according to your requirements\noverlap = 0  # Adjust this if you want overlapping patches\n\nimage_data_1 = []\nempty_img=0\nfor img_id, tma in zip(test_df['image_id'], test_df['is_tma']):\n    #print(img_id, label,  tma)\n    if tma==0:\n        img_name = str(img_id)+\"_thumbnail.png\"\n        large_image = Image.open(\"/kaggle/input/UBC-OCEAN/test_thumbnails/\"+img_name)\n        single_image_patches = []\n        for y in range(0, large_image.height, patch_size[0] - overlap): # (0,2523,192)\n            for x in range(0, large_image.width, patch_size[1] - overlap):  # (0,3000,192)  224-32=192\n                patch = large_image.crop((x, y, x+patch_size[1], y+patch_size[0]))\n                image = np.array(patch)\n                if np.sum(image)==0:\n                    empty_img+=1\n                elif (np.sum(image[0:,0:50])==0) or (np.sum(image[0:,50:])==0) or (np.sum(image[0:,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,0:])==0) or (np.sum(image[50:,0:])==0) or (np.sum(image[100:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,0:50])==0) or (np.sum(image[0:50,50:])==0):\n                    empty_img+=1\n                elif (np.sum(image[50:100,0:50])==0) or (np.sum(image[50:100,50:])==0):\n                    empty_img+=1\n                elif (np.sum(image[50:,0:100])==0) or (np.sum(image[80:,80:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,75:])==0) or (np.sum(image[50:,75:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:40,80:])==0) or (np.sum(image[80:,80:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:35,0:])==0) or (np.sum(image[80:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:25,0:50])==0) or (np.sum(image[0:25,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:20,0:25])==0) or (np.sum(image[0:20,25:50])==0) or (np.sum(image[0:20,50:80])==0) or (np.sum(image[0:20,90:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:25,0:25])==0) or (np.sum(image[0:25,100:])==0) or (np.sum(image[100:,0:25])==0) or (np.sum(image[100:,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:15,0:])==0) or (np.sum(image[115:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:,0:15])==0) or (np.sum(image[0:,115:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:20,0:20])==0) or (np.sum(image[0:20,110:])==0) or (np.sum(image[110:,0:20])==0) or (np.sum(image[110:,110:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:10,0:10])==0) or (np.sum(image[0:10,115:])==0) or (np.sum(image[0:10,40:60])==0) or (np.sum(image[0:10,80:100])==0):\n                    empty_img+=1\n                elif (np.sum(image[40:50,0:10])==0) or (np.sum(image[110:,80:100])==0) or (np.sum(image[70:85,70:90])==0) or (np.sum(image[50:60,110:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:10,0:10])==0) or (np.sum(image[0:10,0:])==0) or (np.sum(image[0:10,120:])==0):\n                    empty_img+=1\n                elif (np.sum(image[120:,0:10])==0) or (np.sum(image[120:,120:])==0) or (np.sum(image[0:20,60:75])==0) or (np.sum(image[115:,60:75])==0):\n                    empty_img+=1\n\n                else:\n                    single_image_patches.append(image)\n        image_data_1.append(single_image_patches)\n        \n    elif tma==1:\n        img_name = str(img_id)+\".png\"\n        large_image = Image.open(\"/kaggle/input/UBC-OCEAN/test_images/\"+img_name)\n        for y in range(0, large_image.height, patch_size[0] - overlap): # (0,2523,192)\n            for x in range(0, large_image.width, patch_size[1] - overlap):  # (0,3000,192)  224-32=192\n                patch = large_image.crop((x, y, x+patch_size[1], y+patch_size[0]))\n                image = np.array(patch)\n                if np.sum(image)==0:\n                    empty_img+=1\n                elif (np.sum(image[0:,0:50])==0) or (np.sum(image[0:,50:])==0) or (np.sum(image[0:,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,0:])==0) or (np.sum(image[50:,0:])==0) or (np.sum(image[100:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,0:50])==0) or (np.sum(image[0:50,50:])==0):\n                    empty_img+=1\n                elif (np.sum(image[50:100,0:50])==0) or (np.sum(image[50:100,50:])==0):\n                    empty_img+=1\n                elif (np.sum(image[50:,0:100])==0) or (np.sum(image[80:,80:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,75:])==0) or (np.sum(image[50:,75:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:40,80:])==0) or (np.sum(image[80:,80:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:35,0:])==0) or (np.sum(image[80:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:25,0:50])==0) or (np.sum(image[0:25,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:20,0:25])==0) or (np.sum(image[0:20,25:50])==0) or (np.sum(image[0:20,50:80])==0) or (np.sum(image[0:20,90:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:25,0:25])==0) or (np.sum(image[0:25,100:])==0) or (np.sum(image[100:,0:25])==0) or (np.sum(image[100:,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:15,0:])==0) or (np.sum(image[115:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:,0:15])==0) or (np.sum(image[0:,115:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:20,0:20])==0) or (np.sum(image[0:20,110:])==0) or (np.sum(image[110:,0:20])==0) or (np.sum(image[110:,110:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:10,0:10])==0) or (np.sum(image[0:10,115:])==0) or (np.sum(image[0:10,40:60])==0) or (np.sum(image[0:10,80:100])==0):\n                    empty_img+=1\n                elif (np.sum(image[40:50,0:10])==0) or (np.sum(image[110:,80:100])==0) or (np.sum(image[70:85,70:90])==0) or (np.sum(image[50:60,110:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:10,0:10])==0) or (np.sum(image[0:10,0:])==0) or (np.sum(image[0:10,120:])==0):\n                    empty_img+=1\n                elif (np.sum(image[120:,0:10])==0) or (np.sum(image[120:,120:])==0) or (np.sum(image[0:20,60:75])==0) or (np.sum(image[115:,60:75])==0):\n                    empty_img+=1\n                    \n                else:\n                    single_image_patches.append(image)\n        image_data_1.append(single_image_patches)","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:52:43.48521Z","iopub.execute_input":"2023-10-19T14:52:43.485633Z","iopub.status.idle":"2023-10-19T14:52:44.052508Z","shell.execute_reply.started":"2023-10-19T14:52:43.485595Z","shell.execute_reply":"2023-10-19T14:52:44.05133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image.MAX_IMAGE_PIXELS = 10000000000\n\n# image_data_1 = []\n# #empty_img=0\n# for img_id, tma in zip(test_df['image_id'], test_df['is_tma']):\n#     #print(img_id, label,  tma)\n#     if tma==0:\n#         img_name = str(img_id)+\"_thumbnail.png\"\n#         image = Image.open(\"/kaggle/input/UBC-OCEAN/test_thumbnails/\"+img_name)\n#         image = image.resize((128,128))\n#         image = np.array(image)\n#         image_data_1.append(image)\n        \n        \n#     elif tma==1:\n#         img_name = str(img_id)+\".png\"\n#         image = Image.open(\"/kaggle/input/UBC-OCEAN/test_images/\"+img_name)\n#         image = image.resize((128,128))\n#         image = np.array(image)\n#         image_data_1.append(image)","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:16:06.303514Z","iopub.execute_input":"2023-10-19T14:16:06.303761Z","iopub.status.idle":"2023-10-19T14:16:06.30782Z","shell.execute_reply.started":"2023-10-19T14:16:06.30374Z","shell.execute_reply":"2023-10-19T14:16:06.30691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(image_data_1)","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:52:44.498521Z","iopub.execute_input":"2023-10-19T14:52:44.500006Z","iopub.status.idle":"2023-10-19T14:52:44.507171Z","shell.execute_reply.started":"2023-10-19T14:52:44.499969Z","shell.execute_reply":"2023-10-19T14:52:44.50534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(image_data_1[0])","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:52:45.27545Z","iopub.execute_input":"2023-10-19T14:52:45.277111Z","iopub.status.idle":"2023-10-19T14:52:45.28417Z","shell.execute_reply.started":"2023-10-19T14:52:45.277061Z","shell.execute_reply":"2023-10-19T14:52:45.282711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(image_data_1[0])","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:52:46.272555Z","iopub.execute_input":"2023-10-19T14:52:46.273431Z","iopub.status.idle":"2023-10-19T14:52:46.281448Z","shell.execute_reply.started":"2023-10-19T14:52:46.273389Z","shell.execute_reply":"2023-10-19T14:52:46.280277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## list of patch image array\ntest_patch_image_array_folder = []\nfor img in image_data_1:\n    single_image = np.array(img)\n    test_patch_image_array_folder.append(single_image)","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:52:51.268839Z","iopub.execute_input":"2023-10-19T14:52:51.26979Z","iopub.status.idle":"2023-10-19T14:52:51.282186Z","shell.execute_reply.started":"2023-10-19T14:52:51.269753Z","shell.execute_reply":"2023-10-19T14:52:51.280752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(40,80))\ncount=0\nj=1\nfor i in test_patch_image_array_folder[0]:\n    plt.subplot(20,10,j)\n    plt.imshow(i)\n    j+=1","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:42:22.965859Z","iopub.execute_input":"2023-10-19T14:42:22.966314Z","iopub.status.idle":"2023-10-19T14:42:59.129438Z","shell.execute_reply.started":"2023-10-19T14:42:22.966277Z","shell.execute_reply":"2023-10-19T14:42:59.127671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(image_data_1))\nprint(empty_img)","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:53:01.14759Z","iopub.execute_input":"2023-10-19T14:53:01.148028Z","iopub.status.idle":"2023-10-19T14:53:01.154503Z","shell.execute_reply.started":"2023-10-19T14:53:01.147998Z","shell.execute_reply":"2023-10-19T14:53:01.153611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_patch_image_array_folder[0].shape","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:53:01.15711Z","iopub.execute_input":"2023-10-19T14:53:01.157463Z","iopub.status.idle":"2023-10-19T14:53:01.170924Z","shell.execute_reply.started":"2023-10-19T14:53:01.157433Z","shell.execute_reply":"2023-10-19T14:53:01.170041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test_patch_image_array_folder[0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Convert into Array","metadata":{}},{"cell_type":"code","source":"#test_data = np.array(image_data_1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test_data.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load The Model ","metadata":{}},{"cell_type":"code","source":"# Recreate the exact same model, including its weights and the optimizer\n# path = \"/kaggle/input/mobilenet-model/mobilenet_model_1.h5\"\n# path2 = \"/kaggle/input/mobilenet-model-2/mobilenet_model_2.h5\"\n# path3 = \"/kaggle/input/mobilenet-model-3/mobilenet_model_3.h5\"\npath4 = \"/kaggle/input/new-mobilenet-model-128x128/new_mobilenet_model.h5\"\n\nmy_model = tf.keras.models.load_model(\n       (path4),\n       custom_objects={'KerasLayer':hub.KerasLayer}\n)\n# Show the model architecture\nmy_model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:53:46.86406Z","iopub.execute_input":"2023-10-19T14:53:46.86444Z","iopub.status.idle":"2023-10-19T14:53:49.046317Z","shell.execute_reply.started":"2023-10-19T14:53:46.864409Z","shell.execute_reply":"2023-10-19T14:53:49.044817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predicting Testing Images for Submission","metadata":{}},{"cell_type":"code","source":"#test_patch_image_array_folder[0].shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import Counter\nclass_labels = ['CC', 'EC', 'HGSC', 'LGSC', 'MC','Other']","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:55:53.868303Z","iopub.execute_input":"2023-10-19T14:55:53.868782Z","iopub.status.idle":"2023-10-19T14:55:53.875559Z","shell.execute_reply.started":"2023-10-19T14:55:53.868749Z","shell.execute_reply":"2023-10-19T14:55:53.874014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_submission_label = []\nfor i in test_patch_image_array_folder:\n    output = my_model.predict(i)\n    output = [np.argmax(j) for j in output]\n    \n    ## Use Counter to find most common\n    count = Counter(output)\n    result = count.most_common(1)[0][0]\n    test_submission_label.append(class_labels[result])","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:56:41.256813Z","iopub.execute_input":"2023-10-19T14:56:41.257204Z","iopub.status.idle":"2023-10-19T14:56:43.662404Z","shell.execute_reply.started":"2023-10-19T14:56:41.257176Z","shell.execute_reply":"2023-10-19T14:56:43.661514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#count.most_common()","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:56:46.716519Z","iopub.execute_input":"2023-10-19T14:56:46.717594Z","iopub.status.idle":"2023-10-19T14:56:46.72465Z","shell.execute_reply.started":"2023-10-19T14:56:46.717548Z","shell.execute_reply":"2023-10-19T14:56:46.723266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test_submission_label","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:56:56.747341Z","iopub.execute_input":"2023-10-19T14:56:56.748615Z","iopub.status.idle":"2023-10-19T14:56:56.755778Z","shell.execute_reply.started":"2023-10-19T14:56:56.748563Z","shell.execute_reply":"2023-10-19T14:56:56.754671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Final submission","metadata":{}},{"cell_type":"code","source":"# y_pred = my_model.predict(test_data)\n# y_pred_test = [class_labels[np.argmax(i)] for i in y_pred]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#y_pred_test","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming 'test_ids' are the IDs of test samples, and 'predictions' are the predicted values\nsubmission = pd.DataFrame({'image_id': test_df['image_id'] , 'label': test_submission_label})\nsubmission.to_csv('submission.csv', index=False)  # Save the CSV file","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:58:49.627405Z","iopub.execute_input":"2023-10-19T14:58:49.627836Z","iopub.status.idle":"2023-10-19T14:58:49.638822Z","shell.execute_reply.started":"2023-10-19T14:58:49.627805Z","shell.execute_reply":"2023-10-19T14:58:49.637282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_submission = pd.read_csv(\"submission.csv\")\nfinal_submission","metadata":{"execution":{"iopub.status.busy":"2023-10-19T14:58:50.864009Z","iopub.execute_input":"2023-10-19T14:58:50.864879Z","iopub.status.idle":"2023-10-19T14:58:50.878415Z","shell.execute_reply.started":"2023-10-19T14:58:50.864841Z","shell.execute_reply":"2023-10-19T14:58:50.877315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}