{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# UBC Ovarian Cancer Subtype Classification and Outlier Detection (UBC-OCEAN)","metadata":{}},{"cell_type":"markdown","source":"### Import Libraries","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\n\nfrom statsmodels.tools.sm_exceptions import ConvergenceWarning\nwarnings.simplefilter(\"ignore\", ConvergenceWarning)\n\nimport pandas as pd\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\nfrom matplotlib.image import imread\n%matplotlib inline\nimport seaborn as sns\n\nfrom sklearn.metrics import classification_report , confusion_matrix , accuracy_score , auc\nfrom sklearn.model_selection import train_test_split\n\nimport cv2\n#from google.colab.patches import cv2_imshow\nfrom PIL import Image \nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras import Sequential\nfrom keras.layers import Input, Dense,Conv2D , MaxPooling2D, Flatten,BatchNormalization,Dropout\nfrom tensorflow.keras.preprocessing import image_dataset_from_directory\nimport tensorflow_hub as hub \n\nfrom keras.applications.vgg19 import VGG19\n\nimport joblib","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:01.85794Z","iopub.execute_input":"2024-05-18T20:07:01.85833Z","iopub.status.idle":"2024-05-18T20:07:14.80304Z","shell.execute_reply.started":"2024-05-18T20:07:01.858284Z","shell.execute_reply":"2024-05-18T20:07:14.801973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Import Training Data","metadata":{"toc-hr-collapsed":true}},{"cell_type":"code","source":"learn = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ndata = learn.copy()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-05-18T20:07:14.805067Z","iopub.execute_input":"2024-05-18T20:07:14.805749Z","iopub.status.idle":"2024-05-18T20:07:14.824267Z","shell.execute_reply.started":"2024-05-18T20:07:14.805718Z","shell.execute_reply":"2024-05-18T20:07:14.823249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-05-18T20:07:14.825501Z","iopub.execute_input":"2024-05-18T20:07:14.825877Z","iopub.status.idle":"2024-05-18T20:07:14.83323Z","shell.execute_reply.started":"2024-05-18T20:07:14.825849Z","shell.execute_reply":"2024-05-18T20:07:14.832068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model 1\nVGG-16 based model.","metadata":{}},{"cell_type":"code","source":"model1_train_data = data[data[\"is_tma\"]==False]\nmodel1_train_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:14.8346Z","iopub.execute_input":"2024-05-18T20:07:14.835017Z","iopub.status.idle":"2024-05-18T20:07:14.863921Z","shell.execute_reply.started":"2024-05-18T20:07:14.834986Z","shell.execute_reply":"2024-05-18T20:07:14.862873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_labels = ['CC', 'EC', 'HGSC', 'LGSC', 'MC']\nmodel1_train_data['label'] = model1_train_data['label'].replace({'CC':0, 'EC':1, 'HGSC':2, 'LGSC':3, 'MC':4})\nmodel1_train_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:14.866806Z","iopub.execute_input":"2024-05-18T20:07:14.86713Z","iopub.status.idle":"2024-05-18T20:07:14.88334Z","shell.execute_reply.started":"2024-05-18T20:07:14.867102Z","shell.execute_reply":"2024-05-18T20:07:14.882234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1_test_data = data[data[\"is_tma\"]==True]\nmodel1_test_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:14.884771Z","iopub.execute_input":"2024-05-18T20:07:14.885681Z","iopub.status.idle":"2024-05-18T20:07:14.89863Z","shell.execute_reply.started":"2024-05-18T20:07:14.885641Z","shell.execute_reply":"2024-05-18T20:07:14.897431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_labels = ['CC', 'EC', 'HGSC', 'LGSC', 'MC']\nmodel1_test_data['label'] = model1_test_data['label'].replace({'CC':0, 'EC':1, 'HGSC':2, 'LGSC':3, 'MC':4})\nmodel1_test_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:14.899861Z","iopub.execute_input":"2024-05-18T20:07:14.900272Z","iopub.status.idle":"2024-05-18T20:07:14.917255Z","shell.execute_reply.started":"2024-05-18T20:07:14.900232Z","shell.execute_reply":"2024-05-18T20:07:14.916372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tma_total_count = len(data[data[\"is_tma\"]==True])\ntma_total_count","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:14.918507Z","iopub.execute_input":"2024-05-18T20:07:14.918912Z","iopub.status.idle":"2024-05-18T20:07:14.930287Z","shell.execute_reply.started":"2024-05-18T20:07:14.918872Z","shell.execute_reply":"2024-05-18T20:07:14.929231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1_test_data[0:12]","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:14.93122Z","iopub.execute_input":"2024-05-18T20:07:14.931551Z","iopub.status.idle":"2024-05-18T20:07:14.948771Z","shell.execute_reply.started":"2024-05-18T20:07:14.931523Z","shell.execute_reply":"2024-05-18T20:07:14.947955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wsi_total_count = len(data[data[\"is_tma\"]==False])\nwsi_total_count","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:14.949816Z","iopub.execute_input":"2024-05-18T20:07:14.950599Z","iopub.status.idle":"2024-05-18T20:07:14.958441Z","shell.execute_reply.started":"2024-05-18T20:07:14.950567Z","shell.execute_reply":"2024-05-18T20:07:14.957196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1_train_data[0:50]","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:14.959682Z","iopub.execute_input":"2024-05-18T20:07:14.960514Z","iopub.status.idle":"2024-05-18T20:07:14.980214Z","shell.execute_reply.started":"2024-05-18T20:07:14.960483Z","shell.execute_reply":"2024-05-18T20:07:14.978991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1_train_data = model1_train_data[0:50] # Only first 50 elements will be used to train Model 1\nmodel1_test_data = model1_test_data[0:12] # Only first 12 of 25 TMAs available are used to test the model","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:14.981805Z","iopub.execute_input":"2024-05-18T20:07:14.982195Z","iopub.status.idle":"2024-05-18T20:07:14.991988Z","shell.execute_reply.started":"2024-05-18T20:07:14.982164Z","shell.execute_reply":"2024-05-18T20:07:14.990422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Image.MAX_IMAGE_PIXELS = 10000000000\n# Define patch size and overlap (if needed)\npatch_size = (128,128)  # Adjust this according to your requirements\noverlap = 0  # Adjust this if you want overlapping patches\n\nimage_data = []\nimage_label = []\nempty_img=0\nfor img_id, label , tma in zip(model1_train_data['image_id'],model1_train_data['label'], model1_train_data['is_tma']):\n    #print(img_id, label,  tma)\n    if tma==0:\n        img_name = str(img_id)+\"_thumbnail.png\"\n        large_image = Image.open(\"/kaggle/input/UBC-OCEAN/train_thumbnails/\"+img_name)\n        for y in range(0, large_image.height, patch_size[0] - overlap): # (0,2523,192)\n            for x in range(0, large_image.width, patch_size[1] - overlap):  # (0,3000,192)  224-32=192\n                patch = large_image.crop((x, y, x+patch_size[1], y+patch_size[0]))\n                image = np.array(patch)\n                if np.sum(image)==0:\n                    empty_img+=1\n                elif (np.sum(image[0:,0:50])==0) or (np.sum(image[0:,50:])==0) or (np.sum(image[0:,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,0:])==0) or (np.sum(image[50:,0:])==0) or (np.sum(image[100:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,0:50])==0) or (np.sum(image[0:50,50:])==0):\n                    empty_img+=1\n                elif (np.sum(image[50:100,0:50])==0) or (np.sum(image[50:100,50:])==0):\n                    empty_img+=1\n                elif (np.sum(image[50:,0:100])==0) or (np.sum(image[80:,80:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,75:])==0) or (np.sum(image[50:,75:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:40,80:])==0) or (np.sum(image[80:,80:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:35,0:])==0) or (np.sum(image[80:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:25,0:50])==0) or (np.sum(image[0:25,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:20,0:25])==0) or (np.sum(image[0:20,25:50])==0) or (np.sum(image[0:20,50:80])==0) or (np.sum(image[0:20,90:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:25,0:25])==0) or (np.sum(image[0:25,100:])==0) or (np.sum(image[100:,0:25])==0) or (np.sum(image[100:,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:15,0:])==0) or (np.sum(image[115:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:,0:15])==0) or (np.sum(image[0:,115:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:20,0:20])==0) or (np.sum(image[0:20,110:])==0) or (np.sum(image[110:,0:20])==0) or (np.sum(image[110:,110:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:10,0:10])==0) or (np.sum(image[0:10,115:])==0) or (np.sum(image[0:10,40:60])==0) or (np.sum(image[0:10,80:100])==0):\n                    empty_img+=1\n                elif (np.sum(image[40:50,0:10])==0) or (np.sum(image[110:,80:100])==0) or (np.sum(image[70:85,70:90])==0) or (np.sum(image[50:60,110:])==0):\n                    empty_img+=1\n\n                else:\n                    image_data.append(image)\n                    image_label.append(label)\n        \n    elif tma==1:\n        img_name = str(img_id)+\".png\"\n        large_image = Image.open(\"/kaggle/input/UBC-OCEAN/train_images/\"+img_name)\n        for y in range(0, large_image.height, patch_size[0] - overlap): # (0,2523,192)\n            for x in range(0, large_image.width, patch_size[1] - overlap):  # (0,3000,192)  224-32=192\n                patch = large_image.crop((x, y, x+patch_size[1], y+patch_size[0]))\n                image = np.array(patch)\n                if np.sum(image)==0:\n                    empty_img+=1\n                elif (np.sum(image[0:,0:50])==0) or (np.sum(image[0:,50:])==0) or (np.sum(image[0:,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,0:])==0) or (np.sum(image[50:,0:])==0) or (np.sum(image[100:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,0:50])==0) or (np.sum(image[0:50,50:])==0):\n                    empty_img+=1\n                elif (np.sum(image[50:100,0:50])==0) or (np.sum(image[50:100,50:])==0):\n                    empty_img+=1\n                elif (np.sum(image[50:,0:100])==0) or (np.sum(image[80:,80:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,75:])==0) or (np.sum(image[50:,75:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:40,80:])==0) or (np.sum(image[80:,80:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:35,0:])==0) or (np.sum(image[80:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:25,0:50])==0) or (np.sum(image[0:25,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:20,0:25])==0) or (np.sum(image[0:20,25:50])==0) or (np.sum(image[0:20,50:80])==0) or (np.sum(image[0:20,90:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:25,0:25])==0) or (np.sum(image[0:25,100:])==0) or (np.sum(image[100:,0:25])==0) or (np.sum(image[100:,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:15,0:])==0) or (np.sum(image[115:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:,0:15])==0) or (np.sum(image[0:,115:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:20,0:20])==0) or (np.sum(image[0:20,110:])==0) or (np.sum(image[110:,0:20])==0) or (np.sum(image[110:,110:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:10,0:10])==0) or (np.sum(image[0:10,115:])==0) or (np.sum(image[0:10,40:60])==0) or (np.sum(image[0:10,80:100])==0):\n                    empty_img+=1\n                elif (np.sum(image[40:50,0:10])==0) or (np.sum(image[110:,80:100])==0) or (np.sum(image[70:85,70:90])==0) or (np.sum(image[50:60,110:])==0):\n                    empty_img+=1\n                \n                    \n                else:\n                    image_data.append(image)\n                    image_label.append(label)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:14.994004Z","iopub.execute_input":"2024-05-18T20:07:14.994465Z","iopub.status.idle":"2024-05-18T20:07:38.082724Z","shell.execute_reply.started":"2024-05-18T20:07:14.994421Z","shell.execute_reply":"2024-05-18T20:07:38.081534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1_train_image_data = image_data\nprint(len(model1_train_image_data))\n\nmodel1_train_image_label = image_label\nprint(len(model1_train_image_label))\n\nprint(empty_img)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:38.090281Z","iopub.execute_input":"2024-05-18T20:07:38.09066Z","iopub.status.idle":"2024-05-18T20:07:38.096654Z","shell.execute_reply.started":"2024-05-18T20:07:38.09063Z","shell.execute_reply":"2024-05-18T20:07:38.09571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1_train_image_data[0].shape","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:38.098094Z","iopub.execute_input":"2024-05-18T20:07:38.098442Z","iopub.status.idle":"2024-05-18T20:07:38.108063Z","shell.execute_reply.started":"2024-05-18T20:07:38.098405Z","shell.execute_reply":"2024-05-18T20:07:38.107103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1_train_image_data[0]","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:38.109703Z","iopub.execute_input":"2024-05-18T20:07:38.110104Z","iopub.status.idle":"2024-05-18T20:07:38.120708Z","shell.execute_reply.started":"2024-05-18T20:07:38.110067Z","shell.execute_reply":"2024-05-18T20:07:38.119643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = np.array(model1_train_image_data) \ny_train = np.array(model1_train_image_label)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:38.122098Z","iopub.execute_input":"2024-05-18T20:07:38.122504Z","iopub.status.idle":"2024-05-18T20:07:38.31092Z","shell.execute_reply.started":"2024-05-18T20:07:38.122468Z","shell.execute_reply":"2024-05-18T20:07:38.309985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x_train.shape)\nprint(y_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:38.312233Z","iopub.execute_input":"2024-05-18T20:07:38.312652Z","iopub.status.idle":"2024-05-18T20:07:38.316913Z","shell.execute_reply.started":"2024-05-18T20:07:38.312623Z","shell.execute_reply":"2024-05-18T20:07:38.316129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the NumPy array to a file\n# joblib.dump(x3, 'x_24000.joblib')\n\njoblib.dump(x_train, 'x_data_train.joblib')\njoblib.dump(y_train, 'y_data_train.joblib')","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:38.318186Z","iopub.execute_input":"2024-05-18T20:07:38.318516Z","iopub.status.idle":"2024-05-18T20:07:38.87064Z","shell.execute_reply.started":"2024-05-18T20:07:38.318489Z","shell.execute_reply":"2024-05-18T20:07:38.869384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_data_train = joblib.load(\"/kaggle/working/x_data_train.joblib\")\ny_data_train = joblib.load(\"/kaggle/working/y_data_train.joblib\")\nprint(x_data_train.shape)\nprint(y_data_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:38.871887Z","iopub.execute_input":"2024-05-18T20:07:38.872731Z","iopub.status.idle":"2024-05-18T20:07:39.104384Z","shell.execute_reply.started":"2024-05-18T20:07:38.872695Z","shell.execute_reply":"2024-05-18T20:07:39.103298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Proceed the same preprocessing, but with testing data (corresponding to is_tma=True).","metadata":{}},{"cell_type":"code","source":"model1_test_data = data[data[\"is_tma\"]==True]\nmodel1_test_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:39.10578Z","iopub.execute_input":"2024-05-18T20:07:39.10612Z","iopub.status.idle":"2024-05-18T20:07:39.118964Z","shell.execute_reply.started":"2024-05-18T20:07:39.106092Z","shell.execute_reply":"2024-05-18T20:07:39.117771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_labels = ['CC', 'EC', 'HGSC', 'LGSC', 'MC']\nmodel1_test_data['label'] = model1_test_data['label'].replace({'CC':0, 'EC':1, 'HGSC':2, 'LGSC':3, 'MC':4})\nmodel1_test_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:39.120929Z","iopub.execute_input":"2024-05-18T20:07:39.121342Z","iopub.status.idle":"2024-05-18T20:07:39.135548Z","shell.execute_reply.started":"2024-05-18T20:07:39.121312Z","shell.execute_reply":"2024-05-18T20:07:39.134388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Image.MAX_IMAGE_PIXELS = 10000000000\n# Define patch size and overlap (if needed)\npatch_size = (128,128)  # Adjust this according to your requirements\noverlap = 0  # Adjust this if you want overlapping patches\n\nimage_data = []\nimage_label = []\nempty_img=0\nfor img_id, label , tma in zip(model1_test_data['image_id'],model1_test_data['label'], model1_test_data['is_tma']):\n    #print(img_id, label,  tma)\n    if tma==0:\n        img_name = str(img_id)+\"_thumbnail.png\"\n        large_image = Image.open(\"/kaggle/input/UBC-OCEAN/train_thumbnails/\"+img_name)\n        for y in range(0, large_image.height, patch_size[0] - overlap): # (0,2523,192)\n            for x in range(0, large_image.width, patch_size[1] - overlap):  # (0,3000,192)  224-32=192\n                patch = large_image.crop((x, y, x+patch_size[1], y+patch_size[0]))\n                image = np.array(patch)\n                if np.sum(image)==0:\n                    empty_img+=1\n                elif (np.sum(image[0:,0:50])==0) or (np.sum(image[0:,50:])==0) or (np.sum(image[0:,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,0:])==0) or (np.sum(image[50:,0:])==0) or (np.sum(image[100:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,0:50])==0) or (np.sum(image[0:50,50:])==0):\n                    empty_img+=1\n                elif (np.sum(image[50:100,0:50])==0) or (np.sum(image[50:100,50:])==0):\n                    empty_img+=1\n                elif (np.sum(image[50:,0:100])==0) or (np.sum(image[80:,80:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,75:])==0) or (np.sum(image[50:,75:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:40,80:])==0) or (np.sum(image[80:,80:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:35,0:])==0) or (np.sum(image[80:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:25,0:50])==0) or (np.sum(image[0:25,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:20,0:25])==0) or (np.sum(image[0:20,25:50])==0) or (np.sum(image[0:20,50:80])==0) or (np.sum(image[0:20,90:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:25,0:25])==0) or (np.sum(image[0:25,100:])==0) or (np.sum(image[100:,0:25])==0) or (np.sum(image[100:,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:15,0:])==0) or (np.sum(image[115:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:,0:15])==0) or (np.sum(image[0:,115:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:20,0:20])==0) or (np.sum(image[0:20,110:])==0) or (np.sum(image[110:,0:20])==0) or (np.sum(image[110:,110:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:10,0:10])==0) or (np.sum(image[0:10,115:])==0) or (np.sum(image[0:10,40:60])==0) or (np.sum(image[0:10,80:100])==0):\n                    empty_img+=1\n                elif (np.sum(image[40:50,0:10])==0) or (np.sum(image[110:,80:100])==0) or (np.sum(image[70:85,70:90])==0) or (np.sum(image[50:60,110:])==0):\n                    empty_img+=1\n\n                else:\n                    image_data.append(image)\n                    image_label.append(label)\n        \n    elif tma==1:\n        img_name = str(img_id)+\".png\"\n        large_image = Image.open(\"/kaggle/input/UBC-OCEAN/train_images/\"+img_name)\n        for y in range(0, large_image.height, patch_size[0] - overlap): # (0,2523,192)\n            for x in range(0, large_image.width, patch_size[1] - overlap):  # (0,3000,192)  224-32=192\n                patch = large_image.crop((x, y, x+patch_size[1], y+patch_size[0]))\n                image = np.array(patch)\n                if np.sum(image)==0:\n                    empty_img+=1\n                elif (np.sum(image[0:,0:50])==0) or (np.sum(image[0:,50:])==0) or (np.sum(image[0:,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,0:])==0) or (np.sum(image[50:,0:])==0) or (np.sum(image[100:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,0:50])==0) or (np.sum(image[0:50,50:])==0):\n                    empty_img+=1\n                elif (np.sum(image[50:100,0:50])==0) or (np.sum(image[50:100,50:])==0):\n                    empty_img+=1\n                elif (np.sum(image[50:,0:100])==0) or (np.sum(image[80:,80:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:50,75:])==0) or (np.sum(image[50:,75:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:40,80:])==0) or (np.sum(image[80:,80:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:35,0:])==0) or (np.sum(image[80:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:25,0:50])==0) or (np.sum(image[0:25,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:20,0:25])==0) or (np.sum(image[0:20,25:50])==0) or (np.sum(image[0:20,50:80])==0) or (np.sum(image[0:20,90:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:25,0:25])==0) or (np.sum(image[0:25,100:])==0) or (np.sum(image[100:,0:25])==0) or (np.sum(image[100:,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:15,0:])==0) or (np.sum(image[115:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:,0:15])==0) or (np.sum(image[0:,115:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:20,0:20])==0) or (np.sum(image[0:20,110:])==0) or (np.sum(image[110:,0:20])==0) or (np.sum(image[110:,110:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:10,0:10])==0) or (np.sum(image[0:10,115:])==0) or (np.sum(image[0:10,40:60])==0) or (np.sum(image[0:10,80:100])==0):\n                    empty_img+=1\n                elif (np.sum(image[40:50,0:10])==0) or (np.sum(image[110:,80:100])==0) or (np.sum(image[70:85,70:90])==0) or (np.sum(image[50:60,110:])==0):\n                    empty_img+=1\n                \n                    \n                else:\n                    image_data.append(image)\n                    image_label.append(label)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:07:39.137162Z","iopub.execute_input":"2024-05-18T20:07:39.137491Z","iopub.status.idle":"2024-05-18T20:08:09.45731Z","shell.execute_reply.started":"2024-05-18T20:07:39.137455Z","shell.execute_reply":"2024-05-18T20:08:09.456392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1_test_image_data = image_data\nprint(len(model1_test_image_data))\n\nmodel1_test_image_label = image_label\nprint(len(model1_test_image_label))\n\nprint(empty_img)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:08:09.458363Z","iopub.execute_input":"2024-05-18T20:08:09.45893Z","iopub.status.idle":"2024-05-18T20:08:09.46498Z","shell.execute_reply.started":"2024-05-18T20:08:09.458899Z","shell.execute_reply":"2024-05-18T20:08:09.463939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1_test_image_data[0].shape","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:08:09.466535Z","iopub.execute_input":"2024-05-18T20:08:09.466987Z","iopub.status.idle":"2024-05-18T20:08:09.476622Z","shell.execute_reply.started":"2024-05-18T20:08:09.466943Z","shell.execute_reply":"2024-05-18T20:08:09.47547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1_test_image_data[0]","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:08:09.478182Z","iopub.execute_input":"2024-05-18T20:08:09.479245Z","iopub.status.idle":"2024-05-18T20:08:09.488107Z","shell.execute_reply.started":"2024-05-18T20:08:09.479201Z","shell.execute_reply":"2024-05-18T20:08:09.487154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# converting variables on numoy arrays.\nx_test = np.array(model1_test_image_data) \ny_test = np.array(model1_test_image_label)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:08:09.489858Z","iopub.execute_input":"2024-05-18T20:08:09.490663Z","iopub.status.idle":"2024-05-18T20:08:09.827239Z","shell.execute_reply.started":"2024-05-18T20:08:09.490623Z","shell.execute_reply":"2024-05-18T20:08:09.826038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking variables shape\nprint(x_test.shape)\nprint(y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:08:09.828806Z","iopub.execute_input":"2024-05-18T20:08:09.829252Z","iopub.status.idle":"2024-05-18T20:08:09.834765Z","shell.execute_reply.started":"2024-05-18T20:08:09.829222Z","shell.execute_reply":"2024-05-18T20:08:09.833775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# exporting variables as job.lib\njoblib.dump(x_test, 'x_data_test.joblib')\njoblib.dump(y_test, 'y_data_test.joblib')","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:08:09.83633Z","iopub.execute_input":"2024-05-18T20:08:09.836701Z","iopub.status.idle":"2024-05-18T20:08:10.755915Z","shell.execute_reply.started":"2024-05-18T20:08:09.836672Z","shell.execute_reply":"2024-05-18T20:08:10.754842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# loading .joblib file on working folder\nx_data_test = joblib.load(\"/kaggle/working/x_data_test.joblib\")\ny_data_test = joblib.load(\"/kaggle/working/y_data_test.joblib\")\nprint(x_data_test.shape)\nprint(y_data_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:08:10.757646Z","iopub.execute_input":"2024-05-18T20:08:10.758067Z","iopub.status.idle":"2024-05-18T20:08:11.13482Z","shell.execute_reply.started":"2024-05-18T20:08:10.758027Z","shell.execute_reply":"2024-05-18T20:08:11.133645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# matching independent and target variables of training and testing datasets for model 1\nx_train = x_data_train \nx_test  = x_data_test\ny_train = y_data_train\ny_test  = y_data_test","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:08:11.13634Z","iopub.execute_input":"2024-05-18T20:08:11.136686Z","iopub.status.idle":"2024-05-18T20:08:11.143849Z","shell.execute_reply.started":"2024-05-18T20:08:11.136658Z","shell.execute_reply":"2024-05-18T20:08:11.143088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Scaling the data for train images\nx_train_scaled = x_train/255\nx_test_scaled = x_test/255","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:08:11.144692Z","iopub.execute_input":"2024-05-18T20:08:11.144985Z","iopub.status.idle":"2024-05-18T20:08:15.092703Z","shell.execute_reply.started":"2024-05-18T20:08:11.144959Z","shell.execute_reply":"2024-05-18T20:08:15.09147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model building using VGG16 Model available on Keras\n\nfrom tensorflow.keras.applications import VGG16\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, Flatten\nfrom tensorflow.keras.optimizers import Adam\nnum_classes = 5\n# Load the VGG16 model with ImageNet weights and exclude the top classification layer\nbase_model = VGG16(weights='imagenet', include_top=False, input_shape=(128,128, 3))\n\n# Customize the top classification layers\nx = base_model.output\nx = Flatten()(x)\nx = Dense(1000, activation='relu')(x) #original is 4096\nx = Dense(1000, activation='relu')(x) #original is 4096\npredictions = Dense(num_classes, activation='softmax')(x)  # Replace num_classes with the number of classes in your dataset\n\n# Create the VGG16 model with your custom top layer\nmodel = Model(inputs=base_model.input, outputs=predictions)\n\n# Freeze pre-trained layers (optional)\nfor layer in base_model.layers:\n    layer.trainable = False\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:08:15.093934Z","iopub.execute_input":"2024-05-18T20:08:15.094267Z","iopub.status.idle":"2024-05-18T20:08:16.578354Z","shell.execute_reply.started":"2024-05-18T20:08:15.094239Z","shell.execute_reply":"2024-05-18T20:08:16.570475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compile the model\nmodel.compile(optimizer=Adam(lr=0.0001), loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n# Train the model on your new dataset\nhistory = model.fit(x_train_scaled, y_train, epochs=5, batch_size=64,\n                   validation_data=(x_test_scaled,y_test))","metadata":{"execution":{"iopub.status.busy":"2024-05-18T20:08:16.579707Z","iopub.execute_input":"2024-05-18T20:08:16.580035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model evaluation on train % test data\nloss ,acc = model.evaluate(x_train_scaled, y_train)\nprint(\"Accuracy on Train Data:\",acc)\nprint()\nloss ,acc = model.evaluate(x_test_scaled, y_test )\nprint(\"Accuracy on Test Data:\",acc)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(x_test_scaled)\ny_pred_test = [np.argmax(i) for i in y_pred]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(y_pred_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the model to a file\nmodel.save(\"UBC-OCEAN-CHL1-model1.h5\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Metrics evaluation on test data\nprint(\"Confusion Matrix:\\n\",confusion_matrix(y_test,y_pred_test))\nprint()\nprint(\"Classification Report:\\n\",classification_report(y_test,y_pred_test))","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}