{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        break\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-15T07:41:35.359654Z","iopub.execute_input":"2023-10-15T07:41:35.360009Z","iopub.status.idle":"2023-10-15T07:41:35.903864Z","shell.execute_reply.started":"2023-10-15T07:41:35.359983Z","shell.execute_reply":"2023-10-15T07:41:35.902747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import All Libraries","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\nimport os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nfrom sklearn.metrics import classification_report , confusion_matrix , accuracy_score , auc\nfrom sklearn.model_selection import train_test_split\n\nimport cv2\n#from google.colab.patches import cv2_imshow\nfrom PIL import Image \nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras import Sequential\nfrom keras.layers import Input, Dense,Conv2D , MaxPooling2D, Flatten,BatchNormalization,Dropout\nfrom tensorflow.keras.preprocessing import image_dataset_from_directory\nimport tensorflow_hub as hub ","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:01:46.823452Z","iopub.execute_input":"2023-10-16T13:01:46.823826Z","iopub.status.idle":"2023-10-16T13:01:58.913429Z","shell.execute_reply.started":"2023-10-16T13:01:46.8238Z","shell.execute_reply":"2023-10-16T13:01:58.912333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\n#test_csv_path = \"/kaggle/input/UBC-OCEAN/test.csv\"\n\ntrain_df = pd.read_csv(train_csv_path)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:01:58.915159Z","iopub.execute_input":"2023-10-16T13:01:58.916385Z","iopub.status.idle":"2023-10-16T13:01:58.969088Z","shell.execute_reply.started":"2023-10-16T13:01:58.916333Z","shell.execute_reply":"2023-10-16T13:01:58.96768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=train_df['label'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:02:03.212411Z","iopub.execute_input":"2023-10-16T13:02:03.212858Z","iopub.status.idle":"2023-10-16T13:02:03.460897Z","shell.execute_reply.started":"2023-10-16T13:02:03.212824Z","shell.execute_reply":"2023-10-16T13:02:03.459513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.scatter(train_df['image_width'],train_df['image_height'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:02:12.022232Z","iopub.execute_input":"2023-10-16T13:02:12.023809Z","iopub.status.idle":"2023-10-16T13:02:12.260747Z","shell.execute_reply.started":"2023-10-16T13:02:12.023724Z","shell.execute_reply":"2023-10-16T13:02:12.259332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['is_tma'] = train_df['is_tma'].astype('int8')","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:02:16.942392Z","iopub.execute_input":"2023-10-16T13:02:16.942892Z","iopub.status.idle":"2023-10-16T13:02:16.950306Z","shell.execute_reply.started":"2023-10-16T13:02:16.942853Z","shell.execute_reply":"2023-10-16T13:02:16.949217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['is_tma'].value_counts().plot(kind=\"pie\",autopct=\"%.1f%%\")\nplt.title(\"Image Distributions on Train Data\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:02:19.410406Z","iopub.execute_input":"2023-10-16T13:02:19.410926Z","iopub.status.idle":"2023-10-16T13:02:19.620043Z","shell.execute_reply.started":"2023-10-16T13:02:19.410886Z","shell.execute_reply":"2023-10-16T13:02:19.618424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_thumbnails = os.listdir(\"/kaggle/input/UBC-OCEAN/train_thumbnails\")\ntrain_images = os.listdir(\"/kaggle/input/UBC-OCEAN/train_images\")","metadata":{"execution":{"iopub.status.busy":"2023-10-15T07:42:04.103852Z","iopub.execute_input":"2023-10-15T07:42:04.104536Z","iopub.status.idle":"2023-10-15T07:42:04.111181Z","shell.execute_reply.started":"2023-10-15T07:42:04.104505Z","shell.execute_reply":"2023-10-15T07:42:04.110016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_labels = ['CC', 'EC', 'HGSC', 'LGSC', 'MC']\ntrain_df['label'] = train_df['label'].replace({'CC':0, 'EC':1, 'HGSC':2, 'LGSC':3, 'MC':4})\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:02:27.131584Z","iopub.execute_input":"2023-10-16T13:02:27.132051Z","iopub.status.idle":"2023-10-16T13:02:27.146424Z","shell.execute_reply.started":"2023-10-16T13:02:27.132018Z","shell.execute_reply":"2023-10-16T13:02:27.145309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:02:31.816434Z","iopub.execute_input":"2023-10-16T13:02:31.816843Z","iopub.status.idle":"2023-10-16T13:02:31.827782Z","shell.execute_reply.started":"2023-10-16T13:02:31.816812Z","shell.execute_reply":"2023-10-16T13:02:31.82643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training Images Preprocessing","metadata":{}},{"cell_type":"code","source":"Image.MAX_IMAGE_PIXELS = 10000000000\n# Define patch size and overlap (if needed)\npatch_size = (224, 224)  # Adjust this according to your requirements\noverlap = 32  # Adjust this if you want overlapping patches\n\nimage_data = []\nimage_label = []\nempty_img=0\nfor img_id, label , tma in zip(train_df['image_id'],train_df['label'], train_df['is_tma']):\n    #print(img_id, label,  tma)\n    if tma==0:\n        img_name = str(img_id)+\"_thumbnail.png\"\n        large_image = Image.open(\"/kaggle/input/UBC-OCEAN/train_thumbnails/\"+img_name)\n        for y in range(0, large_image.height, patch_size[0] - overlap): # (0,2523,192)\n            for x in range(0, large_image.width, patch_size[1] - overlap):  # (0,3000,192)  224-32=192\n                patch = large_image.crop((x, y, x+patch_size[1], y+patch_size[0]))\n                image = np.array(patch)\n                if np.sum(image)==0:\n                    empty_img+=1\n                elif (np.sum(image[0:,0:100])==0) or (np.sum(image[0:,100:])==0) or (np.sum(image[0:,150:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:100,0:])==0) or (np.sum(image[100:,0:])==0) or (np.sum(image[150:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:100,0:100])==0) or (np.sum(image[0:100,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[100:,0:100])==0) or (np.sum(image[100:,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[150:,0:150])==0) or (np.sum(image[150:,150:])==0):\n                    empty_img+=1\n\n                else:\n                    image_data.append(image)\n                    image_label.append(label)\n        \n    elif tma==1:\n        img_name = str(img_id)+\".png\"\n        large_image = Image.open(\"/kaggle/input/UBC-OCEAN/train_images/\"+img_name)\n        for y in range(0, large_image.height, patch_size[0] - overlap): # (0,2523,192)\n            for x in range(0, large_image.width, patch_size[1] - overlap):  # (0,3000,192)  224-32=192\n                patch = large_image.crop((x, y, x+patch_size[1], y+patch_size[0]))\n                image = np.array(patch)\n                if np.sum(image)==0:\n                    empty_img+=1\n                elif (np.sum(image[0:,0:100])==0) or (np.sum(image[0:,100:])==0) or (np.sum(image[0:,150:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:100,0:])==0) or (np.sum(image[100:,0:])==0) or (np.sum(image[150:,0:])==0):\n                    empty_img+=1\n                elif (np.sum(image[0:100,0:100])==0) or (np.sum(image[0:100,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[100:,0:100])==0) or (np.sum(image[100:,100:])==0):\n                    empty_img+=1\n                elif (np.sum(image[150:,0:150])==0) or (np.sum(image[150:,150:])==0):\n                    empty_img+=1\n\n                else:\n                    image_data.append(image)\n                    image_label.append(label)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:04:53.932559Z","iopub.execute_input":"2023-10-16T13:04:53.933061Z","iopub.status.idle":"2023-10-16T13:08:25.147139Z","shell.execute_reply.started":"2023-10-16T13:04:53.933021Z","shell.execute_reply":"2023-10-16T13:08:25.145714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(image_data))\nprint(len(image_label))\nprint(empty_img)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:08:25.149545Z","iopub.execute_input":"2023-10-16T13:08:25.150008Z","iopub.status.idle":"2023-10-16T13:08:25.15783Z","shell.execute_reply.started":"2023-10-16T13:08:25.14997Z","shell.execute_reply":"2023-10-16T13:08:25.156276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set(image_label)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:08:25.159487Z","iopub.execute_input":"2023-10-16T13:08:25.159905Z","iopub.status.idle":"2023-10-16T13:08:25.188494Z","shell.execute_reply.started":"2023-10-16T13:08:25.159874Z","shell.execute_reply":"2023-10-16T13:08:25.187508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Covert image data into array for training","metadata":{}},{"cell_type":"code","source":"x = np.array(image_data)\ny = np.array(image_label)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:08:25.191033Z","iopub.execute_input":"2023-10-16T13:08:25.191384Z","iopub.status.idle":"2023-10-16T13:08:29.756913Z","shell.execute_reply.started":"2023-10-16T13:08:25.191356Z","shell.execute_reply":"2023-10-16T13:08:29.755467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x.shape)\nprint(y.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:08:29.758542Z","iopub.execute_input":"2023-10-16T13:08:29.758931Z","iopub.status.idle":"2023-10-16T13:08:29.765143Z","shell.execute_reply.started":"2023-10-16T13:08:29.758901Z","shell.execute_reply":"2023-10-16T13:08:29.764178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1 = x[:12000]\ny1 = y[:12000]","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:09:01.614748Z","iopub.execute_input":"2023-10-16T13:09:01.615749Z","iopub.status.idle":"2023-10-16T13:09:01.623467Z","shell.execute_reply.started":"2023-10-16T13:09:01.615704Z","shell.execute_reply":"2023-10-16T13:09:01.62204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split The Data","metadata":{}},{"cell_type":"code","source":"x_train, x_test ,y_train, y_test = train_test_split(x1, y1,test_size=0.10, shuffle=True)\nprint(x_train.shape)\nprint(x_test.shape)\nprint(y_train.shape)\nprint(y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:09:33.784454Z","iopub.execute_input":"2023-10-16T13:09:33.784876Z","iopub.status.idle":"2023-10-16T13:09:34.718731Z","shell.execute_reply.started":"2023-10-16T13:09:33.784848Z","shell.execute_reply":"2023-10-16T13:09:34.717363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train Image Visualization","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(30,50))\nfor i in range(60):\n    plt.subplot(10,6,i+1)\n    plt.imshow(x_train[i])\n    plt.title(f\"Label:{class_labels[y_train[i]]}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:14:40.974667Z","iopub.execute_input":"2023-10-16T13:14:40.975093Z","iopub.status.idle":"2023-10-16T13:14:59.566861Z","shell.execute_reply.started":"2023-10-16T13:14:40.975061Z","shell.execute_reply":"2023-10-16T13:14:59.56443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test Image Visualization","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(30,50))\nfor i in range(60):\n    plt.subplot(10,6,i+1)\n    plt.imshow(x_test[i])\n    plt.title(f\"Label:{class_labels[y_test[i]]}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:15:47.072421Z","iopub.execute_input":"2023-10-16T13:15:47.073088Z","iopub.status.idle":"2023-10-16T13:16:05.907429Z","shell.execute_reply.started":"2023-10-16T13:15:47.073046Z","shell.execute_reply":"2023-10-16T13:16:05.905104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Scale The Data","metadata":{}},{"cell_type":"code","source":"# x_train_scaled = x_train/255\n# x_test_scaled = x_test/255","metadata":{"execution":{"iopub.status.busy":"2023-10-15T07:49:38.630007Z","iopub.execute_input":"2023-10-15T07:49:38.631122Z","iopub.status.idle":"2023-10-15T07:49:39.886255Z","shell.execute_reply.started":"2023-10-15T07:49:38.631078Z","shell.execute_reply":"2023-10-15T07:49:39.884928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Building Using Pre-trained Model","metadata":{}},{"cell_type":"code","source":"## MobileNet V2 = 224x224x3\n#path = \"https://tfhub.dev/google/imagenet/mobilenet_v1_100_224/classification/5\"\npath = \"https://tfhub.dev/google/imagenet/mobilenet_v3_large_100_224/classification/5\"\nmobilenet_path = \"/kaggle/input/mobilenet-v2/tensorflow2/100-224-classification/2\"\n\nmobilenet_model_hub = hub.KerasLayer(path,  input_shape=(224,224,3), trainable=False)\n\nnum_class = 5\nmobilenet_model = Sequential()\nmobilenet_model.add(mobilenet_model_hub)\nmobilenet_model.add(Dense(500,activation='relu'))\nmobilenet_model.add(Dropout(0.2))\nmobilenet_model.add(Dense(500,activation='relu'))\nmobilenet_model.add(Dropout(0.2))\nmobilenet_model.add(Dense(units=num_class, activation=\"softmax\"))\n\n\nmobilenet_model.compile(optimizer=\"adam\",loss=\"sparse_categorical_crossentropy\",\n             metrics=[\"accuracy\"])\n\n\nmobilenet_model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:24:21.170111Z","iopub.execute_input":"2023-10-16T13:24:21.171288Z","iopub.status.idle":"2023-10-16T13:24:24.932782Z","shell.execute_reply.started":"2023-10-16T13:24:21.171234Z","shell.execute_reply":"2023-10-16T13:24:24.931308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = mobilenet_model.fit(x_train,y_train,epochs=10,\n         batch_size=64,validation_data=(x_test,y_test))","metadata":{"execution":{"iopub.status.busy":"2023-10-16T13:25:06.095366Z","iopub.execute_input":"2023-10-16T13:25:06.095886Z","iopub.status.idle":"2023-10-16T14:13:37.896601Z","shell.execute_reply.started":"2023-10-16T13:25:06.095848Z","shell.execute_reply":"2023-10-16T14:13:37.894549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Evaluation on Train & Test Data","metadata":{}},{"cell_type":"code","source":"loss ,acc = mobilenet_model.evaluate(x_train, y_train)\nprint(\"Accuracy on Train Data:\",acc)\nprint()\nloss ,acc = mobilenet_model.evaluate(x_test, y_test)\nprint(\"Accuracy on Test Data:\",acc)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T14:18:02.87525Z","iopub.execute_input":"2023-10-16T14:18:02.875686Z","iopub.status.idle":"2023-10-16T14:21:54.859306Z","shell.execute_reply.started":"2023-10-16T14:18:02.875655Z","shell.execute_reply":"2023-10-16T14:21:54.857686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = mobilenet_model.predict(x_test)\ny_pred_test = [np.argmax(i) for i in y_pred]","metadata":{"execution":{"iopub.status.busy":"2023-10-16T14:21:54.861249Z","iopub.execute_input":"2023-10-16T14:21:54.861572Z","iopub.status.idle":"2023-10-16T14:22:17.856263Z","shell.execute_reply.started":"2023-10-16T14:21:54.861547Z","shell.execute_reply":"2023-10-16T14:22:17.854434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(y_pred_test)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T14:22:17.858278Z","iopub.execute_input":"2023-10-16T14:22:17.858731Z","iopub.status.idle":"2023-10-16T14:22:17.869719Z","shell.execute_reply.started":"2023-10-16T14:22:17.85869Z","shell.execute_reply":"2023-10-16T14:22:17.867669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Metrics Evaluation on Test Data","metadata":{}},{"cell_type":"code","source":"print(\"Confusion Matrix:\\n\",confusion_matrix(y_test,y_pred_test))\nprint()\nprint(\"Classification Report:\\n\",classification_report(y_test,y_pred_test))","metadata":{"execution":{"iopub.status.busy":"2023-10-16T14:22:17.872074Z","iopub.execute_input":"2023-10-16T14:22:17.87252Z","iopub.status.idle":"2023-10-16T14:22:17.900061Z","shell.execute_reply.started":"2023-10-16T14:22:17.872485Z","shell.execute_reply":"2023-10-16T14:22:17.898898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Compare Actual & Predicted Labels","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(30,50))\nfor i in range(60):\n    plt.subplot(10,6,i+1)\n    plt.imshow(x_test[i])\n    plt.title(f\"Actual Label:{class_labels[y_test[i]]}\\nPredicted Label:{class_labels[y_pred_test[i]]}\")\n    plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2023-10-16T14:34:48.465938Z","iopub.execute_input":"2023-10-16T14:34:48.466437Z","iopub.status.idle":"2023-10-16T14:35:01.261067Z","shell.execute_reply.started":"2023-10-16T14:34:48.4664Z","shell.execute_reply":"2023-10-16T14:35:01.259795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}