{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n#         break\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-21T17:01:30.345381Z","iopub.execute_input":"2023-10-21T17:01:30.346087Z","iopub.status.idle":"2023-10-21T17:01:30.351229Z","shell.execute_reply.started":"2023-10-21T17:01:30.346054Z","shell.execute_reply":"2023-10-21T17:01:30.350113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import All Libraries","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\nimport os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nfrom sklearn.metrics import classification_report , confusion_matrix , accuracy_score , auc\nfrom sklearn.model_selection import train_test_split\n\nimport cv2\n#from google.colab.patches import cv2_imshow\nfrom PIL import Image \nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras import Sequential\nfrom keras.layers import Input, Dense,Conv2D , MaxPooling2D, Flatten,BatchNormalization,Dropout\nfrom tensorflow.keras.preprocessing import image_dataset_from_directory\nimport tensorflow_hub as hub","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:02:02.152341Z","iopub.execute_input":"2023-10-21T17:02:02.152648Z","iopub.status.idle":"2023-10-21T17:02:14.291179Z","shell.execute_reply.started":"2023-10-21T17:02:02.152622Z","shell.execute_reply":"2023-10-21T17:02:14.290398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\n\ntrain_df = pd.read_csv(train_csv_path)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:02:14.292745Z","iopub.execute_input":"2023-10-21T17:02:14.293239Z","iopub.status.idle":"2023-10-21T17:02:14.333785Z","shell.execute_reply.started":"2023-10-21T17:02:14.293213Z","shell.execute_reply":"2023-10-21T17:02:14.332923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=train_df['label'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:02:14.334813Z","iopub.execute_input":"2023-10-21T17:02:14.335055Z","iopub.status.idle":"2023-10-21T17:02:14.582442Z","shell.execute_reply.started":"2023-10-21T17:02:14.335034Z","shell.execute_reply":"2023-10-21T17:02:14.581463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.scatter(train_df['image_width'],train_df['image_height'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:02:14.583487Z","iopub.execute_input":"2023-10-21T17:02:14.583802Z","iopub.status.idle":"2023-10-21T17:02:14.769577Z","shell.execute_reply.started":"2023-10-21T17:02:14.583778Z","shell.execute_reply":"2023-10-21T17:02:14.768772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['is_tma'] = train_df['is_tma'].astype('int8')","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:02:14.771957Z","iopub.execute_input":"2023-10-21T17:02:14.77223Z","iopub.status.idle":"2023-10-21T17:02:14.777304Z","shell.execute_reply.started":"2023-10-21T17:02:14.772207Z","shell.execute_reply":"2023-10-21T17:02:14.776407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['is_tma'].value_counts().plot(kind=\"pie\",autopct=\"%.1f%%\")\nplt.title(\"Image Distributions on Train Data\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:02:14.77829Z","iopub.execute_input":"2023-10-21T17:02:14.778569Z","iopub.status.idle":"2023-10-21T17:02:14.928703Z","shell.execute_reply.started":"2023-10-21T17:02:14.778538Z","shell.execute_reply":"2023-10-21T17:02:14.927803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_thumbnails = os.listdir(\"/kaggle/input/UBC-OCEAN/train_thumbnails\")\n# train_images = os.listdir(\"/kaggle/input/UBC-OCEAN/train_images\")","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:02:14.929861Z","iopub.execute_input":"2023-10-21T17:02:14.930395Z","iopub.status.idle":"2023-10-21T17:02:14.934383Z","shell.execute_reply.started":"2023-10-21T17:02:14.930363Z","shell.execute_reply":"2023-10-21T17:02:14.933481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_labels = ['CC', 'EC', 'HGSC', 'LGSC', 'MC','Other']\ntrain_df['label'] = train_df['label'].replace({'CC':0, 'EC':1, 'HGSC':2, 'LGSC':3, 'MC':4,'Other':5})\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:02:14.935705Z","iopub.execute_input":"2023-10-21T17:02:14.93626Z","iopub.status.idle":"2023-10-21T17:02:14.953439Z","shell.execute_reply.started":"2023-10-21T17:02:14.93623Z","shell.execute_reply":"2023-10-21T17:02:14.9522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['image_id']=train_df['image_id'].astype(\"int32\")\ntrain_df['label']=train_df['label'].astype(\"int8\")\ntrain_df['image_width']=train_df['image_width'].astype(\"int32\")\ntrain_df['image_height']=train_df['image_height'].astype(\"int32\")","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:02:14.955047Z","iopub.execute_input":"2023-10-21T17:02:14.955684Z","iopub.status.idle":"2023-10-21T17:02:14.964072Z","shell.execute_reply.started":"2023-10-21T17:02:14.955653Z","shell.execute_reply":"2023-10-21T17:02:14.963162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_df['label'].value_counts()\ntrain_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:02:14.967123Z","iopub.execute_input":"2023-10-21T17:02:14.967414Z","iopub.status.idle":"2023-10-21T17:02:14.989035Z","shell.execute_reply.started":"2023-10-21T17:02:14.967391Z","shell.execute_reply":"2023-10-21T17:02:14.988112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:02:14.990075Z","iopub.execute_input":"2023-10-21T17:02:14.990649Z","iopub.status.idle":"2023-10-21T17:02:15.005281Z","shell.execute_reply.started":"2023-10-21T17:02:14.990605Z","shell.execute_reply":"2023-10-21T17:02:15.004065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training Images Preprocessing","metadata":{}},{"cell_type":"code","source":"Image.MAX_IMAGE_PIXELS = 10000000000\n\nimage_data = []\nimage_label = []\n\nfor img_id, label, tma in zip(train_df['image_id'],train_df['label'] ,train_df['is_tma']):\n    #print(img_id, label,  tma)\n    if tma==0:\n        img_name = str(img_id)+\"_thumbnail.png\"\n        image = Image.open(\"/kaggle/input/UBC-OCEAN/train_thumbnails/\"+img_name)\n        image = image.resize((512,512))\n        image = np.array(image)\n        image_data.append(image)\n        image_label.append(label)\n        \n        \n    elif tma==1:\n        img_name = str(img_id)+\".png\"\n        image = Image.open(\"/kaggle/input/UBC-OCEAN/train_images/\"+img_name)\n        image = image.resize((512,512))\n        image = np.array(image)\n        image_data.append(image)\n        image_label.append(label)","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:02:15.007133Z","iopub.execute_input":"2023-10-21T17:02:15.007669Z","iopub.status.idle":"2023-10-21T17:05:03.426657Z","shell.execute_reply.started":"2023-10-21T17:02:15.007623Z","shell.execute_reply":"2023-10-21T17:05:03.425856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(image_data))\nprint(len(image_label))","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:05:03.427848Z","iopub.execute_input":"2023-10-21T17:05:03.428693Z","iopub.status.idle":"2023-10-21T17:05:03.433702Z","shell.execute_reply.started":"2023-10-21T17:05:03.428654Z","shell.execute_reply":"2023-10-21T17:05:03.432839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_data[0].shape","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:05:03.436574Z","iopub.execute_input":"2023-10-21T17:05:03.436846Z","iopub.status.idle":"2023-10-21T17:05:03.447009Z","shell.execute_reply.started":"2023-10-21T17:05:03.436824Z","shell.execute_reply":"2023-10-21T17:05:03.446121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Images Visualization","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(30,50))\nj=1\nfor i in range(60):\n    plt.subplot(10,6,j)\n    plt.imshow(image_data[i])\n    plt.title(f\"Label:{class_labels[image_label[i]]}\")\n    j+=1","metadata":{"execution":{"iopub.status.busy":"2023-10-21T14:24:35.063701Z","iopub.execute_input":"2023-10-21T14:24:35.064202Z","iopub.status.idle":"2023-10-21T14:24:55.23725Z","shell.execute_reply.started":"2023-10-21T14:24:35.064159Z","shell.execute_reply":"2023-10-21T14:24:55.235958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-10-21T06:47:14.693006Z","iopub.execute_input":"2023-10-21T06:47:14.693323Z","iopub.status.idle":"2023-10-21T06:47:14.702763Z","shell.execute_reply.started":"2023-10-21T06:47:14.693294Z","shell.execute_reply":"2023-10-21T06:47:14.701784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Covert image data into array for training","metadata":{}},{"cell_type":"code","source":"x = np.array(image_data)\ny = np.array(image_label)","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:05:49.603344Z","iopub.execute_input":"2023-10-21T17:05:49.604237Z","iopub.status.idle":"2023-10-21T17:05:49.729841Z","shell.execute_reply.started":"2023-10-21T17:05:49.604204Z","shell.execute_reply":"2023-10-21T17:05:49.72903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x.shape)\nprint(y.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:05:50.032569Z","iopub.execute_input":"2023-10-21T17:05:50.033009Z","iopub.status.idle":"2023-10-21T17:05:50.038043Z","shell.execute_reply.started":"2023-10-21T17:05:50.032971Z","shell.execute_reply":"2023-10-21T17:05:50.037109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import joblib\n\n# # Save the NumPy array to a file\n# joblib.dump(x, 'x_data.joblib')\n# joblib.dump(y, 'y_data.joblib')","metadata":{"execution":{"iopub.status.busy":"2023-10-21T14:26:27.042382Z","iopub.execute_input":"2023-10-21T14:26:27.042821Z","iopub.status.idle":"2023-10-21T14:26:27.594613Z","shell.execute_reply.started":"2023-10-21T14:26:27.042789Z","shell.execute_reply":"2023-10-21T14:26:27.593139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import joblib\n# x_data = joblib.load(\"/kaggle/input/ovarian-training-image-data-1024x1024/x_scaled_data.joblib\")\n# y_data = joblib.load(\"/kaggle/input/ovarian-training-image-data-1024x1024/y_data.joblib\")","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:00:42.580557Z","iopub.execute_input":"2023-10-21T17:00:42.581253Z","iopub.status.idle":"2023-10-21T17:00:44.198792Z","shell.execute_reply.started":"2023-10-21T17:00:42.581218Z","shell.execute_reply":"2023-10-21T17:00:44.197748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(x_data.shape)\n# print(y_data.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:00:44.3099Z","iopub.execute_input":"2023-10-21T17:00:44.310235Z","iopub.status.idle":"2023-10-21T17:00:44.315223Z","shell.execute_reply.started":"2023-10-21T17:00:44.310207Z","shell.execute_reply":"2023-10-21T17:00:44.314197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split The Data","metadata":{}},{"cell_type":"code","source":"x_train, x_test ,y_train, y_test = train_test_split(x , y , test_size=0.10, shuffle=True)\nprint(x_train.shape)\nprint(x_test.shape)\nprint(y_train.shape)\nprint(y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:06:05.69799Z","iopub.execute_input":"2023-10-21T17:06:05.69839Z","iopub.status.idle":"2023-10-21T17:06:05.822265Z","shell.execute_reply.started":"2023-10-21T17:06:05.698348Z","shell.execute_reply":"2023-10-21T17:06:05.82115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train Image Visualization","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(30,50))\nfor i in range(60):\n    plt.subplot(10,6,i+1)\n    plt.imshow(x_train[i])\n    plt.title(f\"Label:{class_labels[y_train[i]]}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-21T14:27:45.13762Z","iopub.execute_input":"2023-10-21T14:27:45.138091Z","iopub.status.idle":"2023-10-21T14:28:05.472099Z","shell.execute_reply.started":"2023-10-21T14:27:45.13806Z","shell.execute_reply":"2023-10-21T14:28:05.470866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Scale The Data","metadata":{}},{"cell_type":"code","source":"x_train_scaled = x_train/255\nx_test_scaled = x_test/255","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:06:16.67819Z","iopub.execute_input":"2023-10-21T17:06:16.679397Z","iopub.status.idle":"2023-10-21T17:06:17.74792Z","shell.execute_reply.started":"2023-10-21T17:06:16.679355Z","shell.execute_reply":"2023-10-21T17:06:17.747035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:06:21.142437Z","iopub.execute_input":"2023-10-21T17:06:21.142802Z","iopub.status.idle":"2023-10-21T17:06:21.148549Z","shell.execute_reply.started":"2023-10-21T17:06:21.142776Z","shell.execute_reply":"2023-10-21T17:06:21.1476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-10-21T06:47:33.864733Z","iopub.execute_input":"2023-10-21T06:47:33.865093Z","iopub.status.idle":"2023-10-21T06:47:33.869957Z","shell.execute_reply.started":"2023-10-21T06:47:33.865063Z","shell.execute_reply":"2023-10-21T06:47:33.868772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Building Using ResNet 152 Layer ","metadata":{}},{"cell_type":"markdown","source":"## Resnet 152","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import ResNet152\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Dense, Flatten\nfrom tensorflow.keras.models import Sequential\n\n# Define data augmentation parameters\ndatagen = ImageDataGenerator(\n    #rescale=1.0 / 255.0,    # Rescale pixel values to [0, 1]\n    rotation_range=20,      # Randomly rotate images by up to 20 degrees\n    width_shift_range=0.2,  # Randomly shift width by up to 20% of the image width\n    height_shift_range=0.2, # Randomly shift height by up to 20% of the image height\n    shear_range=0.2,        # Apply shear transformations\n    zoom_range=0.2,         # Apply random zoom in/out\n    horizontal_flip=True,   # Randomly flip images horizontally\n    fill_mode='nearest'     # Fill empty areas with the nearest pixel value\n)\n\n# Load the ResNet-152 model with pretrained weights (excluding top classification layers)\nbase_model = ResNet152(weights='imagenet', include_top=False, input_shape=(512, 512, 3))\n\n# Create a new Sequential model and add the ResNet-152 base\nmodel = Sequential()\nmodel.add(base_model)\n\n# Add a global average pooling layer to reduce the spatial dimensions\n#model.add(GlobalAveragePooling2D())\n\n# Add a custom Flatten layer\nmodel.add(Flatten())\n\n# Add a dense layer for classification (replace 'num_classes' with your number of classes)\nmodel.add(Dense(1024, activation='relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(units=6, activation='softmax'))\n\n\n# Make the base model layers non-trainable\nfor layer in base_model.layers:\n    layer.trainable = False\n\n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-10-21T14:59:27.834532Z","iopub.execute_input":"2023-10-21T14:59:27.835426Z","iopub.status.idle":"2023-10-21T14:59:42.658483Z","shell.execute_reply.started":"2023-10-21T14:59:27.83539Z","shell.execute_reply":"2023-10-21T14:59:42.657473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compile the model\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n# Train the model with data augmentation\nhistory1 = model.fit(datagen.flow(x_train, y_train, batch_size=64), \n                     validation_data=(x_test, y_test), epochs=6)","metadata":{"execution":{"iopub.status.busy":"2023-10-21T14:59:42.660399Z","iopub.execute_input":"2023-10-21T14:59:42.661219Z","iopub.status.idle":"2023-10-21T15:03:19.731314Z","shell.execute_reply.started":"2023-10-21T14:59:42.661181Z","shell.execute_reply":"2023-10-21T15:03:19.730409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the model to a file\nmodel.save(\"resnet_152_model_new.h5\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Evaluation on Test & Train Data","metadata":{}},{"cell_type":"code","source":"loss , acc = model.evaluate(x_test,y_test)\nprint(\"Accuracy on Test Data:\",acc)\nprint()\nloss , acc = model.evaluate(x_train,y_train)\nprint(\"Accuracy on Train Data:\",acc)","metadata":{"execution":{"iopub.status.busy":"2023-10-21T17:00:12.505741Z","iopub.execute_input":"2023-10-21T17:00:12.506139Z","iopub.status.idle":"2023-10-21T17:00:12.510378Z","shell.execute_reply.started":"2023-10-21T17:00:12.50611Z","shell.execute_reply":"2023-10-21T17:00:12.509493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}