{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import time\nimport cv2 as cv\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport numpy as np\n\nfrom sklearn.svm import SVC\nfrom sklearn.model_selection import train_test_split, GridSearchCV, StratifiedKFold\nfrom sklearn.metrics import classification_report\nfrom sklearn.preprocessing import LabelEncoder\n\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, MaxPooling2D, Dropout, Conv2D, Flatten\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom keras.models import Sequential\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2023-11-07T12:30:59.568661Z","iopub.execute_input":"2023-11-07T12:30:59.568982Z","iopub.status.idle":"2023-11-07T12:31:03.640853Z","shell.execute_reply.started":"2023-11-07T12:30:59.568951Z","shell.execute_reply":"2023-11-07T12:31:03.640053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load the Data","metadata":{}},{"cell_type":"code","source":"# Load the Data\ntrain = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\n\n# is_ma = True\nistma_False = train[train[\"is_tma\"]==False]\nistma_False.tail()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T12:31:03.642252Z","iopub.execute_input":"2023-11-07T12:31:03.642749Z","iopub.status.idle":"2023-11-07T12:31:03.659902Z","shell.execute_reply.started":"2023-11-07T12:31:03.64272Z","shell.execute_reply":"2023-11-07T12:31:03.65893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"istma_False.count()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T12:31:03.661019Z","iopub.execute_input":"2023-11-07T12:31:03.661345Z","iopub.status.idle":"2023-11-07T12:31:03.669758Z","shell.execute_reply.started":"2023-11-07T12:31:03.661318Z","shell.execute_reply":"2023-11-07T12:31:03.66881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"code","source":"# Sample images for plotting\npath = \"/kaggle/input/UBC-OCEAN/train_thumbnails/65371_thumbnail.png\"\nimage = cv.imread(path, 0)\nimage = cv.resize(image, (224, 224), interpolation=cv.INTER_AREA)\n\n# CLAHE (Contrast Limited Adaptive Histogram Equalization)\nclahe = cv.createCLAHE(clipLimit=40.0, tileGridSize=(2, 2))\nclahe_img = clahe.apply(image)\n\n# Canny\n#sigma = 0.3\n#median = np.median(image)\n#lower = int(max(0, (1.0 - sigma) * median))\n#upper = int(min(255, (1.0 + sigma) * median))\n#auto_canny = cv.Canny(clahe_img, lower, upper)\n\n#  Plotting\nfig, ax = plt.subplots(1,2, figsize=(10,7))\nax[0].imshow(image, cmap='gray'),ax[0].set_title('grayscale'), ax[0].axis('off')\nax[1].imshow(clahe_img, cmap='gray'), ax[1].set_title('CLAHE'), ax[1].axis('off')\n#ax[2].imshow(auto_canny, cmap='gray'), ax[2].set_title('Canny'), ax[2].axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T13:21:50.604457Z","iopub.execute_input":"2023-11-07T13:21:50.605196Z","iopub.status.idle":"2023-11-07T13:21:51.273658Z","shell.execute_reply.started":"2023-11-07T13:21:50.605159Z","shell.execute_reply":"2023-11-07T13:21:51.272613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocessing(path):\n    image = cv.imread(path, 0)\n    image = cv.resize(image, (224, 224), interpolation = cv.INTER_AREA)\n    \n    # CLAHE (Contrast Limited Adaptive Histogram Equalization)\n    clahe = cv.createCLAHE(clipLimit=40.0, tileGridSize=(2,2))\n    clahe_img = clahe.apply(image)\n    \n    # Canny\n    #sigma = 0.3\n    #median = np.median(image)\n    #lower = int(max(0, (1.0 - sigma) * median))\n    #upper = int(min(255, (1.0 + sigma) * median))\n    #auto_canny = cv.Canny(clahe_img, lower, upper)\n    \n    #otsu = cv.threshold(clahe_img, 0, 255, cv.THRESH_BINARY+cv.THRESH_OTSU)[1]\n    \n    w,h = clahe_img.shape\n    rgb_img = np.zeros((w,h,3), dtype=np.uint8)\n    rgb_img[:,:,0] = clahe_img\n    rgb_img[:,:,1] = clahe_img\n    rgb_img[:,:,2] = clahe_img\n            \n    return rgb_img","metadata":{"execution":{"iopub.status.busy":"2023-11-07T12:53:31.498145Z","iopub.execute_input":"2023-11-07T12:53:31.499251Z","iopub.status.idle":"2023-11-07T12:53:31.506677Z","shell.execute_reply.started":"2023-11-07T12:53:31.499212Z","shell.execute_reply":"2023-11-07T12:53:31.505467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_img(dataframe, path):\n    classes = []\n    images = []\n    for data in dataframe.iterrows():\n        img_path = os.path.join(path,f\"{str(data[1][0])}_thumbnail.png\")\n        \n        classes.append(data[1][1])\n        images.append(preprocessing(img_path))\n       \n    # Encode classes\n    le = LabelEncoder()\n    label = le.fit_transform(classes)\n    label_list = list(le.classes_)\n        \n    return images, label, label_list","metadata":{"execution":{"iopub.status.busy":"2023-11-07T13:18:37.53652Z","iopub.execute_input":"2023-11-07T13:18:37.536915Z","iopub.status.idle":"2023-11-07T13:18:37.543539Z","shell.execute_reply.started":"2023-11-07T13:18:37.536884Z","shell.execute_reply":"2023-11-07T13:18:37.542527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\nimages, label, label_list = data_img(istma_False, path)\n\nimages = np.array(images)\nprint(images.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-07T13:18:39.7278Z","iopub.execute_input":"2023-11-07T13:18:39.728582Z","iopub.status.idle":"2023-11-07T13:20:16.831588Z","shell.execute_reply.started":"2023-11-07T13:18:39.728546Z","shell.execute_reply":"2023-11-07T13:20:16.830593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Check what is the dataset balance or not?","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nplt.figure(figsize=(10, 5))\nsns.countplot(istma_False, x=\"label\")\n#plt.xticks(rotation=90)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T12:32:42.299104Z","iopub.execute_input":"2023-11-07T12:32:42.29937Z","iopub.status.idle":"2023-11-07T12:32:42.794069Z","shell.execute_reply.started":"2023-11-07T12:32:42.299347Z","shell.execute_reply":"2023-11-07T12:32:42.793061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model","metadata":{}},{"cell_type":"markdown","source":"### VGG16","metadata":{}},{"cell_type":"code","source":"# Sample\nrandom_n = np.random.randint(0,512,3)\n\nfig, ax = plt.subplots(1,3, figsize=(12, 8))\nj=0\nfor i in random_n:\n    ax[j].imshow(images[i], cmap='gray')\n    j+=1\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T12:55:17.350013Z","iopub.execute_input":"2023-11-07T12:55:17.350344Z","iopub.status.idle":"2023-11-07T12:55:17.976902Z","shell.execute_reply.started":"2023-11-07T12:55:17.350316Z","shell.execute_reply":"2023-11-07T12:55:17.975876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Augmentation\nimages = np.array(images)\n\n# split dataset\nX_train, X_test, y_train, y_test = train_test_split(images, label, test_size=0.2, \n                                                    random_state=42)\n\ndatagen = ImageDataGenerator(rescale=1.0 / 255.0)\ntrain_data = datagen.flow(\n    X_train,\n    y=y_train,\n    batch_size = 32)","metadata":{"execution":{"iopub.status.busy":"2023-11-07T12:55:17.978462Z","iopub.execute_input":"2023-11-07T12:55:17.97879Z","iopub.status.idle":"2023-11-07T12:55:18.091645Z","shell.execute_reply.started":"2023-11-07T12:55:17.978761Z","shell.execute_reply":"2023-11-07T12:55:18.090742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import VGG19\nbase_model = VGG19(\n    include_top=False,\n    input_shape=(224, 224, 3),\n    weights='imagenet'\n)\n\nfor layer in base_model.layers:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2023-11-07T12:55:18.09413Z","iopub.execute_input":"2023-11-07T12:55:18.094438Z","iopub.status.idle":"2023-11-07T12:55:18.545841Z","shell.execute_reply.started":"2023-11-07T12:55:18.09441Z","shell.execute_reply":"2023-11-07T12:55:18.544715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = tf.keras.layers.GlobalAveragePooling2D()(base_model.output)\nx = tf.keras.layers.Dense(64, activation='relu')(x)\nx = tf.keras.layers.Dense(5, activation='softmax')(x)\n\nmodel = tf.keras.Model(inputs=base_model.input, outputs=x)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T12:55:18.547304Z","iopub.execute_input":"2023-11-07T12:55:18.547731Z","iopub.status.idle":"2023-11-07T12:55:18.629853Z","shell.execute_reply.started":"2023-11-07T12:55:18.547692Z","shell.execute_reply":"2023-11-07T12:55:18.62897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n    loss='sparse_categorical_crossentropy',\n    metrics= [\"accuracy\"]\n)","metadata":{"execution":{"iopub.status.busy":"2023-11-07T12:55:18.631031Z","iopub.execute_input":"2023-11-07T12:55:18.631332Z","iopub.status.idle":"2023-11-07T12:55:18.643218Z","shell.execute_reply.started":"2023-11-07T12:55:18.631306Z","shell.execute_reply":"2023-11-07T12:55:18.642345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_callback = EarlyStopping(\n    monitor = 'loss', mode = 'min',\n    patience = 3\n)\n\nhistory = model.fit(\n    train_data,\n    #validation_split = 0.2,\n    epochs = 100,\n    callbacks = [model_callback]\n)","metadata":{"execution":{"iopub.status.busy":"2023-11-07T12:55:18.644516Z","iopub.execute_input":"2023-11-07T12:55:18.644963Z","iopub.status.idle":"2023-11-07T12:58:24.293985Z","shell.execute_reply.started":"2023-11-07T12:55:18.64493Z","shell.execute_reply":"2023-11-07T12:58:24.293094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.evaluate(X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2023-11-07T12:58:24.295275Z","iopub.execute_input":"2023-11-07T12:58:24.295619Z","iopub.status.idle":"2023-11-07T12:58:25.125453Z","shell.execute_reply.started":"2023-11-07T12:58:24.295588Z","shell.execute_reply":"2023-11-07T12:58:25.124471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(8, 3))\nplt.plot(history.epoch, history.history['loss'])\n#plt.plot(history.epoch, history.history['val_loss'])\nplt.legend(['train loss', 'val loss'])\nplt.title('Loss Diagram')\nplt.xlabel('Epoch(s)')\nplt.ylabel('Loss')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T12:58:25.126866Z","iopub.execute_input":"2023-11-07T12:58:25.127305Z","iopub.status.idle":"2023-11-07T12:58:25.35044Z","shell.execute_reply.started":"2023-11-07T12:58:25.127267Z","shell.execute_reply":"2023-11-07T12:58:25.349394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Testing","metadata":{}},{"cell_type":"code","source":"test_path = \"/kaggle/input/UBC-OCEAN/test_thumbnails/41_thumbnail.png\"\nimage = preprocessing(test_path)\nimg_array = np.expand_dims(image,0)\nclasses = model.predict(img_array/255)\n\nplt.imshow(image, cmap='gray')\nplt.title(label_list[np.argmax(classes)])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T13:21:05.86301Z","iopub.execute_input":"2023-11-07T13:21:05.863394Z","iopub.status.idle":"2023-11-07T13:21:06.458284Z","shell.execute_reply.started":"2023-11-07T13:21:05.863363Z","shell.execute_reply":"2023-11-07T13:21:06.45732Z"},"trusted":true},"execution_count":null,"outputs":[]}]}