{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import time\nimport cv2 as cv\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport numpy as np\n\nfrom sklearn.svm import SVC\nfrom sklearn.model_selection import train_test_split, GridSearchCV, StratifiedKFold\nfrom sklearn.metrics import classification_report\nfrom sklearn.preprocessing import LabelEncoder\n\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, MaxPooling2D, Dropout, Conv2D, Flatten\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom keras.models import Sequential\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:35:36.642548Z","iopub.execute_input":"2023-11-09T10:35:36.643276Z","iopub.status.idle":"2023-11-09T10:35:53.517539Z","shell.execute_reply.started":"2023-11-09T10:35:36.643235Z","shell.execute_reply":"2023-11-09T10:35:53.516447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load the Data","metadata":{}},{"cell_type":"code","source":"# Load the Data\ntrain = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\n\n# is_ma = True\nistma_False = train[train[\"is_tma\"]==False]\nistma_False.tail()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:35:53.519741Z","iopub.execute_input":"2023-11-09T10:35:53.520976Z","iopub.status.idle":"2023-11-09T10:35:53.56553Z","shell.execute_reply.started":"2023-11-09T10:35:53.520935Z","shell.execute_reply":"2023-11-09T10:35:53.564514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"istma_False.count()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:35:53.566576Z","iopub.execute_input":"2023-11-09T10:35:53.56684Z","iopub.status.idle":"2023-11-09T10:35:53.579229Z","shell.execute_reply.started":"2023-11-09T10:35:53.566817Z","shell.execute_reply":"2023-11-09T10:35:53.578363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"code","source":"# Sample images for plotting\npath = \"/kaggle/input/UBC-OCEAN/train_thumbnails/65371_thumbnail.png\"\nimage = cv.imread(path, 0)\nimage = cv.resize(image, (224, 224), interpolation=cv.INTER_AREA)\n\n# CLAHE (Contrast Limited Adaptive Histogram Equalization)\nclahe = cv.createCLAHE(clipLimit=40.0, tileGridSize=(2, 2))\nclahe_img = clahe.apply(image)\n\n# Canny\n#sigma = 0.3\n#median = np.median(image)\n#lower = int(max(0, (1.0 - sigma) * median))\n#upper = int(min(255, (1.0 + sigma) * median))\n#auto_canny = cv.Canny(clahe_img, lower, upper)\n\n#  Plotting\nfig, ax = plt.subplots(1,2, figsize=(10,7))\nax[0].imshow(image, cmap='gray'),ax[0].set_title('grayscale'), ax[0].axis('off')\nax[1].imshow(clahe_img, cmap='gray'), ax[1].set_title('CLAHE'), ax[1].axis('off')\n#ax[2].imshow(auto_canny, cmap='gray'), ax[2].set_title('Canny'), ax[2].axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:35:53.581706Z","iopub.execute_input":"2023-11-09T10:35:53.581964Z","iopub.status.idle":"2023-11-09T10:35:54.501838Z","shell.execute_reply.started":"2023-11-09T10:35:53.581932Z","shell.execute_reply":"2023-11-09T10:35:54.500913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocessing(path):\n    image = cv.imread(path, 0)\n    image = cv.resize(image, (224, 224), interpolation = cv.INTER_AREA)\n    \n    # CLAHE (Contrast Limited Adaptive Histogram Equalization)\n    clahe = cv.createCLAHE(clipLimit=40.0, tileGridSize=(2,2))\n    clahe_img = clahe.apply(image)\n    \n    # Canny\n    sigma = 0.3\n    median = np.median(image)\n    lower = int(max(0, (1.0 - sigma) * median))\n    upper = int(min(255, (1.0 + sigma) * median))\n    auto_canny = cv.Canny(clahe_img, lower, upper)\n    \n    #otsu = cv.threshold(clahe_img, 0, 255, cv.THRESH_BINARY+cv.THRESH_OTSU)[1]\n    \n    w,h = clahe_img.shape\n    rgb_img = np.zeros((w,h,3), dtype=np.uint8)\n    rgb_img[:,:,0] = auto_canny\n    rgb_img[:,:,1] = auto_canny\n    rgb_img[:,:,2] = auto_canny\n            \n    return rgb_img","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:35:54.503017Z","iopub.execute_input":"2023-11-09T10:35:54.503307Z","iopub.status.idle":"2023-11-09T10:35:54.51113Z","shell.execute_reply.started":"2023-11-09T10:35:54.50328Z","shell.execute_reply":"2023-11-09T10:35:54.510344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_img(dataframe, path):\n    classes = []\n    images = []\n    for data in dataframe.iterrows():\n        img_path = os.path.join(path,f\"{str(data[1][0])}_thumbnail.png\")\n        \n        classes.append(data[1][1])\n        images.append(preprocessing(img_path))\n       \n    # Encode classes\n    le = LabelEncoder()\n    label = le.fit_transform(classes)\n    label_list = list(le.classes_)\n        \n    return images, label, label_list","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:35:54.512382Z","iopub.execute_input":"2023-11-09T10:35:54.513128Z","iopub.status.idle":"2023-11-09T10:35:54.531028Z","shell.execute_reply.started":"2023-11-09T10:35:54.513092Z","shell.execute_reply":"2023-11-09T10:35:54.530125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\nimages, label, label_list = data_img(istma_False, path)\n\nimages = np.array(images)\nprint(images.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:35:54.532019Z","iopub.execute_input":"2023-11-09T10:35:54.532259Z","iopub.status.idle":"2023-11-09T10:37:56.149244Z","shell.execute_reply.started":"2023-11-09T10:35:54.532238Z","shell.execute_reply":"2023-11-09T10:37:56.148214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Check what is the dataset balance or not?","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nplt.figure(figsize=(10, 4))\nsns.countplot(istma_False, x=\"label\")\n#plt.xticks(rotation=90)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:37:56.150659Z","iopub.execute_input":"2023-11-09T10:37:56.151516Z","iopub.status.idle":"2023-11-09T10:37:56.935855Z","shell.execute_reply.started":"2023-11-09T10:37:56.151476Z","shell.execute_reply":"2023-11-09T10:37:56.934911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model","metadata":{}},{"cell_type":"markdown","source":"### VGG16","metadata":{}},{"cell_type":"code","source":"# Sample\nrandom_n = np.random.randint(0,512,3)\n\nfig, ax = plt.subplots(1,3, figsize=(12, 8))\nj=0\nfor i in random_n:\n    ax[j].imshow(images[i], cmap='gray')\n    j+=1\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:37:56.937364Z","iopub.execute_input":"2023-11-09T10:37:56.937768Z","iopub.status.idle":"2023-11-09T10:37:57.553773Z","shell.execute_reply.started":"2023-11-09T10:37:56.937731Z","shell.execute_reply":"2023-11-09T10:37:57.552822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''# Augmentation\nimages = np.array(images)\n\n# split dataset\nX_train, X_test, y_train, y_test = train_test_split(images, label, test_size=0.2, \n                                                    random_state=42)\n\ndatagen = ImageDataGenerator(rescale=1.0 / 255.0)\ntrain_data = datagen.flow(\n    X_train,\n    y=y_train,\n    batch_size = 32)'''","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:37:57.556789Z","iopub.execute_input":"2023-11-09T10:37:57.55709Z","iopub.status.idle":"2023-11-09T10:37:57.563778Z","shell.execute_reply.started":"2023-11-09T10:37:57.557063Z","shell.execute_reply":"2023-11-09T10:37:57.56271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications.efficientnet import EfficientNetB0\nbase_model = EfficientNetB0(\n    include_top=False,\n    input_shape=(224, 224, 3),\n    weights='imagenet'\n)\n\nfor layer in base_model.layers:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:37:57.565101Z","iopub.execute_input":"2023-11-09T10:37:57.565365Z","iopub.status.idle":"2023-11-09T10:38:06.32499Z","shell.execute_reply.started":"2023-11-09T10:37:57.565335Z","shell.execute_reply":"2023-11-09T10:38:06.324034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = tf.keras.layers.GlobalAveragePooling2D()(base_model.output)\nx = tf.keras.layers.Dense(64, activation='relu')(x)\nx = tf.keras.layers.Dense(5, activation='softmax')(x)\n\nmodel = tf.keras.Model(inputs=base_model.input, outputs=x)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:38:06.326245Z","iopub.execute_input":"2023-11-09T10:38:06.326595Z","iopub.status.idle":"2023-11-09T10:38:06.910512Z","shell.execute_reply.started":"2023-11-09T10:38:06.326565Z","shell.execute_reply":"2023-11-09T10:38:06.908972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n    loss='sparse_categorical_crossentropy',\n    metrics= [\"accuracy\"]\n)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:38:06.911599Z","iopub.execute_input":"2023-11-09T10:38:06.911891Z","iopub.status.idle":"2023-11-09T10:38:06.967592Z","shell.execute_reply.started":"2023-11-09T10:38:06.911865Z","shell.execute_reply":"2023-11-09T10:38:06.966523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_callback = EarlyStopping(\n    monitor = 'val_loss', mode = 'min',\n    patience = 3\n)\n\nhistory = model.fit(\n    images, label,\n    validation_split = 0.2,\n    epochs = 100,\n    callbacks = [model_callback]\n)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:38:06.968753Z","iopub.execute_input":"2023-11-09T10:38:06.969032Z","iopub.status.idle":"2023-11-09T10:38:35.93876Z","shell.execute_reply.started":"2023-11-09T10:38:06.969008Z","shell.execute_reply":"2023-11-09T10:38:35.937892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model.evaluate(X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:38:35.940644Z","iopub.execute_input":"2023-11-09T10:38:35.941338Z","iopub.status.idle":"2023-11-09T10:38:35.945619Z","shell.execute_reply.started":"2023-11-09T10:38:35.941277Z","shell.execute_reply":"2023-11-09T10:38:35.94465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(8, 3))\nplt.plot(history.epoch, history.history['loss'])\n#plt.plot(history.epoch, history.history['val_loss'])\nplt.legend(['train loss', 'val loss'])\nplt.title('Loss Diagram')\nplt.xlabel('Epoch(s)')\nplt.ylabel('Loss')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:38:35.947057Z","iopub.execute_input":"2023-11-09T10:38:35.947457Z","iopub.status.idle":"2023-11-09T10:38:55.649893Z","shell.execute_reply.started":"2023-11-09T10:38:35.947419Z","shell.execute_reply":"2023-11-09T10:38:55.648963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Testing","metadata":{}},{"cell_type":"code","source":"test_path = \"/kaggle/input/UBC-OCEAN/test_thumbnails/41_thumbnail.png\"\nimage = preprocessing(test_path)\nimg_array = np.expand_dims(image,0)\nclasses = model.predict(img_array/255)\n\nplt.imshow(image, cmap='gray')\nplt.title(label_list[np.argmax(classes)])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:38:55.65106Z","iopub.execute_input":"2023-11-09T10:38:55.651371Z","iopub.status.idle":"2023-11-09T10:38:58.348739Z","shell.execute_reply.started":"2023-11-09T10:38:55.651344Z","shell.execute_reply":"2023-11-09T10:38:58.347777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/UBC-OCEAN/sample_submission.csv\")\nsample_submission['label'] = label_list[np.argmax(classes)]\n\n# Save the updated DataFrame to a CSV file\nsample_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:41:33.988965Z","iopub.execute_input":"2023-11-09T10:41:33.989702Z","iopub.status.idle":"2023-11-09T10:41:34.002376Z","shell.execute_reply.started":"2023-11-09T10:41:33.989667Z","shell.execute_reply":"2023-11-09T10:41:34.001609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T10:41:36.495562Z","iopub.execute_input":"2023-11-09T10:41:36.496485Z","iopub.status.idle":"2023-11-09T10:41:36.507775Z","shell.execute_reply.started":"2023-11-09T10:41:36.496442Z","shell.execute_reply":"2023-11-09T10:41:36.506486Z"},"trusted":true},"execution_count":null,"outputs":[]}]}