{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom tqdm.auto import tqdm\nfrom skimage import io\nfrom skimage.transform import rescale, resize\n\nimport matplotlib.pyplot as plt\nimport pandas as pd\nfrom tqdm.auto import tqdm\nfrom skimage import io\nfrom skimage.transform import rescale, resize\n\nimport matplotlib.pyplot as plt\nfrom __future__ import absolute_import\nfrom __future__ import division\nfrom __future__ import print_function\n\nimport tensorflow as tf\nimport numpy as np\nfrom keras.models import Model\nfrom keras.layers import Input, Conv2D, Activation, Concatenate, MaxPooling2D, GlobalAveragePooling2D, Dense\nfrom tensorflow.keras.preprocessing import image\n\nimport torch\nfrom torch.utils.data import DataLoader\nfrom torchvision import transforms, datasets\nimport pandas as pd\nimport os\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.losses import CategoricalCrossentropy\nlayers =  tf.keras.layers\nmodels =tf.keras.models","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:31:41.450112Z","iopub.execute_input":"2023-11-16T11:31:41.450613Z","iopub.status.idle":"2023-11-16T11:31:53.886936Z","shell.execute_reply.started":"2023-11-16T11:31:41.450573Z","shell.execute_reply":"2023-11-16T11:31:53.886164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\nPath('/kaggle/working/Cancer_Data').mkdir(parents=True, exist_ok=True)\nPath('/kaggle/working/Cancer_Data/HGSC').mkdir(parents=True, exist_ok=True)\nPath('/kaggle/working/Cancer_Data/EC').mkdir(parents=True, exist_ok=True)\nPath('/kaggle/working/Cancer_Data/CC').mkdir(parents=True, exist_ok=True)\nPath('/kaggle/working/Cancer_Data/MC').mkdir(parents=True, exist_ok=True)\nPath('/kaggle/working/Cancer_Data/LGSC').mkdir(parents=True, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:31:53.888554Z","iopub.execute_input":"2023-11-16T11:31:53.889069Z","iopub.status.idle":"2023-11-16T11:31:53.895662Z","shell.execute_reply.started":"2023-11-16T11:31:53.889043Z","shell.execute_reply":"2023-11-16T11:31:53.894776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport os\n\ndef process_images_with_keyword(keyword):\n    #  read the CSV file into the DataFrame 'df'\n    df = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\n\n    # Compare the text in the column with the provided keyword\n    comparison_result = df['label'].str.contains(keyword, case=False)\n\n    # Use the comparison result\n    filtered_data = df[comparison_result]\n\n    # Get the \"image_id\" column and convert it to strings\n    image_ids = filtered_data['image_id'].astype(str)\n\n    # Define the source and destination directories\n    source_directory = '/kaggle/input/UBC-OCEAN/train_thumbnails/'\n    destination_directory = f'/kaggle/working/Cancer_Data/{keyword}/'\n\n    # Create the destination directory if it doesn't exist\n    os.makedirs(destination_directory, exist_ok=True)\n\n    # Iterate through image_ids, read images, and save resized images\n    for image_id in image_ids:\n        source_path = os.path.join(source_directory, f\"{image_id}_thumbnail.png\")\n        destination_path = os.path.join(destination_directory, f\"{image_id}.png\")\n        \n        # Load the image\n        image = cv2.imread(source_path)\n        \n        if image is not None:\n            # Resize the image\n            resized_image = cv2.resize(image, (224, 224))\n            \n            # Save the resized image\n            cv2.imwrite(destination_path, resized_image)\n        \n        # Print the paths for reference\n      #  print(\"Source Path:\", source_path)\n      #  print(\"Destination Path:\", destination_path)","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:31:53.896825Z","iopub.execute_input":"2023-11-16T11:31:53.897458Z","iopub.status.idle":"2023-11-16T11:31:54.096908Z","shell.execute_reply.started":"2023-11-16T11:31:53.897425Z","shell.execute_reply":"2023-11-16T11:31:54.096196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process_images_with_keyword(\"CC\")\nprocess_images_with_keyword(\"EC\")\nprocess_images_with_keyword(\"HGSC\")\nprocess_images_with_keyword(\"LGSC\")\nprocess_images_with_keyword(\"MC\")","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:31:54.099349Z","iopub.execute_input":"2023-11-16T11:31:54.099709Z","iopub.status.idle":"2023-11-16T11:33:29.42698Z","shell.execute_reply.started":"2023-11-16T11:31:54.099677Z","shell.execute_reply":"2023-11-16T11:33:29.426156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_set= '/kaggle/working/Cancer_Data/'\nfor i,d in enumerate([image_set]):\n    filepaths=[]\n    labels=[]\n    classlist=os.listdir(d)\n    for klass in classlist:\n        classpath=os.path.join(d,klass)\n        if os.path.isdir(classpath):\n            flist=os.listdir(classpath)\n            for f in flist:\n                fpath=os.path.join(classpath,f)\n                filepaths.append(fpath)\n                labels.append(klass)\n    Fseries= pd.Series(filepaths, name='filepaths')\n    Lseries=pd.Series(labels, name='labels')\n    lung_df=pd.concat([Fseries, Lseries], axis=1)\ndf=pd.concat([lung_df], axis =0).reset_index(drop=True)# make a combined dataframe\n\nprint(df['labels'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:33:29.428091Z","iopub.execute_input":"2023-11-16T11:33:29.428376Z","iopub.status.idle":"2023-11-16T11:33:29.444945Z","shell.execute_reply.started":"2023-11-16T11:33:29.428351Z","shell.execute_reply":"2023-11-16T11:33:29.444045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_split=.5\ntest_split=.25\ndummy_split=test_split/(1-train_split)\ntrain_df, dummy_df=train_test_split(df, train_size=train_split, shuffle=True, random_state=123)\ntest_df, valid_df=train_test_split(dummy_df, train_size=dummy_split, shuffle=True, random_state=123)\nprint ('train_df length: ', len(train_df), ' _test_df length: ', len(test_df), '  valid_df length: ', len(valid_df))","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:33:29.446049Z","iopub.execute_input":"2023-11-16T11:33:29.446342Z","iopub.status.idle":"2023-11-16T11:33:29.601614Z","shell.execute_reply.started":"2023-11-16T11:33:29.446318Z","shell.execute_reply":"2023-11-16T11:33:29.600788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\nheight=224\nwidth=224\nchannels=3\nbatch_size=8\nimg_shape=(height, width, channels)\nimg_size=(height, width)\nlength=len(test_df)\ntest_batch_size=sorted([int(length/n) for n in range(1,length+1) if length % n ==0 and length/n<=80],reverse=True)[0]\ntest_steps=int(length/test_batch_size)\nprint ( 'test batch size: ' ,test_batch_size, '  test steps: ', test_steps)\ndef scalar(img):\n    return img/127.5-1  # scale pixel between -1 and +1\ngen=ImageDataGenerator(preprocessing_function=scalar)\ntrain_set=gen.flow_from_dataframe( train_df, x_col='filepaths', y_col='labels', target_size=img_size, class_mode='categorical',\n                                    color_mode='rgb', shuffle=True, batch_size=batch_size)\ntest_set=gen.flow_from_dataframe( test_df, x_col='filepaths', y_col='labels', target_size=img_size, class_mode='categorical',\n                                    color_mode='rgb', shuffle=False, batch_size=test_batch_size)\nvalidate_set=gen.flow_from_dataframe( valid_df, x_col='filepaths', y_col='labels', target_size=img_size, class_mode='categorical',\n                                    color_mode='rgb', shuffle=True, batch_size=batch_size)","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:33:29.602883Z","iopub.execute_input":"2023-11-16T11:33:29.603293Z","iopub.status.idle":"2023-11-16T11:33:29.632964Z","shell.execute_reply.started":"2023-11-16T11:33:29.603255Z","shell.execute_reply":"2023-11-16T11:33:29.63222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**CNN**","metadata":{}},{"cell_type":"code","source":"import keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Activation\nfrom keras.optimizers import SGD\nmodel= keras.models.Sequential()\n\nmodel.add(keras.layers.Conv2D(32, 3, activation='relu', input_shape=(224, 224, 3)))\nmodel.add(keras.layers.Dropout(0.1))\nmodel.add(keras.layers.MaxPooling2D())\n\nmodel.add(keras.layers.Conv2D(64, 3, activation='relu'))\nmodel.add(keras.layers.Dropout(0.15))\nmodel.add(keras.layers.MaxPooling2D())\n\nmodel.add(keras.layers.Conv2D(128, 3, activation='relu'))\nmodel.add(keras.layers.Dropout(0.15))\nmodel.add(keras.layers.MaxPooling2D())\n\nmodel.add(keras.layers.Flatten())\nmodel.add(keras.layers.Dense(4096, activation='relu'))\nmodel.add(keras.layers.Dense(4096, activation='relu'))\nmodel.add(keras.layers.Dense(5, activation='softmax'))\n\nmodel.compile(optimizer='adam',loss='categorical_crossentropy', metrics=['accuracy'])\nmodel.summary()\nmodel.compile(loss = 'categorical_crossentropy', optimizer = 'adam', metrics = ['accuracy'])\n#executing the model\nhistory = model.fit(train_set, validation_data = (validate_set), epochs = 30, verbose = 1)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:33:29.634219Z","iopub.execute_input":"2023-11-16T11:33:29.634571Z","iopub.status.idle":"2023-11-16T11:34:40.435723Z","shell.execute_reply.started":"2023-11-16T11:33:29.63454Z","shell.execute_reply":"2023-11-16T11:34:40.43485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import f1_score\nY_pred = model.predict(test_set)\ny_pred = np.argmax(Y_pred ,axis =1)\n\npreds = model.predict(test_set,verbose=1)\npredictions = preds.copy()\npredictions[predictions <= 0.5] = 0\npredictions[predictions > 0.5] = 1\n\nprint('Classification Report')\ntarget_names = ['CC','EC','HGSC','LGSC','MC']\nprint(classification_report(test_set.classes, y_pred, target_names=target_names))\n\nfrom sklearn.metrics import classification_report,confusion_matrix\ncm = pd.DataFrame(data=confusion_matrix( y_true= test_set.classes, y_pred= y_pred, labels=[0, 1,2,3,4]), index=['Actual CC','Actual EC','Actual HGSC','Actual LGSC', 'Actual MC'],columns=['Predicted CC','Predicted EC','Predicted HGSC','Predicted LGSC', 'Predicted MC'])\nimport seaborn as sns\nsns.heatmap(cm,annot=True,fmt=\"d\",cmap=\"YlGn\")","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:34:40.437411Z","iopub.execute_input":"2023-11-16T11:34:40.438275Z","iopub.status.idle":"2023-11-16T11:34:42.72264Z","shell.execute_reply.started":"2023-11-16T11:34:40.438233Z","shell.execute_reply":"2023-11-16T11:34:42.721681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Applying Augmentation","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nheight = 224\nwidth = 224\nchannels = 3\nbatch_size = 8\nimg_shape = (height, width, channels)\nimg_size = (height, width)\nlength = len(test_df)\ntest_batch_size = sorted([int(length / n) for n in range(1, length + 1) if length % n == 0 and length / n <= 80], reverse=True)[0]\ntest_steps = int(length / test_batch_size)\nprint('test batch size:', test_batch_size, '  test steps:', test_steps)\n\n# Defining the augmentation function\ndef augment(image):\n    # Apply augmentation techniques here\n    # Example: use the ImageDataGenerator for augmentation\n    datagen = ImageDataGenerator(\n        rotation_range=40,\n        width_shift_range=0.2,\n        height_shift_range=0.2,\n        shear_range=0.2,\n        zoom_range=0.2,\n        horizontal_flip=True,\n        fill_mode='nearest',\n        preprocessing_function=scalar\n    )\n    return datagen.random_transform(image)\n\n# Define the scalar function\ndef scalar(img):\n    return img / 127.5 - 1  # scale pixel between -1 and +1\n\ngen = ImageDataGenerator(preprocessing_function=augment)\n\ntrain_set = gen.flow_from_dataframe(train_df, x_col='filepaths', y_col='labels', target_size=img_size,\n                                    class_mode='categorical', color_mode='rgb', shuffle=True, batch_size=batch_size)\n\ntest_set = gen.flow_from_dataframe(test_df, x_col='filepaths', y_col='labels', target_size=img_size,\n                                   class_mode='categorical', color_mode='rgb', shuffle=False, batch_size=test_batch_size)\n\nvalidate_set = gen.flow_from_dataframe(valid_df, x_col='filepaths', y_col='labels', target_size=img_size,\n                                       class_mode='categorical', color_mode='rgb', shuffle=True, batch_size=batch_size)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:34:42.72664Z","iopub.execute_input":"2023-11-16T11:34:42.726911Z","iopub.status.idle":"2023-11-16T11:34:42.75669Z","shell.execute_reply.started":"2023-11-16T11:34:42.726888Z","shell.execute_reply":"2023-11-16T11:34:42.755884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n# import random\n# from PIL import Image, ImageFilter, ImageEnhance \n# import numpy as np\n# import cv2\n\n# # Define the source directory containing class folders\n# source_directory = '/kaggle/working/Cancer_Data/'\n\n# # Define the directory to save augmented images\n# output_base_directory = '/kaggle/working/Augmented_Data/'\n\n# # Define the number of augmented images to generate per original image\n# num_augmented_images = 500\n\n# # Define a list of available augmentation techniques\n# augmentation_techniques = [\"rotate\", \"flip\", \"blur\", \"brightness\", \"contrast\"]\n\n# # Iterate through class folders and augment images\n# class_folders = os.listdir(source_directory)\n\n# for class_folder in class_folders:\n#     class_directory = os.path.join(source_directory, class_folder)\n#     output_directory = os.path.join(output_base_directory, f'{class_folder}_Augmented')\n\n#     # Create the output directory if it doesn't exist\n#     os.makedirs(output_directory, exist_ok=True)\n\n#     # Iterate through original images\n#     original_images = os.listdir(class_directory)\n#     for image_filename in original_images:\n#         image_path = os.path.join(class_directory, image_filename)\n#         img = Image.open(image_path)\n        \n#         # Apply random augmentations\n#         for _ in range(num_augmented_images):\n#             augmented_img = img.copy()\n\n#             # Randomly select an augmentation technique\n#             selected_technique = random.choice(augmentation_techniques)\n\n#             if selected_technique == \"rotate\":\n#                 angle = random.randint(-25, 25)\n#                 augmented_img = augmented_img.rotate(angle)\n\n#             elif selected_technique == \"flip\":\n#                 augmented_img = augmented_img.transpose(Image.FLIP_LEFT_RIGHT)\n\n#             elif selected_technique == \"blur\":\n#                 augmented_img = augmented_img.filter(ImageFilter.BLUR)\n\n#             elif selected_technique == \"brightness\":\n#                 enhancer = ImageEnhance.Brightness(augmented_img)\n#                 factor = random.uniform(0.7, 1.3)\n#                 augmented_img = enhancer.enhance(factor)\n\n#             elif selected_technique == \"contrast\":\n#                 enhancer = ImageEnhance.Contrast(augmented_img)\n#                 factor = random.uniform(0.7, 1.3)\n#                 augmented_img = enhancer.enhance(factor)\n\n#             # Save the augmented image\n#             output_filename = f\"{os.path.splitext(image_filename)[0]}_aug_{_}.jpg\"\n#             output_path = os.path.join(output_directory, output_filename)\n#             augmented_img.save(output_path)\n\n#     print(f\"Augmented {class_folder} class.\")\n\n# print(\"Augmentation complete. Augmented images saved in the 'Augmented_Data' folder.\")","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:34:42.758148Z","iopub.execute_input":"2023-11-16T11:34:42.758413Z","iopub.status.idle":"2023-11-16T11:34:42.76446Z","shell.execute_reply.started":"2023-11-16T11:34:42.758389Z","shell.execute_reply":"2023-11-16T11:34:42.763536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n# import pandas as pd\n\n# # Define the path to the output directory\n# output_base_directory = '/kaggle/working/Augmented_Data'\n\n# # Initialize lists to store file paths and labels\n# filepaths = []\n# labels = []\n\n# classlist = os.listdir(output_base_directory)\n\n# for klass in classlist:\n#     classpath = os.path.join(output_base_directory, klass)\n#     if os.path.isdir(classpath):\n#         flist = os.listdir(classpath)\n#         for f in flist:\n#             fpath = os.path.join(classpath, f)\n#             filepaths.append(fpath)\n#             labels.append(klass)\n\n# Fseries = pd.Series(filepaths, name='filepaths')\n# Lseries = pd.Series(labels, name='labels')\n\n# lung_df = pd.concat([Fseries, Lseries], axis=1)\n# df = pd.concat([lung_df], axis=0).reset_index(drop=True)\n\n# print(df['labels'].value_counts())\n","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:34:42.765539Z","iopub.execute_input":"2023-11-16T11:34:42.765829Z","iopub.status.idle":"2023-11-16T11:34:42.779323Z","shell.execute_reply.started":"2023-11-16T11:34:42.765805Z","shell.execute_reply":"2023-11-16T11:34:42.778602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_split=.5\n# test_split=.25\n# dummy_split=test_split/(1-train_split)\n# train_df, dummy_df=train_test_split(df, train_size=train_split, shuffle=True, random_state=123)\n# test_df, valid_df=train_test_split(dummy_df, train_size=dummy_split, shuffle=True, random_state=123)\n# print ('train_df length: ', len(train_df), ' _test_df length: ', len(test_df), '  valid_df length: ', len(valid_df))","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:34:42.78044Z","iopub.execute_input":"2023-11-16T11:34:42.780777Z","iopub.status.idle":"2023-11-16T11:34:42.789542Z","shell.execute_reply.started":"2023-11-16T11:34:42.780746Z","shell.execute_reply":"2023-11-16T11:34:42.788649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from tensorflow.keras.preprocessing.image import ImageDataGenerator\n# height=224\n# width=224\n# channels=3\n# batch_size=8\n# img_shape=(height, width, channels)\n# img_size=(height, width)\n# length=len(test_df)\n# test_batch_size=sorted([int(length/n) for n in range(1,length+1) if length % n ==0 and length/n<=80],reverse=True)[0]\n# test_steps=int(length/test_batch_size)\n# print ( 'test batch size: ' ,test_batch_size, '  test steps: ', test_steps)\n# def scalar(img):\n#     return img/127.5-1  # scale pixel between -1 and +1\n# gen=ImageDataGenerator(preprocessing_function=scalar)\n# train_set=gen.flow_from_dataframe( train_df, x_col='filepaths', y_col='labels', target_size=img_size, class_mode='categorical',\n#                                     color_mode='rgb', shuffle=True, batch_size=batch_size)\n# test_set=gen.flow_from_dataframe( test_df, x_col='filepaths', y_col='labels', target_size=img_size, class_mode='categorical',\n#                                     color_mode='rgb', shuffle=False, batch_size=test_batch_size)\n# validate_set=gen.flow_from_dataframe( valid_df, x_col='filepaths', y_col='labels', target_size=img_size, class_mode='categorical',\n#                                     color_mode='rgb', shuffle=True, batch_size=batch_size)","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:34:42.790597Z","iopub.execute_input":"2023-11-16T11:34:42.790924Z","iopub.status.idle":"2023-11-16T11:34:42.801146Z","shell.execute_reply.started":"2023-11-16T11:34:42.790897Z","shell.execute_reply":"2023-11-16T11:34:42.800221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Activation\nfrom keras.optimizers import SGD\nmodel= keras.models.Sequential()\n\nmodel.add(keras.layers.Conv2D(32, 3, activation='relu', input_shape=(224, 224, 3)))\nmodel.add(keras.layers.Dropout(0.1))\nmodel.add(keras.layers.MaxPooling2D())\n\nmodel.add(keras.layers.Conv2D(64, 3, activation='relu'))\nmodel.add(keras.layers.Dropout(0.15))\nmodel.add(keras.layers.MaxPooling2D())\n\nmodel.add(keras.layers.Conv2D(128, 3, activation='relu'))\nmodel.add(keras.layers.Dropout(0.15))\nmodel.add(keras.layers.MaxPooling2D())\n\nmodel.add(keras.layers.Flatten())\nmodel.add(keras.layers.Dense(256, activation='relu'))\nmodel.add(keras.layers.Dense(5, activation='softmax'))\n\nmodel.compile(optimizer='adam',loss='categorical_crossentropy', metrics=['accuracy'])\nmodel.summary()\nmodel.compile(loss = 'categorical_crossentropy', optimizer = 'adam', metrics = ['accuracy'])\n#executing the model\nhistory = model.fit(train_set, validation_data = (validate_set), epochs = 30, verbose = 1)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:34:42.802343Z","iopub.execute_input":"2023-11-16T11:34:42.802702Z","iopub.status.idle":"2023-11-16T11:37:00.052295Z","shell.execute_reply.started":"2023-11-16T11:34:42.802678Z","shell.execute_reply":"2023-11-16T11:37:00.051518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import f1_score\nY_pred = model.predict(test_set)\ny_pred = np.argmax(Y_pred ,axis =1)\n\npreds = model.predict(test_set,verbose=1)\npredictions = preds.copy()\npredictions[predictions <= 0.5] = 0\npredictions[predictions > 0.5] = 1\n\nprint('Classification Report')\ntarget_names = ['CC','EC','HGSC','LGSC','MC']\nprint(classification_report(test_set.classes, y_pred, target_names=target_names))\n\nfrom sklearn.metrics import classification_report,confusion_matrix\ncm = pd.DataFrame(data=confusion_matrix( y_true= test_set.classes, y_pred= y_pred, labels=[0, 1,2,3,4]), index=['Actual CC','Actual EC','Actual HGSC','Actual LGSC', 'Actual MC'],columns=['Predicted CC','Predicted EC','Predicted HGSC','Predicted LGSC', 'Predicted MC'])\nimport seaborn as sns\nsns.heatmap(cm,annot=True,fmt=\"d\",cmap=\"YlGn\")","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:37:00.053606Z","iopub.execute_input":"2023-11-16T11:37:00.053883Z","iopub.status.idle":"2023-11-16T11:37:05.397712Z","shell.execute_reply.started":"2023-11-16T11:37:00.053858Z","shell.execute_reply":"2023-11-16T11:37:05.396716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # plotting the loss\n# plt.plot(history.history['loss'],label = 'Training loss')\n# plt.plot(history.history['val_loss'], label = 'Validation loss')\n# #plt.title('loss')\n# plt.legend()\n# plt.show()\n\n# # Both Validation and Training accuracy is shown here\n# plt.plot(history.history['accuracy'], label='Training accuracy')\n# plt.plot(history.history['val_accuracy'], label='Validation accuracy')\n# #plt.title('Accuracy')\n# plt.legend()\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:37:05.398966Z","iopub.execute_input":"2023-11-16T11:37:05.399307Z","iopub.status.idle":"2023-11-16T11:37:05.403484Z","shell.execute_reply.started":"2023-11-16T11:37:05.399278Z","shell.execute_reply":"2023-11-16T11:37:05.402615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from keras.applications import MobileNet\n\n# from keras.layers import Dense, Flatten, Conv2D, MaxPooling2D, Dropout, Input\n\n# img_size = [224, 224]\n# MobileNet = MobileNet(input_shape=img_size + [3], weights='imagenet', include_top=False)# don't train existing weights\n# # don't train existing weights\n# for layer in MobileNet.layers:\n#     layer.trainable = False\n\n# flatten = Flatten()(MobileNet.output)\n# dense = Dense(256, activation = 'relu')(flatten)\n# dense = Dense(128, activation = 'relu')(dense)\n# prediction = Dense(5, activation = 'softmax')(dense)\n    \n\n# #creating a model\n# model_2 = Model(inputs = MobileNet.input, outputs = prediction )\n# model_2.summary()\n\n# model_2.compile(loss = 'categorical_crossentropy', optimizer = 'adam', metrics = ['accuracy'])\n# #executing the model\n# history = model_2.fit(train_set, validation_data = (validate_set), epochs = 50, verbose = 1)\n\n# from sklearn.metrics import classification_report\n# from sklearn.metrics import confusion_matrix\n# from sklearn.metrics import f1_score\n# Y_pred = model_2.predict(test_set)\n# y_pred = np.argmax(Y_pred ,axis =1)\n# preds = model_2.predict(test_set,verbose=1)\n# predictions = preds.copy()\n# predictions[predictions <= 0.5] = 0\n# predictions[predictions > 0.5] = 1\n\n# print('Classification Report')\n# target_names = ['CC','EC','HGSC','LGSC','MC']\n# print(classification_report(test_set.classes, y_pred, target_names=target_names))\n\n# from sklearn.metrics import classification_report,confusion_matrix\n# cm = pd.DataFrame(data=confusion_matrix( y_true= test_set.classes, y_pred= y_pred, labels=[0, 1,2,3,4]), index=['Actual CC','Actual EC','Actual HGSC','Actual LGSC', 'Actual MC'],columns=['Predicted CC','Predicted EC','Predicted HGSC','Predicted LGSC', 'Predicted MC'])\n# import seaborn as sns\n# sns.heatmap(cm,annot=True,fmt=\"d\",cmap=\"YlGn\")","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:37:05.404559Z","iopub.execute_input":"2023-11-16T11:37:05.404814Z","iopub.status.idle":"2023-11-16T11:37:05.428756Z","shell.execute_reply.started":"2023-11-16T11:37:05.404792Z","shell.execute_reply":"2023-11-16T11:37:05.428023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **TESTING**","metadata":{}},{"cell_type":"code","source":"#Test thumbnail\nfrom skimage.io import imread\n\nfrom matplotlib.pyplot import imshow\nimg = imread(\"/kaggle/input/UBC-OCEAN/test_thumbnails/41_thumbnail.png\")\nimshow(img)\nprint(f\"Original Dimensions : {img.shape}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:43:06.181142Z","iopub.execute_input":"2023-11-16T11:43:06.181529Z","iopub.status.idle":"2023-11-16T11:43:07.60197Z","shell.execute_reply.started":"2023-11-16T11:43:06.181497Z","shell.execute_reply":"2023-11-16T11:43:07.601032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a function to import an image and resize it to be able to be used with our model\ndef load_and_prep_image(filename, img_shape=224):\n\n  # Read in target file (an image)\n  img = tf.io.read_file(filename)\n\n  # Decode the read file into a tensor & ensure 3 colour channels\n  # (our model is trained on images with 3 colour channels and sometimes images have 4 colour channels)\n  img = tf.image.decode_image(img, channels=3)\n\n  # Resize the image (to the same size our model was trained on)\n  img = tf.image.resize(img, size = [img_shape, img_shape])\n\n  # Rescale the image (get all values between 0 and 1)\n  img = img/255.\n  return img\n\ndef pred_and_plot(model, filename, target_names):\n  \"\"\"\n  Imports an image located at filename, makes a prediction on it with\n  a trained model and plots the image with the predicted class as the title.\n  \"\"\"\n  # Import the target image and preprocess it\n  img = load_and_prep_image(filename)\n\n  # Make a prediction\n  pred = model.predict(tf.expand_dims(img, axis=0))\n\n  # Get the predicted class\n  pred_class = target_names[int(tf.round(pred)[0][0])]\n\n  # Plot the image and predicted class\n  plt.imshow(img)\n  plt.title(f\"Prediction: {pred_class}\")\n  plt.axis(False)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:55:26.98087Z","iopub.execute_input":"2023-11-16T11:55:26.981229Z","iopub.status.idle":"2023-11-16T11:55:26.989377Z","shell.execute_reply.started":"2023-11-16T11:55:26.981201Z","shell.execute_reply":"2023-11-16T11:55:26.988354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test our model on a test image\npred_and_plot(model, \"/kaggle/input/UBC-OCEAN/test_thumbnails/41_thumbnail.png\", target_names)","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:52:01.078128Z","iopub.execute_input":"2023-11-16T11:52:01.078452Z","iopub.status.idle":"2023-11-16T11:52:01.531358Z","shell.execute_reply.started":"2023-11-16T11:52:01.078427Z","shell.execute_reply":"2023-11-16T11:52:01.53044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_class(model, filename, target_names):\n    \"\"\"\n    Imports an image located at filename, makes a prediction on it with\n    a trained model, and returns the predicted class.\n    \"\"\"\n    # Import the target image and preprocess it\n    img = load_and_prep_image(filename)\n\n    # Make a prediction\n    pred = model.predict(tf.expand_dims(img, axis=0))\n\n    # Get the predicted class\n    pred_class = target_names[int(tf.round(pred)[0][0])]\n\n    return pred_class","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:56:56.302761Z","iopub.execute_input":"2023-11-16T11:56:56.303198Z","iopub.status.idle":"2023-11-16T11:56:56.313308Z","shell.execute_reply.started":"2023-11-16T11:56:56.303151Z","shell.execute_reply":"2023-11-16T11:56:56.309543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filename = \"/kaggle/input/UBC-OCEAN/test_thumbnails/41_thumbnail.png\"  # Replace with the path to your image\npred_class = predict_class(model, filename, target_names)\n\n# Print the predicted class\nprint(\"Predicted Class:\", pred_class)","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:57:36.767881Z","iopub.execute_input":"2023-11-16T11:57:36.76826Z","iopub.status.idle":"2023-11-16T11:57:36.971046Z","shell.execute_reply.started":"2023-11-16T11:57:36.76823Z","shell.execute_reply":"2023-11-16T11:57:36.970097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 4. Save the predictions to a submission.csv file\ntest_df = pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')\nsubmission_df = pd.DataFrame({\n    'image_id': test_df['image_id'],\n    'label': pred_class\n})\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-16T11:58:49.058063Z","iopub.execute_input":"2023-11-16T11:58:49.058478Z","iopub.status.idle":"2023-11-16T11:58:49.067326Z","shell.execute_reply.started":"2023-11-16T11:58:49.058447Z","shell.execute_reply":"2023-11-16T11:58:49.0662Z"},"trusted":true},"execution_count":null,"outputs":[]}]}