{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-11T04:02:56.037848Z","iopub.execute_input":"2023-11-11T04:02:56.038697Z","iopub.status.idle":"2023-11-11T04:02:56.398158Z","shell.execute_reply.started":"2023-11-11T04:02:56.03866Z","shell.execute_reply":"2023-11-11T04:02:56.397068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom tqdm.auto import tqdm\nfrom skimage import io\nfrom skimage.transform import rescale, resize\n\nimport matplotlib.pyplot as plt\nimport pandas as pd\nfrom tqdm.auto import tqdm\nfrom skimage.transform import rescale, resize\n\n\nfrom __future__ import absolute_import\nfrom __future__ import division\nfrom __future__ import print_function\n\nimport tensorflow as tf\nimport numpy as np\nfrom keras.models import Model\nfrom keras.layers import Input, Conv2D, Activation, Concatenate, MaxPooling2D, GlobalAveragePooling2D, Dense\nfrom tensorflow.keras.preprocessing import image\n\nimport torch\nfrom torch.utils.data import DataLoader\nfrom torchvision import transforms, datasets\nimport pandas as pd\nimport os\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.losses import CategoricalCrossentropy\nlayers =  tf.keras.layers\nmodels =tf.keras.models","metadata":{"execution":{"iopub.status.busy":"2023-11-11T04:02:56.400124Z","iopub.execute_input":"2023-11-11T04:02:56.40061Z","iopub.status.idle":"2023-11-11T04:03:19.037449Z","shell.execute_reply.started":"2023-11-11T04:02:56.400573Z","shell.execute_reply":"2023-11-11T04:03:19.036646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\nPath('/kaggle/working/Cancer_Data').mkdir(parents=True, exist_ok=True)\nPath('/kaggle/working/Cancer_Data/HGSC').mkdir(parents=True, exist_ok=True)\nPath('/kaggle/working/Cancer_Data/EC').mkdir(parents=True, exist_ok=True)\nPath('/kaggle/working/Cancer_Data/CC').mkdir(parents=True, exist_ok=True)\nPath('/kaggle/working/Cancer_Data/MC').mkdir(parents=True, exist_ok=True)\nPath('/kaggle/working/Cancer_Data/LGSC').mkdir(parents=True, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T04:03:19.038521Z","iopub.execute_input":"2023-11-11T04:03:19.039048Z","iopub.status.idle":"2023-11-11T04:03:19.045857Z","shell.execute_reply.started":"2023-11-11T04:03:19.039022Z","shell.execute_reply":"2023-11-11T04:03:19.04491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport os\n\ndef process_images_with_keyword(keyword):\n    #  read the CSV file into the DataFrame 'df'\n    df = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\n\n    # Compare the text in the column with the provided keyword\n    comparison_result = df['label'].str.contains(keyword, case=False)\n\n    # Use the comparison result\n    filtered_data = df[comparison_result]\n\n    # Get the \"image_id\" column and convert it to strings\n    image_ids = filtered_data['image_id'].astype(str)\n\n    # Define the source and destination directories\n    source_directory = '/kaggle/input/UBC-OCEAN/train_thumbnails/'\n    destination_directory = f'/kaggle/working/Cancer_Data/{keyword}/'\n\n    # Create the destination directory if it doesn't exist\n    os.makedirs(destination_directory, exist_ok=True)\n\n    # Iterate through image_ids, read images, and save resized images\n    for image_id in image_ids:\n        source_path = os.path.join(source_directory, f\"{image_id}_thumbnail.png\")\n        destination_path = os.path.join(destination_directory, f\"{image_id}.png\")\n        \n        # Load the image\n        image = cv2.imread(source_path)\n        \n        if image is not None:\n            # Resize the image\n            resized_image = cv2.resize(image, (224, 224))\n            \n            # Save the resized image\n            cv2.imwrite(destination_path, resized_image)\n        \n        # Print the paths for reference\n      #  print(\"Source Path:\", source_path)\n      #  print(\"Destination Path:\", destination_path)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T04:03:19.048152Z","iopub.execute_input":"2023-11-11T04:03:19.048448Z","iopub.status.idle":"2023-11-11T04:03:19.479897Z","shell.execute_reply.started":"2023-11-11T04:03:19.048424Z","shell.execute_reply":"2023-11-11T04:03:19.479088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process_images_with_keyword(\"CC\")\nprocess_images_with_keyword(\"EC\")\nprocess_images_with_keyword(\"HGSC\")\nprocess_images_with_keyword(\"LGSC\")\nprocess_images_with_keyword(\"MC\")","metadata":{"execution":{"iopub.status.busy":"2023-11-11T04:03:19.480972Z","iopub.execute_input":"2023-11-11T04:03:19.481238Z","iopub.status.idle":"2023-11-11T04:05:03.119989Z","shell.execute_reply.started":"2023-11-11T04:03:19.481213Z","shell.execute_reply":"2023-11-11T04:05:03.119179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_set= '/kaggle/working/Cancer_Data/'\nfor i,d in enumerate([image_set]):\n    filepaths=[]\n    labels=[]\n    classlist=os.listdir(d)\n    for klass in classlist:\n        classpath=os.path.join(d,klass)\n        if os.path.isdir(classpath):\n            flist=os.listdir(classpath)\n            for f in flist:\n                fpath=os.path.join(classpath,f)\n                filepaths.append(fpath)\n                labels.append(klass)\n    Fseries= pd.Series(filepaths, name='filepaths')\n    Lseries=pd.Series(labels, name='labels')\n    ov_df=pd.concat([Fseries, Lseries], axis=1)\ndf=pd.concat([ov_df], axis =0).reset_index(drop=True)# make a combined dataframe\n\nprint(df['labels'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-11-11T04:05:03.12118Z","iopub.execute_input":"2023-11-11T04:05:03.121795Z","iopub.status.idle":"2023-11-11T04:05:03.140926Z","shell.execute_reply.started":"2023-11-11T04:05:03.121754Z","shell.execute_reply":"2023-11-11T04:05:03.139846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_split=.5\ntest_split=.25\ndummy_split=test_split/(1-train_split)\ntrain_df, dummy_df=train_test_split(df, train_size=train_split, shuffle=True, random_state=123)\ntest_df, valid_df=train_test_split(dummy_df, train_size=dummy_split, shuffle=True, random_state=123)\nprint ('train_df length: ', len(train_df), ' _test_df length: ', len(test_df), '  valid_df length: ', len(valid_df))","metadata":{"execution":{"iopub.status.busy":"2023-11-11T04:05:03.144573Z","iopub.execute_input":"2023-11-11T04:05:03.145114Z","iopub.status.idle":"2023-11-11T04:05:03.39691Z","shell.execute_reply.started":"2023-11-11T04:05:03.145088Z","shell.execute_reply":"2023-11-11T04:05:03.395984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\nheight=224\nwidth=224\nchannels=3\nbatch_size=8\nimg_shape=(height, width, channels)\nimg_size=(height, width)\nlength=len(test_df)\ntest_batch_size=sorted([int(length/n) for n in range(1,length+1) if length % n ==0 and length/n<=80],reverse=True)[0]\ntest_steps=int(length/test_batch_size)\nprint ( 'test batch size: ' ,test_batch_size, '  test steps: ', test_steps)\ndef scalar(img):\n    return img/127.5-1  # scale pixel between -1 and +1\ngen=ImageDataGenerator(preprocessing_function=scalar)\ntrain_set=gen.flow_from_dataframe( train_df, x_col='filepaths', y_col='labels', target_size=img_size, class_mode='categorical',\n                                    color_mode='rgb', shuffle=True, batch_size=batch_size)\ntest_set=gen.flow_from_dataframe( test_df, x_col='filepaths', y_col='labels', target_size=img_size, class_mode='categorical',\n                                    color_mode='rgb', shuffle=False, batch_size=test_batch_size)\nvalidate_set=gen.flow_from_dataframe( valid_df, x_col='filepaths', y_col='labels', target_size=img_size, class_mode='categorical',\n                                    color_mode='rgb', shuffle=True, batch_size=batch_size)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T04:05:03.398946Z","iopub.execute_input":"2023-11-11T04:05:03.39923Z","iopub.status.idle":"2023-11-11T04:05:03.429292Z","shell.execute_reply.started":"2023-11-11T04:05:03.399205Z","shell.execute_reply":"2023-11-11T04:05:03.428456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Activation\nfrom keras.optimizers import SGD\nmodel= keras.models.Sequential()\n\nmodel.add(keras.layers.Conv2D(32, 3, activation='relu', input_shape=(224, 224, 3)))\nmodel.add(keras.layers.Dropout(0.1))\nmodel.add(keras.layers.MaxPooling2D())\n\nmodel.add(keras.layers.Conv2D(64, 3, activation='relu'))\nmodel.add(keras.layers.Dropout(0.25))\nmodel.add(keras.layers.MaxPooling2D())\n\nmodel.add(keras.layers.Conv2D(128, 3, activation='relu'))\nmodel.add(keras.layers.Dropout(0.5))\nmodel.add(keras.layers.MaxPooling2D())\n\nmodel.add(keras.layers.Conv2D(256, 3, activation='relu'))\nmodel.add(keras.layers.Dropout(0.25))\nmodel.add(keras.layers.MaxPooling2D())\n\nmodel.add(keras.layers.Flatten())\nmodel.add(keras.layers.Dense(512, activation='relu'))\nmodel.add(keras.layers.Dense(256, activation='relu'))\nmodel.add(keras.layers.Dense(5, activation='softmax'))\n\nmodel.compile(optimizer='adam',loss='categorical_crossentropy', metrics=['accuracy'])\nmodel.summary()\nmodel.compile(loss = 'categorical_crossentropy', optimizer = 'adam', metrics = ['accuracy'])\n#executing the model\nhistory = model.fit(train_set, validation_data = (validate_set), epochs = 50, verbose = 1)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-11T04:05:03.43059Z","iopub.execute_input":"2023-11-11T04:05:03.431188Z","iopub.status.idle":"2023-11-11T04:06:40.188364Z","shell.execute_reply.started":"2023-11-11T04:05:03.431154Z","shell.execute_reply":"2023-11-11T04:06:40.187555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import precision_score, recall_score, f1_score\nY_pred = model.predict(test_set)\ny_pred = np.argmax(Y_pred ,axis =1)\n\npreds = model.predict(test_set,verbose=1)\npredictions = preds.copy()\npredictions[predictions <= 0.5] = 0\npredictions[predictions > 0.5] = 1\n\nprint('Classification Report')\ntarget_names = ['CC','EC','HGSC','LGSC','MC']\nprint(classification_report(test_set.classes, y_pred, target_names=target_names))\n\nfrom sklearn.metrics import classification_report,confusion_matrix\ncm = pd.DataFrame(data=confusion_matrix( y_true= test_set.classes, y_pred= y_pred, labels=[0, 1,2,3,4]), index=['Actual CC','Actual EC','Actual HGSC','Actual LGSC', 'Actual MC'],columns=['Predicted CC','Predicted EC','Predicted HGSC','Predicted LGSC', 'Predicted MC'])\nimport seaborn as sns\nsns.heatmap(cm,annot=True,fmt=\"d\",cmap=\"YlGn\")","metadata":{"execution":{"iopub.status.busy":"2023-11-11T04:06:40.191881Z","iopub.execute_input":"2023-11-11T04:06:40.192255Z","iopub.status.idle":"2023-11-11T04:06:43.051647Z","shell.execute_reply.started":"2023-11-11T04:06:40.192226Z","shell.execute_reply":"2023-11-11T04:06:43.050683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **PREDICTION**","metadata":{}},{"cell_type":"code","source":"#Test thumbnail\nfrom skimage.io import imread\n\nfrom matplotlib.pyplot import imshow\nimg = imread(\"/kaggle/input/UBC-OCEAN/test_thumbnails/41_thumbnail.png\")\nimshow(img)\nprint(f\"Original Dimensions : {img.shape}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-11T04:06:43.052853Z","iopub.execute_input":"2023-11-11T04:06:43.053236Z","iopub.status.idle":"2023-11-11T04:06:44.698268Z","shell.execute_reply.started":"2023-11-11T04:06:43.053199Z","shell.execute_reply":"2023-11-11T04:06:44.697393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a function to import an image and resize it to be able to be used with our model\ndef load_and_prep_image(filename, img_shape=224):\n\n  # Read in target file (an image)\n  img = tf.io.read_file(filename)\n\n  # Decode the read file into a tensor & ensure 3 colour channels\n  # (our model is trained on images with 3 colour channels and sometimes images have 4 colour channels)\n  img = tf.image.decode_image(img, channels=3)\n\n  # Resize the image (to the same size our model was trained on)\n  img = tf.image.resize(img, size = [img_shape, img_shape])\n\n  # Rescale the image (get all values between 0 and 1)\n  img = img/255.\n  return img\n\ndef pred_and_plot(model, filename, target_names):\n  \"\"\"\n  Imports an image located at filename, makes a prediction on it with\n  a trained model and plots the image with the predicted class as the title.\n  \"\"\"\n  # Import the target image and preprocess it\n  img = load_and_prep_image(filename)\n\n  # Make a prediction\n  pred = model.predict(tf.expand_dims(img, axis=0))\n\n  # Get the predicted class\n  pred_class = target_names[int(tf.round(pred)[0][0])]\n\n  # Plot the image and predicted class\n  plt.imshow(img)\n  plt.title(f\"Prediction: {pred_class}\")\n  plt.axis(False)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-11T04:06:44.699371Z","iopub.execute_input":"2023-11-11T04:06:44.699636Z","iopub.status.idle":"2023-11-11T04:06:44.707534Z","shell.execute_reply.started":"2023-11-11T04:06:44.69961Z","shell.execute_reply":"2023-11-11T04:06:44.706552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test our model on a test image\npred_and_plot(model, \"/kaggle/input/UBC-OCEAN/test_thumbnails/41_thumbnail.png\", target_names)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T04:06:44.708513Z","iopub.execute_input":"2023-11-11T04:06:44.708756Z","iopub.status.idle":"2023-11-11T04:06:45.31686Z","shell.execute_reply.started":"2023-11-11T04:06:44.708733Z","shell.execute_reply":"2023-11-11T04:06:45.315718Z"},"trusted":true},"execution_count":null,"outputs":[]}]}