{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30627,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/UBC-OCEAN/train_thumbnails'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-15T14:44:30.721895Z","iopub.execute_input":"2023-12-15T14:44:30.722608Z","iopub.status.idle":"2023-12-15T14:44:30.736159Z","shell.execute_reply.started":"2023-12-15T14:44:30.722576Z","shell.execute_reply":"2023-12-15T14:44:30.735135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\nimport os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nfrom sklearn.metrics import classification_report , confusion_matrix , accuracy_score , auc\nfrom sklearn.model_selection import train_test_split\n\nimport cv2\n#from google.colab.patches import cv2_imshow\nfrom PIL import Image \nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras import Sequential\nfrom keras.layers import Input, Dense,Conv2D , MaxPooling2D, Flatten,BatchNormalization,Dropout\nfrom tensorflow.keras.preprocessing import image_dataset_from_directory\nimport tensorflow_hub as hub ","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:44:36.87683Z","iopub.execute_input":"2023-12-15T14:44:36.877672Z","iopub.status.idle":"2023-12-15T14:44:36.886776Z","shell.execute_reply.started":"2023-12-15T14:44:36.877639Z","shell.execute_reply":"2023-12-15T14:44:36.885789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:44:37.215097Z","iopub.execute_input":"2023-12-15T14:44:37.215382Z","iopub.status.idle":"2023-12-15T14:44:37.23255Z","shell.execute_reply.started":"2023-12-15T14:44:37.215356Z","shell.execute_reply":"2023-12-15T14:44:37.231666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:44:37.567852Z","iopub.execute_input":"2023-12-15T14:44:37.568142Z","iopub.status.idle":"2023-12-15T14:44:37.576774Z","shell.execute_reply.started":"2023-12-15T14:44:37.568119Z","shell.execute_reply":"2023-12-15T14:44:37.575704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:44:37.879671Z","iopub.execute_input":"2023-12-15T14:44:37.879983Z","iopub.status.idle":"2023-12-15T14:44:37.887215Z","shell.execute_reply.started":"2023-12-15T14:44:37.879957Z","shell.execute_reply":"2023-12-15T14:44:37.88628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x=\"label\")\nplt.title(\"Ovarian Cancer Types Distributions\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:44:38.124902Z","iopub.execute_input":"2023-12-15T14:44:38.125165Z","iopub.status.idle":"2023-12-15T14:44:38.372654Z","shell.execute_reply.started":"2023-12-15T14:44:38.125142Z","shell.execute_reply":"2023-12-15T14:44:38.371761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_train = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\npath_test = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"\ntrain_folder = os.listdir(path_train)\ntest_folder = os.listdir(path_test)\n\nprint(len(train_folder))\nprint(len(test_folder))","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:44:38.382919Z","iopub.execute_input":"2023-12-15T14:44:38.383219Z","iopub.status.idle":"2023-12-15T14:44:38.390071Z","shell.execute_reply.started":"2023-12-15T14:44:38.383193Z","shell.execute_reply":"2023-12-15T14:44:38.389138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_folder[:5]","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:44:38.61848Z","iopub.execute_input":"2023-12-15T14:44:38.618791Z","iopub.status.idle":"2023-12-15T14:44:38.624858Z","shell.execute_reply.started":"2023-12-15T14:44:38.618765Z","shell.execute_reply":"2023-12-15T14:44:38.623913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_folder","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:44:38.867134Z","iopub.execute_input":"2023-12-15T14:44:38.867425Z","iopub.status.idle":"2023-12-15T14:44:38.874181Z","shell.execute_reply.started":"2023-12-15T14:44:38.8674Z","shell.execute_reply":"2023-12-15T14:44:38.87324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_tma = train_df[train_df['is_tma']==True]\ntrain_df_tma","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:44:39.116204Z","iopub.execute_input":"2023-12-15T14:44:39.116468Z","iopub.status.idle":"2023-12-15T14:44:39.132598Z","shell.execute_reply.started":"2023-12-15T14:44:39.116445Z","shell.execute_reply":"2023-12-15T14:44:39.131472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_no_tma = train_df[train_df['is_tma']==False]\ntrain_df_no_tma","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:44:39.180099Z","iopub.execute_input":"2023-12-15T14:44:39.180419Z","iopub.status.idle":"2023-12-15T14:44:39.195039Z","shell.execute_reply.started":"2023-12-15T14:44:39.180384Z","shell.execute_reply":"2023-12-15T14:44:39.194162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_tma['image_id_path'] = [f\"{i}.png\" for i in train_df_tma['image_id']]","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:44:39.226899Z","iopub.execute_input":"2023-12-15T14:44:39.227162Z","iopub.status.idle":"2023-12-15T14:44:39.232213Z","shell.execute_reply.started":"2023-12-15T14:44:39.227139Z","shell.execute_reply":"2023-12-15T14:44:39.231198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_tma","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:44:39.276166Z","iopub.execute_input":"2023-12-15T14:44:39.276411Z","iopub.status.idle":"2023-12-15T14:44:39.289644Z","shell.execute_reply.started":"2023-12-15T14:44:39.276389Z","shell.execute_reply":"2023-12-15T14:44:39.288746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_no_tma['image_id_path'] = [f\"{i}_thumbnail.png\" for i in train_df_no_tma['image_id']]","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:44:39.416325Z","iopub.execute_input":"2023-12-15T14:44:39.416568Z","iopub.status.idle":"2023-12-15T14:44:39.421445Z","shell.execute_reply.started":"2023-12-15T14:44:39.416547Z","shell.execute_reply":"2023-12-15T14:44:39.420488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_no_tma","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:44:39.50965Z","iopub.execute_input":"2023-12-15T14:44:39.509931Z","iopub.status.idle":"2023-12-15T14:44:39.523473Z","shell.execute_reply.started":"2023-12-15T14:44:39.509907Z","shell.execute_reply":"2023-12-15T14:44:39.522635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16,24))\npath = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\nj=1\nfor img, lb in zip(train_df_no_tma['image_id_path'][:24],train_df_no_tma['label'][:24]):\n    plt.subplot(6,4,j)\n    path = os.path.join(\"/kaggle/input/UBC-OCEAN/train_thumbnails/\",img)\n    image = plt.imread(path)\n    image = plt.imshow(image)\n    plt.title(f\"Label:{lb}\")\n    j+=1","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:44:39.82191Z","iopub.execute_input":"2023-12-15T14:44:39.822257Z","iopub.status.idle":"2023-12-15T14:45:21.880276Z","shell.execute_reply.started":"2023-12-15T14:44:39.82222Z","shell.execute_reply":"2023-12-15T14:45:21.879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Start try Extract Patches for one image","metadata":{}},{"cell_type":"code","source":"train_folder[0]","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:45:21.882006Z","iopub.execute_input":"2023-12-15T14:45:21.882316Z","iopub.status.idle":"2023-12-15T14:45:21.888062Z","shell.execute_reply.started":"2023-12-15T14:45:21.882289Z","shell.execute_reply":"2023-12-15T14:45:21.887234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"first_image_path = \"/kaggle/input/UBC-OCEAN/train_thumbnails/\" + train_folder[0]\nfirst_image = plt.imread(first_image_path)\nplt.imshow(first_image)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:45:21.889296Z","iopub.execute_input":"2023-12-15T14:45:21.88957Z","iopub.status.idle":"2023-12-15T14:45:23.943551Z","shell.execute_reply.started":"2023-12-15T14:45:21.889546Z","shell.execute_reply":"2023-12-15T14:45:23.942652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"first_image = tf.io.read_file(first_image_path)\nfirst_image = tf.image.decode_jpeg(first_image, channels=3)\n\nfirst_image_tensor = tf.convert_to_tensor(first_image)\n\n\npatch_height = 100\npatch_width = 100\n\nfirst_image_patch = tf.image.extract_patches([first_image_tensor], \n                                             sizes = [1, patch_height, patch_width, 1], \n                                             strides=[1, patch_height, patch_width, 1], \n                                             rates=[1, 1, 1, 1], \n                                             padding='VALID')\nfirst_image_patch","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:45:23.94576Z","iopub.execute_input":"2023-12-15T14:45:23.946059Z","iopub.status.idle":"2023-12-15T14:45:24.247713Z","shell.execute_reply.started":"2023-12-15T14:45:23.946028Z","shell.execute_reply":"2023-12-15T14:45:24.246754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(first_image.shape)\nprint(first_image_patch.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:45:24.248859Z","iopub.execute_input":"2023-12-15T14:45:24.249162Z","iopub.status.idle":"2023-12-15T14:45:24.254261Z","shell.execute_reply.started":"2023-12-15T14:45:24.249135Z","shell.execute_reply":"2023-12-15T14:45:24.253278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"first_image_patch[0][0].shape","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:45:24.255468Z","iopub.execute_input":"2023-12-15T14:45:24.256119Z","iopub.status.idle":"2023-12-15T14:45:26.834937Z","shell.execute_reply.started":"2023-12-15T14:45:24.256073Z","shell.execute_reply":"2023-12-15T14:45:26.833855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Try With Patches","metadata":{}},{"cell_type":"code","source":"!pip install patchify","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:45:26.836071Z","iopub.execute_input":"2023-12-15T14:45:26.836337Z","iopub.status.idle":"2023-12-15T14:45:38.692942Z","shell.execute_reply.started":"2023-12-15T14:45:26.836314Z","shell.execute_reply":"2023-12-15T14:45:38.691699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import patchify\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:45:38.694576Z","iopub.execute_input":"2023-12-15T14:45:38.694956Z","iopub.status.idle":"2023-12-15T14:45:38.700305Z","shell.execute_reply.started":"2023-12-15T14:45:38.694923Z","shell.execute_reply":"2023-12-15T14:45:38.699261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# IMG = Image.open(\"/kaggle/input/UBC-OCEAN/train_thumbnails/\" + train_folder[0])\n# IMG","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:45:38.701524Z","iopub.execute_input":"2023-12-15T14:45:38.701867Z","iopub.status.idle":"2023-12-15T14:45:40.120354Z","shell.execute_reply.started":"2023-12-15T14:45:38.701843Z","shell.execute_reply":"2023-12-15T14:45:40.118634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG = cv2.imread(\"/kaggle/input/UBC-OCEAN/train_thumbnails/\" + train_folder[0])\nIMG = cv2.cvtColor(IMG, cv2.COLOR_BGR2RGB)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:08:34.455163Z","iopub.execute_input":"2023-12-15T17:08:34.455505Z","iopub.status.idle":"2023-12-15T17:08:34.647255Z","shell.execute_reply.started":"2023-12-15T17:08:34.455479Z","shell.execute_reply":"2023-12-15T17:08:34.646225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG.size # (width, height)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:03:39.488049Z","iopub.execute_input":"2023-12-15T17:03:39.488422Z","iopub.status.idle":"2023-12-15T17:03:39.494627Z","shell.execute_reply.started":"2023-12-15T17:03:39.48839Z","shell.execute_reply":"2023-12-15T17:03:39.493688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG.width","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:03:40.066127Z","iopub.execute_input":"2023-12-15T17:03:40.066875Z","iopub.status.idle":"2023-12-15T17:03:40.108485Z","shell.execute_reply.started":"2023-12-15T17:03:40.066846Z","shell.execute_reply":"2023-12-15T17:03:40.107349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG.height","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:45:40.14438Z","iopub.execute_input":"2023-12-15T14:45:40.144951Z","iopub.status.idle":"2023-12-15T14:45:40.152574Z","shell.execute_reply.started":"2023-12-15T14:45:40.144918Z","shell.execute_reply":"2023-12-15T14:45:40.151761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG.mode","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:45:40.153696Z","iopub.execute_input":"2023-12-15T14:45:40.154012Z","iopub.status.idle":"2023-12-15T14:45:40.161007Z","shell.execute_reply.started":"2023-12-15T14:45:40.153983Z","shell.execute_reply":"2023-12-15T14:45:40.160079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from albumentations.pytorch import ToTensorV2\nimport albumentations as A\n\ntransform = A.Compose(\n            [\n                A.Resize(512, 512),\n#                 A.ShiftScaleRotate(\n#                     shift_limit=0.1, scale_limit=0.15, rotate_limit=60, p=0.5\n#                 ),\n#                 A.HueSaturationValue(\n#                     hue_shift_limit=0.2, sat_shift_limit=0.2, val_shift_limit=0.2, p=0.5\n#                 ),\n#                 A.RandomBrightnessContrast(\n#                     brightness_limit=(-0.1, 0.1), contrast_limit=(-0.1, 0.1), p=0.5\n#                 ),\n#                 A.Normalize(\n#                     mean=[0.485, 0.456, 0.406],\n#                     std=[0.229, 0.224, 0.225],\n#                     max_pixel_value=255.0,\n#                     p=1.0,\n#                 ),\n#                 ToTensorV2(),\n            ],\n            p=1,\n        )","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:09:27.323361Z","iopub.execute_input":"2023-12-15T17:09:27.324205Z","iopub.status.idle":"2023-12-15T17:09:27.329586Z","shell.execute_reply.started":"2023-12-15T17:09:27.324169Z","shell.execute_reply":"2023-12-15T17:09:27.328637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# numpy_array = np.array(IMG)\nimage = transform(image=IMG)[\"image\"]","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:09:32.310792Z","iopub.execute_input":"2023-12-15T17:09:32.311417Z","iopub.status.idle":"2023-12-15T17:09:32.317932Z","shell.execute_reply.started":"2023-12-15T17:09:32.311387Z","shell.execute_reply":"2023-12-15T17:09:32.31701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image.shape","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:09:33.969764Z","iopub.execute_input":"2023-12-15T17:09:33.970339Z","iopub.status.idle":"2023-12-15T17:09:33.976562Z","shell.execute_reply.started":"2023-12-15T17:09:33.970303Z","shell.execute_reply":"2023-12-15T17:09:33.975503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:09:34.831242Z","iopub.execute_input":"2023-12-15T17:09:34.8321Z","iopub.status.idle":"2023-12-15T17:09:35.122026Z","shell.execute_reply.started":"2023-12-15T17:09:34.832064Z","shell.execute_reply":"2023-12-15T17:09:35.120983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gather = []\n\npatches = patchify.patchify(np.asarray(image), patch_size=(32, 32, 3), step=32)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:09:36.005662Z","iopub.execute_input":"2023-12-15T17:09:36.006053Z","iopub.status.idle":"2023-12-15T17:09:36.011358Z","shell.execute_reply.started":"2023-12-15T17:09:36.006023Z","shell.execute_reply":"2023-12-15T17:09:36.010272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patches.shape","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:09:42.355785Z","iopub.execute_input":"2023-12-15T17:09:42.356579Z","iopub.status.idle":"2023-12-15T17:09:42.362491Z","shell.execute_reply.started":"2023-12-15T17:09:42.356546Z","shell.execute_reply":"2023-12-15T17:09:42.361525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(patches.shape[0]):\n    for j in range(patches.shape[1]):\n        get_patches = patches[i, j, 0, :, :, :]\n        gather.append(get_patches)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:09:44.187909Z","iopub.execute_input":"2023-12-15T17:09:44.188631Z","iopub.status.idle":"2023-12-15T17:09:44.19379Z","shell.execute_reply.started":"2023-12-15T17:09:44.188599Z","shell.execute_reply":"2023-12-15T17:09:44.192795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(gather)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:09:45.907763Z","iopub.execute_input":"2023-12-15T17:09:45.908326Z","iopub.status.idle":"2023-12-15T17:09:45.913908Z","shell.execute_reply.started":"2023-12-15T17:09:45.908293Z","shell.execute_reply":"2023-12-15T17:09:45.912883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(6, 4))\nplt.imshow(IMG)\nplt.axis('off')\n\nfig1, ax1 = plt.subplots(nrows=16, ncols=16, figsize=(6, 3))\n\na = 0\nfor R1 in range(16):\n    for C1 in range(16):\n        ax1[R1, C1].imshow(gather[a])\n        ax1[R1, C1].axis('off')\n        a += 1","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:09:47.778821Z","iopub.execute_input":"2023-12-15T17:09:47.779556Z","iopub.status.idle":"2023-12-15T17:09:55.53024Z","shell.execute_reply.started":"2023-12-15T17:09:47.779525Z","shell.execute_reply":"2023-12-15T17:09:55.529144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(gather)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:09:55.532134Z","iopub.execute_input":"2023-12-15T17:09:55.532498Z","iopub.status.idle":"2023-12-15T17:09:55.53975Z","shell.execute_reply.started":"2023-12-15T17:09:55.532464Z","shell.execute_reply":"2023-12-15T17:09:55.53897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(gather[0])","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:10:01.100098Z","iopub.execute_input":"2023-12-15T17:10:01.10047Z","iopub.status.idle":"2023-12-15T17:10:01.104511Z","shell.execute_reply.started":"2023-12-15T17:10:01.10044Z","shell.execute_reply":"2023-12-15T17:10:01.103582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(gather[0].shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:10:01.788025Z","iopub.execute_input":"2023-12-15T17:10:01.788945Z","iopub.status.idle":"2023-12-15T17:10:01.793524Z","shell.execute_reply.started":"2023-12-15T17:10:01.788912Z","shell.execute_reply":"2023-12-15T17:10:01.792594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum(sum(sum(gather[0])))","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:10:03.168852Z","iopub.execute_input":"2023-12-15T17:10:03.169211Z","iopub.status.idle":"2023-12-15T17:10:03.175633Z","shell.execute_reply.started":"2023-12-15T17:10:03.169181Z","shell.execute_reply":"2023-12-15T17:10:03.174655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum(sum(sum(gather[200])))","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:43:27.354437Z","iopub.status.idle":"2023-12-15T14:43:27.354851Z","shell.execute_reply.started":"2023-12-15T14:43:27.354612Z","shell.execute_reply":"2023-12-15T14:43:27.354633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patches_not_black = []\n\nfor i in gather:\n    if sum(sum(sum(i))) != 0:\n        patches_not_black.append(i)\n\nprint(len(patches_not_black))","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:05:57.198297Z","iopub.execute_input":"2023-12-15T17:05:57.199219Z","iopub.status.idle":"2023-12-15T17:05:57.227698Z","shell.execute_reply.started":"2023-12-15T17:05:57.199186Z","shell.execute_reply":"2023-12-15T17:05:57.226672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum(sum(sum(patches_not_black[10])))","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:43:27.358077Z","iopub.status.idle":"2023-12-15T14:43:27.358509Z","shell.execute_reply.started":"2023-12-15T14:43:27.358284Z","shell.execute_reply":"2023-12-15T14:43:27.358305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Start try Extract Patches For All Training Data","metadata":{}},{"cell_type":"code","source":"train_df['image_id_path'] = [f\"{i}_thumbnail.png\" for i in train_df['image_id']]\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-12-15T15:06:44.334654Z","iopub.execute_input":"2023-12-15T15:06:44.335494Z","iopub.status.idle":"2023-12-15T15:06:44.350998Z","shell.execute_reply.started":"2023-12-15T15:06:44.335464Z","shell.execute_reply":"2023-12-15T15:06:44.350111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x=\"label\")\nplt.title(\"Ovarian Cancer Types Distributions\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-15T15:06:46.066105Z","iopub.execute_input":"2023-12-15T15:06:46.066439Z","iopub.status.idle":"2023-12-15T15:06:46.305364Z","shell.execute_reply.started":"2023-12-15T15:06:46.066414Z","shell.execute_reply":"2023-12-15T15:06:46.304392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for i in train_df[\"image_id_path\"]:\n#     print(i)\n    \n# print(len(train_df[\"image_id_path\"]))","metadata":{"execution":{"iopub.status.busy":"2023-12-15T15:06:49.18233Z","iopub.execute_input":"2023-12-15T15:06:49.182687Z","iopub.status.idle":"2023-12-15T15:06:49.187031Z","shell.execute_reply.started":"2023-12-15T15:06:49.182657Z","shell.execute_reply":"2023-12-15T15:06:49.186075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# result = train_df.loc[train_df['image_id_path'] == \"91_thumbnail.png\", 'label'][train_df.loc[train_df['image_id_path'] == \"91_thumbnail.png\"]]\n# print(result)\n\n# result2 = train_df.loc[train_df['image_id_path'] == \"91_thumbnail.png\", 'label'][0]\n# print(result2)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-15T15:06:49.189142Z","iopub.execute_input":"2023-12-15T15:06:49.189747Z","iopub.status.idle":"2023-12-15T15:06:49.195484Z","shell.execute_reply.started":"2023-12-15T15:06:49.189696Z","shell.execute_reply":"2023-12-15T15:06:49.19452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from albumentations.pytorch import ToTensorV2\nimport albumentations as A\n\n\ntraining_patches_hgsc = []\ntraining_patches_lgsc = []\ntraining_patches_ec = []\ntraining_patches_cc = []\ntraining_patches_mc = []\nother = []\ncount = 0\n\n\ntransform = A.Compose(\n            [\n                A.Resize(512, 512),\n                A.ShiftScaleRotate(\n                    shift_limit=0.1, scale_limit=0.15, rotate_limit=60, p=0.5\n                ),\n                A.HueSaturationValue(\n                    hue_shift_limit=0.2, sat_shift_limit=0.2, val_shift_limit=0.2, p=0.5\n                ),\n                A.RandomBrightnessContrast(\n                    brightness_limit=(-0.1, 0.1), contrast_limit=(-0.1, 0.1), p=0.5\n                ),\n                A.Normalize(\n                    mean=[0.485, 0.456, 0.406],\n                    std=[0.229, 0.224, 0.225],\n                    max_pixel_value=255.0,\n                    p=1.0,\n                ),\n                ToTensorV2(),\n            ],\n            p=1,\n        )\n\nfor i in train_df[\"image_id_path\"]:   \n    try:\n        IMG = cv2.imread(\"/kaggle/input/UBC-OCEAN/train_thumbnails/\" + i)\n        IMG = cv2.cvtColor(IMG, cv2.COLOR_BGR2RGB)\n#         IMG = Image.open(\"/kaggle/input/UBC-OCEAN/train_thumbnails/\" + i)\n# FileNotFoundError\n    except:\n        print(f\"File not found: {i}\")\n        count += 1\n        continue\n    ###########################\n    gather = []\n\n    patches = patchify.patchify(np.asarray(IMG), patch_size=(32, 32, 3), step=32)\n    ###########################\n    for j in range(patches.shape[0]):\n        for k in range(patches.shape[1]):\n            get_patches = patches[j, k, 0, :, :, :]\n            gather.append(get_patches)\n    ###########################\n        \n    if (train_df.loc[train_df['image_id_path'] == f\"{i}\", 'label'][count]) == \"HGSC\":\n        for j in gather:\n            if sum(sum(sum(j))) != 0:\n                training_patches_hgsc.append(j)\n        gather[:] = []\n        \n    elif (train_df.loc[train_df['image_id_path'] == f\"{i}\", 'label'][count]) == \"LGSC\":\n        for j in gather:\n            if sum(sum(sum(j))) != 0:\n                training_patches_lgsc.append(j)   \n        gather[:] = []\n                \n    elif (train_df.loc[train_df['image_id_path'] == f\"{i}\", 'label'][count]) == \"EC\":\n        for j in gather:\n            if sum(sum(sum(j))) != 0:\n                training_patches_ec.append(j)   \n        gather[:] = []\n                \n    elif (train_df.loc[train_df['image_id_path'] == f\"{i}\", 'label'][count]) == \"CC\":\n        for j in gather:\n            if sum(sum(sum(j))) != 0:\n                training_patches_cc.append(j)\n        gather[:] = []\n                \n    elif (train_df.loc[train_df['image_id_path'] == f\"{i}\", 'label'][count]) == \"MC\":\n        for j in gather:\n            if sum(sum(sum(j))) != 0:\n                training_patches_mc.append(j)\n        gather[:] = []\n        \n    else: \n        print(\"hoba\", i)\n        print(\"ho\", gather)\n        for j in gather:\n            if sum(sum(sum(j))) != 0:\n                other.append(j)\n        gather[:] = []\n    \n    count += 1\n    \n\n# image = transform(image=IMG)[\"image\"]\nprint(\"hgsc\", len(training_patches_hgsc))\nprint(\"lgsc\", len(training_patches_lgsc))\nprint(\"ec\", len(training_patches_ec))\nprint(\"cc\", len(training_patches_cc))\nprint(\"mc\", len(training_patches_mc))\nprint(\"other\", len(other))","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:15:44.544868Z","iopub.execute_input":"2023-12-15T17:15:44.545624Z","iopub.status.idle":"2023-12-15T17:20:59.280669Z","shell.execute_reply.started":"2023-12-15T17:15:44.54559Z","shell.execute_reply":"2023-12-15T17:20:59.279462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(\"hgsc\", len(training_patches_hgsc))\n# print(\"lgsc\", len(training_patches_lgsc))\n# print(\"ec\", len(training_patches_ec))\n# print(\"cc\", len(training_patches_cc))\n# print(\"mc\", len(training_patches_mc))\n# print(\"other\", len(other))\ntype(training_patches_hgsc[0].shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T15:15:16.35303Z","iopub.execute_input":"2023-12-15T15:15:16.353806Z","iopub.status.idle":"2023-12-15T15:15:16.359887Z","shell.execute_reply.started":"2023-12-15T15:15:16.353776Z","shell.execute_reply":"2023-12-15T15:15:16.358871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# hgsc 87672\n# lgsc 16321\n# ec 50889\n# cc 37171\n# mc 19196\n# other 0","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:43:27.368979Z","iopub.status.idle":"2023-12-15T14:43:27.369314Z","shell.execute_reply.started":"2023-12-15T14:43:27.369142Z","shell.execute_reply":"2023-12-15T14:43:27.369158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Try With Transformer","metadata":{}},{"cell_type":"code","source":"columns = ['original_image_path', 'order', 'label']\ndf = pd.DataFrame(columns=columns)\ndf","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:45:08.563121Z","iopub.execute_input":"2023-12-15T17:45:08.563529Z","iopub.status.idle":"2023-12-15T17:45:08.573564Z","shell.execute_reply.started":"2023-12-15T17:45:08.563497Z","shell.execute_reply":"2023-12-15T17:45:08.572602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.loc[len(df)] = [1, 2, 5]\ndf","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:43:09.836263Z","iopub.execute_input":"2023-12-15T17:43:09.836672Z","iopub.status.idle":"2023-12-15T17:43:09.860465Z","shell.execute_reply.started":"2023-12-15T17:43:09.83664Z","shell.execute_reply":"2023-12-15T17:43:09.859319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.drop(df.index)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:44:51.054785Z","iopub.execute_input":"2023-12-15T17:44:51.055742Z","iopub.status.idle":"2023-12-15T17:44:51.06948Z","shell.execute_reply.started":"2023-12-15T17:44:51.05569Z","shell.execute_reply":"2023-12-15T17:44:51.068688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:44:54.111437Z","iopub.execute_input":"2023-12-15T17:44:54.111808Z","iopub.status.idle":"2023-12-15T17:44:54.119973Z","shell.execute_reply.started":"2023-12-15T17:44:54.111777Z","shell.execute_reply":"2023-12-15T17:44:54.119108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import albumentations as A\ntransform = A.Compose(\n            [\n                A.Resize(512, 512),\n                A.ShiftScaleRotate(\n                    shift_limit=0.1, scale_limit=0.15, rotate_limit=60, p=0.5\n                ),\n                A.HueSaturationValue(\n                    hue_shift_limit=0.2, sat_shift_limit=0.2, val_shift_limit=0.2, p=0.5\n                ),\n                A.RandomBrightnessContrast(\n                    brightness_limit=(-0.1, 0.1), contrast_limit=(-0.1, 0.1), p=0.5\n                ),","metadata":{"execution":{"iopub.status.busy":"2023-12-15T14:43:27.370463Z","iopub.status.idle":"2023-12-15T14:43:27.370807Z","shell.execute_reply.started":"2023-12-15T14:43:27.370616Z","shell.execute_reply":"2023-12-15T14:43:27.370631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_patches_hgsc = []\ntraining_patches_lgsc = []\ntraining_patches_ec = []\ntraining_patches_cc = []\ntraining_patches_mc = []\nother = []\n\ntraining_patches_hgsc_temp = []\ntraining_patches_lgsc_temp = []\ntraining_patches_ec_temp = []\ntraining_patches_cc_temp = []\ntraining_patches_mc_temp = []\n\ncount = 0\n\n\nfor i in train_df[\"image_id_path\"]:\n    try:\n        IMG = Image.open(\"/kaggle/input/UBC-OCEAN/train_thumbnails/\" + i)\n        \n    except FileNotFoundError:\n        print(f\"File not found: {i}\")\n        count += 1\n        continue\n    ###########################\n    gather = []\n\n    patches = patchify.patchify(np.asarray(IMG), patch_size=(32, 32, 3), step=32)\n    ###########################\n    for j in range(patches.shape[0]):\n        for k in range(patches.shape[1]):\n            get_patches = patches[j, k, 0, :, :, :]\n            gather.append(get_patches)\n    ###########################\n    if (train_df.loc[train_df['image_id_path'] == f\"{i}\", 'label'][count]) == \"HGSC\":\n        for j in gather:\n            if sum(sum(sum(j))) != 0:\n                training_patches_hgsc.append(j)\n                training_patches_hgsc_temp.append(j)\n        for indexo, patch in enumerate(training_patches_hgsc_temp):\n            df.loc[len(df)] = [i, indexo, (train_df.loc[train_df['image_id_path'] == f\"{i}\", 'label'][count])]\n            # Creating an image from the NumPy array\n            img = Image.fromarray(patch)\n\n            # Specify the path where you want to save the image in your working folder\n            patch_path = f\"{indexo}\" + \"_\" + i\n            img_path = os.path.join(folder_path, patch_path)\n\n            # Save the image to the working folder\n            img.save(img_path)\n            \n        gather[:] = []\n        training_patches_hgsc_temp[:] = []\n        \n    elif (train_df.loc[train_df['image_id_path'] == f\"{i}\", 'label'][count]) == \"LGSC\":\n        for j in gather:\n            if sum(sum(sum(j))) != 0:\n                training_patches_lgsc.append(j)\n                training_patches_lgsc_temp.append(j)\n        for indexo, patch in enumerate(training_patches_lgsc_temp):\n            df.loc[len(df)] = [i, indexo, (train_df.loc[train_df['image_id_path'] == f\"{i}\", 'label'][count])]\n            # Creating an image from the NumPy array\n            img = Image.fromarray(patch)\n\n            # Specify the path where you want to save the image in your working folder\n            patch_path = f\"{indexo}\" + \"_\" + i\n            img_path = os.path.join(folder_path, patch_path)\n\n            # Save the image to the working folder\n            img.save(img_path)\n            \n        gather[:] = []\n        training_patches_lgsc_temp[:] = []\n                \n    elif (train_df.loc[train_df['image_id_path'] == f\"{i}\", 'label'][count]) == \"EC\":\n        for j in gather:\n            if sum(sum(sum(j))) != 0:\n                training_patches_ec.append(j) \n                training_patches_ec_temp.append(j)\n        for indexo, patch in enumerate(training_patches_ec_temp):\n            df.loc[len(df)] = [i, indexo, (train_df.loc[train_df['image_id_path'] == f\"{i}\", 'label'][count])]\n            # Creating an image from the NumPy array\n            img = Image.fromarray(patch)\n\n            # Specify the path where you want to save the image in your working folder\n            patch_path = f\"{indexo}\" + \"_\" + i\n            img_path = os.path.join(folder_path, patch_path)\n\n            # Save the image to the working folder\n            img.save(img_path)\n            \n        gather[:] = []\n        training_patches_ec_temp[:] = []\n                \n    elif (train_df.loc[train_df['image_id_path'] == f\"{i}\", 'label'][count]) == \"CC\":\n        for j in gather:\n            if sum(sum(sum(j))) != 0:\n                training_patches_cc.append(j)\n                training_patches_cc_temp.append(j)\n        for indexo, patch in enumerate(training_patches_cc_temp):\n            df.loc[len(df)] = [i, indexo, (train_df.loc[train_df['image_id_path'] == f\"{i}\", 'label'][count])]\n            # Creating an image from the NumPy array\n            img = Image.fromarray(patch)\n\n            # Specify the path where you want to save the image in your working folder\n            patch_path = f\"{indexo}\" + \"_\" + i\n            img_path = os.path.join(folder_path, patch_path)\n\n            # Save the image to the working folder\n            img.save(img_path)\n            \n        gather[:] = []\n        training_patches_cc_temp[:] = []\n                \n    elif (train_df.loc[train_df['image_id_path'] == f\"{i}\", 'label'][count]) == \"MC\":\n        for j in gather:\n            if sum(sum(sum(j))) != 0:\n                training_patches_mc.append(j)\n                training_patches_mc_temp.append(j)\n        for indexo, patch in enumerate(training_patches_mc_temp):\n            df.loc[len(df)] = [i, indexo, (train_df.loc[train_df['image_id_path'] == f\"{i}\", 'label'][count])]\n            # Creating an image from the NumPy array\n            img = Image.fromarray(patch)\n\n            # Specify the path where you want to save the image in your working folder\n            patch_path = f\"{indexo}\" + \"_\" + i\n            img_path = os.path.join(folder_path, patch_path)\n\n            # Save the image to the working folder\n            img.save(img_path)\n            \n        gather[:] = []\n        training_patches_mc_temp[:] = []\n        \n    else: \n        print(\"hoba\", i)\n        print(\"ho\", gather)\n        for j in gather:\n            if sum(sum(sum(j))) != 0:\n                other.append(j)\n        gather[:] = []\n    \n    count += 1\n    print(df)\n    \nprint(\"hgsc\", len(training_patches_hgsc))\nprint(\"lgsc\", len(training_patches_lgsc))\nprint(\"ec\", len(training_patches_ec))\nprint(\"cc\", len(training_patches_cc))\nprint(\"mc\", len(training_patches_mc))\nprint(\"other\", len(other))","metadata":{"execution":{"iopub.status.busy":"2023-12-15T17:39:36.740889Z","iopub.execute_input":"2023-12-15T17:39:36.741657Z","iopub.status.idle":"2023-12-15T17:42:17.627838Z","shell.execute_reply.started":"2023-12-15T17:39:36.741626Z","shell.execute_reply":"2023-12-15T17:42:17.626807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n#create DataFrame\ndf = pd.DataFrame({'points': [10, 12, 12, 14, 13, 18],\n                   'rebounds': [7, 7, 8, 13, 7, 4],\n                   'assists': [11, 8, 10, 6, 6, 5]})\n\n\n#add new row to end of DataFrame\ndf.loc[len(df.index)] = [20, 7, 5]\n\n#view updated DataFrame\ndf","metadata":{"execution":{"iopub.status.busy":"2023-12-15T16:22:48.711447Z","iopub.execute_input":"2023-12-15T16:22:48.711853Z","iopub.status.idle":"2023-12-15T16:22:48.725655Z","shell.execute_reply.started":"2023-12-15T16:22:48.711822Z","shell.execute_reply":"2023-12-15T16:22:48.724658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n\n# Specify the folder name you want to zip\nfolder_name = 'patches'\n\n# Specify the path to the folder\nfolder_path = '/kaggle/working/' + folder_name\n\n# Specify the path for the zip file\nzip_file_path = '/kaggle/working/' + folder_name + \"ho\" + '.zip'\n\n# Create a zip file of the folder\nshutil.make_archive(zip_file_path, 'zip', folder_path)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-15T22:35:47.428927Z","iopub.execute_input":"2023-12-15T22:35:47.429358Z","iopub.status.idle":"2023-12-15T22:36:05.679322Z","shell.execute_reply.started":"2023-12-15T22:35:47.429305Z","shell.execute_reply":"2023-12-15T22:36:05.678232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import FileLink\n\n# Provide a download link for the zip file\nFileLink(rf'{zip_file_path}.zip')\n","metadata":{"execution":{"iopub.status.busy":"2023-12-15T22:36:15.133012Z","iopub.execute_input":"2023-12-15T22:36:15.13339Z","iopub.status.idle":"2023-12-15T22:36:15.140071Z","shell.execute_reply.started":"2023-12-15T22:36:15.133359Z","shell.execute_reply":"2023-12-15T22:36:15.13907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# # Specify the folder name\n# folder_name = 'my_working_folder'\n\n# # Create the working folder in Kaggle\n# working_directory = '/kaggle/working/'\n# folder_path = os.path.join(working_directory, folder_name)\n\n# # Create the folder if it doesn't exist\n# os.makedirs(folder_path, exist_ok=True)\n\n# # Now, you can save files to this folder\n# # Example: Creating a NumPy array (replace this with your array)\n# array_data = np.random.randint(0, 255, size=(256, 256, 3), dtype=np.uint8)\n\n# Creating an image from the NumPy array\nimage = Image.fromarray(array_data)\n\n# Specify the path where you want to save the image in your working folder\nimage_path = os.path.join(folder_path, 'output_image.png')\n\n# Save the image to the working folder\nimage.save(image_path)\n\n# Optional: Display the saved image path\nprint(f\"Image saved to: {image_path}\")\n","metadata":{},"execution_count":null,"outputs":[]}]}