{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":193,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":137}],"dockerImageVersionId":30626,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        break\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-28T23:31:28.879716Z","iopub.execute_input":"2024-01-28T23:31:28.88123Z","iopub.status.idle":"2024-01-28T23:31:29.398533Z","shell.execute_reply.started":"2024-01-28T23:31:28.881179Z","shell.execute_reply":"2024-01-28T23:31:29.397587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Importing Libraries","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\nimport os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\nfrom sklearn.metrics import classification_report , confusion_matrix , accuracy_score , auc\nfrom sklearn.model_selection import train_test_split\nimport seaborn as sns\nimport cv2 as cv\nfrom PIL import Image \nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras import Sequential\nfrom keras.layers import Input, Dense,Conv2D , MaxPooling2D, Flatten,BatchNormalization,Dropout \nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, ReLU, MaxPool2D, GlobalAvgPool2D\nfrom tensorflow.keras.layers import Input, Add, ZeroPadding2D, Activation, BatchNormalization, Flatten, Conv2D, AveragePooling2D, GlobalMaxPooling2D, GlobalAveragePooling2D\nfrom tensorflow.keras.initializers import glorot_uniform\nfrom tensorflow.keras.models import Model\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OneHotEncoder\nfrom tensorflow.keras.preprocessing import image_dataset_from_directory\nimport tensorflow_hub as hub ","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:29.40058Z","iopub.execute_input":"2024-01-28T23:31:29.401422Z","iopub.status.idle":"2024-01-28T23:31:36.042162Z","shell.execute_reply.started":"2024-01-28T23:31:29.401364Z","shell.execute_reply":"2024-01-28T23:31:36.040823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df=pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:36.043955Z","iopub.execute_input":"2024-01-28T23:31:36.044674Z","iopub.status.idle":"2024-01-28T23:31:36.070492Z","shell.execute_reply.started":"2024-01-28T23:31:36.044637Z","shell.execute_reply":"2024-01-28T23:31:36.069185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:36.07195Z","iopub.execute_input":"2024-01-28T23:31:36.072308Z","iopub.status.idle":"2024-01-28T23:31:36.083492Z","shell.execute_reply.started":"2024-01-28T23:31:36.072279Z","shell.execute_reply":"2024-01-28T23:31:36.082146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:36.087923Z","iopub.execute_input":"2024-01-28T23:31:36.088714Z","iopub.status.idle":"2024-01-28T23:31:36.120206Z","shell.execute_reply.started":"2024-01-28T23:31:36.088662Z","shell.execute_reply":"2024-01-28T23:31:36.11904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:36.122039Z","iopub.execute_input":"2024-01-28T23:31:36.122514Z","iopub.status.idle":"2024-01-28T23:31:36.1375Z","shell.execute_reply.started":"2024-01-28T23:31:36.122469Z","shell.execute_reply":"2024-01-28T23:31:36.136329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['label'].value_counts().plot(kind='pie',autopct=\"%.1f%%\")\nplt.title(\"Ovarian Cancer Types Distributions\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:36.138452Z","iopub.execute_input":"2024-01-28T23:31:36.138774Z","iopub.status.idle":"2024-01-28T23:31:36.432458Z","shell.execute_reply.started":"2024-01-28T23:31:36.138748Z","shell.execute_reply":"2024-01-28T23:31:36.43107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\nsns.countplot(data=df, x='label', order=df['label'].value_counts().index)\nplt.title('Ovarian Cancer Types Distributions')\nplt.xlabel('Label')\nplt.ylabel('Nombre')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:36.435153Z","iopub.execute_input":"2024-01-28T23:31:36.437174Z","iopub.status.idle":"2024-01-28T23:31:36.690429Z","shell.execute_reply.started":"2024-01-28T23:31:36.437111Z","shell.execute_reply":"2024-01-28T23:31:36.689288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image Width and Height Distributions","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\nplt.subplot(1,2,1)\nsns.kdeplot(train_df['image_width'])\nplt.subplot(1,2,2)\nsns.kdeplot(train_df['image_height'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:36.692208Z","iopub.execute_input":"2024-01-28T23:31:36.692591Z","iopub.status.idle":"2024-01-28T23:31:37.25416Z","shell.execute_reply.started":"2024-01-28T23:31:36.692558Z","shell.execute_reply":"2024-01-28T23:31:37.252964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(18,6))\nplt.subplot(1,2,1)\nplt.hist(x=train_df['image_width'])\nplt.title(\"Image Width Distributions\")\nplt.subplot(1,2,2)\nplt.hist(x=train_df['image_height'])\nplt.title(\"Image Height Distributions\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:37.255681Z","iopub.execute_input":"2024-01-28T23:31:37.256407Z","iopub.status.idle":"2024-01-28T23:31:37.736961Z","shell.execute_reply.started":"2024-01-28T23:31:37.256353Z","shell.execute_reply":"2024-01-28T23:31:37.736054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['image_width'].describe()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:37.7381Z","iopub.execute_input":"2024-01-28T23:31:37.739134Z","iopub.status.idle":"2024-01-28T23:31:37.749647Z","shell.execute_reply.started":"2024-01-28T23:31:37.739098Z","shell.execute_reply":"2024-01-28T23:31:37.748485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['image_height'].describe()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:37.75174Z","iopub.execute_input":"2024-01-28T23:31:37.752364Z","iopub.status.idle":"2024-01-28T23:31:37.76859Z","shell.execute_reply.started":"2024-01-28T23:31:37.752321Z","shell.execute_reply":"2024-01-28T23:31:37.767261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:37.770963Z","iopub.execute_input":"2024-01-28T23:31:37.771465Z","iopub.status.idle":"2024-01-28T23:31:37.788289Z","shell.execute_reply.started":"2024-01-28T23:31:37.77142Z","shell.execute_reply":"2024-01-28T23:31:37.786854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:37.794524Z","iopub.execute_input":"2024-01-28T23:31:37.794909Z","iopub.status.idle":"2024-01-28T23:31:37.809285Z","shell.execute_reply.started":"2024-01-28T23:31:37.794878Z","shell.execute_reply":"2024-01-28T23:31:37.807883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  Outliers","metadata":{}},{"cell_type":"code","source":"q1 =train_df[\"image_id\"].quantile(0.25)\nq3 =train_df[\"image_id\"].quantile(0.75)\niqr = q3 - q1\noutlier_threshold = 1.5 * iqr\noutliers_mask = (train_df[\"image_id\"]< (q1 - outlier_threshold)) | (train_df[\"image_id\"] > (q3+ outlier_threshold))\nig, ax = plt.subplots()\nsns.boxplot(data=train_df[\"image_id\"], orient='v' , ax=ax)\nax.scatter(outliers_mask.sum(), ax.get_ylim()[0], c='red' , label='outliers')\nax.set_ylabel('')\nax.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:37.810822Z","iopub.execute_input":"2024-01-28T23:31:37.811183Z","iopub.status.idle":"2024-01-28T23:31:38.042997Z","shell.execute_reply.started":"2024-01-28T23:31:37.811149Z","shell.execute_reply":"2024-01-28T23:31:38.041781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q1 =train_df[\"image_width\"].quantile(0.25)\nq3 =train_df[\"image_width\"].quantile(0.75)\niqr = q3 - q1\noutlier_threshold = 1.5 * iqr\noutliers_mask = (train_df[\"image_width\"]< (q1 - outlier_threshold)) | (train_df[\"image_width\"] > (q3+ outlier_threshold))\nig, ax = plt.subplots()\nsns.boxplot(data=train_df[\"image_width\"], orient='v' , ax=ax)\nax.scatter(outliers_mask.sum(), ax.get_ylim()[0], c='red' , label='outliers')\nax.set_ylabel('')\nax.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:38.04424Z","iopub.execute_input":"2024-01-28T23:31:38.044567Z","iopub.status.idle":"2024-01-28T23:31:38.262839Z","shell.execute_reply.started":"2024-01-28T23:31:38.044539Z","shell.execute_reply":"2024-01-28T23:31:38.26183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q1 =train_df[\"image_height\"].quantile(0.25)\nq3 =train_df[\"image_height\"].quantile(0.75)\niqr = q3 - q1\noutlier_threshold = 1.5 * iqr\noutliers_mask = (train_df[\"image_height\"]< (q1 - outlier_threshold)) | (train_df[\"image_height\"] > (q3+ outlier_threshold))\nig, ax = plt.subplots()\nsns.boxplot(data=train_df[\"image_height\"], orient='v' , ax=ax)\nax.scatter(outliers_mask.sum(), ax.get_ylim()[0], c='red' , label='outliers')\nax.set_ylabel('')\nax.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:38.26416Z","iopub.execute_input":"2024-01-28T23:31:38.264713Z","iopub.status.idle":"2024-01-28T23:31:38.446426Z","shell.execute_reply.started":"2024-01-28T23:31:38.264679Z","shell.execute_reply":"2024-01-28T23:31:38.445048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image Data Collection","metadata":{}},{"cell_type":"code","source":"#large size\npath_train =\"/kaggle/input/UBC-OCEAN/train_images\"\npath_test =\"/kaggle/input/UBC-OCEAN/test_images\"\ntrain_folder = os.listdir(path_train)\ntest_folder = os.listdir(path_test)\nprint(len(train_folder))\nprint(len(test_folder))","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:38.448028Z","iopub.execute_input":"2024-01-28T23:31:38.448416Z","iopub.status.idle":"2024-01-28T23:31:38.460733Z","shell.execute_reply.started":"2024-01-28T23:31:38.448381Z","shell.execute_reply":"2024-01-28T23:31:38.459216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#images in small size\npath_train_copy = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\npath_test_copy = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"\ntrain_folder_copy = os.listdir(path_train_copy)\ntest_folder_copy = os.listdir(path_test_copy)\nprint(len(train_folder_copy))\nprint(len(test_folder_copy))","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:38.461954Z","iopub.execute_input":"2024-01-28T23:31:38.462401Z","iopub.status.idle":"2024-01-28T23:31:38.47499Z","shell.execute_reply.started":"2024-01-28T23:31:38.462365Z","shell.execute_reply":"2024-01-28T23:31:38.474029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport pandas as pd\ndf = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\nx = df.iloc[:,0:4].values\ny = df.iloc[:,0:1].values\ntrain_x, val_x, train_y, val_y = train_test_split(x, y, random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:38.476335Z","iopub.execute_input":"2024-01-28T23:31:38.477199Z","iopub.status.idle":"2024-01-28T23:31:38.489755Z","shell.execute_reply.started":"2024-01-28T23:31:38.477158Z","shell.execute_reply":"2024-01-28T23:31:38.488251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"val_y shape: {val_y.shape}\")\nprint(f\"train_x shape: {train_x.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:38.492203Z","iopub.execute_input":"2024-01-28T23:31:38.492723Z","iopub.status.idle":"2024-01-28T23:31:38.499505Z","shell.execute_reply.started":"2024-01-28T23:31:38.492672Z","shell.execute_reply":"2024-01-28T23:31:38.498229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reading 1 image","metadata":{}},{"cell_type":"code","source":"img = cv.imread('/kaggle/input/UBC-OCEAN/train_thumbnails/10077_thumbnail.png')\nplt.figure(figsize=(15, 5))#(15inch, 5inch)= (15*80 pixels, 5*80 pixels)\nplt.title('input image')\nplt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:38.502127Z","iopub.execute_input":"2024-01-28T23:31:38.502908Z","iopub.status.idle":"2024-01-28T23:31:40.727219Z","shell.execute_reply.started":"2024-01-28T23:31:38.50286Z","shell.execute_reply":"2024-01-28T23:31:40.723649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Normalies the image","metadata":{}},{"cell_type":"code","source":"img_normalized = cv.normalize(img, None, 0, 255,cv.NORM_MINMAX, dtype=cv.CV_8U)\nplt.figure(figsize=(15, 5))\nplt.title('Normalized Image')\nplt.imshow(img_normalized)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:40.728663Z","iopub.execute_input":"2024-01-28T23:31:40.729056Z","iopub.status.idle":"2024-01-28T23:31:42.50165Z","shell.execute_reply.started":"2024-01-28T23:31:40.729Z","shell.execute_reply":"2024-01-28T23:31:42.500178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. locate TMA","metadata":{}},{"cell_type":"code","source":"import os\n\ntrain_images_folder = '/kaggle/input/UBC-OCEAN/train_images'\ntrain_thumbnails_folder = '/kaggle/input/UBC-OCEAN/train_thumbnails'\n\nimages_files = set(os.listdir(train_images_folder))\nthumbnails_files = set(os.listdir(train_thumbnails_folder))\n\nimages_filenames = set([filename.split('.')[0] for filename in images_files])\n\nthumbnails_filenames = set([filename.split('_')[0] for filename in thumbnails_files])\n\nmissing_thumbnails = images_filenames - thumbnails_filenames\nmissing_thumbnails=[name+'.png' for name in missing_thumbnails]\n\nprint(\"TMA list:\")\nfor file_name in missing_thumbnails:\n    print(file_name)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:42.503273Z","iopub.execute_input":"2024-01-28T23:31:42.504287Z","iopub.status.idle":"2024-01-28T23:31:42.516578Z","shell.execute_reply.started":"2024-01-28T23:31:42.504235Z","shell.execute_reply":"2024-01-28T23:31:42.515414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Read train_csv","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ndf=pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ndf","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:42.517907Z","iopub.execute_input":"2024-01-28T23:31:42.518609Z","iopub.status.idle":"2024-01-28T23:31:42.541934Z","shell.execute_reply.started":"2024-01-28T23:31:42.518568Z","shell.execute_reply":"2024-01-28T23:31:42.540648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Visualize HGSC(TMA)","metadata":{}},{"cell_type":"code","source":"is_tma_df=df[df['is_tma']]\nis_tma_df","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:42.543249Z","iopub.execute_input":"2024-01-28T23:31:42.543586Z","iopub.status.idle":"2024-01-28T23:31:42.560733Z","shell.execute_reply.started":"2024-01-28T23:31:42.543556Z","shell.execute_reply":"2024-01-28T23:31:42.559307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Visualize HGSC(WSI thumbnail)","metadata":{}},{"cell_type":"code","source":"is_not_tma_df=df[df['is_tma']==False]\nis_not_tma_df","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:42.562953Z","iopub.execute_input":"2024-01-28T23:31:42.563764Z","iopub.status.idle":"2024-01-28T23:31:42.588148Z","shell.execute_reply.started":"2024-01-28T23:31:42.56371Z","shell.execute_reply":"2024-01-28T23:31:42.586698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_tma = train_df[train_df['is_tma']==True]","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:42.59028Z","iopub.execute_input":"2024-01-28T23:31:42.59077Z","iopub.status.idle":"2024-01-28T23:31:42.599393Z","shell.execute_reply.started":"2024-01-28T23:31:42.590724Z","shell.execute_reply":"2024-01-28T23:31:42.59786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_no_tma = train_df[train_df['is_tma']==False]\ntrain_df_no_tma","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:42.601234Z","iopub.execute_input":"2024-01-28T23:31:42.601781Z","iopub.status.idle":"2024-01-28T23:31:42.627696Z","shell.execute_reply.started":"2024-01-28T23:31:42.601736Z","shell.execute_reply":"2024-01-28T23:31:42.626284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_no_tma['image_id_path'] = [f\"{i}_thumbnail.png\" for i in train_df_no_tma['image_id']]","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:42.630556Z","iopub.execute_input":"2024-01-28T23:31:42.631163Z","iopub.status.idle":"2024-01-28T23:31:42.639624Z","shell.execute_reply.started":"2024-01-28T23:31:42.631115Z","shell.execute_reply":"2024-01-28T23:31:42.638525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_no_tma","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:42.641173Z","iopub.execute_input":"2024-01-28T23:31:42.641558Z","iopub.status.idle":"2024-01-28T23:31:42.67003Z","shell.execute_reply.started":"2024-01-28T23:31:42.641527Z","shell.execute_reply":"2024-01-28T23:31:42.668849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_tma['image_id_path'] = [f\"{i}.png\" for i in train_df_tma['image_id']]","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:42.671769Z","iopub.execute_input":"2024-01-28T23:31:42.672515Z","iopub.status.idle":"2024-01-28T23:31:42.681406Z","shell.execute_reply.started":"2024-01-28T23:31:42.672465Z","shell.execute_reply":"2024-01-28T23:31:42.680311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_tma","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:42.682938Z","iopub.execute_input":"2024-01-28T23:31:42.68387Z","iopub.status.idle":"2024-01-28T23:31:42.704811Z","shell.execute_reply.started":"2024-01-28T23:31:42.683833Z","shell.execute_reply":"2024-01-28T23:31:42.703943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Processing data","metadata":{}},{"cell_type":"code","source":"s=512","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:42.706219Z","iopub.execute_input":"2024-01-28T23:31:42.707132Z","iopub.status.idle":"2024-01-28T23:31:42.717421Z","shell.execute_reply.started":"2024-01-28T23:31:42.707094Z","shell.execute_reply":"2024-01-28T23:31:42.71655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_data = []\nimage_label = []\npath = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\npath1=\"/kaggle/input/UBC-OCEAN/train_images/\"\nfor img , label in zip(train_df_no_tma['image_id_path'],train_df_no_tma['label']):\n    image = Image.open(\"/kaggle/input/UBC-OCEAN/train_thumbnails/\"+img)\n    image = image.resize((s,s))\n    image = image.convert(\"RGB\")\n    image = np.array(image)\n    image_data.append(image)\n    image_label.append(label)\n\nfor img , label in zip(train_df_tma['image_id_path'],train_df_tma['label']):\n    image = Image.open(\"/kaggle/input/UBC-OCEAN/train_images/\"+img)\n    image = image.resize((s,s))\n    image = image.convert(\"RGB\")\n    image = np.array(image)\n    image_data.append(image)\n    image_label.append(label)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:31:42.719015Z","iopub.execute_input":"2024-01-28T23:31:42.719791Z","iopub.status.idle":"2024-01-28T23:34:30.458334Z","shell.execute_reply.started":"2024-01-28T23:31:42.719755Z","shell.execute_reply":"2024-01-28T23:34:30.456921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(image_data))\nprint(len(image_label))","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:34:30.459956Z","iopub.execute_input":"2024-01-28T23:34:30.46095Z","iopub.status.idle":"2024-01-28T23:34:30.466642Z","shell.execute_reply.started":"2024-01-28T23:34:30.460899Z","shell.execute_reply":"2024-01-28T23:34:30.465474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set(image_label)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:34:30.467913Z","iopub.execute_input":"2024-01-28T23:34:30.468291Z","iopub.status.idle":"2024-01-28T23:34:30.48399Z","shell.execute_reply.started":"2024-01-28T23:34:30.468257Z","shell.execute_reply":"2024-01-28T23:34:30.482846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_label_1 = []\nfor i in image_label:\n    if i==\"CC\":\n        image_label_1.append(0)\n    elif i==\"EC\":\n        image_label_1.append(1)\n    elif i==\"HGSC\":\n        image_label_1.append(2)\n    elif i==\"LGSC\":\n        image_label_1.append(3)\n    elif i==\"MC\":\n        image_label_1.append(4)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:34:30.485352Z","iopub.execute_input":"2024-01-28T23:34:30.48569Z","iopub.status.idle":"2024-01-28T23:34:30.495029Z","shell.execute_reply.started":"2024-01-28T23:34:30.485661Z","shell.execute_reply":"2024-01-28T23:34:30.49387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set(image_label_1)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:34:30.496252Z","iopub.execute_input":"2024-01-28T23:34:30.496597Z","iopub.status.idle":"2024-01-28T23:34:30.509823Z","shell.execute_reply.started":"2024-01-28T23:34:30.496568Z","shell.execute_reply":"2024-01-28T23:34:30.508709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(image_label_1)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:34:30.51161Z","iopub.execute_input":"2024-01-28T23:34:30.511986Z","iopub.status.idle":"2024-01-28T23:34:30.522641Z","shell.execute_reply.started":"2024-01-28T23:34:30.51194Z","shell.execute_reply":"2024-01-28T23:34:30.521458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"supp_dir = '/kaggle/input/ubc-ovarian-cancer-competition-supplemental-masks'\ndata_dir = '/kaggle/input/UBC-OCEAN'\n\ntrain_csv = pd.read_csv(data_dir + '/train.csv')\ntest_csv = pd.read_csv(data_dir + '/test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:34:30.535349Z","iopub.execute_input":"2024-01-28T23:34:30.536142Z","iopub.status.idle":"2024-01-28T23:34:30.547716Z","shell.execute_reply.started":"2024-01-28T23:34:30.536092Z","shell.execute_reply":"2024-01-28T23:34:30.546639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#filter for WSI\ntrain_csv = train_csv[train_csv['is_tma'] == False]\ntrain_data, val_data = train_test_split(train_csv, test_size=0.2, random_state=42)\n\n#image paths\ntrain_image_paths = [data_dir + '/train_thumbnails/' + str(img_id) + '_thumbnail.png' for img_id in train_data['image_id']]\nval_image_paths = [data_dir + '/train_thumbnails/' + str(img_id) + '_thumbnail.png' for img_id in val_data['image_id']]\ntest_image_paths = [data_dir + '/test_thumbnails/' + str(img_id) + '_thumbnail.png' for img_id in test_csv['image_id']]\n\n#multi-class classification: encoding labels for model (one-hot encoding)\none_hot_encoder = OneHotEncoder(sparse_output=False)\n\n# Reshape the labels to a 2D array before applying OneHotEncoder\ntrain_labels = np.array(train_data['label'])\nval_labels = np.array(val_data['label'])\n\ntrain_labels_reshaped = train_labels.reshape(-1, 1)\nval_labels_reshaped = val_labels.reshape(-1, 1)\n\ntrain_labels_one_hot = one_hot_encoder.fit_transform(train_labels_reshaped)\nval_labels_one_hot = one_hot_encoder.transform(val_labels_reshaped)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:34:30.549182Z","iopub.execute_input":"2024-01-28T23:34:30.549545Z","iopub.status.idle":"2024-01-28T23:34:30.5644Z","shell.execute_reply.started":"2024-01-28T23:34:30.549513Z","shell.execute_reply":"2024-01-28T23:34:30.563516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#(1) feature scales first (all pixel values are now between 0 and 1), image augmentation transforms (shear_range, zoom_range, horizontal_flip) to prevent overfitting\ndatagen = ImageDataGenerator(rescale = 1./255,\n                                   shear_range = 0.2,\n                                   #zoom_range = 0.2,\n                                   horizontal_flip = True\n                            )\n\ndef load_and_augment_img(img_path):\n    img = Image.open(img_path)\n    img = img.resize((224, 224))  # Resize to desired dimensions\n    img = np.array(img)  # Convert to numpy array\n    img = img.reshape((1,) + img.shape)  # Reshape to (1, height, width, channels) for flow()\n    img = datagen.flow(img, batch_size=1).next()  # Apply data augmentation\n    return img[0]\n\n# Apply data augmentation to training, validation, and test images\ntrain_images_augmented = [load_and_augment_img(path) for path in train_image_paths]\nval_images_augmented = [load_and_augment_img(path) for path in val_image_paths]\ntest_images_augmented = [load_and_augment_img(path) for path in test_image_paths]","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:34:30.566091Z","iopub.execute_input":"2024-01-28T23:34:30.566852Z","iopub.status.idle":"2024-01-28T23:36:40.415848Z","shell.execute_reply.started":"2024-01-28T23:34:30.566821Z","shell.execute_reply":"2024-01-28T23:36:40.414543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ndef visualize(image):\n    plt.figure(figsize=(10, 10))\n    plt.axis('off')\n    plt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:40.417719Z","iopub.execute_input":"2024-01-28T23:36:40.418131Z","iopub.status.idle":"2024-01-28T23:36:40.423808Z","shell.execute_reply.started":"2024-01-28T23:36:40.418077Z","shell.execute_reply":"2024-01-28T23:36:40.422483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize(train_images_augmented[0])","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:40.42532Z","iopub.execute_input":"2024-01-28T23:36:40.426994Z","iopub.status.idle":"2024-01-28T23:36:40.624792Z","shell.execute_reply.started":"2024-01-28T23:36:40.426936Z","shell.execute_reply":"2024-01-28T23:36:40.623404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\nnow we need to build the model to train our data\n\nfirst to convert the data into arrays using numpy","metadata":{}},{"cell_type":"markdown","source":"# Split The Data For Training ang Testing Purpose","metadata":{}},{"cell_type":"code","source":"x = np.array(image_data)\ny = np.array(image_label_1)\nx_train , x_test, y_train, y_test = train_test_split(x,y,test_size=0.2,shuffle=True)\nprint(f'X_train shape  is {x_train.shape}')\nprint(f'X_test shape  is {x_test.shape}')\nprint(f'y_train shape  is {y_train.shape}')\nprint(f'y_test shape  is {y_test.shape}')","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:40.626818Z","iopub.execute_input":"2024-01-28T23:36:40.627646Z","iopub.status.idle":"2024-01-28T23:36:41.103083Z","shell.execute_reply.started":"2024-01-28T23:36:40.627597Z","shell.execute_reply":"2024-01-28T23:36:41.101521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_image_data = []\npath = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:41.104633Z","iopub.execute_input":"2024-01-28T23:36:41.105607Z","iopub.status.idle":"2024-01-28T23:36:41.110791Z","shell.execute_reply.started":"2024-01-28T23:36:41.105558Z","shell.execute_reply":"2024-01-28T23:36:41.109503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_thumbnail_folder = os.listdir(path)\n\nfor img in test_thumbnail_folder:\n    image = Image.open(\"/kaggle/input/UBC-OCEAN/test_thumbnails/\"+img)\n    image = image.resize((600,600))\n    image = image.convert(\"RGB\")\n    image = np.array(image)\n    test_image_data.append(image)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:41.114189Z","iopub.execute_input":"2024-01-28T23:36:41.11466Z","iopub.status.idle":"2024-01-28T23:36:41.394644Z","shell.execute_reply.started":"2024-01-28T23:36:41.114612Z","shell.execute_reply":"2024-01-28T23:36:41.393236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_image_data[0].shape","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:41.39643Z","iopub.execute_input":"2024-01-28T23:36:41.396807Z","iopub.status.idle":"2024-01-28T23:36:41.404092Z","shell.execute_reply.started":"2024-01-28T23:36:41.396773Z","shell.execute_reply":"2024-01-28T23:36:41.402812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_test_array = np.array(test_image_data)\n\n\n## Scaling  The test Images\npredict_test_array_scaled = predict_test_array/255","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:41.405997Z","iopub.execute_input":"2024-01-28T23:36:41.406402Z","iopub.status.idle":"2024-01-28T23:36:41.416708Z","shell.execute_reply.started":"2024-01-28T23:36:41.406365Z","shell.execute_reply":"2024-01-28T23:36:41.415389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:41.418434Z","iopub.execute_input":"2024-01-28T23:36:41.418823Z","iopub.status.idle":"2024-01-28T23:36:41.437079Z","shell.execute_reply.started":"2024-01-28T23:36:41.418787Z","shell.execute_reply":"2024-01-28T23:36:41.435744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:41.439071Z","iopub.execute_input":"2024-01-28T23:36:41.439439Z","iopub.status.idle":"2024-01-28T23:36:41.447941Z","shell.execute_reply.started":"2024-01-28T23:36:41.439402Z","shell.execute_reply":"2024-01-28T23:36:41.446794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:41.449846Z","iopub.execute_input":"2024-01-28T23:36:41.450242Z","iopub.status.idle":"2024-01-28T23:36:41.463639Z","shell.execute_reply.started":"2024-01-28T23:36:41.450209Z","shell.execute_reply":"2024-01-28T23:36:41.462337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,16))\nclass_labels = ['CC', 'EC', 'HGSC', 'LGSC', 'MC']\nfor i in range(12):\n    plt.subplot(4,3,i+1)\n    plt.imshow(x_train[i])\n    plt.title(f\"Label:{class_labels[y_train[i]]}\")","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:41.465495Z","iopub.execute_input":"2024-01-28T23:36:41.466146Z","iopub.status.idle":"2024-01-28T23:36:45.131802Z","shell.execute_reply.started":"2024-01-28T23:36:41.466099Z","shell.execute_reply":"2024-01-28T23:36:45.127705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Scale the data","metadata":{}},{"cell_type":"code","source":"x_train_scaled = x_train/255\nx_test_scaled = x_test/255","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:45.133348Z","iopub.execute_input":"2024-01-28T23:36:45.134072Z","iopub.status.idle":"2024-01-28T23:36:46.52762Z","shell.execute_reply.started":"2024-01-28T23:36:45.134028Z","shell.execute_reply":"2024-01-28T23:36:46.526347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Building The Model\nnow to build the CNN model by Keras , using Conv2D layers , MaxPooling & Denses","metadata":{}},{"cell_type":"code","source":"#KerasModel = keras.models.Sequential([\n        #keras.layers.Conv2D(200,kernel_size=(3,3),activation='relu',input_shape=(s,s,3)),\n        #keras.layers.Conv2D(150,kernel_size=(3,3),activation='relu'),\n        #keras.layers.MaxPool2D(2,2),\n        #keras.layers.Conv2D(100,kernel_size=(3,3),activation='relu'),    \n        #keras.layers.Conv2D(80,kernel_size=(3,3),activation='relu'),\n        #keras.layers.MaxPool2D(2,2),\n        #keras.layers.Conv2D(64,kernel_size=(3,3),activation='relu'),\n        #keras.layers.MaxPool2D(2,2),\n        #keras.layers.Flatten() ,    \n        #keras.layers.Dense(120,activation='relu') ,    \n        #keras.layers.Dense(80,activation='relu') ,    \n        #keras.layers.Dense(50,activation='relu') ,        \n        #keras.layers.Dropout(rate=0.2) ,            \n        #keras.layers.Dense(5,activation='softmax') ,    \n        #])","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:46.529547Z","iopub.execute_input":"2024-01-28T23:36:46.530182Z","iopub.status.idle":"2024-01-28T23:36:46.536804Z","shell.execute_reply.started":"2024-01-28T23:36:46.53013Z","shell.execute_reply":"2024-01-28T23:36:46.535579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"now to compile the model , using adam optimizer , & sparse categorical crossentropy loss","metadata":{}},{"cell_type":"code","source":"#KerasModel.compile(optimizer ='adam',loss='sparse_categorical_crossentropy',metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:46.538546Z","iopub.execute_input":"2024-01-28T23:36:46.539094Z","iopub.status.idle":"2024-01-28T23:36:46.547834Z","shell.execute_reply.started":"2024-01-28T23:36:46.539043Z","shell.execute_reply":"2024-01-28T23:36:46.546646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"so how the model looks like ?","metadata":{}},{"cell_type":"code","source":"#print('Model Details are : ')\n#print(KerasModel.summary())","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:46.5492Z","iopub.execute_input":"2024-01-28T23:36:46.54966Z","iopub.status.idle":"2024-01-28T23:36:46.559808Z","shell.execute_reply.started":"2024-01-28T23:36:46.549623Z","shell.execute_reply":"2024-01-28T23:36:46.558507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"now to train the model , lets use 0 epochs now","metadata":{}},{"cell_type":"code","source":"#epochs = 10\n#ThisModel = KerasModel.fit(x_train, y_train, epochs=epochs,batch_size=64,verbose=1)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:46.561299Z","iopub.execute_input":"2024-01-28T23:36:46.56164Z","iopub.status.idle":"2024-01-28T23:36:46.570759Z","shell.execute_reply.started":"2024-01-28T23:36:46.561612Z","shell.execute_reply":"2024-01-28T23:36:46.569581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"how is the final loss & accuracy","metadata":{}},{"cell_type":"code","source":"#ModelLoss, ModelAccuracy = KerasModel.evaluate(x_test, y_test)\n\n#print('Test Loss is {}'.format(ModelLoss))\n#print('Test Accuracy is {}'.format(ModelAccuracy ))","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:46.572326Z","iopub.execute_input":"2024-01-28T23:36:46.572676Z","iopub.status.idle":"2024-01-28T23:36:46.581507Z","shell.execute_reply.started":"2024-01-28T23:36:46.572646Z","shell.execute_reply":"2024-01-28T23:36:46.580526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EfficientNet","metadata":{}},{"cell_type":"code","source":"eff_path = \"/kaggle/input/efficientnet-v2/tensorflow2/imagenet1k-b0-classification/2\"\npath1 = \"/kaggle/input/efficientnet-v2/tensorflow2/imagenet1k-b0-classification/2\"","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:46.583181Z","iopub.execute_input":"2024-01-28T23:36:46.583545Z","iopub.status.idle":"2024-01-28T23:36:46.595223Z","shell.execute_reply.started":"2024-01-28T23:36:46.583511Z","shell.execute_reply":"2024-01-28T23:36:46.59368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eff_model = hub.KerasLayer(path1,  input_shape=(s,s,3), trainable=False)\n\nnum_class = 5\nefficientnet_model = Sequential()\n\nefficientnet_model.add(eff_model)\nefficientnet_model.add(Dense(units=600, activation=\"relu\"))\nefficientnet_model.add(Dropout(0.2))\nefficientnet_model.add(Dense(units=600, activation=\"relu\"))\nefficientnet_model.add(Dropout(0.2))\nefficientnet_model.add(Dense(units=num_class, activation=\"softmax\"))\n\nefficientnet_model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:46.597151Z","iopub.execute_input":"2024-01-28T23:36:46.597558Z","iopub.status.idle":"2024-01-28T23:36:53.483678Z","shell.execute_reply.started":"2024-01-28T23:36:46.597524Z","shell.execute_reply":"2024-01-28T23:36:53.48239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"efficientnet_model.compile(optimizer=\"adam\",loss=\"sparse_categorical_crossentropy\",metrics=[\"accuracy\"])\n\nhistory_2 = efficientnet_model.fit(x_train_scaled,y_train,epochs=20,batch_size=64,\nvalidation_data=(x_test_scaled,y_test))","metadata":{"execution":{"iopub.status.busy":"2024-01-28T23:36:53.486244Z","iopub.execute_input":"2024-01-28T23:36:53.486731Z","iopub.status.idle":"2024-01-29T00:09:00.863995Z","shell.execute_reply.started":"2024-01-28T23:36:53.486686Z","shell.execute_reply":"2024-01-29T00:09:00.861421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss, acc = efficientnet_model.evaluate(x_test_scaled,y_test)\nprint(\"Accuracy on Test Data:\",acc) \nloss ,acc = efficientnet_model.evaluate(x_train_scaled,y_train)\nprint(\"Accuracy on Train Data:\",acc)","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:09:00.871053Z","iopub.execute_input":"2024-01-29T00:09:00.872608Z","iopub.status.idle":"2024-01-29T00:11:06.139464Z","shell.execute_reply.started":"2024-01-29T00:09:00.872525Z","shell.execute_reply":"2024-01-29T00:11:06.138452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the accuracy and loss\nacc = history_2.history['accuracy']\nloss = history_2.history['loss']\nepochs = range(1, len(acc) + 1)\n\nplt.plot(epochs, acc, 'b', label='Training Accuracy')\nplt.plot(epochs, loss, 'r', label='Training Loss')\nplt.title('Training Accuracy and Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy / Loss')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:06.141092Z","iopub.execute_input":"2024-01-29T00:11:06.141698Z","iopub.status.idle":"2024-01-29T00:11:06.52525Z","shell.execute_reply.started":"2024-01-29T00:11:06.141663Z","shell.execute_reply":"2024-01-29T00:11:06.523884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#PLOTTING ACCURACY AND LOSS GRAPHS\nfig, axes = plt.subplots(nrows=1, ncols=2, figsize=(10, 4))\n\naxes[0].plot(history_2.history['accuracy'], label = 'Training')\naxes[0].plot(history_2.history['val_accuracy'], label = 'Validation')\n\naxes[0].set_title(\"Model accuracy\")\naxes[0].set_xlabel('Epoch')\naxes[0].set_ylabel('Accuracy rate')\n\naxes[0].legend()  \n\naxes[1].plot(history_2.history['loss'], label = 'Training')\naxes[1].plot(history_2.history['val_loss'], label = 'Validation')\n\naxes[1].set_title(\"Model loss\")\naxes[1].set_xlabel('Epoch')\naxes[1].set_ylabel('Loss')\n\nplt.legend()  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:06.526887Z","iopub.execute_input":"2024-01-29T00:11:06.527305Z","iopub.status.idle":"2024-01-29T00:11:07.137735Z","shell.execute_reply.started":"2024-01-29T00:11:06.527269Z","shell.execute_reply":"2024-01-29T00:11:07.136472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = efficientnet_model.predict(x_test_scaled)","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:07.139627Z","iopub.execute_input":"2024-01-29T00:11:07.140037Z","iopub.status.idle":"2024-01-29T00:11:27.378471Z","shell.execute_reply.started":"2024-01-29T00:11:07.139993Z","shell.execute_reply":"2024-01-29T00:11:27.377338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_label = [np.argmax(i) for i in y_pred]","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:27.379967Z","iopub.execute_input":"2024-01-29T00:11:27.380865Z","iopub.status.idle":"2024-01-29T00:11:27.3861Z","shell.execute_reply.started":"2024-01-29T00:11:27.38082Z","shell.execute_reply":"2024-01-29T00:11:27.385229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_label[:10]   # Predicted Label","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:27.387264Z","iopub.execute_input":"2024-01-29T00:11:27.388216Z","iopub.status.idle":"2024-01-29T00:11:27.409348Z","shell.execute_reply.started":"2024-01-29T00:11:27.388177Z","shell.execute_reply":"2024-01-29T00:11:27.40816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test[:10]  # Actual Label","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:27.411333Z","iopub.execute_input":"2024-01-29T00:11:27.411715Z","iopub.status.idle":"2024-01-29T00:11:27.422014Z","shell.execute_reply.started":"2024-01-29T00:11:27.411681Z","shell.execute_reply":"2024-01-29T00:11:27.420907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"-----Metrics Evaluation on Test Data-----\")\nprint()\nprint(\"Confusion Matrix:\\n\",confusion_matrix(y_test,y_pred_label))\nprint()\nprint(\"Classification Report:\\n\",classification_report(y_test,y_pred_label))","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:27.423885Z","iopub.execute_input":"2024-01-29T00:11:27.424298Z","iopub.status.idle":"2024-01-29T00:11:27.448165Z","shell.execute_reply.started":"2024-01-29T00:11:27.424265Z","shell.execute_reply":"2024-01-29T00:11:27.447038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions on Test Data","metadata":{}},{"cell_type":"code","source":"class_labels = ['CC', 'EC', 'HGSC', 'LGSC', 'MC']","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:27.449669Z","iopub.execute_input":"2024-01-29T00:11:27.450055Z","iopub.status.idle":"2024-01-29T00:11:27.456259Z","shell.execute_reply.started":"2024-01-29T00:11:27.450021Z","shell.execute_reply":"2024-01-29T00:11:27.455044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = efficientnet_model.predict(predict_test_array_scaled)\npredictions = [np.argmax(i) for i in predictions]\npredictions","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:27.457865Z","iopub.execute_input":"2024-01-29T00:11:27.458279Z","iopub.status.idle":"2024-01-29T00:11:29.068214Z","shell.execute_reply.started":"2024-01-29T00:11:27.458245Z","shell.execute_reply":"2024-01-29T00:11:29.06725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"/kaggle/input/UBC-OCEAN/sample_submission.csv\"\np = pd.read_csv(path)\np","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:29.069504Z","iopub.execute_input":"2024-01-29T00:11:29.069879Z","iopub.status.idle":"2024-01-29T00:11:29.114008Z","shell.execute_reply.started":"2024-01-29T00:11:29.069846Z","shell.execute_reply":"2024-01-29T00:11:29.112526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"/kaggle/input/UBC-OCEAN/test.csv\"\ntest = pd.read_csv(path)\ntest","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:29.115819Z","iopub.execute_input":"2024-01-29T00:11:29.117209Z","iopub.status.idle":"2024-01-29T00:11:29.132465Z","shell.execute_reply.started":"2024-01-29T00:11:29.117162Z","shell.execute_reply":"2024-01-29T00:11:29.131243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_file = pd.DataFrame()\nsubmission_file['image_id'] = [41]\nsubmission_file['label'] = [class_labels[i] for i in predictions]\n\nsubmission_file.to_csv(\"submission.csv\",index_label=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:29.134311Z","iopub.execute_input":"2024-01-29T00:11:29.134698Z","iopub.status.idle":"2024-01-29T00:11:29.153146Z","shell.execute_reply.started":"2024-01-29T00:11:29.134662Z","shell.execute_reply":"2024-01-29T00:11:29.151689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"submission.csv\")\ndf","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:29.155096Z","iopub.execute_input":"2024-01-29T00:11:29.155487Z","iopub.status.idle":"2024-01-29T00:11:29.169405Z","shell.execute_reply.started":"2024-01-29T00:11:29.155453Z","shell.execute_reply":"2024-01-29T00:11:29.168087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" # ResNet_50","metadata":{}},{"cell_type":"code","source":"unique_classes = np.unique(image_label)\nuni_classes = list(unique_classes)\nlength = len(unique_classes)\n\nprint(uni_classes)","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:29.170749Z","iopub.execute_input":"2024-01-29T00:11:29.171177Z","iopub.status.idle":"2024-01-29T00:11:29.18088Z","shell.execute_reply.started":"2024-01-29T00:11:29.171137Z","shell.execute_reply":"2024-01-29T00:11:29.179384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_tensor = (224,224, 3)\nNUMBER_OF_CLASSES = 5","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:29.182358Z","iopub.execute_input":"2024-01-29T00:11:29.183046Z","iopub.status.idle":"2024-01-29T00:11:29.190362Z","shell.execute_reply.started":"2024-01-29T00:11:29.182992Z","shell.execute_reply":"2024-01-29T00:11:29.189001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DEFINE THE RESNET-50 ARCHITECTURE *************************************************************\n\ndef conv_batchnorm_relu(x, filters, kernel_size, strides=1):\n    x = Conv2D(filters=filters, kernel_size=kernel_size, strides=strides, padding = 'same')(x)\n    x = BatchNormalization()(x)\n    x = ReLU()(x)\n    return x\n\ndef identity_block(tensor, filters):\n    x = conv_batchnorm_relu(tensor, filters=filters, kernel_size=1, strides=1)\n    x = conv_batchnorm_relu(x, filters=filters, kernel_size=3, strides=1)\n    x = Conv2D(filters=4*filters, kernel_size=1, strides=1)(x)\n    x = BatchNormalization()(x)\n    x = Add()([tensor,x]) \n    x = ReLU()(x)\n    return x\n\ndef projection_block(tensor, filters, strides):\n    x = conv_batchnorm_relu(tensor, filters=filters, kernel_size=1, strides=strides)     \n    x = conv_batchnorm_relu(x, filters=filters, kernel_size=3, strides=1)     \n    x = Conv2D(filters=4*filters, kernel_size=1, strides=1)(x)     \n    x = BatchNormalization()(x) \n    shortcut = Conv2D(filters=4*filters, kernel_size=1, strides=strides)(tensor)     \n    shortcut = BatchNormalization()(shortcut)          \n    x = Add()([shortcut,x])       \n    x = ReLU()(x)          \n    return x \n    \ndef resnet_block(x, filters, reps, strides):\n    x = projection_block(x, filters, strides)\n    for _ in range(reps-1):\n        x = identity_block(x,filters)\n    return x ","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:29.192901Z","iopub.execute_input":"2024-01-29T00:11:29.193449Z","iopub.status.idle":"2024-01-29T00:11:29.21074Z","shell.execute_reply.started":"2024-01-29T00:11:29.193402Z","shell.execute_reply":"2024-01-29T00:11:29.209151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input = Input(shape=input_tensor)\n\n\nx = conv_batchnorm_relu(input, filters=64, kernel_size=7, strides=2)\nx = MaxPool2D(pool_size=3, strides=2)(x)\nx = resnet_block(x, filters=64, reps=3, strides=1)\nx = resnet_block(x, filters=128, reps=4, strides=2)\nx = resnet_block(x, filters=256, reps=6, strides=2)\nx = resnet_block(x, filters=512, reps=3, strides=2)\nx = GlobalAvgPool2D()(x)\n\n\noutput = Dense(NUMBER_OF_CLASSES, activation ='softmax')(x)\n\n\nmodel = Model(inputs=input, outputs=output)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:29.213033Z","iopub.execute_input":"2024-01-29T00:11:29.21354Z","iopub.status.idle":"2024-01-29T00:11:31.815887Z","shell.execute_reply.started":"2024-01-29T00:11:29.21349Z","shell.execute_reply":"2024-01-29T00:11:31.814427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nfrom tensorflow.keras.utils import plot_model\n\nplot_model(model)\n'''","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:11:31.817905Z","iopub.execute_input":"2024-01-29T00:11:31.81835Z","iopub.status.idle":"2024-01-29T00:11:31.826636Z","shell.execute_reply.started":"2024-01-29T00:11:31.818311Z","shell.execute_reply":"2024-01-29T00:11:31.825308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.optimizers import Adam\n#from tensorflow.keras.callbacks import EarlyStopping\n\nmodel.compile(optimizer=Adam(learning_rate=0.001), loss=\"categorical_crossentropy\", metrics=['accuracy'])\n\nresults = model.fit(\n    np.array(train_images_augmented), train_labels_one_hot,\n    epochs=10, \n    batch_size=32,\n    validation_data=(np.array(val_images_augmented), val_labels_one_hot ),\n)","metadata":{"execution":{"iopub.status.busy":"2024-01-29T00:41:22.365206Z","iopub.execute_input":"2024-01-29T00:41:22.365753Z","iopub.status.idle":"2024-01-29T01:05:07.290376Z","shell.execute_reply.started":"2024-01-29T00:41:22.365711Z","shell.execute_reply":"2024-01-29T01:05:07.288857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#SUBMISSION FILE\n\npred = model.predict(np.array(test_images_augmented))\ntest = np.argmax(pred,axis=1)\npredicted_labels = [uni_classes[i] for i in test]\n\nsubmission = [[test_csv[\"image_id\"][i], predicted_labels[i]] for i in range(len(test_csv)) ]\ndf = pd.DataFrame(submission,columns = [\"image_id\",\"label\"])\n\ndf.to_csv(\"submission.csv\", index=False)\n\n#PLOTTING ACCURACY AND LOSS GRAPHS\nfig, axes = plt.subplots(nrows=1, ncols=2, figsize=(10, 4))\n\naxes[0].plot(results.history['accuracy'], label = 'Training')\naxes[0].plot(results.history['val_accuracy'], label = 'Validation')\n\naxes[0].set_title(\"Model accuracy\")\naxes[0].set_xlabel('Epoch')\naxes[0].set_ylabel('Accuracy rate')\n\naxes[0].legend()  \n\naxes[1].plot(results.history['loss'], label = 'Training')\naxes[1].plot(results.history['val_loss'], label = 'Validation')\n\naxes[1].set_title(\"Model loss\")\naxes[1].set_xlabel('Epoch')\naxes[1].set_ylabel('Loss')\n\nplt.legend()  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-29T01:05:09.092818Z","iopub.execute_input":"2024-01-29T01:05:09.093335Z","iopub.status.idle":"2024-01-29T01:05:09.775339Z","shell.execute_reply.started":"2024-01-29T01:05:09.093296Z","shell.execute_reply":"2024-01-29T01:05:09.774066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}