{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n#import numpy as np # linear algebra\n#import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    print(dirname)\n    #for filename in filenames:\n        #print(os.path.join(dirname, filename))\n    \n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-28T22:50:41.560532Z","iopub.execute_input":"2023-10-28T22:50:41.560851Z","iopub.status.idle":"2023-10-28T22:50:41.581842Z","shell.execute_reply.started":"2023-10-28T22:50:41.560825Z","shell.execute_reply":"2023-10-28T22:50:41.580683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport seaborn as sns\n\nfrom PIL import Image\nimport tensorflow as tf\nimport tensorflow_hub as hub\nfrom tensorflow import keras\nfrom sklearn.model_selection import train_test_split\nfrom keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:50:41.586896Z","iopub.execute_input":"2023-10-28T22:50:41.587142Z","iopub.status.idle":"2023-10-28T22:50:46.291047Z","shell.execute_reply.started":"2023-10-28T22:50:41.587111Z","shell.execute_reply":"2023-10-28T22:50:46.290066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\ntest_csv = pd.read_csv(\"/kaggle/input/UBC-OCEAN/test.csv\")\nprint(train_csv.shape, test_csv.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:50:46.292333Z","iopub.execute_input":"2023-10-28T22:50:46.293007Z","iopub.status.idle":"2023-10-28T22:50:46.306892Z","shell.execute_reply.started":"2023-10-28T22:50:46.292975Z","shell.execute_reply":"2023-10-28T22:50:46.305733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv.sample(10)","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:50:46.310914Z","iopub.execute_input":"2023-10-28T22:50:46.311239Z","iopub.status.idle":"2023-10-28T22:50:46.327238Z","shell.execute_reply.started":"2023-10-28T22:50:46.311212Z","shell.execute_reply":"2023-10-28T22:50:46.326331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv.info()","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:50:46.328292Z","iopub.execute_input":"2023-10-28T22:50:46.32861Z","iopub.status.idle":"2023-10-28T22:50:46.34419Z","shell.execute_reply.started":"2023-10-28T22:50:46.328585Z","shell.execute_reply":"2023-10-28T22:50:46.343092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(train_csv, x='label')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:50:46.345405Z","iopub.execute_input":"2023-10-28T22:50:46.345767Z","iopub.status.idle":"2023-10-28T22:50:46.869834Z","shell.execute_reply.started":"2023-10-28T22:50:46.345735Z","shell.execute_reply":"2023-10-28T22:50:46.868807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:50:46.871603Z","iopub.execute_input":"2023-10-28T22:50:46.872323Z","iopub.status.idle":"2023-10-28T22:50:46.881881Z","shell.execute_reply.started":"2023-10-28T22:50:46.872283Z","shell.execute_reply":"2023-10-28T22:50:46.880724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_counts = train_csv['label'].value_counts()\nfig = px.pie(label_counts, values=label_counts.values, names=label_counts.index)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:50:46.884037Z","iopub.execute_input":"2023-10-28T22:50:46.88441Z","iopub.status.idle":"2023-10-28T22:50:46.957Z","shell.execute_reply.started":"2023-10-28T22:50:46.884378Z","shell.execute_reply":"2023-10-28T22:50:46.956084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"is_tma_counts = train_csv['is_tma'].value_counts()\nfig = px.pie(is_tma_counts, values=is_tma_counts.values, names=is_tma_counts.index)\nfig.update_layout(title=\"Image Distributions on Train Data\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:50:46.958336Z","iopub.execute_input":"2023-10-28T22:50:46.958641Z","iopub.status.idle":"2023-10-28T22:50:47.01356Z","shell.execute_reply.started":"2023-10-28T22:50:46.958615Z","shell.execute_reply":"2023-10-28T22:50:47.012581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.histplot(data=train_csv, x='label', kde=True)\nplt.title('Distribution des labels')\nplt.show()\n\nplt.figure(figsize=(8, 6))\nsns.histplot(data=train_csv, x='image_width', kde=True)\nplt.title('Distribution de la largeur des images')\nplt.show()\n\nplt.figure(figsize=(8, 6))\nsns.histplot(data=train_csv, x='image_height', kde=True)\nplt.title('Distribution de la hauteur des images')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:50:47.014877Z","iopub.execute_input":"2023-10-28T22:50:47.01523Z","iopub.status.idle":"2023-10-28T22:50:48.013413Z","shell.execute_reply.started":"2023-10-28T22:50:47.015196Z","shell.execute_reply":"2023-10-28T22:50:48.012454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Image.MAX_IMAGE_PIXELS = 10000000000\n\nimages_data = []\nimages_label = []\n\nfor img_id, label, tma in zip(train_csv['image_id'],train_csv['label'] ,train_csv['is_tma']):\n    #print(img_id, label,  tma)\n    if not tma:\n        img_name = str(img_id)+\"_thumbnail.png\"\n        image = Image.open(\"/kaggle/input/UBC-OCEAN/train_thumbnails/\"+img_name)\n        image = image.resize((512,512))\n        image = np.array(image)\n        images_data.append(image)\n        images_label.append(label)\n        \n        \n    elif tma:\n        img_name = str(img_id)+\".png\"\n        image = Image.open(\"/kaggle/input/UBC-OCEAN/train_images/\"+img_name)\n        image = image.resize((512,512))\n        image = np.array(image)\n        images_data.append(image)\n        images_label.append(label)\n\nX = np.array(images_data)\nlabels = np.array(images_label)\nX.shape, labels.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:50:48.014781Z","iopub.execute_input":"2023-10-28T22:50:48.015147Z","iopub.status.idle":"2023-10-28T22:53:02.099774Z","shell.execute_reply.started":"2023-10-28T22:50:48.015112Z","shell.execute_reply":"2023-10-28T22:53:02.09851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(X[0])\nplt.title(labels[0])","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:53:02.10119Z","iopub.execute_input":"2023-10-28T22:53:02.101542Z","iopub.status.idle":"2023-10-28T22:53:02.504897Z","shell.execute_reply.started":"2023-10-28T22:53:02.101498Z","shell.execute_reply":"2023-10-28T22:53:02.503773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_row = 6\nn_col = 5\nplt.figure(figsize=(10, 7))\nfor i in range(n_row*n_col):\n    idx = np.random.randint(538)\n    plt.subplot(n_row, n_col,i+1)\n    plt.imshow(X[idx])\n    plt.title(labels[idx])\n    plt.axis('off')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:53:02.509849Z","iopub.execute_input":"2023-10-28T22:53:02.510186Z","iopub.status.idle":"2023-10-28T22:53:05.547382Z","shell.execute_reply.started":"2023-10-28T22:53:02.510159Z","shell.execute_reply":"2023-10-28T22:53:05.546267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classes = {\n    'CC' : 0, 'EC' : 1, 'HGSC' : 2, 'LGSC' : 3, 'MC' : 4\n}\n\nY = []\nfor elt in labels:\n    Y.append(classes[elt])\nY = np.array(Y)\nY.shape, Y[0], labels[0]","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:53:47.161358Z","iopub.execute_input":"2023-10-28T22:53:47.16176Z","iopub.status.idle":"2023-10-28T22:53:47.170476Z","shell.execute_reply.started":"2023-10-28T22:53:47.161726Z","shell.execute_reply":"2023-10-28T22:53:47.16943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.unique(Y)","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:53:50.143593Z","iopub.execute_input":"2023-10-28T22:53:50.14403Z","iopub.status.idle":"2023-10-28T22:53:50.151637Z","shell.execute_reply.started":"2023-10-28T22:53:50.143995Z","shell.execute_reply":"2023-10-28T22:53:50.150532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_c = 5\ny = keras.utils.to_categorical(\n    Y, num_classes=5\n)\n\nx_train, x_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=0,stratify=y)","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:53:52.06932Z","iopub.execute_input":"2023-10-28T22:53:52.069735Z","iopub.status.idle":"2023-10-28T22:53:52.227641Z","shell.execute_reply.started":"2023-10-28T22:53:52.069702Z","shell.execute_reply":"2023-10-28T22:53:52.226726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"X_train shape : {x_train.shape}\\ny_train shape : {y_train.shape}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:53:53.65818Z","iopub.execute_input":"2023-10-28T22:53:53.658577Z","iopub.status.idle":"2023-10-28T22:53:53.66424Z","shell.execute_reply.started":"2023-10-28T22:53:53.658541Z","shell.execute_reply":"2023-10-28T22:53:53.66321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train[:10]","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:53:54.88544Z","iopub.execute_input":"2023-10-28T22:53:54.88641Z","iopub.status.idle":"2023-10-28T22:53:54.894217Z","shell.execute_reply.started":"2023-10-28T22:53:54.886372Z","shell.execute_reply":"2023-10-28T22:53:54.893171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train_scaled = x_train/255\nx_val_scaled = x_val/255","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:53:58.792281Z","iopub.execute_input":"2023-10-28T22:53:58.793322Z","iopub.status.idle":"2023-10-28T22:53:59.8927Z","shell.execute_reply.started":"2023-10-28T22:53:58.793275Z","shell.execute_reply":"2023-10-28T22:53:59.891702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"url1 = \"https://tfhub.dev/google/imagenet/efficientnet_v2_imagenet21k_ft1k_m/classification/2\"\nurl2 = \"https://tfhub.dev/google/imagenet/efficientnet_v2_imagenet21k_ft1k_xl/classification/2\"\n\nmodel_hub = hub.KerasLayer(url2,  input_shape=(512,512,3), trainable=False)\n\nnum_class = 5\nmodel = keras.models.Sequential()\nmodel.add(model_hub)\n#model.add(keras.layers.Dense(64,activation='relu'))\n#model.add(keras.layers.Dropout(0.2))\n#model.add(keras.layers.Dense(32,activation='relu'))\n#model.add(keras.layers.Dropout(0.2))\nmodel.add(keras.layers.Dense(units=num_class, activation=\"softmax\"))\n\n\nmodel.compile(optimizer=\"adam\",loss=\"categorical_crossentropy\",\n             metrics=[\"accuracy\"])\n\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-10-28T23:24:32.577313Z","iopub.execute_input":"2023-10-28T23:24:32.578025Z","iopub.status.idle":"2023-10-28T23:25:08.144769Z","shell.execute_reply.started":"2023-10-28T23:24:32.577991Z","shell.execute_reply":"2023-10-28T23:25:08.143802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_generator = ImageDataGenerator(\n    rotation_range=10, \n    width_shift_range=0.1,  \n    height_shift_range=0.1, \n    zoom_range=0.1,  \n    horizontal_flip=True \n)\naugmented_data_generator = data_generator.flow(x_train_scaled, y_train, batch_size=16)","metadata":{"execution":{"iopub.status.busy":"2023-10-28T23:25:51.497612Z","iopub.execute_input":"2023-10-28T23:25:51.498298Z","iopub.status.idle":"2023-10-28T23:25:51.984731Z","shell.execute_reply.started":"2023-10-28T23:25:51.498261Z","shell.execute_reply":"2023-10-28T23:25:51.983583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(augmented_data_generator,epochs=20, validation_data=(x_val_scaled,y_val))","metadata":{"execution":{"iopub.status.busy":"2023-10-28T23:25:56.579134Z","iopub.execute_input":"2023-10-28T23:25:56.579879Z","iopub.status.idle":"2023-10-28T23:40:45.887599Z","shell.execute_reply.started":"2023-10-28T23:25:56.579843Z","shell.execute_reply":"2023-10-28T23:40:45.886382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}