{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfrom tensorflow.keras.preprocessing import image\nfrom keras import layers\nimport keras\nimport cv2\nimport matplotlib.pyplot as plt \nimport matplotlib.image as mpimg\nimport plotly.express as px\nimport plotly.graph_objects as go","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Showing train_thumbnails Image","metadata":{}},{"cell_type":"markdown","source":"containing smaller .png copies of the whole slide images. Thumbnails are not provided for TMAs.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\nimg = plt.imread('/kaggle/input/UBC-OCEAN/train_thumbnails/10077_thumbnail.png')\nimplt = plt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T08:15:38.680476Z","iopub.execute_input":"2023-10-08T08:15:38.681332Z","iopub.status.idle":"2023-10-08T08:15:40.651867Z","shell.execute_reply.started":"2023-10-08T08:15:38.6813Z","shell.execute_reply":"2023-10-08T08:15:40.650845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Showing test_image ","metadata":{}},{"cell_type":"markdown","source":"[train/test]_images A folder containing the relevant images. There are two categories of images: whole slide images (WSI) and tissue microarray (TMA). Whole slide images are at 20x magnification and can be quite large. The TMAs are smaller (roughly 4,000x4,000 pixels) but at 40x magnification.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\nimg = plt.imread('/kaggle/input/UBC-OCEAN/test_images/41.png')\nimplt = plt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T08:18:00.896665Z","iopub.execute_input":"2023-10-08T08:18:00.897143Z","iopub.status.idle":"2023-10-08T08:19:50.459835Z","shell.execute_reply.started":"2023-10-08T08:18:00.897108Z","shell.execute_reply":"2023-10-08T08:19:50.458504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* **image_id** - A unique ID code for each image.\n\n* **label** - The target class. One of these subtypes of ovarian cancer: CC, EC, HGSC, LGSC, MC, Other. The Other class is not present in the training set; identifying outliers is one of the challenges of this competition. Only available for the train set.\n\n* **image_width** - The image width in pixels.\n\n* **image_height** - The image height in pixels.\n\n* **is_tma** - True if the slide is a tissue microarray. Only available for the train set.","metadata":{}},{"cell_type":"code","source":"ubc_train = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-08T08:15:40.653339Z","iopub.execute_input":"2023-10-08T08:15:40.653638Z","iopub.status.idle":"2023-10-08T08:15:40.663413Z","shell.execute_reply.started":"2023-10-08T08:15:40.653611Z","shell.execute_reply":"2023-10-08T08:15:40.662121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ubc_train","metadata":{"execution":{"iopub.status.busy":"2023-10-08T08:15:40.665996Z","iopub.execute_input":"2023-10-08T08:15:40.666692Z","iopub.status.idle":"2023-10-08T08:15:40.684672Z","shell.execute_reply.started":"2023-10-08T08:15:40.666647Z","shell.execute_reply":"2023-10-08T08:15:40.683559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(ubc_train,x=\"label\", title=\"label\", color=\"label\")\n# Update the layout and add box plots\nfig.update_layout(\n    bargap=0.2\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-08T08:16:10.955135Z","iopub.execute_input":"2023-10-08T08:16:10.955534Z","iopub.status.idle":"2023-10-08T08:16:13.016281Z","shell.execute_reply.started":"2023-10-08T08:16:10.955506Z","shell.execute_reply":"2023-10-08T08:16:13.01519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax = px.histogram(ubc_train,x=\"is_tma\",marginal=\"box\",title=\"is_tma\")\nax.update_layout(bargap=0.2)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T08:17:07.700825Z","iopub.execute_input":"2023-10-08T08:17:07.702618Z","iopub.status.idle":"2023-10-08T08:17:07.826353Z","shell.execute_reply.started":"2023-10-08T08:17:07.702567Z","shell.execute_reply":"2023-10-08T08:17:07.825135Z"},"trusted":true},"execution_count":null,"outputs":[]}]}