{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip -q install tensorflow==2.3.0","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Basics / Data manipulation\nimport numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nimport zipfile\nimport os\n\n# Visualization\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport cv2\nimport skimage.io\n\n# ML\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:42.100996Z","iopub.execute_input":"2022-01-20T19:24:42.101279Z","iopub.status.idle":"2022-01-20T19:24:48.366399Z","shell.execute_reply.started":"2022-01-20T19:24:42.101247Z","shell.execute_reply":"2022-01-20T19:24:48.365709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data\n10k+ of .tiff images\n*    **80%** for training \n*    **20%** for internal testing\n            *  10% Validation\n            *  10% Testing","metadata":{}},{"cell_type":"code","source":"# Folder paths\nTRAIN = '../input/prostate-cancer-grade-assessment/train_images'\nMASKS = '../input/prostate-cancer-grade-assessment/train_label_masks'\nOUT_TRAIN = './train.zip'\nOUT_VALIDATION = './validation.zip'\nOUT_TEST = './test.zip'\nOUT_MASKS_TRAIN = './masks_train.zip'\nOUT_MASKS_VALIDATION = './masks_validation.zip'\nOUT_MASKS_TEST = './masks_test.zip'\n\nBASE_FOLDER = \"/kaggle/input/prostate-cancer-grade-assessment/\"\n!ls {BASE_FOLDER}\nBASE_FOLDER2 =\"/kaggle/input/panda-tiles/\"\n!ls {BASE_FOLDER2}","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:48.367593Z","iopub.execute_input":"2022-01-20T19:24:48.368094Z","iopub.status.idle":"2022-01-20T19:24:49.934549Z","shell.execute_reply.started":"2022-01-20T19:24:48.368055Z","shell.execute_reply":"2022-01-20T19:24:49.933715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(BASE_FOLDER+\"train.csv\")\ntest = pd.read_csv(BASE_FOLDER+\"test.csv\")\nsub = pd.read_csv(BASE_FOLDER+\"sample_submission.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:49.936185Z","iopub.execute_input":"2022-01-20T19:24:49.936747Z","iopub.status.idle":"2022-01-20T19:24:50.009888Z","shell.execute_reply.started":"2022-01-20T19:24:49.936698Z","shell.execute_reply":"2022-01-20T19:24:50.009103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking for all the \"negative\" labels in the label of gleason_score\ntrain[train['gleason_score'] == 'negative']","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:50.013184Z","iopub.execute_input":"2022-01-20T19:24:50.013411Z","iopub.status.idle":"2022-01-20T19:24:50.032251Z","shell.execute_reply.started":"2022-01-20T19:24:50.013387Z","shell.execute_reply":"2022-01-20T19:24:50.031305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Deleting from the dataset a mislabeled row and converting the \"negative\" labels to \"0+0\" in order to have an standard\ntrain.drop([7273],inplace=True)\ntrain['gleason_score'] = train['gleason_score'].apply(lambda x: \"0+0\" if x == \"negative\" else x)","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:50.03533Z","iopub.execute_input":"2022-01-20T19:24:50.03556Z","iopub.status.idle":"2022-01-20T19:24:50.045716Z","shell.execute_reply.started":"2022-01-20T19:24:50.035538Z","shell.execute_reply":"2022-01-20T19:24:50.044859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = np.array(train) # Converting the DataFrame to an array to take the column\nlabels = data[:, 3] # Labels of interest (GLEASON SCORE)\nlabels","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:50.047016Z","iopub.execute_input":"2022-01-20T19:24:50.04731Z","iopub.status.idle":"2022-01-20T19:24:50.055127Z","shell.execute_reply.started":"2022-01-20T19:24:50.047282Z","shell.execute_reply":"2022-01-20T19:24:50.054244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = data[:, :3] # Features of interest (ID, PROVIDER, ISUP GRADE)\nfeatures","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:50.056798Z","iopub.execute_input":"2022-01-20T19:24:50.057113Z","iopub.status.idle":"2022-01-20T19:24:50.064794Z","shell.execute_reply.started":"2022-01-20T19:24:50.057077Z","shell.execute_reply":"2022-01-20T19:24:50.064115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = features\ny = labels","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:50.065853Z","iopub.execute_input":"2022-01-20T19:24:50.066189Z","iopub.status.idle":"2022-01-20T19:24:50.07296Z","shell.execute_reply.started":"2022-01-20T19:24:50.066152Z","shell.execute_reply":"2022-01-20T19:24:50.072322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:50.074094Z","iopub.execute_input":"2022-01-20T19:24:50.074341Z","iopub.status.idle":"2022-01-20T19:24:50.082914Z","shell.execute_reply.started":"2022-01-20T19:24:50.074315Z","shell.execute_reply":"2022-01-20T19:24:50.082266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_check, y_train, y_check = train_test_split(X, y, test_size=0.2, random_state = 42) ","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:50.08497Z","iopub.execute_input":"2022-01-20T19:24:50.085343Z","iopub.status.idle":"2022-01-20T19:24:50.094339Z","shell.execute_reply.started":"2022-01-20T19:24:50.085253Z","shell.execute_reply":"2022-01-20T19:24:50.093563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_validation, X_test, y_validation, y_test = train_test_split(X_check, y_check, test_size=0.5, random_state = 84)","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:50.095522Z","iopub.execute_input":"2022-01-20T19:24:50.095728Z","iopub.status.idle":"2022-01-20T19:24:50.102448Z","shell.execute_reply.started":"2022-01-20T19:24:50.095706Z","shell.execute_reply":"2022-01-20T19:24:50.101751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train-Validation-Test","metadata":{}},{"cell_type":"code","source":"X_train = pd.DataFrame(X_train, columns=[\"image_id\", \"data_provider\", \"isup_grade\"])\nX_train[\"gleason_score\"] = y_train\n\nX_validation = pd.DataFrame(X_validation, columns=[\"image_id\", \"data_provider\", \"isup_grade\"])\nX_validation[\"gleason_score\"] = y_validation\n\nX_test = pd.DataFrame(X_test, columns=[\"image_id\", \"data_provider\", \"isup_grade\"])\nX_test[\"gleason_score\"] = y_test","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:50.103639Z","iopub.execute_input":"2022-01-20T19:24:50.103834Z","iopub.status.idle":"2022-01-20T19:24:50.120461Z","shell.execute_reply.started":"2022-01-20T19:24:50.103812Z","shell.execute_reply":"2022-01-20T19:24:50.119808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_eda = X_train.groupby(\"gleason_score\").count()[\"image_id\"].reset_index().sort_values(by=\"image_id\", ascending=False)\ntrain_eda.style.background_gradient(cmap=\"Greens\")","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:50.121733Z","iopub.execute_input":"2022-01-20T19:24:50.122051Z","iopub.status.idle":"2022-01-20T19:24:50.22064Z","shell.execute_reply.started":"2022-01-20T19:24:50.122011Z","shell.execute_reply":"2022-01-20T19:24:50.219968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_eda = X_validation.groupby(\"gleason_score\").count()[\"image_id\"].reset_index().sort_values(by=\"image_id\", ascending=False)\nvalidation_eda.style.background_gradient(cmap=\"Reds\")","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:50.221504Z","iopub.execute_input":"2022-01-20T19:24:50.221701Z","iopub.status.idle":"2022-01-20T19:24:50.244521Z","shell.execute_reply.started":"2022-01-20T19:24:50.221679Z","shell.execute_reply":"2022-01-20T19:24:50.243955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_eda = X_test.groupby(\"gleason_score\").count()[\"image_id\"].reset_index().sort_values(by=\"image_id\", ascending=False)\ntest_eda.style.background_gradient(cmap=\"Blues\")","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:50.2454Z","iopub.execute_input":"2022-01-20T19:24:50.245727Z","iopub.status.idle":"2022-01-20T19:24:50.269068Z","shell.execute_reply.started":"2022-01-20T19:24:50.245701Z","shell.execute_reply":"2022-01-20T19:24:50.268373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objects as go\n\nfig = go.Figure(data=[\n    go.Bar(name=\"Test\", x=train_eda[\"gleason_score\"], y=train_eda[\"image_id\"]),\n    go.Bar(name=\"Validation\", x=validation_eda[\"gleason_score\"], y=validation_eda[\"image_id\"]),\n    go.Bar(name=\"Train\", x=test_eda[\"gleason_score\"], y=test_eda[\"image_id\"]),\n])\n\n# Change the bar mode\nfig.update_layout(barmode='group')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:50.270297Z","iopub.execute_input":"2022-01-20T19:24:50.270507Z","iopub.status.idle":"2022-01-20T19:24:52.610257Z","shell.execute_reply.started":"2022-01-20T19:24:50.270485Z","shell.execute_reply":"2022-01-20T19:24:52.609526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:52.61177Z","iopub.execute_input":"2022-01-20T19:24:52.612111Z","iopub.status.idle":"2022-01-20T19:24:52.686143Z","shell.execute_reply.started":"2022-01-20T19:24:52.612072Z","shell.execute_reply":"2022-01-20T19:24:52.685441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = train_eda\nfig = px.pie(df, values='image_id', names='gleason_score', title = 'Training Images')\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:52.687421Z","iopub.execute_input":"2022-01-20T19:24:52.687628Z","iopub.status.idle":"2022-01-20T19:24:52.970266Z","shell.execute_reply.started":"2022-01-20T19:24:52.687605Z","shell.execute_reply":"2022-01-20T19:24:52.969385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = validation_eda\nfig = px.pie(df, values='image_id', names='gleason_score', title = 'Validation Images')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:52.971589Z","iopub.execute_input":"2022-01-20T19:24:52.97183Z","iopub.status.idle":"2022-01-20T19:24:53.257017Z","shell.execute_reply.started":"2022-01-20T19:24:52.971803Z","shell.execute_reply":"2022-01-20T19:24:53.256114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = test_eda\nfig = px.pie(df, values='image_id', names='gleason_score', title = 'Testing Images')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:53.258222Z","iopub.execute_input":"2022-01-20T19:24:53.258431Z","iopub.status.idle":"2022-01-20T19:24:53.728082Z","shell.execute_reply.started":"2022-01-20T19:24:53.258408Z","shell.execute_reply":"2022-01-20T19:24:53.727293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = 'Training Images','Validation Images', 'Testing Images'\nsizes_features = [len(X_train), len(X_validation), len(X_test)]\n# sizes_labels = [len(y_train), len(y_validation)]\n\nfig, ax = plt.subplots(figsize=(30,7))\n\nax.pie(sizes_features, labels=labels, autopct='%1.1f%%',\n          shadow=True, startangle=60)\nax.axis('equal')  # Equal aspect ratio ensures that pie is drawn as a circle\nax.set_title(f\"Distribution of the dataset\\n Total Images - {(len(X) / len(X) * 100)}%: {len(X)}\\n Training images - {(len(X_train) / len(X) * 100)}%: {len(X_train)}\\n Validation images - {(len(X_validation) / len(X) * 100)}%: {len(X_validation)}\\n Testing images - {(len(X_test) / len(X) * 100)}%: {len(X_test)} \\n \"\n                                                                          ,weight=\"bold\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:53.729247Z","iopub.execute_input":"2022-01-20T19:24:53.729454Z","iopub.status.idle":"2022-01-20T19:24:53.880992Z","shell.execute_reply.started":"2022-01-20T19:24:53.729432Z","shell.execute_reply":"2022-01-20T19:24:53.880083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving the datasets\nX_train = pd.DataFrame(X_train, columns=[\"image_id\", \"data_provider\", \"isup_grade\", \"gleason_score\"])\nX_validation = pd.DataFrame(X_validation, columns=[\"image_id\", \"data_provider\", \"isup_grade\", \"gleason_score\"])\nX_test = pd.DataFrame(X_test, columns=[\"image_id\", \"data_provider\", \"isup_grade\", \"gleason_score\"])\n\nX_train.to_csv(\"./training.csv\")\nX_validation.to_csv(\"./validation.csv\")\nX_test.to_csv(\"./testing.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:53.882204Z","iopub.execute_input":"2022-01-20T19:24:53.88241Z","iopub.status.idle":"2022-01-20T19:24:54.095384Z","shell.execute_reply.started":"2022-01-20T19:24:53.882387Z","shell.execute_reply":"2022-01-20T19:24:54.094706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SIZE_IMG = 112\nN = 16\ndef tile(img, mask):\n    result = []\n    shape = img.shape\n    pad0,pad1 = (SIZE_IMG - shape[0]%SIZE_IMG)%SIZE_IMG, (SIZE_IMG - shape[1]%SIZE_IMG)%SIZE_IMG\n    img = np.pad(img, [[pad0//2, pad0-pad0//2], [pad1//2, pad1 - pad1//2],[0,0]],\n                constant_values=255)\n    mask = np.pad(mask,[[pad0//2, pad0-pad0//2], [pad1//2,pad1-pad1//2], [0,0]],\n                constant_values=0)\n    img = img.reshape(img.shape[0]//SIZE_IMG, SIZE_IMG, img.shape[1]//SIZE_IMG,SIZE_IMG, 3)\n    img = img.transpose(0,2,1,3,4).reshape(-1, SIZE_IMG,SIZE_IMG,3)\n    mask = mask.reshape(mask.shape[0]//SIZE_IMG, SIZE_IMG,mask.shape[1]//SIZE_IMG, SIZE_IMG, 3)\n    mask = mask.transpose(0, 2, 1, 3, 4).reshape(-1, SIZE_IMG,SIZE_IMG, 3)\n    if len(img) < N:\n        mask = np.pad(mask, [[0, N-len(img)], [0, 0], [0, 0],[0, 0]], constant_values=0)\n        img = np.pad(img, [[0, N-len(img)],[0, 0],[0, 0], [0, 0]], constant_values=255)\n    idxs = np.argsort(img.reshape(img.shape[0], -1).sum(-1))[: N]\n    img = img[idxs]\n    mask = mask[idxs]\n    for i in range(len(img)):\n        result.append({'img':img[i], 'mask':mask[i], 'idx':i})\n    return result","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:54.097031Z","iopub.execute_input":"2022-01-20T19:24:54.097331Z","iopub.status.idle":"2022-01-20T19:24:54.111619Z","shell.execute_reply.started":"2022-01-20T19:24:54.097297Z","shell.execute_reply":"2022-01-20T19:24:54.110877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import openslide\nimg=openslide.OpenSlide('/kaggle/input/prostate-cancer-grade-assessment/train_images/2fd1c7dc4a0f3a546a59717d8e9d28c3.tiff')\ndisplay(img.get_thumbnail(size=(512,512)))\n","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:54.112779Z","iopub.execute_input":"2022-01-20T19:24:54.113098Z","iopub.status.idle":"2022-01-20T19:24:54.556198Z","shell.execute_reply.started":"2022-01-20T19:24:54.113059Z","shell.execute_reply":"2022-01-20T19:24:54.555408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Image Preview\n","metadata":{}},{"cell_type":"code","source":"train_dataset = pd.read_csv(\"./training.csv\", usecols=[\"image_id\", \"data_provider\", \"isup_grade\", \"gleason_score\"])\n# validation_dataset = pd.read_csv(\"./validation.csv\", usecols=[\"image_id\", \"data_provider\", \"isup_grade\", \"gleason_score\"])\n\nf, ax = plt.subplots(4,4, figsize=(10, 10))\n\n# Mapping to the original dataset\n# ../input/prostate-cancer-grade-assessment/train_images/0005f7aaab2800f6170c399693a96917.tiff\n# ../input/prostate-cancer-grade-assessment/train_label_masks/0005f7aaab2800f6170c399693a96917_mask.tiff\nimg = skimage.io.MultiImage(os.path.join(TRAIN,\"0005f7aaab2800f6170c399693a96917\"+'.tiff'))[1]\nmask = skimage.io.MultiImage(os.path.join(MASKS,\"0005f7aaab2800f6170c399693a96917\"+'_mask.tiff'))[1]\ntiles = tile(img, mask)\nfor t in range(len(tiles)):\n    ax[t//4, t%4].imshow(tiles[t][\"img\"]) # Displaying Image    \n    ax[t//4, t%4].axis('off')      ","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:54.558053Z","iopub.execute_input":"2022-01-20T19:24:54.558383Z","iopub.status.idle":"2022-01-20T19:24:56.244271Z","shell.execute_reply.started":"2022-01-20T19:24:54.558343Z","shell.execute_reply":"2022-01-20T19:24:56.243569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Concatenate Images - 16:1\nThe function ```concat_tile()``` concatenates 16 tiles in one single image, which will be saved later on.","metadata":{}},{"cell_type":"code","source":"id_train = train_dataset[\"image_id\"][163]\nid_validation = validation_dataset[\"image_id\"][163]\nid_test = validation_dataset[\"image_id\"][163]\n# Testing the function\ndef concat_tile(im_list_2d):\n    return cv2.vconcat([cv2.hconcat(im_list_h) for im_list_h in im_list_2d])\n\ndef mosaic(tiles):\n\n    im1 = tiles[0][\"img\"]\n    im2 = tiles[1][\"img\"]\n    im3 = tiles[2][\"img\"]\n    im4 = tiles[3][\"img\"]\n\n    im5 = tiles[4][\"img\"]\n    im6 = tiles[5][\"img\"]\n    im7 = tiles[6][\"img\"]\n    im8 = tiles[7][\"img\"]\n\n    im9 = tiles[8][\"img\"]\n    im10 = tiles[9][\"img\"]\n    im11 = tiles[10][\"img\"]\n    im12 = tiles[11][\"img\"]\n\n    im13 = tiles[12][\"img\"]\n    im14 = tiles[13][\"img\"]\n    im15 = tiles[14][\"img\"]\n    im16 = tiles[15][\"img\"]\n\n    im_tile = concat_tile([[im1, im2, im3, im4],\n                           [im5, im6, im7, im8],\n                           [im9, im10, im11, im12],\n                           [im13, im14, im15, im16]])\n    return im_tile\n\nimg = skimage.io.MultiImage(os.path.join(TRAIN, f\"{id_train}.tiff\"))[1]\nmask = skimage.io.MultiImage(os.path.join(MASKS, f\"{id_train}_mask.tiff\"))[1]\ntiles = tile(img, mask)\n\nmosaic_img = mosaic(tiles)\nplt.title(f\"ID: {id_train}\")\nplt.imshow(mosaic_img)","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:56.245371Z","iopub.execute_input":"2022-01-20T19:24:56.245755Z","iopub.status.idle":"2022-01-20T19:24:56.755457Z","shell.execute_reply.started":"2022-01-20T19:24:56.245717Z","shell.execute_reply":"2022-01-20T19:24:56.753791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generating the Dataset\n\n* Iterate through the train and test dataset\n    * Map the for both the train and the test dataset to the base folder\n    * Zip the 16 subimages\n    * Save the 16 subimages in their correspondant GLEASON_SCORE folder","metadata":{}},{"cell_type":"code","source":"train_IDs = train_dataset[\"image_id\"]\nvalidation_IDs = validation_dataset[\"image_id\"]\ntest_IDs = test_dataset[\"image_id\"]\n\nnot_found_train = []\nnot_found_validation = []\nnot_found_test = []\n\ndef generate_dataset(ids, dataset_type):\n    if dataset_type == \"train\":\n        x_tot,x2_tot = [], []\n        with zipfile.ZipFile(OUT_TRAIN, 'w') as img_out,\\\n         zipfile.ZipFile(OUT_MASKS_TRAIN, 'w') as mask_out:\n            for gleason_score, id in enumerate(tqdm(ids)):\n                try:\n                    img = skimage.io.MultiImage(os.path.join(TRAIN,id+'.tiff'))[1]\n                    mask = skimage.io.MultiImage(os.path.join(MASKS,id+'_mask.tiff'))[1]\n                    tiles = tile(img,mask)                    \n                    img = mosaic(tiles)\n                    \n                    x_tot.append((img/255.0).reshape(-1,3).mean(0))\n                    x2_tot.append(((img/255.0)**2).reshape(-1,3).mean(0))\n                    # If read with PIL RGB turns into BGR\n                    img = cv2.imencode('.png',cv2.cvtColor(img, cv2.COLOR_RGB2BGR))[1]\n                    # Uncomment to classify by ISUP GRADE \n                    # img_out.writestr(f'train/ISUP_GRADE_{train_dataset[\"isup_grade\"][isup_grade]}/{id}_{idx}.png', img)\n                    img_out.writestr(f'train/GLEASON_SCORE_{train_dataset[\"gleason_score\"][gleason_score]}/{id}.png', img)\n                except Exception as e:\n                    not_found_train.append(id)\n        print(f\"INFO: Not images found in train: {len(not_found_train)}\")\n        \n    elif dataset_type == \"valid\": \n        x_tot,x2_tot = [], []\n        with zipfile.ZipFile(OUT_VALIDATION, 'w') as img_out,\\\n         zipfile.ZipFile(OUT_MASKS_VALIDATION, 'w') as mask_out:\n            for gleason_score, id in enumerate(tqdm(ids)):\n                try:\n                    img = skimage.io.MultiImage(os.path.join(TRAIN,id+'.tiff'))[1]\n                    mask = skimage.io.MultiImage(os.path.join(MASKS,id+'_mask.tiff'))[1]\n                    tiles = tile(img,mask)\n                    img = mosaic(tiles)\n                    \n                    x_tot.append((img/255.0).reshape(-1,3).mean(0))\n                    x2_tot.append(((img/255.0)**2).reshape(-1,3).mean(0)) \n                    # If read with PIL RGB turns into BGR\n                    img = cv2.imencode('.png',cv2.cvtColor(img, cv2.COLOR_RGB2BGR))[1]\n                    # Uncomment to classify by ISUP GRADE \n                    # img_out.writestr(f'test/ISUP_GRADE_{train_dataset[\"isup_grade\"][isup_grade]}/{id}_{idx}.png', img)\n                    img_out.writestr(f'validation/GLEASON_SCORE_{validation_dataset[\"gleason_score\"][gleason_score]}/{id}.png', img)\n                except Exception as e:\n                    not_found_validation.append(id)\n\n        print(f\"INFO: Not images found in validation: {len(not_found_validation)}\")\n        \n    elif dataset_type == \"test\":  \n        x_tot,x2_tot = [], []\n        with zipfile.ZipFile(OUT_TEST, 'w') as img_out,\\\n         zipfile.ZipFile(OUT_MASKS_TEST, 'w') as mask_out:\n            for gleason_score, id in enumerate(tqdm(ids)):\n                try:\n                    img = skimage.io.MultiImage(os.path.join(TRAIN,id+'.tiff'))[1]\n                    mask = skimage.io.MultiImage(os.path.join(MASKS,id+'_mask.tiff'))[1]\n                    tiles = tile(img,mask)\n                    img = mosaic(tiles)\n                    \n                    x_tot.append((img/255.0).reshape(-1,3).mean(0))\n                    x2_tot.append(((img/255.0)**2).reshape(-1,3).mean(0)) \n                    # If read with PIL RGB turns into BGR\n                    img = cv2.imencode('.png',cv2.cvtColor(img, cv2.COLOR_RGB2BGR))[1]\n                    # Uncomment to classify by ISUP GRADE \n                    # img_out.writestr(f'test/ISUP_GRADE_{train_dataset[\"isup_grade\"][isup_grade]}/{id}_{idx}.png', img)\n                    img_out.writestr(f'test/GLEASON_SCORE_{test_dataset[\"gleason_score\"][gleason_score]}/{id}.png', img)\n                except Exception as e:\n                    not_found_test.append(id)\n\n        print(f\"INFO: Not images found in test: {len(not_found_test)}\")","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:56.756436Z","iopub.status.idle":"2022-01-20T19:24:56.756752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"generate_dataset(train_IDs, datset_type='train')\ngenerate_dataset(train_IDs, datset_type='valid')\ngenerate_dataset(validation_IDs, datset_type='test')","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:56.757564Z","iopub.status.idle":"2022-01-20T19:24:56.757853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Lost/Corrupted Data\nSome images were not successfully processed for some reason: 80 in the training set, 20 in the testing set.","metadata":{}},{"cell_type":"code","source":"labels = \"Training Images\", \"Validation Images\",\"Testing Images\", \"Loss\"\nsizes_features = [len(X_train), len(X_validation), len(X_test), len(not_found_train) + len(not_found_validation) +len(not_found_test)]\n# sizes_labels = [len(X_train), len(X_validation), 20]\n\nfig, ax = plt.subplots(figsize=(30,7))\n\nax.pie(sizes_features, labels=labels, autopct='%1.1f%%',\n          shadow=True, startangle=60)\nax.axis('equal')  # Equal aspect ratio ensures that pie is drawn as a circle\nax.set_title(f\"Distribution of the dataset\\n\" /\n             f\"Total Images - {(len(X) / len(X) * 100)}%: {len(X)} \\n\" / \n             f\"Training images - {(len(X_train) / len(X) * 100)}%: {len(X_train)} \\n\" /\n             f\"Validation images - {(len(X_validation) / len(X) * 100)}%: {len(X_validation)} \\n\" /\n             f\"Testing images - {(len(X_test) / len(X) * 100)}%: {len(X_test)} \\n \"\n             f\"Loss - : {len(not_found_train) + len(not_found_validation) +len(not_found_test)} / {len(X)} images\", weight=\"bold\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:56.75874Z","iopub.status.idle":"2022-01-20T19:24:56.759041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Removing Lost/Corrupted Data\n\nSince there are 100 images not found (80 images for training and 20 for testing) we want to make sure this does not affect at the end. Given that there is a possibility that the 80 images not found may belong to the classes with the less images provided by the original dataset","metadata":{}},{"cell_type":"code","source":"not_found_train_eda = []\nif not_found_train:\n    for not_found in not_found_train:\n        not_found_train_eda.append(train_dataset[train_dataset[\"image_id\"] == not_found])\n    not_found_train_eda = pd.concat(not_found_train_eda)","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:56.759837Z","iopub.status.idle":"2022-01-20T19:24:56.760149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"not_found_validation_eda = []\nif not_found_validation:\n    for not_found in not_found_validation:\n        not_found_validation_eda.append(validation_dataset[validation_dataset[\"image_id\"] == not_found])\n    not_found_validation_eda = pd.concat(not_found_validation_eda)","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:56.760935Z","iopub.status.idle":"2022-01-20T19:24:56.761287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"not_found_test_eda = []\nif not_found_test:\n    for not_found in not_found_test:\n        not_found_test_eda.append(test_dataset[test_dataset[\"image_id\"] == not_found])\n    not_found_test_eda = pd.concat(not_found_test_eda)","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:56.762239Z","iopub.status.idle":"2022-01-20T19:24:56.762517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if not_found_train_eda.empty:\nnot_found_train_eda = not_found_train_eda.groupby('gleason_score').count()['image_id'].reset_index().sort_values(by='image_id', ascending=False)\nnot_found_train_eda.style.background_gradient(cmap='Greens')","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:56.763314Z","iopub.status.idle":"2022-01-20T19:24:56.763592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if not_found_validation_eda.empty:\nnot_found_validation_eda = not_found_validation_eda.groupby('gleason_score').count()['image_id'].reset_index().sort_values(by='image_id', ascending=False)\nnot_found_validation_eda.style.background_gradient(cmap='Reds')","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:56.764301Z","iopub.status.idle":"2022-01-20T19:24:56.764576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if not not_found_test_eda.empty:\nnot_found_test_eda = not_found_test_eda.groupby('gleason_score').count()['image_id'].reset_index().sort_values(by='image_id', ascending=False)\nnot_found_test_eda.style.background_gradient(cmap='Blues')","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:56.765358Z","iopub.status.idle":"2022-01-20T19:24:56.765648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if not_found_train_eda or not_found_validation_eda or not_found_test_eda:\nfig = go.Figure(data=[\n    go.Bar(name=\"Not found test\", x=not_found_train_eda[\"gleason_score\"], y=not_found_train_eda[\"image_id\"]),\n    go.Bar(name=\"Not found validation\", x=not_found_validation_eda[\"gleason_score\"], y=not_found_validation_eda[\"image_id\"]),\n    go.Bar(name=\"Not found train\", x=not_found_test_eda[\"gleason_score\"], y=not_found_test_eda[\"image_id\"])\n])\n\n# Change the bar mode\nfig.update_layout(barmode='group')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:56.766528Z","iopub.status.idle":"2022-01-20T19:24:56.766828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if not_found_train_eda or not_found_test_eda:\nfig = go.Figure(data=[\n    go.Bar(name=\"Not found train\", x=not_found_train_eda[\"gleason_score\"], y=not_found_train_eda[\"image_id\"]),\n    go.Bar(name=\"Not found validation\", x=not_found_validation_eda[\"gleason_score\"], y=not_found_validation_eda[\"image_id\"]),\n    go.Bar(name=\"Not found test\", x=not_found_test_eda[\"gleason_score\"], y=not_found_test_eda[\"image_id\"]),\n    go.Bar(name=\"Found test\", x=train_eda[\"gleason_score\"], y=train_eda[\"image_id\"]),\n    go.Bar(name=\"Found validation\", x=validation_eda[\"gleason_score\"], y=validation_eda[\"image_id\"]),\n    go.Bar(name=\"Found train\", x=test_eda[\"gleason_score\"], y=test_eda[\"image_id\"])\n])\n\n# Change the bar mode\nfig.update_layout(barmode='group')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:56.76758Z","iopub.status.idle":"2022-01-20T19:24:56.76787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Distribution of the loss images\nIn the next pie charts, it is shown the loss in the images for both the testing and training datasets ","metadata":{}},{"cell_type":"code","source":"# if not_found_train_eda:\ndf = not_found_train_eda\nfig = px.pie(df, values='image_id', names='gleason_score')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:56.768725Z","iopub.status.idle":"2022-01-20T19:24:56.769033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if not_found_validation_eda:\ndf = not_found_validation_eda\nfig = px.pie(df, values='image_id', names='gleason_score')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:56.769783Z","iopub.status.idle":"2022-01-20T19:24:56.770097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if not_found_test_eda:\ndf = not_found_test_eda\nfig = px.pie(df, values='image_id', names='gleason_score')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-20T19:24:56.770968Z","iopub.status.idle":"2022-01-20T19:24:56.771267Z"},"trusted":true},"execution_count":null,"outputs":[]}]}