{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!mkdir ./train ./test","metadata":{"execution":{"iopub.status.busy":"2021-07-21T05:12:51.604488Z","iopub.execute_input":"2021-07-21T05:12:51.604925Z","iopub.status.idle":"2021-07-21T05:12:52.390819Z","shell.execute_reply.started":"2021-07-21T05:12:51.604895Z","shell.execute_reply":"2021-07-21T05:12:52.388705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN = '../input/prostate-cancer-grade-assessment/train_images/'\nMASKS = '../input/prostate-cancer-grade-assessment/train_label_masks/'\nOUT_TRAIN = '/train/train.zip'\nOUT_TEST = '/train/test.zip'\nOUT_MASKS_TRAIN = 'masks_train.zip'\nOUT_MASKS_TEST = 'masks_test.zip'\n\nSIZE_IMG = 128\nN = 16","metadata":{"execution":{"iopub.status.busy":"2021-07-21T05:11:37.262336Z","iopub.execute_input":"2021-07-21T05:11:37.262646Z","iopub.status.idle":"2021-07-21T05:11:37.26769Z","shell.execute_reply.started":"2021-07-21T05:11:37.262618Z","shell.execute_reply":"2021-07-21T05:11:37.26674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tile(img, mask):\n    result = []\n    shape = img.shape\n    pad0,pad1 = (SIZE_IMG - shape[0]%SIZE_IMG)%SIZE_IMG, (SIZE_IMG - shape[1]%SIZE_IMG)%SIZE_IMG\n    img = np.pad(img,[[pad0//2,pad0-pad0//2],[pad1//2,pad1-pad1//2],[0,0]],\n                constant_values=255)\n    mask = np.pad(mask,[[pad0//2,pad0-pad0//2],[pad1//2,pad1-pad1//2],[0,0]],\n                constant_values=0)\n    img = img.reshape(img.shape[0]//SIZE_IMG,SIZE_IMG,img.shape[1]//SIZE_IMG,SIZE_IMG,3)\n    img = img.transpose(0,2,1,3,4).reshape(-1,SIZE_IMG,SIZE_IMG,3)\n    mask = mask.reshape(mask.shape[0]//SIZE_IMG,SIZE_IMG,mask.shape[1]//SIZE_IMG,SIZE_IMG,3)\n    mask = mask.transpose(0,2,1,3,4).reshape(-1,SIZE_IMG,SIZE_IMG,3)\n    if len(img) < N:\n        mask = np.pad(mask,[[0,N-len(img)],[0,0],[0,0],[0,0]],constant_values=0)\n        img = np.pad(img,[[0,N-len(img)],[0,0],[0,0],[0,0]],constant_values=255)\n    idxs = np.argsort(img.reshape(img.shape[0],-1).sum(-1))[:N]\n    img = img[idxs]\n    mask = mask[idxs]\n    for i in range(len(img)):\n        result.append({'img':img[i], 'mask':mask[i], 'idx':i})\n    return result","metadata":{"execution":{"iopub.status.busy":"2021-07-21T00:14:01.220318Z","iopub.execute_input":"2021-07-21T00:14:01.220735Z","iopub.status.idle":"2021-07-21T00:14:01.236887Z","shell.execute_reply.started":"2021-07-21T00:14:01.220705Z","shell.execute_reply":"2021-07-21T00:14:01.236089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_tot,x2_tot = [],[]\nids = [id[:-10] for id in os.listdir(MASKS)]\nwith zipfile.ZipFile(OUT_TRAIN, 'w') as img_out,\\\n zipfile.ZipFile(OUT_MASKS, 'w') as mask_out:\n    for id in tqdm(ids):\n        img = skimage.io.MultiImage(os.path.join(TRAIN,id+'.tiff'))[-1]\n        mask = skimage.io.MultiImage(os.path.join(MASKS,id+'_mask.tiff'))[-1]\n        tiles = tile(img,mask)\n        for t in tiles:\n            img,mask,idx = t['img'],t['mask'],t['idx']\n            x_tot.append((img/255.0).reshape(-1,3).mean(0))\n            x2_tot.append(((img/255.0)**2).reshape(-1,3).mean(0)) \n            #if read with PIL RGB turns into BGR\n            img = cv2.imencode('.png',cv2.cvtColor(img, cv2.COLOR_RGB2BGR))[1]\n            img_out.writestr(f'{id}_{idx}.png', img)\n            mask = cv2.imencode('.png',mask[:,:,0])[1]\n            mask_out.writestr(f'{id}_{idx}.png', mask)","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2021-07-20T22:46:48.610768Z","iopub.execute_input":"2021-07-20T22:46:48.611115Z","iopub.status.idle":"2021-07-20T22:46:48.805924Z","shell.execute_reply.started":"2021-07-20T22:46:48.611074Z","shell.execute_reply":"2021-07-20T22:46:48.804117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#image stats\nimg_avr =  np.array(x_tot).mean(0)\nimg_std =  np.sqrt(np.array(x2_tot).mean(0) - img_avr**2)\nprint('mean:',img_avr, ', std:', np.sqrt(img_std))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = skimage.io.MultiImage(os.path.join(TRAIN,\"003d4dd6bd61221ebc0bfb9350db333f\"+'.tiff'))[-1]\nmask = skimage.io.MultiImage(os.path.join(MASKS,\"003d4dd6bd61221ebc0bfb9350db333f\"+'_mask.tiff'))[-1]\n\ntile_ = tile(img, mask)","metadata":{"execution":{"iopub.status.busy":"2021-07-21T01:15:43.319323Z","iopub.execute_input":"2021-07-21T01:15:43.319661Z","iopub.status.idle":"2021-07-21T01:15:43.393938Z","shell.execute_reply.started":"2021-07-21T01:15:43.319614Z","shell.execute_reply":"2021-07-21T01:15:43.393118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nprint(f\"Number of \")\nprint(len(tile_))\n# print(tile_)\nplt.imshow(tile_[0][\"img\"])","metadata":{"execution":{"iopub.status.busy":"2021-07-21T01:15:45.813138Z","iopub.execute_input":"2021-07-21T01:15:45.813448Z","iopub.status.idle":"2021-07-21T01:15:46.006159Z","shell.execute_reply.started":"2021-07-21T01:15:45.813419Z","shell.execute_reply":"2021-07-21T01:15:46.005393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2021-07-21T01:05:43.887819Z","iopub.execute_input":"2021-07-21T01:05:43.888076Z","iopub.status.idle":"2021-07-21T01:05:44.316682Z","shell.execute_reply.started":"2021-07-21T01:05:43.888051Z","shell.execute_reply":"2021-07-21T01:05:44.316069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Basics / Data manipulation\nimport numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nimport zipfile\nimport os\n\n# Visualization\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport openslide # Delete in the future\nimport cv2\nimport skimage.io\n\n# ML\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:54:41.873093Z","iopub.execute_input":"2021-07-22T04:54:41.873504Z","iopub.status.idle":"2021-07-22T04:54:45.485686Z","shell.execute_reply.started":"2021-07-22T04:54:41.873453Z","shell.execute_reply":"2021-07-22T04:54:45.484172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN = '../input/prostate-cancer-grade-assessment/train_images/'\nMASKS = '../input/prostate-cancer-grade-assessment/train_label_masks/'\nOUT_TRAIN = 'train.zip'\nOUT_TEST = 'test.zip'\nOUT_MASKS_TRAIN = 'masks_train.zip'\nOUT_MASKS_TEST = 'masks_test.zip'\n\nBASE_FOLDER = \"/kaggle/input/prostate-cancer-grade-assessment/\"\n!ls {BASE_FOLDER}","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:54:46.493759Z","iopub.execute_input":"2021-07-22T04:54:46.494158Z","iopub.status.idle":"2021-07-22T04:54:47.257441Z","shell.execute_reply.started":"2021-07-22T04:54:46.494116Z","shell.execute_reply":"2021-07-22T04:54:47.255969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(BASE_FOLDER+\"train.csv\")\ntest = pd.read_csv(BASE_FOLDER+\"test.csv\")\nsub = pd.read_csv(BASE_FOLDER+\"sample_submission.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:54:49.257451Z","iopub.execute_input":"2021-07-22T04:54:49.257906Z","iopub.status.idle":"2021-07-22T04:54:49.323566Z","shell.execute_reply.started":"2021-07-22T04:54:49.257863Z","shell.execute_reply":"2021-07-22T04:54:49.322259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train['gleason_score'] == 'negative']","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:54:49.691211Z","iopub.execute_input":"2021-07-22T04:54:49.69161Z","iopub.status.idle":"2021-07-22T04:54:49.718271Z","shell.execute_reply.started":"2021-07-22T04:54:49.691576Z","shell.execute_reply":"2021-07-22T04:54:49.716985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop([7273],inplace=True)\ntrain['gleason_score'] = train['gleason_score'].apply(lambda x: \"0+0\" if x == \"negative\" else x)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:54:50.066979Z","iopub.execute_input":"2021-07-22T04:54:50.067317Z","iopub.status.idle":"2021-07-22T04:54:50.080902Z","shell.execute_reply.started":"2021-07-22T04:54:50.067285Z","shell.execute_reply":"2021-07-22T04:54:50.079557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = np.array(train) # Converting the DataFrame to an array to take the column\nlabels = data[:, 3]\nlabels","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:54:50.780052Z","iopub.execute_input":"2021-07-22T04:54:50.780465Z","iopub.status.idle":"2021-07-22T04:54:50.79337Z","shell.execute_reply.started":"2021-07-22T04:54:50.78043Z","shell.execute_reply":"2021-07-22T04:54:50.791357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = data[:, 0:3]\nfeatures","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:54:51.306043Z","iopub.execute_input":"2021-07-22T04:54:51.306435Z","iopub.status.idle":"2021-07-22T04:54:51.316656Z","shell.execute_reply.started":"2021-07-22T04:54:51.3064Z","shell.execute_reply":"2021-07-22T04:54:51.31503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = features\ny = labels","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:54:51.745439Z","iopub.execute_input":"2021-07-22T04:54:51.746111Z","iopub.status.idle":"2021-07-22T04:54:51.755035Z","shell.execute_reply.started":"2021-07-22T04:54:51.746071Z","shell.execute_reply":"2021-07-22T04:54:51.753024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:54:52.818431Z","iopub.execute_input":"2021-07-22T04:54:52.818798Z","iopub.status.idle":"2021-07-22T04:54:52.826643Z","shell.execute_reply.started":"2021-07-22T04:54:52.818762Z","shell.execute_reply":"2021-07-22T04:54:52.825134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.20, random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:54:53.185139Z","iopub.execute_input":"2021-07-22T04:54:53.185501Z","iopub.status.idle":"2021-07-22T04:54:53.193547Z","shell.execute_reply.started":"2021-07-22T04:54:53.185468Z","shell.execute_reply":"2021-07-22T04:54:53.192873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"-Features-\")\nprint(f\"{(len(X) / len(X) * 100)}%: {len(X)}\")\nprint(f\"{(len(X_train) / len(X) * 100)}%: {len(X_train)}\")\nprint(f\"{(len(X_test) / len(X) * 100)}%: {len(X_test)}\")\n\nprint(\"-Labels-\")\nprint(f\"{(len(y) / len(y) * 100)}%: {len(y)}\")\nprint(f\"{(len(y_train) / len(y) * 100)}%: {len(y_train)}\")\nprint(f\"{(len(y_test) / len(y) * 100)}%: {len(y_test)}\")\n\n\nlabels = 'Training Images', 'Testing Images'\nsizes_features = [len(X_train), len(X_test)]\nsizes_labels = [len(y_train), len(y_test)]\n\nfig, ax = plt.subplots(1, 2, figsize=(20,5))\n\nax[0].pie(sizes_features, labels=labels, autopct='%1.1f%%',\n          shadow=True, startangle=60)\nax[0].axis('equal')  # Equal aspect ratio ensures that pie is drawn as a circle\nax[0].set_title(f\"Features\")\n\nax[1].pie(sizes_labels, labels=labels, autopct='%1.1f%%',\n          shadow=True, startangle=60)\nax[1].axis('equal')  # Equal aspect ratio ensures that pie is drawn as a circle\nax[1].set_title(f\"Labels\")\n\nfig.tight_layout()\nfig.suptitle(\"Distribution of the dataset\", weight=\"bold\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:54:53.77159Z","iopub.execute_input":"2021-07-22T04:54:53.772121Z","iopub.status.idle":"2021-07-22T04:54:54.010356Z","shell.execute_reply.started":"2021-07-22T04:54:53.772083Z","shell.execute_reply":"2021-07-22T04:54:54.009109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = pd.DataFrame(X_train, columns=[\"image_id\", \"data_provider\", \"isup_grade\"])\nX_test = pd.DataFrame(X_test, columns=[\"image_id\", \"data_provider\", \"isup_grade\"])\n\nX_train.to_csv(\"./training.csv\")\nX_test.to_csv(\"./testing.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:54:55.065805Z","iopub.execute_input":"2021-07-22T04:54:55.066169Z","iopub.status.idle":"2021-07-22T04:54:55.16649Z","shell.execute_reply.started":"2021-07-22T04:54:55.066138Z","shell.execute_reply":"2021-07-22T04:54:55.165302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SIZE_IMG = 112\nN = 16\ndef tile(img, mask):\n    result = []\n    shape = img.shape\n    pad0,pad1 = (SIZE_IMG - shape[0]%SIZE_IMG)%SIZE_IMG, (SIZE_IMG - shape[1]%SIZE_IMG)%SIZE_IMG\n    img = np.pad(img,[[pad0//2,pad0-pad0//2],[pad1//2,pad1-pad1//2],[0,0]],\n                constant_values=255)\n    mask = np.pad(mask,[[pad0//2,pad0-pad0//2],[pad1//2,pad1-pad1//2],[0,0]],\n                constant_values=0)\n    img = img.reshape(img.shape[0]//SIZE_IMG,SIZE_IMG,img.shape[1]//SIZE_IMG,SIZE_IMG,3)\n    img = img.transpose(0,2,1,3,4).reshape(-1,SIZE_IMG,SIZE_IMG,3)\n    mask = mask.reshape(mask.shape[0]//SIZE_IMG,SIZE_IMG,mask.shape[1]//SIZE_IMG,SIZE_IMG,3)\n    mask = mask.transpose(0,2,1,3,4).reshape(-1,SIZE_IMG,SIZE_IMG,3)\n    if len(img) < N:\n        mask = np.pad(mask,[[0,N-len(img)],[0,0],[0,0],[0,0]],constant_values=0)\n        img = np.pad(img,[[0,N-len(img)],[0,0],[0,0],[0,0]],constant_values=255)\n    idxs = np.argsort(img.reshape(img.shape[0],-1).sum(-1))[:N]\n    img = img[idxs]\n    mask = mask[idxs]\n    for i in range(len(img)):\n        result.append({'img':img[i], 'mask':mask[i], 'idx':i})\n    return result","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:54:55.938916Z","iopub.execute_input":"2021-07-22T04:54:55.939502Z","iopub.status.idle":"2021-07-22T04:54:55.957448Z","shell.execute_reply.started":"2021-07-22T04:54:55.939455Z","shell.execute_reply":"2021-07-22T04:54:55.95653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = pd.read_csv(\"./training.csv\", usecols=[\"image_id\", \"data_provider\", \"isup_grade\"])\ntest_dataset = pd.read_csv(\"./testing.csv\", usecols=[\"image_id\", \"data_provider\", \"isup_grade\"])\n\nf, ax = plt.subplots(4,4, figsize=(10, 10))\n\n# Mapping to the original dataset\nimg = skimage.io.MultiImage(os.path.join(TRAIN,\"5801d2195cdcd8d336e8fc097c51a788\"+'.tiff'))[-1]\nmask = skimage.io.MultiImage(os.path.join(MASKS,\"5801d2195cdcd8d336e8fc097c51a788\"+'_mask.tiff'))[-1]\ntiles = tile(img, mask)\nfor t in range(len(tiles)):\n    ax[t//4, t%4].imshow(tiles[t][\"img\"]) # Displaying Image    \n    ax[t//4, t%4].axis('off')      ","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:54:57.487122Z","iopub.execute_input":"2021-07-22T04:54:57.487705Z","iopub.status.idle":"2021-07-22T04:54:58.862085Z","shell.execute_reply.started":"2021-07-22T04:54:57.487664Z","shell.execute_reply":"2021-07-22T04:54:58.861026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_IDs = train_dataset[\"image_id\"]\ntest_IDs = test_dataset[\"image_id\"]\n\nnot_found_train = []\nnot_found_test = []\n\ndef generate_dataset(ids, train=True):\n    if train:\n        x_tot,x2_tot = [], []\n        with zipfile.ZipFile(OUT_TRAIN, 'w') as img_out,\\\n         zipfile.ZipFile(OUT_MASKS_TRAIN, 'w') as mask_out:\n            for isup_grade, id in enumerate(tqdm(ids)):\n                try:\n                    img = skimage.io.MultiImage(os.path.join(TRAIN,id+'.tiff'))[-1]\n                    mask = skimage.io.MultiImage(os.path.join(MASKS,id+'_mask.tiff'))[-1]\n                    tiles = tile(img,mask)\n                    for t in tiles:\n                        img,mask,idx = t['img'],t['mask'],t['idx']\n                        x_tot.append((img/255.0).reshape(-1,3).mean(0))\n                        x2_tot.append(((img/255.0)**2).reshape(-1,3).mean(0)) \n                        #if read with PIL RGB turns into BGR\n                        img = cv2.imencode('.png',cv2.cvtColor(img, cv2.COLOR_RGB2BGR))[1]\n                        img_out.writestr(f'test/ISUP_GRADE_{train_dataset[\"isup_grade\"][isup_grade]}/{id}_{idx}.png', img)\n                        mask = cv2.imencode('.png',mask[:,:,0])[1]\n                        mask_out.writestr(f'test/ISUP_GRADE_{train_dataset[\"isup_grade\"][isup_grade]}/{id}_{idx}.png', mask)\n                except Exception as e:\n                    not_found_train.append(id)\n        print(f\"Not images found in train: {len(not_found_train)}\")\n    if not train: \n        x_tot,x2_tot = [], []\n        with zipfile.ZipFile(OUT_TEST, 'w') as img_out,\\\n         zipfile.ZipFile(OUT_MASKS_TEST, 'w') as mask_out:\n            for isup_grade, id in enumerate(tqdm(ids)):\n                try:\n                    img = skimage.io.MultiImage(os.path.join(TRAIN,id+'.tiff'))[-1]\n                    mask = skimage.io.MultiImage(os.path.join(MASKS,id+'_mask.tiff'))[-1]\n                    tiles = tile(img,mask)\n                except Exception as e:\n                    not_found_test.append(id)\n                for t in tiles:\n                    img,mask,idx = t['img'],t['mask'],t['idx']\n                    x_tot.append((img/255.0).reshape(-1,3).mean(0))\n                    x2_tot.append(((img/255.0)**2).reshape(-1,3).mean(0)) \n                    #if read with PIL RGB turns into BGR\n                    img = cv2.imencode('.png',cv2.cvtColor(img, cv2.COLOR_RGB2BGR))[1]\n                    img_out.writestr(f'train/ISUP_GRADE_{train_dataset[\"isup_grade\"][isup_grade]}/{id}_{idx}.png', img)\n                    mask = cv2.imencode('.png',mask[:,:,0])[1]\n                    mask_out.writestr(f'train/ISUP_GRADE_{train_dataset[\"isup_grade\"][isup_grade]}/{id}_{idx}.png', mask)\n        print(f\"Not images found in test: {len(not_found_test)}\")","metadata":{"execution":{"iopub.status.busy":"2021-07-22T05:38:35.534998Z","iopub.execute_input":"2021-07-22T05:38:35.535344Z","iopub.status.idle":"2021-07-22T05:38:35.565279Z","shell.execute_reply.started":"2021-07-22T05:38:35.535312Z","shell.execute_reply":"2021-07-22T05:38:35.564014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"generate_dataset(train_IDs, train=True)\ngenerate_dataset(test_IDs, train=False)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T05:38:36.059797Z","iopub.execute_input":"2021-07-22T05:38:36.060392Z","iopub.status.idle":"2021-07-22T06:02:25.205904Z","shell.execute_reply.started":"2021-07-22T05:38:36.060339Z","shell.execute_reply":"2021-07-22T06:02:25.20429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"-Features-\")\nprint(f\"{(len(X) / len(X) * 100)}%: {len(X)}\")\nprint(f\"{(len(X_train) / len(X) * 100)}%: {len(X_train)}\")\nprint(f\"{(len(X_test) / len(X) * 100)}%: {len(X_test)}\")\n\nprint(\"-Labels-\")\nprint(f\"{(len(y) / len(y) * 100)}%: {len(y)}\")\nprint(f\"{(len(y_train) / len(y) * 100)}%: {len(y_train)}\")\nprint(f\"{(len(y_test) / len(y) * 100)}%: {len(y_test)}\")\n\n\nlabels = \"Training Images\", \"Testing Images\", \"Loss\"\nsizes_features = [len(X_train), len(X_test), len(not_found_train)]\nsizes_labels = [len(y_train), len(y_test), len(not_found_test)]\n\nfig, ax = plt.subplots(1, 2, figsize=(20,5))\n\nax[0].pie(sizes_features, labels=labels, autopct='%1.1f%%',\n          shadow=True, startangle=60)\nax[0].axis('equal')  # Equal aspect ratio ensures that pie is drawn as a circle\nax[0].set_title(f\"Features\")\n\nax[1].pie(sizes_labels, labels=labels, autopct='%1.1f%%',\n          shadow=True, startangle=60)\nax[1].axis('equal')  # Equal aspect ratio ensures that pie is drawn as a circle\nax[1].set_title(f\"Labels\")\n\nfig.tight_layout()\nfig.suptitle(\"Distribution of the dataset\", weight=\"bold\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-22T06:12:29.32471Z","iopub.execute_input":"2021-07-22T06:12:29.325238Z","iopub.status.idle":"2021-07-22T06:12:29.585301Z","shell.execute_reply.started":"2021-07-22T06:12:29.325195Z","shell.execute_reply":"2021-07-22T06:12:29.584445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2021-07-22T02:48:48.577932Z","iopub.execute_input":"2021-07-22T02:48:48.578345Z","iopub.status.idle":"2021-07-22T02:48:48.584358Z","shell.execute_reply.started":"2021-07-22T02:48:48.578302Z","shell.execute_reply":"2021-07-22T02:48:48.58292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()  # TPU detection\n    print(\"Running on TPU \", tpu.cluster_spec().as_dict()[\"worker\"])\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept ValueError:\n    print(\"Not connected to a TPU runtime. Using CPU/GPU strategy\")\n    strategy = tf.distribute.MirroredStrategy()","metadata":{"execution":{"iopub.status.busy":"2021-07-21T23:34:33.657709Z","iopub.execute_input":"2021-07-21T23:34:33.658282Z","iopub.status.idle":"2021-07-21T23:34:33.702104Z","shell.execute_reply.started":"2021-07-21T23:34:33.658233Z","shell.execute_reply":"2021-07-21T23:34:33.700902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:22:00.189686Z","iopub.execute_input":"2021-07-22T04:22:00.190343Z","iopub.status.idle":"2021-07-22T04:22:00.194805Z","shell.execute_reply.started":"2021-07-22T04:22:00.190303Z","shell.execute_reply":"2021-07-22T04:22:00.19394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" # Creating an object that will contain all the changes that will\n # be performed randomly to the images to help the training\n image_gen = ImageDataGenerator(rotation_range=30,\n                                width_shift_range=0.1,\n                                height_shift_range=0.1,\n                                rescale=1/255,\n                                shear_range=0.2,\n                                zoom_range=0.2,\n                                horizontal_flip=True,\n                                fill_mode=\"nearest\"\n                                )","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:22:08.944533Z","iopub.execute_input":"2021-07-22T04:22:08.945172Z","iopub.status.idle":"2021-07-22T04:22:08.95106Z","shell.execute_reply.started":"2021-07-22T04:22:08.945134Z","shell.execute_reply":"2021-07-22T04:22:08.94997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clsmkl","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_gen.flow_from_directory(\"/content/CATS_DOGS/train\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Setting up EfficientNet","metadata":{}},{"cell_type":"code","source":"\nfrom tensorflow.keras.applications import * #Efficient Net included here\nfrom tensorflow.keras import models\nfrom tensorflow.keras import layers\nfrom keras.preprocessing.image import ImageDataGenerator\nimport os\nimport shutil\nimport pandas as pd\nfrom sklearn import model_selection\nfrom tensorflow.keras import optimizers\n#Use this to check if the GPU is configured correctly\nfrom tensorflow.python.client import device_lib\nprint(device_lib.list_local_devices())\n\n\nfrom tensorflow.keras.applications import EfficientNetB0","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:16:22.176917Z","iopub.execute_input":"2021-07-22T04:16:22.177536Z","iopub.status.idle":"2021-07-22T04:16:22.195481Z","shell.execute_reply.started":"2021-07-22T04:16:22.177495Z","shell.execute_reply":"2021-07-22T04:16:22.194066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conv_base = EfficientNetB0(weights=\"imagenet\", include_top=False, input_shape=(224, 224, 3))","metadata":{"execution":{"iopub.status.busy":"2021-07-22T03:59:17.641892Z","iopub.execute_input":"2021-07-22T03:59:17.642259Z","iopub.status.idle":"2021-07-22T03:59:21.274595Z","shell.execute_reply.started":"2021-07-22T03:59:17.642226Z","shell.execute_reply":"2021-07-22T03:59:21.273327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = models.Sequential()\nmodel.add(conv_base)\nmodel.add(layers.GlobalMaxPooling2D(name=\"gap\"))\n#avoid overfitting\nmodel.add(layers.Dropout(rate=0.2))\n# Set NUMBER_OF_CLASSES to the number of your final predictions.\nmodel.add(layers.Dense(5, activation=\"softmax\", name=\"fc_out\"))\nconv_base.trainable = False","metadata":{"execution":{"iopub.status.busy":"2021-07-22T04:20:23.58638Z","iopub.execute_input":"2021-07-22T04:20:23.58687Z","iopub.status.idle":"2021-07-22T04:20:24.348948Z","shell.execute_reply.started":"2021-07-22T04:20:23.586836Z","shell.execute_reply":"2021-07-22T04:20:24.348144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}