{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-10T11:08:23.045845Z","iopub.execute_input":"2023-06-10T11:08:23.046349Z","iopub.status.idle":"2023-06-10T11:08:44.487991Z","shell.execute_reply.started":"2023-06-10T11:08:23.046319Z","shell.execute_reply":"2023-06-10T11:08:44.487208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install numpy==1.22.0\n","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:08:44.497672Z","iopub.execute_input":"2023-06-10T11:08:44.497975Z","iopub.status.idle":"2023-06-10T11:09:02.273538Z","shell.execute_reply.started":"2023-06-10T11:08:44.497944Z","shell.execute_reply":"2023-06-10T11:09:02.272372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEBUG = True\n!pip install warmup_scheduler\n!pip install efficientnet_pytorch\n\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib \nsns.set_theme(font_scale=1.2, rc={\"figure.figsize\": (8, 6)})\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport seaborn as sns\nimport os\nimport openslide\nimport skimage.io\nfrom IPython.display import Image, display\nimport plotly.graph_objs as go\nimport PIL\nimport PIL.Image\nimport sys\nimport time\nimport cv2\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import DataLoader, Dataset\nfrom torch.utils.data.sampler import SubsetRandomSampler, RandomSampler, SequentialSampler\nfrom warmup_scheduler import GradualWarmupScheduler\nfrom efficientnet_pytorch import model as enet\nimport albumentations\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom tqdm import tqdm_notebook as tqdm\nsys.path = [\n    '../input/efficientnet-pytorch/EfficientNet-PyTorch/EfficientNet-PyTorch-master',\n] + sys.path","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:09:22.471727Z","iopub.execute_input":"2023-06-10T11:09:22.472142Z","iopub.status.idle":"2023-06-10T11:10:00.418488Z","shell.execute_reply.started":"2023-06-10T11:09:22.472106Z","shell.execute_reply":"2023-06-10T11:10:00.417484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install git+https://github.com/ildoonet/pytorch-gradual-warmup-lr.git","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:10:26.656167Z","iopub.execute_input":"2023-06-10T11:10:26.656563Z","iopub.status.idle":"2023-06-10T11:10:38.92722Z","shell.execute_reply.started":"2023-06-10T11:10:26.656526Z","shell.execute_reply":"2023-06-10T11:10:38.925996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images = ('/kaggle/input/prostate-cancer-grade-assessment/train_images')\ntrain_label_masks = ('/kaggle/input/prostate-cancer-grade-assessment/train_label_masks')\ntrain = pd.read_csv('/kaggle/input/prostate-cancer-grade-assessment/train.csv').set_index('image_id')\ntest = pd.read_csv('/kaggle/input/prostate-cancer-grade-assessment/test.csv')\nsubmission = pd.read_csv('/kaggle/input/prostate-cancer-grade-assessment/sample_submission.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:13:31.687269Z","iopub.execute_input":"2023-06-10T11:13:31.688071Z","iopub.status.idle":"2023-06-10T11:13:31.746825Z","shell.execute_reply.started":"2023-06-10T11:13:31.688014Z","shell.execute_reply":"2023-06-10T11:13:31.745935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#data visualization\ndisplay(train.head())\nprint(\"Shape of training data :\", train.shape)\nprint(\"unique data provider :\", len(train.data_provider.unique()))\nprint(\"unique isup_grade(target) :\", len(train.isup_grade.unique()))\nprint(\"unique gleason_score :\", len(train.gleason_score.unique()))","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:13:38.613968Z","iopub.execute_input":"2023-06-10T11:13:38.614956Z","iopub.status.idle":"2023-06-10T11:13:38.641832Z","shell.execute_reply.started":"2023-06-10T11:13:38.614921Z","shell.execute_reply":"2023-06-10T11:13:38.640915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isna().sum() #detects missing values","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:13:56.080702Z","iopub.execute_input":"2023-06-10T11:13:56.081077Z","iopub.status.idle":"2023-06-10T11:13:56.096737Z","shell.execute_reply.started":"2023-06-10T11:13:56.08103Z","shell.execute_reply":"2023-06-10T11:13:56.095623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#isup_grade: The target variable. The severity of the cancer on a 0-5 scale.\n#gleason_score: Train only. An alternate cancer severity rating system with more levels than the ISUP scale. For details on how the gleason and ISUP systems compare, see the Additional Resources tab.","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:14:00.995788Z","iopub.execute_input":"2023-06-10T11:14:00.996426Z","iopub.status.idle":"2023-06-10T11:14:01.001201Z","shell.execute_reply.started":"2023-06-10T11:14:00.996383Z","shell.execute_reply":"2023-06-10T11:14:01.00013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(test.head())\nprint(\"Shape of training data :\", test.shape)\nprint(\"unique data provider :\", len(test.data_provider.unique()))","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:14:04.983882Z","iopub.execute_input":"2023-06-10T11:14:04.984294Z","iopub.status.idle":"2023-06-10T11:14:04.995713Z","shell.execute_reply.started":"2023-06-10T11:14:04.98426Z","shell.execute_reply":"2023-06-10T11:14:04.99469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#explanatory data analysis\ndef plot_count(df, feature, title, size=5):\n    f, ax = plt.subplots(1,1, figsize=(4*size,3*size))\n    total = float(len(df))\n    sns.countplot(data = train, x = feature)\n    plt.title(title)\n    for p in ax.patches:\n        height = p.get_height()\n        ax.text(p.get_x()+p.get_width()/2.,\n                height + 3,\n                '{:1.2f}%'.format(100*height/total),\n                ha=\"center\") \n\n    plt.show()\n          \nplot_count(df=train, feature=\"data_provider\", title = \"data_provider count and %age plot\")","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:14:07.29308Z","iopub.execute_input":"2023-06-10T11:14:07.293687Z","iopub.status.idle":"2023-06-10T11:14:07.674004Z","shell.execute_reply.started":"2023-06-10T11:14:07.293653Z","shell.execute_reply":"2023-06-10T11:14:07.673118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_count(df=train, feature='isup_grade', title = 'isup_grade count and %age plot') #check for ISUP","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:14:10.56231Z","iopub.execute_input":"2023-06-10T11:14:10.562673Z","iopub.status.idle":"2023-06-10T11:14:10.969901Z","shell.execute_reply.started":"2023-06-10T11:14:10.562643Z","shell.execute_reply":"2023-06-10T11:14:10.969038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_count(df=train, feature='gleason_score', title = 'gleason_score count and %age plot', size=3) #check for Gleason score distribution","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:14:13.661589Z","iopub.execute_input":"2023-06-10T11:14:13.661966Z","iopub.status.idle":"2023-06-10T11:14:14.052955Z","shell.execute_reply.started":"2023-06-10T11:14:13.661936Z","shell.execute_reply":"2023-06-10T11:14:14.051985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_relative_distribution(df, feature, hue, title='', size=2):\n    f, ax = plt.subplots(1,1, figsize=(4*size,3*size))\n    total = float(len(df))\n    sns.countplot(x=feature, hue=hue, data=df, palette='Set2')\n    plt.title(title)\n    for p in ax.patches:\n        height = p.get_height()\n        ax.text(p.get_x()+p.get_width()/2.,\n                height + 3,\n                '{:1.2f}%'.format(100*height/total),\n                ha=\"center\") \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:14:16.497299Z","iopub.execute_input":"2023-06-10T11:14:16.497672Z","iopub.status.idle":"2023-06-10T11:14:16.50441Z","shell.execute_reply.started":"2023-06-10T11:14:16.497642Z","shell.execute_reply":"2023-06-10T11:14:16.503359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_relative_distribution(df=train, feature='isup_grade', hue='gleason_score', title = 'relative count plot of isup_grade with gleason_score', size=5)\n#All exams with ISUP grade = 0 have Gleason score 0+0 or negative.\n#All exams with ISUP grade = 1 have Gleason score 3+3.\n#All exams with ISUP grade = 2 have Gleason score 3+4.\n#All exams with ISUP grade = 3 have Gleason score 4+3.\n#All exams with ISUP grade = 4 have Gleason score 4+4 (majority), 3+5 or 5+3.\n#All exams with ISUP grade = 5 have Gleason score 4+5 (majority), 5+4 or 5+5.","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:14:21.498203Z","iopub.execute_input":"2023-06-10T11:14:21.498567Z","iopub.status.idle":"2023-06-10T11:14:22.602913Z","shell.execute_reply.started":"2023-06-10T11:14:21.498539Z","shell.execute_reply":"2023-06-10T11:14:22.602026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_images(image_ids): \n    f, ax = plt.subplots(5, 3, figsize=(18, 22))\n    for i, image_id in enumerate(image_ids):\n        image = openslide.OpenSlide(os.path.join(train_images, f'{image_id}.tiff'))\n        spacing = 1 / (float(image.properties['tiff.XResolution']) / 10000)\n        patch = image.read_region((1780, 1950), 0, (256, 256))\n        ax[i // 3, i % 3].imshow(patch) \n        image.close()       \n        ax[i // 3, i % 3].axis('off')\n        \n        data_provider = train.loc[image_id, 'data_provider']\n        isup_grade = train.loc[image_id, 'isup_grade']\n        gleason_score = train.loc[image_id, 'gleason_score']\n        ax[i // 3, i % 3].set_title(f\"ID: {image_id}\\nSource: {data_provider} ISUP: {isup_grade} Gleason: {gleason_score}\")\n\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:14:26.421898Z","iopub.execute_input":"2023-06-10T11:14:26.422605Z","iopub.status.idle":"2023-06-10T11:14:26.431741Z","shell.execute_reply.started":"2023-06-10T11:14:26.422561Z","shell.execute_reply":"2023-06-10T11:14:26.430702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images = [\n    '07a7ef0ba3bb0d6564a73f4f3e1c2293',\n    '037504061b9fba71ef6e24c48c6df44d',\n    '035b1edd3d1aeeffc77ce5d248a01a53',\n    '059cbf902c5e42972587c8d17d49efed',\n    '06a0cbd8fd6320ef1aa6f19342af2e68',\n    '06eda4a6faca84e84a781fee2d5f47e1',\n    '0a4b7a7499ed55c71033cefb0765e93d',\n    '0838c82917cd9af681df249264d2769c',\n    '046b35ae95374bfb48cdca8d7c83233f',\n    '074c3e01525681a275a42282cd21cbde',\n    '05abe25c883d508ecc15b6e857e59f32',\n    '05f4e9415af9fdabc19109c980daf5ad',\n    '060121a06476ef401d8a21d6567dee6d',\n    '068b0e3be4c35ea983f77accf8351cc8',\n    '08f055372c7b8a7e1df97c6586542ac8'\n]\n\ndisplay_images(images)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:14:29.604889Z","iopub.execute_input":"2023-06-10T11:14:29.605709Z","iopub.status.idle":"2023-06-10T11:14:33.105327Z","shell.execute_reply.started":"2023-06-10T11:14:29.605675Z","shell.execute_reply":"2023-06-10T11:14:33.104117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_masks(image_ids): \n    f, ax = plt.subplots(5, 3, figsize=(18, 22))\n    for i, image_id in enumerate(image_ids):\n        mask = openslide.OpenSlide(os.path.join(train_label_masks, f'{image_id}_mask.tiff'))\n        mask_data = mask.read_region((0, 0), mask.level_count - 1, mask.level_dimensions[-1])\n        cmap = matplotlib.colors.ListedColormap(['black', 'gray', 'green', 'yellow', 'orange', 'red'])\n\n        ax[i // 3, i % 3].imshow(np.asarray(mask_data)[:, :, 0], cmap=cmap, interpolation='nearest', vmin=0, vmax=5) \n        mask.close()       \n        ax[i // 3, i % 3].axis('off')\n        \n        data_provider = train.loc[image_id, 'data_provider']\n        isup_grade = train.loc[image_id, 'isup_grade']\n        gleason_score = train.loc[image_id, 'gleason_score']\n        ax[i // 3, i % 3].set_title(f\"ID: {image_id}\\nSource: {data_provider} ISUP: {isup_grade} Gleason: {gleason_score}\")\n        f.tight_layout()\n        \n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:14:47.885859Z","iopub.execute_input":"2023-06-10T11:14:47.886249Z","iopub.status.idle":"2023-06-10T11:14:47.899618Z","shell.execute_reply.started":"2023-06-10T11:14:47.886217Z","shell.execute_reply":"2023-06-10T11:14:47.898669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_masks(images)","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:14:50.445385Z","iopub.execute_input":"2023-06-10T11:14:50.44609Z","iopub.status.idle":"2023-06-10T11:14:58.570652Z","shell.execute_reply.started":"2023-06-10T11:14:50.44603Z","shell.execute_reply":"2023-06-10T11:14:58.565847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we know from PANDAs discussions that 100 images have no masks. \n# ci sono altre 85 immagini che hanno una mask che non mostra lesioni cancerogene anche se ISUP>0, usarle solo per validation e non training.\n#non sappiamo quali immagini siano\ndef overlay_mask_on_slide(images, center='radboud', alpha=0.8, max_size=(800, 800)):\n    \"\"\"Show a mask overlayed on a slide.\"\"\"\n    f, ax = plt.subplots(5,3, figsize=(18,22))\n    \n    \n    for i, image_id in enumerate(images):\n        slide = openslide.OpenSlide(os.path.join(train_images, f'{image_id}.tiff'))\n        mask = openslide.OpenSlide(os.path.join(train_label_masks, f'{image_id}_mask.tiff'))\n        slide_data = slide.read_region((0,0), slide.level_count - 1, slide.level_dimensions[-1])\n        mask_data = mask.read_region((0,0), mask.level_count - 1, mask.level_dimensions[-1])\n        mask_data = mask_data.split()[0]\n        \n        \n        # Create alpha mask\n        alpha_int = int(round(255*alpha))\n        if center == 'radboud':\n            alpha_content = np.less(mask_data.split()[0], 2).astype('uint8') * alpha_int + (255 - alpha_int)\n        elif center == 'karolinska':\n            alpha_content = np.less(mask_data.split()[0], 1).astype('uint8') * alpha_int + (255 - alpha_int)\n\n        alpha_content = PIL.Image.fromarray(alpha_content)\n        preview_palette = np.zeros(shape=768, dtype=int)\n\n        if center == 'radboud':\n            # Mapping: {0: background, 1: stroma, 2: benign epithelium, 3: Gleason 3, 4: Gleason 4, 5: Gleason 5}\n            preview_palette[0:18] = (np.array([0, 0, 0, 0.5, 0.5, 0.5, 0, 1, 0, 1, 1, 0.7, 1, 0.5, 0, 1, 0, 0]) * 255).astype(int)\n        elif center == 'karolinska':\n            # Mapping: {0: background, 1: benign, 2: cancer}\n            preview_palette[0:9] = (np.array([0, 0, 0, 0, 1, 0, 1, 0, 0]) * 255).astype(int)\n        mask_data.putpalette(data=preview_palette.tolist())\n        mask_rgb = mask_data.convert(mode='RGB')\n        overlayed_image = PIL.Image.composite(image1=slide_data, image2=mask_rgb, mask=alpha_content)\n        overlayed_image.thumbnail(size=max_size, resample=0)\n\n        \n        ax[i//3, i%3].imshow(overlayed_image) \n        slide.close()\n        mask.close()       \n        ax[i//3, i%3].axis('off')\n        \n        data_provider = train.loc[image_id, 'data_provider']\n        isup_grade = train.loc[image_id, 'isup_grade']\n        gleason_score = train.loc[image_id, 'gleason_score']\n        ax[i//3, i%3].set_title(f\"ID: {image_id}\\nSource: {data_provider} ISUP: {isup_grade} Gleason: {gleason_score}\")","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:15:13.077511Z","iopub.execute_input":"2023-06-10T11:15:13.077876Z","iopub.status.idle":"2023-06-10T11:15:13.095762Z","shell.execute_reply.started":"2023-06-10T11:15:13.077845Z","shell.execute_reply":"2023-06-10T11:15:13.094621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"      overlay_mask_on_slide(images)","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:15:20.512552Z","iopub.execute_input":"2023-06-10T11:15:20.512916Z","iopub.status.idle":"2023-06-10T11:15:24.720494Z","shell.execute_reply.started":"2023-06-10T11:15:20.512886Z","shell.execute_reply":"2023-06-10T11:15:24.719357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '../input/prostate-cancer-grade-assessment'\ndf_train = pd.read_csv(os.path.join(data_dir, 'train.csv'))\nimage_folder = os.path.join(data_dir, 'train_images')\n\nkernel_type = 'how_to_train_effnet_b0_to_get_LB_0.86'\n\nenet_type = 'efficientnet-b0'\nfold = 0\ntile_size = 256\nimage_size = 256\nn_tiles = 36\nbatch_size = 2\nnum_workers = 4\nout_dim = 5\ninit_lr = 3e-4\nwarmup_factor = 10\n\nwarmup_epo = 1\nn_epochs = 1 if DEBUG else 30\ndf_train = df_train.sample(100).reset_index(drop=True) if DEBUG else df_train\n\ndevice = torch.device('cuda')\n\nprint(image_folder)","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:15:31.367621Z","iopub.execute_input":"2023-06-10T11:15:31.368002Z","iopub.status.idle":"2023-06-10T11:15:31.396717Z","shell.execute_reply.started":"2023-06-10T11:15:31.36797Z","shell.execute_reply":"2023-06-10T11:15:31.395771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create folds\n\nskf = StratifiedKFold(5, shuffle=True, random_state=42)\ndf_train['fold'] = -1\nfor i, (train_idx, valid_idx) in enumerate(skf.split(df_train, df_train['isup_grade'])):\n    df_train.loc[valid_idx, 'fold'] = i\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:15:35.313404Z","iopub.execute_input":"2023-06-10T11:15:35.313788Z","iopub.status.idle":"2023-06-10T11:15:35.336679Z","shell.execute_reply.started":"2023-06-10T11:15:35.313758Z","shell.execute_reply":"2023-06-10T11:15:35.335628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model\npretrained_model = {\n    'efficientnet-b0': '../input/efficientnet-pytorch/efficientnet-b0-08094119.pth'\n}\n","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:15:49.807171Z","iopub.execute_input":"2023-06-10T11:15:49.80755Z","iopub.status.idle":"2023-06-10T11:15:49.812367Z","shell.execute_reply.started":"2023-06-10T11:15:49.807518Z","shell.execute_reply":"2023-06-10T11:15:49.811453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class enetv2(nn.Module):\n    def __init__(self, backbone, out_dim):\n        super(enetv2, self).__init__()\n        self.enet = enet.EfficientNet.from_name(backbone)\n        self.enet.load_state_dict(torch.load(pretrained_model[backbone]))\n\n        self.myfc = nn.Linear(self.enet._fc.in_features, out_dim)\n        self.enet._fc = nn.Identity()\n\n    def extract(self, x):\n        return self.enet(x)\n\n    def forward(self, x):\n        x = self.extract(x)\n        x = self.myfc(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:15:44.212535Z","iopub.execute_input":"2023-06-10T11:15:44.21289Z","iopub.status.idle":"2023-06-10T11:15:44.219802Z","shell.execute_reply.started":"2023-06-10T11:15:44.212862Z","shell.execute_reply":"2023-06-10T11:15:44.218751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_tiles(img, mode=0):\n        result = []\n        h, w, c = img.shape\n        pad_h = (tile_size - h % tile_size) % tile_size + ((tile_size * mode) // 2)\n        pad_w = (tile_size - w % tile_size) % tile_size + ((tile_size * mode) // 2)\n\n        img2 = np.pad(img,[[pad_h // 2, pad_h - pad_h // 2], [pad_w // 2,pad_w - pad_w//2], [0,0]], constant_values=255)\n        img3 = img2.reshape(\n            img2.shape[0] // tile_size,\n            tile_size,\n            img2.shape[1] // tile_size,\n            tile_size,\n            3\n        )\n\n        img3 = img3.transpose(0,2,1,3,4).reshape(-1, tile_size, tile_size,3)\n        n_tiles_with_info = (img3.reshape(img3.shape[0],-1).sum(1) < tile_size ** 2 * 3 * 255).sum()\n        if len(img3) < n_tiles:\n            img3 = np.pad(img3,[[0,n_tiles-len(img3)],[0,0],[0,0],[0,0]], constant_values=255)\n        idxs = np.argsort(img3.reshape(img3.shape[0],-1).sum(-1))[:n_tiles]\n        img3 = img3[idxs]\n        for i in range(len(img3)):\n            result.append({'img':img3[i], 'idx':i})\n        return result, n_tiles_with_info >= n_tiles\n\n\nclass PANDADataset(Dataset):\n    def __init__(self,\n                 df,\n                 image_size,\n                 n_tiles=n_tiles,\n                 tile_mode=0,\n                 rand=False,\n                 transform=None,\n                ):\n\n        self.df = df.reset_index(drop=True)\n        self.image_size = image_size\n        self.n_tiles = n_tiles\n        self.tile_mode = tile_mode\n        self.rand = rand\n        self.transform = transform\n    def __len__(self):\n        return self.df.shape[0]\n    \n    def __getitem__(self, index):\n        row = self.df.iloc[index]\n        img_id = row.image_id\n\n        tiff_file = os.path.join(image_folder, f'{img_id}.tiff')\n        images = skimage.io.MultiImage(tiff_file)\n        image = images[min(1, len(images) - 1)]  # Access the first or only image\n\n        tiles, OK = get_tiles(image, self.tile_mode)\n\n\n        if self.rand:\n            idxes = np.random.choice(list(range(self.n_tiles)), self.n_tiles, replace=False)\n        else:\n            idxes = list(range(self.n_tiles))\n\n        n_row_tiles = int(np.sqrt(self.n_tiles))\n        images = np.zeros((image_size * n_row_tiles, image_size * n_row_tiles, 3))\n        for h in range(n_row_tiles):\n            for w in range(n_row_tiles):\n                i = h * n_row_tiles + w\n                if len(tiles) > idxes[i]:\n                    this_img = tiles[idxes[i]]['img']\n                else:\n                    this_img = np.ones((self.image_size, self.image_size, 3)).astype(np.uint8) * 255\n                this_img = 255 - this_img\n                if self.transform is not None:\n                    this_img = self.transform(image=this_img)['image']\n                h1 = h * image_size\n                w1 = w * image_size\n                images[h1:h1+image_size, w1:w1+image_size] = this_img\n\n        if self.transform is not None:\n            images = self.transform(image=images)['image']\n        images = images.astype(np.float32)\n        images /= 255\n        images = images.transpose(2, 0, 1)\n\n        label = np.zeros(5).astype(np.float32)\n        label[:row.isup_grade] = 1.\n        return torch.tensor(images), torch.tensor(label)","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:16:00.905461Z","iopub.execute_input":"2023-06-10T11:16:00.90588Z","iopub.status.idle":"2023-06-10T11:16:00.938161Z","shell.execute_reply.started":"2023-06-10T11:16:00.905848Z","shell.execute_reply":"2023-06-10T11:16:00.937091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#augmentation\n\ntransforms_train = albumentations.Compose([\n    albumentations.Transpose(p=0.5),\n    albumentations.VerticalFlip(p=0.5),\n    albumentations.HorizontalFlip(p=0.5),\n])\ntransforms_val = albumentations.Compose([])","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:16:07.58599Z","iopub.execute_input":"2023-06-10T11:16:07.587077Z","iopub.status.idle":"2023-06-10T11:16:07.593409Z","shell.execute_reply.started":"2023-06-10T11:16:07.587012Z","shell.execute_reply":"2023-06-10T11:16:07.59221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_show = PANDADataset(df_train, image_size, n_tiles, 0, transform=transforms_train)\nfrom pylab import rcParams\nrcParams['figure.figsize'] = 20,10\nfor i in range(2):\n    f, axarr = plt.subplots(1,5)\n    for p in range(5):\n        idx = np.random.randint(0, len(dataset_show))\n        img, label = dataset_show[idx]\n        axarr[p].imshow(1. - img.transpose(0, 1).transpose(1,2).squeeze())\n        axarr[p].set_title(str(sum(label)))\n","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:16:11.252258Z","iopub.execute_input":"2023-06-10T11:16:11.25263Z","iopub.status.idle":"2023-06-10T11:17:07.643465Z","shell.execute_reply.started":"2023-06-10T11:16:11.252601Z","shell.execute_reply":"2023-06-10T11:17:07.642488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"criterion = nn.BCEWithLogitsLoss()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:17:23.062934Z","iopub.execute_input":"2023-06-10T11:17:23.063821Z","iopub.status.idle":"2023-06-10T11:17:23.067683Z","shell.execute_reply.started":"2023-06-10T11:17:23.063786Z","shell.execute_reply":"2023-06-10T11:17:23.066768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport torch\nfrom torch.utils.data import DataLoader, Dataset\nimport albumentations\nfrom sklearn.model_selection import StratifiedKFold\nimport os\n\n# ...\n\nclass PANDADataset(Dataset):\n    def __init__(self, df, image_size, tile_size, n_tiles, transform=None):\n        self.df = df\n        self.image_size = image_size\n        self.tile_size = tile_size\n        self.n_tiles = n_tiles\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, index):\n        row = self.df.iloc[index]\n        image_id = row[\"image_id\"]\n        image = self.load_image(image_id)\n        tiles = self.get_tiles(image)\n\n        if self.transform is not None:\n            tiles = [self.transform(image=t)[\"image\"] for t in tiles]\n\n        tiles = np.stack(tiles)\n\n        return {\n            \"image_id\": image_id,\n            \"tiles\": tiles,\n            \"target\": row[\"isup_grade\"]\n        }\n\n    def load_image(self, image_id):\n        # Load image from file and return it\n        # You need to implement this method\n        pass\n\n    def get_tiles(self, image):\n        # Extract tiles from the image and return a list of tiles\n        # You need to implement this method\n        pass\n\n# ...\n\ndef get_data_loaders(df_train, image_size, tile_size, n_tiles, batch_size, num_workers):\n    train_transform = albumentations.Compose([\n        # Add your desired transformations here\n    ])\n\n    valid_transform = albumentations.Compose([\n        # Add your desired transformations here\n    ])\n\n    train_dataset = PANDADataset(df_train, image_size, tile_size, n_tiles, transform=train_transform)\n    valid_dataset = PANDADataset(df_train, image_size, tile_size, n_tiles, transform=valid_transform)\n\n    # Adjust the num_workers to the suggested max number for the current system\n    num_workers = min(num_workers, os.cpu_count())\n\n    train_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True, num_workers=num_workers)\n    valid_loader = DataLoader(valid_dataset, batch_size=batch_size, shuffle=False, num_workers=num_workers)\n\n    return train_loader, valid_loader\n\n# ...\n\n# Define the required variables and data loaders\ndata_dir = '/kaggle/input/prostate-cancer-grade-assessment'\ndf_train = pd.read_csv(os.path.join(data_dir, 'train.csv'))\nimage_folder = os.path.join(data_dir, 'train_images')\n\nkernel_type = 'how_to_train_effnet_b0_to_get_LB_0.86'\nenet_type = 'efficientnet-b0'\ntile_size = 256\nimage_size = 256\nn_tiles = 36\nbatch_size = 2\nnum_workers = 4\nout_dim = 5\ninit_lr = 3e-4\nwarmup_factor = 10\nwarmup_epo = 1\nn_epochs = 1 if DEBUG else 30\ndf_train = df_train.sample(100).reset_index(drop=True) if DEBUG else df_train\ndevice = torch.device('cuda')\n\ntrain_loader, valid_loader = get_data_loaders(df_train, image_size, tile_size, n_tiles, batch_size, num_workers)\n\n# Print the values\nprint(image_folder)\nprint(train_loader)\nprint(valid_loader)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:17:26.59455Z","iopub.execute_input":"2023-06-10T11:17:26.59491Z","iopub.status.idle":"2023-06-10T11:17:26.628733Z","shell.execute_reply.started":"2023-06-10T11:17:26.59488Z","shell.execute_reply":"2023-06-10T11:17:26.627689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nclass enetv2(nn.Module):\n    def __init__(self, num_classes):\n        super(enetv2, self).__init__()\n        # Define your model architecture here\n        self.conv1 = nn.Conv2d(3, 32, kernel_size=3, stride=1, padding=1)\n        self.conv2 = nn.Conv2d(32, 64, kernel_size=3, stride=1, padding=1)\n        self.fc1 = nn.Linear(64 * 32 * 32, 128)\n        self.fc2 = nn.Linear(128, num_classes)\n\n    def forward(self, x):\n        # Define the forward pass of your model\n        x = F.relu(self.conv1(x))\n        x = F.max_pool2d(x, 2)\n        x = F.relu(self.conv2(x))\n        x = F.max_pool2d(x, 2)\n        x = x.view(x.size(0), -1)\n        x = F.relu(self.fc1(x))\n        x = self.fc2(x)\n        return x\n\n\ndef train_epoch(loader, optimizer, model):\n\n    model.train()\n    train_loss = []\n    bar = tqdm(loader)\n    for (data, target) in bar:\n        \n        data, target = data.to(device), target.to(device)\n        loss_func = criterion\n        optimizer.zero_grad()\n        logits = model(data)\n        loss = loss_func(logits, target)\n        loss.backward()\n        optimizer.step()\n\n        loss_np = loss.detach().cpu().numpy()\n        train_loss.append(loss_np)\n        smooth_loss = sum(train_loss[-100:]) / min(len(train_loss), 100)\n        bar.set_description('loss: %.5f, smth: %.5f' % (loss_np, smooth_loss))\n    return train_loss\n\ndef val_epoch(loader, model, get_output=False):\n\n    model.eval()\n    val_loss = []\n    LOGITS = []\n    PREDS = []\n    TARGETS = []\n\n    with torch.no_grad():\n        for (data, target) in tqdm(loader):\n            data, target = data.to(device), target.to(device)\n            logits = model(data)\n\n            loss = criterion(logits, target)\n\n            pred = logits.sigmoid().sum(1).detach().round()\n            LOGITS.append(logits)\n            PREDS.append(pred)\n            TARGETS.append(target.sum(1))\n\n            val_loss.append(loss.detach().cpu().numpy())\n        val_loss = np.mean(val_loss)\n    LOGITS = torch.cat(LOGITS).cpu().numpy()\n    PREDS = torch.cat(PREDS).cpu().numpy()\n    TARGETS = torch.cat(TARGETS).cpu().numpy()\n    acc = (PREDS == TARGETS).mean() * 100.\n    \n    qwk = cohen_kappa_score(PREDS, TARGETS, weights='quadratic')\n    qwk_k = cohen_kappa_score(PREDS[df_valid['data_provider'] == 'karolinska'], df_valid[df_valid['data_provider'] == 'karolinska'].isup_grade.values, weights='quadratic')\n    qwk_r = cohen_kappa_score(PREDS[df_valid['data_provider'] == 'radboud'], df_valid[df_valid['data_provider'] == 'radboud'].isup_grade.values, weights='quadratic')\n    print('qwk', qwk, 'qwk_k', qwk_k, 'qwk_r', qwk_r)\n\n    if get_output:\n        return LOGITS\n    else:\n        return val_loss, acc, qwk\n\n\n# Define your model parameters and other required variables\nnum_classes = 6  # Number of classes in the PANDA challenge\n\n# Create an instance of the enetv2 class\nenetv2_model = enetv2(num_classes)\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nenetv2_model = enetv2_model.to(device)\ncriterion = nn.CrossEntropyLoss()\n\n# Define your optimizer\noptimizer = torch.optim.Adam(enetv2_model.parameters(), lr=0.001)\n\n# Training\ntrain_losses = train_epoch(train_loader, optimizer, enetv2_model)\n\n# Validation\nval_loss, accuracy, qwk = val_epoch(val_loader, enetv2_model)\n\n# Print the results\nprint(\"Validation Loss:\", val_loss)\nprint(\"Accuracy:\", accuracy)\nprint(\"QWK:\", qwk)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:44:50.230564Z","iopub.execute_input":"2023-06-10T11:44:50.230972Z","iopub.status.idle":"2023-06-10T11:44:50.882754Z","shell.execute_reply.started":"2023-06-10T11:44:50.230929Z","shell.execute_reply":"2023-06-10T11:44:50.880997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nclass enetv2(nn.Module):\n    def __init__(self):\n        super(enetv2, self).__init__()\n        # Define your model architecture here\n        self.conv1 = nn.Conv2d(3, 32, kernel_size=3, stride=1, padding=1)\n        self.conv2 = nn.Conv2d(32, 64, kernel_size=3, stride=1, padding=1)\n        self.fc1 = nn.Linear(64 * 32 * 32, 128)\n        self.fc2 = nn.Linear(128, num_classes)\n\n    def forward(self, x):\n        # Define the forward pass of your model\n        x = F.relu(self.conv1(x))\n        x = F.max_pool2d(x, 2)\n        x = F.relu(self.conv2(x))\n        x = F.max_pool2d(x, 2)\n        x = x.view(x.size(0), -1)\n        x = F.relu(self.fc1(x))\n        x = self.fc2(x)\n        return x\n\n\ndef train_epoch(loader, optimizer):\n\n    model.train()\n    train_loss = []\n    bar = tqdm(loader)\n    for (data, target) in bar:\n        \n        data, target = data.to(device), target.to(device)\n        loss_func = criterion\n        optimizer.zero_grad()\n        logits = model(data)\n        loss = loss_func(logits, target)\n        loss.backward()\n        optimizer.step()\n\n        loss_np = loss.detach().cpu().numpy()\n        train_loss.append(loss_np)\n        smooth_loss = sum(train_loss[-100:]) / min(len(train_loss), 100)\n        bar.set_description('loss: %.5f, smth: %.5f' % (loss_np, smooth_loss))\n    return train_loss\ndef val_epoch(loader, get_output=False):\n\n    model.eval()\n    val_loss = []\n    LOGITS = []\n    PREDS = []\n    TARGETS = []\n\n    with torch.no_grad():\n        for (data, target) in tqdm(loader):\n            data, target = data.to(device), target.to(device)\n            logits = model(data)\n\n            loss = criterion(logits, target)\n\n            pred = logits.sigmoid().sum(1).detach().round()\n            LOGITS.append(logits)\n            PREDS.append(pred)\n            TARGETS.append(target.sum(1))\n\n            val_loss.append(loss.detach().cpu().numpy())\n        val_loss = np.mean(val_loss)\n    LOGITS = torch.cat(LOGITS).cpu().numpy()\n    PREDS = torch.cat(PREDS).cpu().numpy()\n    TARGETS = torch.cat(TARGETS).cpu().numpy()\n    acc = (PREDS == TARGETS).mean() * 100.\n    \n    qwk = cohen_kappa_score(PREDS, TARGETS, weights='quadratic')\n    qwk_k = cohen_kappa_score(PREDS[df_valid['data_provider'] == 'karolinska'], df_valid[df_valid['data_provider'] == 'karolinska'].isup_grade.values, weights='quadratic')\n    qwk_r = cohen_kappa_score(PREDS[df_valid['data_provider'] == 'radboud'], df_valid[df_valid['data_provider'] == 'radboud'].isup_grade.values, weights='quadratic')\n    print('qwk', qwk, 'qwk_k', qwk_k, 'qwk_r', qwk_r)\n\n    if get_output:\n        return LOGITS\n    else:\n        return val_loss, acc, qwk\nimport torch.optim as optim\n\n# Create an instance of the enetv2 class\nenetv2 = enetv2()\n\n# Define your model parameters and other required variables\nnum_classes = 6  # Number of classes in the PANDA challenge\nenetv2 = enetv2(num_classes)\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nenetv2 = enetv2.to(device)\ncriterion = nn.CrossEntropyLoss()\n\n# Define your optimizer\noptimizer = torch.optim.Adam(enetv2.parameters(), lr=0.001)\n\n# Training\ntrain_losses = train_epoch(train_loader, optimizer)\n\n# Validation\nval_loss, accuracy, qwk = val_epoch(val_loader)\n\n# Print the results\nprint(\"Validation Loss:\", val_loss)\nprint(\"Accuracy:\", accuracy)\nprint(\"QWK:\", qwk)","metadata":{"execution":{"iopub.status.busy":"2023-06-10T11:38:04.540963Z","iopub.execute_input":"2023-06-10T11:38:04.541399Z","iopub.status.idle":"2023-06-10T11:38:04.692991Z","shell.execute_reply.started":"2023-06-10T11:38:04.54137Z","shell.execute_reply":"2023-06-10T11:38:04.691763Z"},"trusted":true},"execution_count":null,"outputs":[]}]}