{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install ../input/fastai017-whl/fastai-2.2.2-py3-none-any.whl\n!pip install ../input/fastai017-whl/fastcore-1.3.18-py3-none-any.whl\n#!pip install ../input/fastai017-whl/fastai2-0.0.17-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:54:49.290846Z","iopub.execute_input":"2021-10-12T05:54:49.291109Z","iopub.status.idle":"2021-10-12T05:55:43.131049Z","shell.execute_reply.started":"2021-10-12T05:54:49.291054Z","shell.execute_reply":"2021-10-12T05:55:43.129965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"#Load the dependancies\nfrom fastai import *\nfrom fastai.callback.all import *\nfrom fastai.vision.all import *\n\nimport seaborn as sns\nimport numpy as np\nimport pandas as pd\nimport os\nimport cv2\nimport openslide\n\nfrom sklearn.model_selection import StratifiedShuffleSplit\n\nsns.set(style=\"whitegrid\")\nsns.set_context(\"paper\")","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:43.134105Z","iopub.execute_input":"2021-10-12T05:55:43.134459Z","iopub.status.idle":"2021-10-12T05:55:45.655541Z","shell.execute_reply.started":"2021-10-12T05:55:43.134418Z","shell.execute_reply":"2021-10-12T05:55:45.654544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Nach Ausschalten der Internetverbindung, muss auch das vortrainierte Modell manuell geladen werden. Dieses können wir aus dem vorherigen`checkpoint` hereinladen.","metadata":{}},{"cell_type":"code","source":"Path('/root/.cache/torch/hub/checkpoints').mkdir(exist_ok=True, parents=True)\n!cp '../input/xresnet50-pretrained-weigths/xrn50_940.pth' '/root/.cache/torch/hub/checkpoints/xrn50_940.pth'","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:45.657356Z","iopub.execute_input":"2021-10-12T05:55:45.657714Z","iopub.status.idle":"2021-10-12T05:55:49.28713Z","shell.execute_reply.started":"2021-10-12T05:55:45.657687Z","shell.execute_reply":"2021-10-12T05:55:49.286037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /root/.cache/torch/hub/checkpoints/","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:49.289099Z","iopub.execute_input":"2021-10-12T05:55:49.289467Z","iopub.status.idle":"2021-10-12T05:55:49.98179Z","shell.execute_reply.started":"2021-10-12T05:55:49.289423Z","shell.execute_reply":"2021-10-12T05:55:49.980821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"source = Path(\"../input/prostate-cancer-grade-assessment\")\nfiles = os.listdir(source)\nprint(files)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:49.98348Z","iopub.execute_input":"2021-10-12T05:55:49.983884Z","iopub.status.idle":"2021-10-12T05:55:49.990843Z","shell.execute_reply.started":"2021-10-12T05:55:49.983839Z","shell.execute_reply":"2021-10-12T05:55:49.989813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = source/'train_images'\nmask = source/'train_label_masks'\ntrain_labels = pd.read_csv(source/'train.csv')","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:49.992754Z","iopub.execute_input":"2021-10-12T05:55:49.993157Z","iopub.status.idle":"2021-10-12T05:55:50.027753Z","shell.execute_reply.started":"2021-10-12T05:55:49.993123Z","shell.execute_reply":"2021-10-12T05:55:50.027008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Verteilung der Datensätze ausgleichen\nWir wollen möglichst die gleiche Anzahl an Trainingsbildern für jede der verschiedenen Tumorklassen (`isup_grade`).","metadata":{}},{"cell_type":"code","source":"train_labels.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:50.028814Z","iopub.execute_input":"2021-10-12T05:55:50.02914Z","iopub.status.idle":"2021-10-12T05:55:50.052861Z","shell.execute_reply.started":"2021-10-12T05:55:50.029101Z","shell.execute_reply":"2021-10-12T05:55:50.051878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_count(df, feature, title='', size=2):\n    f, ax = plt.subplots(1,1, figsize=(3*size,2*size))\n    total = float(len(df))\n    sns.countplot(df[feature],order = df[feature].value_counts().index, palette='Set1')\n    plt.title(title)\n    for p in ax.patches:\n        height = p.get_height()\n        ax.text(p.get_x()+p.get_width()/2.,\n                height + 3,\n                '{:1.2f}%'.format(100*height/total),\n                ha=\"center\") \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:50.055876Z","iopub.execute_input":"2021-10-12T05:55:50.056247Z","iopub.status.idle":"2021-10-12T05:55:50.064255Z","shell.execute_reply.started":"2021-10-12T05:55:50.056212Z","shell.execute_reply":"2021-10-12T05:55:50.06316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_count(train_labels, 'isup_grade','ISUP grade - data count and percent', size=3)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:50.066539Z","iopub.execute_input":"2021-10-12T05:55:50.067184Z","iopub.status.idle":"2021-10-12T05:55:50.273104Z","shell.execute_reply.started":"2021-10-12T05:55:50.067087Z","shell.execute_reply":"2021-10-12T05:55:50.272302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Die Verteilung der Trainingsbilder in den einzelnen Klassen ist recht unausgewogen....\n\nisup_0 = train_labels[train_labels.isup_grade == 0]\nisup_1 = train_labels[train_labels.isup_grade == 1]\nisup_2 = train_labels[train_labels.isup_grade == 2]\nisup_3 = train_labels[train_labels.isup_grade == 3]\nisup_4 = train_labels[train_labels.isup_grade == 4]\nisup_5 = train_labels[train_labels.isup_grade == 5]\n\nprint(f'isup_0: {len(isup_0)}, isup_1: {len(isup_1)}, isup_2: {len(isup_2)}, isup_3: {len(isup_3)}, isup_4: {len(isup_4)}, isup_5: {len(isup_5)}')\n\n","metadata":{}},{"cell_type":"code","source":"isup_0 = train_labels[train_labels.isup_grade == 0]\nisup_1 = train_labels[train_labels.isup_grade == 1]\nisup_2 = train_labels[train_labels.isup_grade == 2]\nisup_3 = train_labels[train_labels.isup_grade == 3]\nisup_4 = train_labels[train_labels.isup_grade == 4]\nisup_5 = train_labels[train_labels.isup_grade == 5]\n\nprint(f'isup_0: {len(isup_0)}, isup_1: {len(isup_1)}, isup_2: {len(isup_2)}, isup_3: {len(isup_3)}, isup_4: {len(isup_4)}, isup_5: {len(isup_5)}')","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:50.274347Z","iopub.execute_input":"2021-10-12T05:55:50.274847Z","iopub.status.idle":"2021-10-12T05:55:50.292292Z","shell.execute_reply.started":"2021-10-12T05:55:50.274806Z","shell.execute_reply":"2021-10-12T05:55:50.291482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Mithilfe der Funktion `sample` wählen wir aus jeder Kategorie `1224` zufällige Bilder aus.","metadata":{}},{"cell_type":"code","source":"isup_sam0 = isup_0.sample(n=1224)\nisup_sam1 = isup_1.sample(n=1224)\nisup_sam2 = isup_2.sample(n=1224)\nisup_sam3 = isup_3.sample(n=1224)\nisup_sam4 = isup_4.sample(n=1224)\nisup_sam5 = isup_5.sample(n=1224)\n\nframes = [isup_sam0, isup_sam1, isup_sam2, isup_sam3, isup_sam4, isup_sam5]\nbalanced_df = pd.concat(frames)\nbalanced_df","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:50.293484Z","iopub.execute_input":"2021-10-12T05:55:50.293858Z","iopub.status.idle":"2021-10-12T05:55:50.321272Z","shell.execute_reply.started":"2021-10-12T05:55:50.293821Z","shell.execute_reply":"2021-10-12T05:55:50.320495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Ein Subsample für schnelleres Training während der Entwicklung erstellen","metadata":{}},{"cell_type":"code","source":"#isup_sam0 = isup_0.sample(n=256)\n#isup_sam1 = isup_1.sample(n=256)\n#isup_sam2 = isup_2.sample(n=256)\n#isup_sam3 = isup_3.sample(n=256)\n#isup_sam4 = isup_4.sample(n=256)\n#isup_sam5 = isup_5.sample(n=256)\n\n#frames = [isup_sam0, isup_sam1, isup_sam2, isup_sam3, isup_sam4, isup_sam5]\n#balanced_df = pd.concat(frames)\n#balanced_df","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:50.322516Z","iopub.execute_input":"2021-10-12T05:55:50.322875Z","iopub.status.idle":"2021-10-12T05:55:50.326901Z","shell.execute_reply.started":"2021-10-12T05:55:50.322839Z","shell.execute_reply":"2021-10-12T05:55:50.3258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_count(balanced_df, 'isup_grade','ISUP grade - data count and percent', size=3)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:50.328299Z","iopub.execute_input":"2021-10-12T05:55:50.328747Z","iopub.status.idle":"2021-10-12T05:55:50.503689Z","shell.execute_reply.started":"2021-10-12T05:55:50.328711Z","shell.execute_reply":"2021-10-12T05:55:50.502744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Erstellen eines eigenen Trainings- und Testset\nWir wollen aus den vorhandenen Trainingsdaten ein eigenes Testdatenset extrahieren, da das Testdatenset von der Kaggle Competition nicht öffentlich verfügbar ist.","metadata":{}},{"cell_type":"code","source":"df_copy = balanced_df.copy()\n\n# 80/20 split or whatever you choose\ntrain_set = df_copy.sample(frac=0.80, random_state=7)\ntest_set = df_copy.drop(train_set.index)\nprint(len(train_set), len(test_set))","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:50.505162Z","iopub.execute_input":"2021-10-12T05:55:50.505511Z","iopub.status.idle":"2021-10-12T05:55:50.518716Z","shell.execute_reply.started":"2021-10-12T05:55:50.505471Z","shell.execute_reply":"2021-10-12T05:55:50.517594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Save train_set to csv for strattification\ntrain_set.to_csv('split.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:50.520007Z","iopub.execute_input":"2021-10-12T05:55:50.520615Z","iopub.status.idle":"2021-10-12T05:55:50.754397Z","shell.execute_reply.started":"2021-10-12T05:55:50.520553Z","shell.execute_reply":"2021-10-12T05:55:50.753608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Stratifizierung des Trainingsdatensets","metadata":{}},{"cell_type":"code","source":"#df = pd.read_csv('split.csv')\n#df.shape[0]","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:50.755724Z","iopub.execute_input":"2021-10-12T05:55:50.756108Z","iopub.status.idle":"2021-10-12T05:55:50.760438Z","shell.execute_reply.started":"2021-10-12T05:55:50.756062Z","shell.execute_reply":"2021-10-12T05:55:50.759346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sss = StratifiedShuffleSplit(n_splits=6, test_size=0.2, random_state=7)\n#strat_df = pd.DataFrame()\n#cols=['isup_grade']\n\n#for i, (train_index, test_index) in enumerate(sss.split(df, df.isup_grade)):\n#    df_split = df.copy()\n#    df_split['fold'] = i\n#    df_split.loc[train_index, 'which'] = 'train'\n#    df_split.loc[test_index, 'which'] = 'valid'\n#    X_train = df_split.loc[train_index]\n#    X_valid = df_split.loc[test_index]\n#    X_train.loc[:, 'which'] = 'train'\n#    X_valid.loc[:, 'which'] = 'valid'\n#    \n#    mult_dis = X_train.loc[X_train.image_id=='isup_grade']\n#    for _ in range(3): X_train = X_train.append(mult_dis)\n#    \n#    strat_df = strat_df.append(X_train).append(X_valid)\n#    print(i, strat_df.shape, [(X_train[c].sum()/len(X_train), X_train[c].sum()) for c in cols])\n#strat_df = strat_df.reset_index()","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:50.761975Z","iopub.execute_input":"2021-10-12T05:55:50.762523Z","iopub.status.idle":"2021-10-12T05:55:50.770184Z","shell.execute_reply.started":"2021-10-12T05:55:50.762484Z","shell.execute_reply":"2021-10-12T05:55:50.769086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Das nun erstellte stratifizierte Datenset enthält zusätzlich die Information, in welchen `fold` es gehört und ob es in das Trainings- oder Testdatenset gehört.","metadata":{}},{"cell_type":"code","source":"#strat_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:50.771567Z","iopub.execute_input":"2021-10-12T05:55:50.772046Z","iopub.status.idle":"2021-10-12T05:55:50.780769Z","shell.execute_reply.started":"2021-10-12T05:55:50.772007Z","shell.execute_reply":"2021-10-12T05:55:50.77984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Anzeigen der Bilder aus dem Trainingsset","metadata":{}},{"cell_type":"code","source":"def view_image(folder, fn):\n    filename = f'{folder}/{fn}.tiff'\n    file = openslide.OpenSlide(str(filename))\n    t = tensor(file.get_thumbnail(size=(512, 512)))\n    pil = PILImage.create(t) \n    return pil","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:50.782087Z","iopub.execute_input":"2021-10-12T05:55:50.782619Z","iopub.status.idle":"2021-10-12T05:55:50.789861Z","shell.execute_reply.started":"2021-10-12T05:55:50.782562Z","shell.execute_reply":"2021-10-12T05:55:50.788989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"glee_35 = train_labels[train_labels.gleason_score == '3+5']\nglee_35[:5]","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:50.791219Z","iopub.execute_input":"2021-10-12T05:55:50.791667Z","iopub.status.idle":"2021-10-12T05:55:50.809562Z","shell.execute_reply.started":"2021-10-12T05:55:50.791631Z","shell.execute_reply":"2021-10-12T05:55:50.808635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_image(train, '05819281002c55258bb3086cc55e3b48')","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:50.811115Z","iopub.execute_input":"2021-10-12T05:55:50.811505Z","iopub.status.idle":"2021-10-12T05:55:51.080263Z","shell.execute_reply.started":"2021-10-12T05:55:50.811466Z","shell.execute_reply":"2021-10-12T05:55:51.079307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DataBlock für das Modell erstellen","metadata":{}},{"cell_type":"markdown","source":"Für die Verwendung der Bilder im `DataBlock` benötigen wir eine Funktion, die das Bild im `tiff`-Format öffnet und in einen `Tensor` konvertiert.","metadata":{}},{"cell_type":"code","source":"def get_i(fn):\n    filename = f'{train}/{fn.image_id}.tiff'\n    example2 = openslide.OpenSlide(str(filename))\n    ee = example2.get_thumbnail(size=(512, 512))\n    return tensor(ee)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:51.081529Z","iopub.execute_input":"2021-10-12T05:55:51.0819Z","iopub.status.idle":"2021-10-12T05:55:51.087111Z","shell.execute_reply.started":"2021-10-12T05:55:51.081862Z","shell.execute_reply":"2021-10-12T05:55:51.086005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Nun können wir unseren `DataBlock` definieren.","metadata":{}},{"cell_type":"code","source":"blocks = (\n          ImageBlock,\n          CategoryBlock\n          )    \ngetters = [\n           get_i,\n           ColReader('isup_grade')\n          ]\ntrends = DataBlock(blocks=blocks,\n                  splitter=RandomSplitter(),\n                  getters=getters,\n                  item_tfms=Resize(512),\n                  batch_tfms=aug_transforms(size=224)\n                  )","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:51.093488Z","iopub.execute_input":"2021-10-12T05:55:51.093762Z","iopub.status.idle":"2021-10-12T05:55:51.101926Z","shell.execute_reply.started":"2021-10-12T05:55:51.093735Z","shell.execute_reply":"2021-10-12T05:55:51.100797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Um unser Modell mit dem `k-fold`-Datenset trainieren zu können, benötigen wir eine eigene Funktion, die für jeden `fold` den `DataLoader` sowie den `Learner`  zurückliefert.","metadata":{}},{"cell_type":"code","source":"def train_on_folds(bs, size, base_lr, folds):\n    learners = []\n    all_val_preds = []\n    all_val_labels = []\n    all_test_preds = []\n    \n    for fold in range(folds):\n        print(f'Processing fold: {fold}....')\n        dls = get_dls(bs=bs, size=size, fold=fold, df=strat_df)\n        \n        learn = get_learner(dls, arch, loss_func, cbs)\n        learn = train_learner(learn, base_lr)\n        learn.save(f'model_fold_{fold}')\n        learners.append(learn)\n        learn.recorder.plot_loss()\n        \n        test_dl = dls.test_dl(train_df)\n        test_preds, _, _ = learn.get_preds(dl=test_dl, with_decoded=True)\n        val_preds, val_labels = learn.get_preds()\n    \n        all_val_preds.append(val_preds)\n        all_val_labels.append(val_labels)\n        all_test_preds.append(test_preds)\n    \n    plt.show()\n    return learners, all_val_preds, all_val_labels, all_test_preds","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2021-10-12T05:55:51.104555Z","iopub.execute_input":"2021-10-12T05:55:51.105137Z","iopub.status.idle":"2021-10-12T05:55:51.114118Z","shell.execute_reply.started":"2021-10-12T05:55:51.105098Z","shell.execute_reply":"2021-10-12T05:55:51.113282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set_seed(7)\ndef get_dls(bs, size, fold, df):\n    df_fold = df.copy()\n    df_fold = df_fold.loc[df_fold.fold==fold].reset_index()\n    \n    trends = DataBlock(blocks=blocks,\n                       splitter=IndexSplitter(df_fold.loc[df_fold.which=='valid'].index),\n                       getters=getters,\n                       item_tfms=Resize(256),\n                       batch_tfms=aug_transforms(size=size)\n                       )\n    dls = trends.dataloaders(df_fold, bs=bs)\n    assert (len(dls.train_ds) + len(dls.valid_ds)) == len(df_fold)\n    return dls","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:51.11544Z","iopub.execute_input":"2021-10-12T05:55:51.115931Z","iopub.status.idle":"2021-10-12T05:55:51.126711Z","shell.execute_reply.started":"2021-10-12T05:55:51.115893Z","shell.execute_reply":"2021-10-12T05:55:51.125667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_learner(dls,arch,loss_func, cbs):\n        return cnn_learner(dls,arch,loss_func=loss_func,\n                           metrics=metrics, cbs=cbs).to_fp16()","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:51.127904Z","iopub.execute_input":"2021-10-12T05:55:51.128208Z","iopub.status.idle":"2021-10-12T05:55:51.136594Z","shell.execute_reply.started":"2021-10-12T05:55:51.128181Z","shell.execute_reply":"2021-10-12T05:55:51.135822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_learner(learn, base_lr):\n    learn.fine_tune(10)\n    learn.unfreeze()\n    learn.fit_one_cycle(10, base_lr)    \n    return learn","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:51.138662Z","iopub.execute_input":"2021-10-12T05:55:51.138923Z","iopub.status.idle":"2021-10-12T05:55:51.146136Z","shell.execute_reply.started":"2021-10-12T05:55:51.138899Z","shell.execute_reply":"2021-10-12T05:55:51.145321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dls = get_dls(bs=128, size=128, fold=2, df=strat_df)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:51.147289Z","iopub.execute_input":"2021-10-12T05:55:51.147544Z","iopub.status.idle":"2021-10-12T05:55:51.156204Z","shell.execute_reply.started":"2021-10-12T05:55:51.147516Z","shell.execute_reply":"2021-10-12T05:55:51.155262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dls.show_batch()","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:51.157388Z","iopub.execute_input":"2021-10-12T05:55:51.157813Z","iopub.status.idle":"2021-10-12T05:55:51.165949Z","shell.execute_reply.started":"2021-10-12T05:55:51.157783Z","shell.execute_reply":"2021-10-12T05:55:51.164969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dls.c","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:51.167255Z","iopub.execute_input":"2021-10-12T05:55:51.167658Z","iopub.status.idle":"2021-10-12T05:55:51.174558Z","shell.execute_reply.started":"2021-10-12T05:55:51.167592Z","shell.execute_reply":"2021-10-12T05:55:51.173734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Basis CNN Learner mit fastai","metadata":{}},{"cell_type":"code","source":"bs = 128\ndls = trends.dataloaders(df_copy, bs=bs)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:51.175988Z","iopub.execute_input":"2021-10-12T05:55:51.176434Z","iopub.status.idle":"2021-10-12T05:55:58.027989Z","shell.execute_reply.started":"2021-10-12T05:55:51.176346Z","shell.execute_reply":"2021-10-12T05:55:58.026995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.c","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:58.029399Z","iopub.execute_input":"2021-10-12T05:55:58.029776Z","iopub.status.idle":"2021-10-12T05:55:58.038682Z","shell.execute_reply.started":"2021-10-12T05:55:58.029736Z","shell.execute_reply":"2021-10-12T05:55:58.037431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.show_batch(max_n=9)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:55:58.040547Z","iopub.execute_input":"2021-10-12T05:55:58.041036Z","iopub.status.idle":"2021-10-12T05:56:18.269642Z","shell.execute_reply.started":"2021-10-12T05:55:58.04097Z","shell.execute_reply":"2021-10-12T05:56:18.268809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cohen`s Kappa","metadata":{}},{"cell_type":"code","source":"arch = xresnet50\nloss_func = LabelSmoothingCrossEntropy(eps=0.3, reduction='mean')\ncbs = [ShowGraphCallback()]\nmetrics=[accuracy,CohenKappa(weights='quadratic')]\ntrain_df = pd.read_csv('split.csv')\nbs = 64\nsize = 196","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:56:18.270739Z","iopub.execute_input":"2021-10-12T05:56:18.271035Z","iopub.status.idle":"2021-10-12T05:56:18.287301Z","shell.execute_reply.started":"2021-10-12T05:56:18.271005Z","shell.execute_reply":"2021-10-12T05:56:18.28649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use this cell for the kfold dataloaders\n#dls = get_dls(bs=bs, size=size, fold=1, df=strat_df)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:56:18.288532Z","iopub.execute_input":"2021-10-12T05:56:18.28899Z","iopub.status.idle":"2021-10-12T05:56:18.292703Z","shell.execute_reply.started":"2021-10-12T05:56:18.288952Z","shell.execute_reply":"2021-10-12T05:56:18.291647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use this cell for the non-stratified dataloaders ","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:56:18.294366Z","iopub.execute_input":"2021-10-12T05:56:18.29488Z","iopub.status.idle":"2021-10-12T05:56:18.300352Z","shell.execute_reply.started":"2021-10-12T05:56:18.294842Z","shell.execute_reply":"2021-10-12T05:56:18.299234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = cnn_learner(dls,\n                    arch=arch,\n                    loss_func=loss_func,\n                    metrics=metrics,\n                    cbs=cbs)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:56:18.301897Z","iopub.execute_input":"2021-10-12T05:56:18.302377Z","iopub.status.idle":"2021-10-12T05:56:19.1892Z","shell.execute_reply.started":"2021-10-12T05:56:18.30234Z","shell.execute_reply":"2021-10-12T05:56:19.188301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.lr_find()","metadata":{"execution":{"iopub.status.busy":"2021-10-12T05:56:19.190556Z","iopub.execute_input":"2021-10-12T05:56:19.190935Z","iopub.status.idle":"2021-10-12T06:16:23.318446Z","shell.execute_reply.started":"2021-10-12T05:56:19.190896Z","shell.execute_reply":"2021-10-12T06:16:23.317633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for the kfold Training\n#base_lr = 10e-2\n#learners, all_val_preds, all_val_labels, all_test_preds = train_on_folds(bs, size, base_lr, folds=1)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T06:16:23.319944Z","iopub.execute_input":"2021-10-12T06:16:23.320292Z","iopub.status.idle":"2021-10-12T06:16:23.323669Z","shell.execute_reply.started":"2021-10-12T06:16:23.320249Z","shell.execute_reply":"2021-10-12T06:16:23.322842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(dls.train_ds)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T06:16:23.324759Z","iopub.execute_input":"2021-10-12T06:16:23.325091Z","iopub.status.idle":"2021-10-12T06:16:23.33532Z","shell.execute_reply.started":"2021-10-12T06:16:23.325054Z","shell.execute_reply":"2021-10-12T06:16:23.334213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for the Training of the Base Learning\nlearn.fine_tune(15)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T06:16:23.336566Z","iopub.execute_input":"2021-10-12T06:16:23.336899Z","iopub.status.idle":"2021-10-12T09:25:54.457015Z","shell.execute_reply.started":"2021-10-12T06:16:23.336865Z","shell.execute_reply":"2021-10-12T09:25:54.456183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.export('./models/prostate1.pth')","metadata":{"execution":{"iopub.status.busy":"2021-10-12T14:47:09.713566Z","iopub.execute_input":"2021-10-12T14:47:09.715551Z","iopub.status.idle":"2021-10-12T14:47:10.065263Z","shell.execute_reply.started":"2021-10-12T14:47:09.71551Z","shell.execute_reply":"2021-10-12T14:47:10.064306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Vollständiges Modell trainineren","metadata":{}},{"cell_type":"code","source":"learn.unfreeze()","metadata":{"execution":{"iopub.status.busy":"2021-10-12T14:47:10.066662Z","iopub.execute_input":"2021-10-12T14:47:10.067025Z","iopub.status.idle":"2021-10-12T14:47:10.075452Z","shell.execute_reply.started":"2021-10-12T14:47:10.066989Z","shell.execute_reply":"2021-10-12T14:47:10.074542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.lr_find()","metadata":{"execution":{"iopub.status.busy":"2021-10-12T09:25:54.764458Z","iopub.execute_input":"2021-10-12T09:25:54.764838Z","iopub.status.idle":"2021-10-12T09:45:23.064731Z","shell.execute_reply.started":"2021-10-12T09:25:54.764801Z","shell.execute_reply":"2021-10-12T09:45:23.06388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fit_one_cycle(25)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T09:45:23.066116Z","iopub.execute_input":"2021-10-12T09:45:23.066451Z","iopub.status.idle":"2021-10-12T14:41:59.248412Z","shell.execute_reply.started":"2021-10-12T09:45:23.066411Z","shell.execute_reply":"2021-10-12T14:41:59.24752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Überprüfen anhand des eigenen Testdatensets","metadata":{}},{"cell_type":"code","source":"test_set","metadata":{"execution":{"iopub.status.busy":"2021-10-12T14:41:59.249938Z","iopub.execute_input":"2021-10-12T14:41:59.250283Z","iopub.status.idle":"2021-10-12T14:41:59.266679Z","shell.execute_reply.started":"2021-10-12T14:41:59.250242Z","shell.execute_reply":"2021-10-12T14:41:59.265723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Beispielbilder aus dem Testdatenset","metadata":{}},{"cell_type":"code","source":"tst_dl = dls.test_dl(test_set)\ntst_dl.show_batch(max_n=9)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T14:41:59.268001Z","iopub.execute_input":"2021-10-12T14:41:59.268355Z","iopub.status.idle":"2021-10-12T14:42:18.463098Z","shell.execute_reply.started":"2021-10-12T14:41:59.268312Z","shell.execute_reply":"2021-10-12T14:42:18.46229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# If you are using kfold Training\n#_, _, pred_classes = learn.get_preds(dl=tst_dl, with_decoded=True)\n#pred_classes","metadata":{"execution":{"iopub.status.busy":"2021-10-12T14:42:18.465133Z","iopub.execute_input":"2021-10-12T14:42:18.465539Z","iopub.status.idle":"2021-10-12T14:42:18.473711Z","shell.execute_reply.started":"2021-10-12T14:42:18.465495Z","shell.execute_reply":"2021-10-12T14:42:18.472733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, _, pred_classes = learn.get_preds(dl=tst_dl, with_decoded=True)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T14:42:18.477324Z","iopub.execute_input":"2021-10-12T14:42:18.477743Z","iopub.status.idle":"2021-10-12T14:44:38.927967Z","shell.execute_reply.started":"2021-10-12T14:42:18.477701Z","shell.execute_reply":"2021-10-12T14:44:38.927049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_set.copy()\ntest_df['isup_grade_pred'] = pred_classes\ntest_df","metadata":{"execution":{"iopub.status.busy":"2021-10-12T14:44:38.929604Z","iopub.execute_input":"2021-10-12T14:44:38.930019Z","iopub.status.idle":"2021-10-12T14:44:38.956401Z","shell.execute_reply.started":"2021-10-12T14:44:38.929973Z","shell.execute_reply":"2021-10-12T14:44:38.95571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Confusion Matrix","metadata":{}},{"cell_type":"code","source":"confusion_matrix = pd.crosstab(test_df['isup_grade'], \n                               test_df['isup_grade_pred'], \n                               rownames=['Actual'], \n                               colnames=['Predicted'])\nsns.heatmap(confusion_matrix, annot=True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-10-12T14:44:38.957685Z","iopub.execute_input":"2021-10-12T14:44:38.958031Z","iopub.status.idle":"2021-10-12T14:44:39.517848Z","shell.execute_reply.started":"2021-10-12T14:44:38.957994Z","shell.execute_reply":"2021-10-12T14:44:39.517034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report \nprint(classification_report(test_df['isup_grade'], test_df['isup_grade_pred']))","metadata":{"execution":{"iopub.status.busy":"2021-10-12T14:44:39.519042Z","iopub.execute_input":"2021-10-12T14:44:39.519392Z","iopub.status.idle":"2021-10-12T14:44:39.532776Z","shell.execute_reply.started":"2021-10-12T14:44:39.519364Z","shell.execute_reply":"2021-10-12T14:44:39.531823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interp = ClassificationInterpretation.from_learner(learn)\nlosses,idxs = interp.top_losses()\nlen(dls.valid_ds)==len(losses)==len(idxs)\ninterp.plot_confusion_matrix(figsize=(7,7))","metadata":{"execution":{"iopub.status.busy":"2021-10-12T14:44:39.533973Z","iopub.execute_input":"2021-10-12T14:44:39.534437Z","iopub.status.idle":"2021-10-12T14:47:07.115431Z","shell.execute_reply.started":"2021-10-12T14:44:39.534397Z","shell.execute_reply":"2021-10-12T14:47:07.114556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Einreichen der Ergebnisse für den Wettbewerb","metadata":{}},{"cell_type":"code","source":"submission_test_path = \"../input/prostate-cancer-grade-assessment/test_images/\"\nsample = '../input/prostate-cancer-grade-assessment/sample_submission.csv'","metadata":{"execution":{"iopub.status.busy":"2021-10-12T14:47:07.117068Z","iopub.execute_input":"2021-10-12T14:47:07.117467Z","iopub.status.idle":"2021-10-12T14:47:07.122517Z","shell.execute_reply.started":"2021-10-12T14:47:07.117417Z","shell.execute_reply":"2021-10-12T14:47:07.121593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ../input/prostate-cancer-grade-assessment","metadata":{"execution":{"iopub.status.busy":"2021-10-12T14:47:07.12391Z","iopub.execute_input":"2021-10-12T14:47:07.124637Z","iopub.status.idle":"2021-10-12T14:47:07.87616Z","shell.execute_reply.started":"2021-10-12T14:47:07.124596Z","shell.execute_reply":"2021-10-12T14:47:07.875132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!touch \"../input/prostate-cancer-grade-assessment/test.csv\"","metadata":{"execution":{"iopub.status.busy":"2021-10-12T14:47:07.877848Z","iopub.execute_input":"2021-10-12T14:47:07.878195Z","iopub.status.idle":"2021-10-12T14:47:08.606506Z","shell.execute_reply.started":"2021-10-12T14:47:07.878151Z","shell.execute_reply":"2021-10-12T14:47:08.605472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.read_csv(sample)\ntest_df = pd.read_csv(source/f'test.csv')\nif os.path.exists(submission_test_path):\n    #learn.load('prostate.pkl')\n    def get_inf(df=test_df):\n        filename = f'{submission_test_path}/{df.image_id}.tiff' \n        example2 = openslide.OpenSlide(str(filename))\n        ee = example2.get_thumbnail(size=(255, 255))\n        return tensor(ee)\n    \n    blocks = (\n          ImageBlock,\n          CategoryBlock\n          )\n    getters = [\n           get_inf,\n           ColReader('isup_grade')\n          ]\n\n    trends = DataBlock(blocks=blocks,\n              getters=getters,\n              item_tfms=Resize(224)\n              )\n    \n    #dls = trends.dataloaders(test_df, bs=32)\n    #learn = cnn_learner(dls, xresnet18)\n    \n    test_dl = dls.test_dl(test_df)\n    _,_,pred_classes = learn.get_preds(dl=test_dl, with_decoded=True)\n    \n    test_df[\"isup_grade\"] = pred_classes\n    sub = test_df[[\"image_id\",\"isup_grade\"]]\n    sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T14:47:08.610152Z","iopub.execute_input":"2021-10-12T14:47:08.610449Z","iopub.status.idle":"2021-10-12T14:47:08.635914Z","shell.execute_reply.started":"2021-10-12T14:47:08.610414Z","shell.execute_reply":"2021-10-12T14:47:08.635083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}