{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"The basic idea of this notebook is to train two separate models for MLO and CC views.\nI strongly believe that it can help in future aggregation of results at object-level.\n\nThis notebook utilises [Catalyst](https://catalyst-team.com/) DL framework. ","metadata":{}},{"cell_type":"code","source":"import catalyst\nfrom catalyst import utils\ncatalyst.utils.torch.get_device()","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:12:08.494733Z","iopub.execute_input":"2023-02-02T14:12:08.495326Z","iopub.status.idle":"2023-02-02T14:12:14.818793Z","shell.execute_reply.started":"2023-02-02T14:12:08.495209Z","shell.execute_reply":"2023-02-02T14:12:14.817638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"First of all we want to setup Hyperparams of pipeline","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision import transforms\nimport os\nimport cv2\nimport torch\nfrom torch import nn, optim\nfrom kaggle_secrets import UserSecretsClient\n\nN_SPLIT = 3","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-02T14:12:14.821361Z","iopub.execute_input":"2023-02-02T14:12:14.823102Z","iopub.status.idle":"2023-02-02T14:12:14.828614Z","shell.execute_reply.started":"2023-02-02T14:12:14.82306Z","shell.execute_reply":"2023-02-02T14:12:14.827768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INTERNET = True\ndebug = False\nPREDICT= False\nDOUBLE_GPU = True\nIMAGE_SIZE = 512\nif DOUBLE_GPU:\n    os.environ[\"CUDA_VISIBLE_DEVICES\"] = \"0,1\"\n\nPROJECT_NAME = f\"RSNAMammo_FS_{IMAGE_SIZE}\"\nuser_secrets = UserSecretsClient()\nWANDB_API_KEY = None\nif INTERNET:\n    WANDB_API_KEY = user_secrets.get_secret(\"VVV\")\n    os.environ['WANDB_API_KEY'] = WANDB_API_KEY\n\n\n\nTRAIN_PATH = \"/kaggle/input/rsna-cut-off-empty-space-from-images\"\nTEST_PATH = \"/kaggle/input/rsna-breast-cancer-detection/test_images\"\n\n\n# SPLIT_FRACTION = 0.95\nBATCH_SIZE = 10\nif DOUBLE_GPU:\n    BATCH_SIZE*=2\nMODEL = 'efficientnet_b4'\nEPOCHS = 20\nLR = 10e-4\nNUM_CLASS = 1\n\nif debug:\n    EPOCHS = 1","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:12:14.830336Z","iopub.execute_input":"2023-02-02T14:12:14.830703Z","iopub.status.idle":"2023-02-02T14:12:15.040122Z","shell.execute_reply.started":"2023-02-02T14:12:14.830667Z","shell.execute_reply":"2023-02-02T14:12:15.039137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def img2roi(img, is_dicom=False):\n    \"\"\"\n    Returns ROI area in other words \n    cuts the image to a desired one\n    \n    Because there are machine label tags,\n    undesired details out of the breast image.\n    \"\"\"\n    if not is_dicom:\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        \n    img = np.array(img * 255, dtype = np.uint8)\n    bin_img = cv2.threshold(img, 20, 255, cv2.THRESH_BINARY)[1]    \n    return img","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:12:15.041534Z","iopub.execute_input":"2023-02-02T14:12:15.041957Z","iopub.status.idle":"2023-02-02T14:12:15.051434Z","shell.execute_reply.started":"2023-02-02T14:12:15.041919Z","shell.execute_reply":"2023-02-02T14:12:15.050516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class RSNADataset(Dataset):\n    def __init__(self, df, img_folder, transform, is_test=False):\n        self.df = df\n        self.img_folder = img_folder\n        self.transform = transform\n        self.is_test = is_test\n    \n    def __getitem__(self, idx):\n        if self.is_test:\n            dcm_path = os.path.join(self.img_folder, self.df[\"dcm_path\"][idx])\n            img = read_dicom(dcm_path)\n            img = img2roi(img, is_dicom=True) \n        else:\n            img_path = os.path.join(self.img_folder, self.df[\"img_name\"][idx])\n            img = cv2.imread(img_path)\n            img = img2roi(img, is_dicom=False)\n        img = cv2.resize(img, (IMAGE_SIZE, IMAGE_SIZE))\n        if self.transform is not None:\n            img = self.transform(img)    \n        img = torch.tensor(img, dtype=torch.float)\n        if not self.is_test:\n            target = self.df[\"cancer\"][idx]\n            target = torch.tensor(target, dtype=torch.float)\n            return img, target\n        img = img.unsqueeze(0)\n        return img, -1.0\n    \n    def __len__(self):\n        return len(self.df)","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:12:15.055069Z","iopub.execute_input":"2023-02-02T14:12:15.05545Z","iopub.status.idle":"2023-02-02T14:12:15.066643Z","shell.execute_reply.started":"2023-02-02T14:12:15.055411Z","shell.execute_reply":"2023-02-02T14:12:15.065723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_transforms():\n    transform = transforms.Compose([\n            transforms.ToPILImage(),\n            transforms.RandomVerticalFlip(0.5),\n            transforms.RandomHorisontalFlip(0.5),\n            transforms.RandomRotation(degrees=(-10, 10)),\n            transforms.ToTensor(), \n            transforms.Normalize(mean=0.2179, std=0.0529)\n        ])\n    return transform","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:12:15.068408Z","iopub.execute_input":"2023-02-02T14:12:15.068787Z","iopub.status.idle":"2023-02-02T14:12:15.08449Z","shell.execute_reply.started":"2023-02-02T14:12:15.068753Z","shell.execute_reply":"2023-02-02T14:12:15.083578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sigmoid(x):\n    return 1/(1+np.exp(-x))","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:12:15.085686Z","iopub.execute_input":"2023-02-02T14:12:15.085971Z","iopub.status.idle":"2023-02-02T14:12:15.097259Z","shell.execute_reply.started":"2023-02-02T14:12:15.085946Z","shell.execute_reply":"2023-02-02T14:12:15.096143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Competition metic. We are going to utilise it for model results logging","metadata":{}},{"cell_type":"code","source":"def pfbeta(predictions, labels, beta=1.):\n    predictions = sigmoid(predictions.cpu().numpy())\n    labels = labels.cpu().numpy()\n\n    y_true_count = 0\n    ctp = 0\n    cfp = 0\n\n    for idx in range(len(labels)):\n        prediction = min(max(predictions[idx], 0), 1)\n        if (labels[idx]):\n            y_true_count += 1\n            ctp += prediction\n        else:\n            cfp += prediction\n\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp)\n    c_recall = ctp / max(y_true_count, 1)  # avoid / 0\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n        return result\n    else:\n        return 0","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:12:15.100511Z","iopub.execute_input":"2023-02-02T14:12:15.101585Z","iopub.status.idle":"2023-02-02T14:12:15.110476Z","shell.execute_reply.started":"2023-02-02T14:12:15.101549Z","shell.execute_reply":"2023-02-02T14:12:15.109521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ndf[\"img_name\"] = df[\"patient_id\"].astype(str) + \"/\" + df[\"image_id\"].astype(str) + \".png\"\ndf = df.sample(frac=1).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:12:15.11198Z","iopub.execute_input":"2023-02-02T14:12:15.112447Z","iopub.status.idle":"2023-02-02T14:12:15.349127Z","shell.execute_reply.started":"2023-02-02T14:12:15.112411Z","shell.execute_reply":"2023-02-02T14:12:15.348151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we are going to setup main class of Catalyst framework: Runner.\n\nRunner is an abstraction of train and inference loop of default training pipeline.\n\nFirst of all we define WandB logger. It just saves all the metrics we define in callbacks.\n\nThen we setup main criterion: BCEWithLogitsLoss.\n\nThen Callbacks come into play. I use Criterion Callback to compute loss.\n\nRunner is abstruction over train loop and every train loop needs it's loss. Criterion Callback just defines which tensors are targets and which- predictions to put into Criterion call. \n\nCheckpointCallback saves weights: it saves last checkpoint to *model.last.pth* at defined folder and based of *val_metric* saves *model.best.pth* ath the same place. \n\ntqdmCallback just shows how model training goes and helps to estimate run speed.\n\nFscore and Auc callbacks help to log additional metrics. Be carefull,  AUCCallback does not support fp16. But it's not that essential as far as we can use our main competitions evaluation metric: pfbeta score. It is logged with *SklearnLoaderCallback* as far, as it has prediction-target intefrace. \n\nAs optimizer I choose just Adam as default nice-to-start optimizer.\n\n","metadata":{}},{"cell_type":"code","source":"import sys\nsys.path.append('../input/timm-pytorch-image-models/pytorch-image-models-master')\n\nimport catalyst\n\ncatalyst.__version__ \n\nimport timm\nimport copy\nimport catalyst\nfrom catalyst.callbacks.metrics.classification import MultilabelPrecisionRecallF1SupportCallback\nif catalyst.__version__ == '21.08':\n    from catalyst.contrib.nn.criterion.focal import FocalLossBinary\nelse:\n    from catalyst.contrib.losses import FocalLossBinary\n\nfrom catalyst.runners import SupervisedRunner\nfrom typing import Any, Mapping\nfrom catalyst.loggers.wandb import WandbLogger\nimport catalyst.callbacks as dl\nfrom catalyst.metrics.functional._f1_score import fbeta_score\n\nfrom catalyst import metrics\n\nclass RSNAMammoRunner(SupervisedRunner):\n    def forward(self, batch: Mapping[str, Any], **kwargs) -> Mapping[str, Any]:\n        output = self._process_input(batch, **kwargs).squeeze()\n        output = self._process_output(output)\n        return output\n    \n    def get_loggers(self):\n        loggers = {}\n        if WANDB_API_KEY:\n            loggers[\"wandb\"] = WandbLogger(project=PROJECT_NAME, \n                                           group=f\"MODEL:{MODEL}|\" + self._type + \"|\" + self._name + self.get_criterion()._get_name(),\n                                           name=f\"LR:{LR}|EPOCHS:{EPOCHS}|{self.split_}\")\n        return loggers\n    \n    def get_criterion(self, **stuff):\n#         return FocalLossBinary()\n        return nn.BCEWithLogitsLoss()\n    \n    def get_callbacks(self, *params):\n        callbacks = {\n            \"criterion\": dl.CriterionCallback(\n                metric_key=\"loss\", input_key=\"logits\", target_key=\"targets\"\n            ),\n            \"checkpoint\": dl.CheckpointCallback(\n                self._logdir, loader_key=\"valid\", metric_key=\"loss\", minimize=True\n            ),\n            \"optimizer\": dl.OptimizerCallback(\n                metric_key=\"loss\",\n            ),\n            \"tqdm\": dl.TqdmCallback(),\n            \"F_score\" : dl.PrecisionRecallF1SupportCallback(\n                input_key=\"logits\", target_key=\"targets\", num_classes=NUM_CLASS\n            ),\n            \"AUC\" : dl.AUCCallback(\n                input_key=\"logits\", target_key=\"targets\", #num_classes=NUM_CLASS\n            ),\n            \"FBETA\": dl.SklearnLoaderCallback(\n                    keys={\"predictions\": \"logits\", \"labels\": \"targets\"},\n                    metric_fn=pfbeta,\n                    metric_key=\"pfbeta\",\n#                     average=\"macro\",\n#                     multi_class=\"ovo\"\n        )\n            \n        }\n        if catalyst.__version__ != '21.08':\n            callbacks['backward'] = dl.BackwardCallback(metric_key=\"loss\")\n        return callbacks\n\n    def get_optimizer(self, model, **stuff):\n#         if '1.7.1+cpu':\n#             return torch.optim.SGD(model.parameters(), lr=LR)\n#         else:\n        return torch.optim.Adam(model.parameters(), lr=LR)\n\ndef get_model():\n    out_dim = 1\n    backbone = timm.create_model(MODEL, pretrained=False, in_chans=1)\n    if 'resnet' in MODEL or 'resnext' in MODEL or 'inception' in MODEL:\n        backbone.fc = nn.Linear(backbone.fc.in_features, \n                             out_dim)\n    else:\n        backbone.classifier = nn.Linear(backbone.classifier.in_features, \n                                 out_dim)\n    return backbone","metadata":{"execution":{"iopub.status.busy":"2023-02-02T19:45:45.776029Z","iopub.execute_input":"2023-02-02T19:45:45.77643Z","iopub.status.idle":"2023-02-02T19:45:45.7952Z","shell.execute_reply.started":"2023-02-02T19:45:45.776397Z","shell.execute_reply":"2023-02-02T19:45:45.793631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_loader(df, sampler=None, shuffle=False, drop_last=False):\n    dataset = RSNADataset(df=df.reset_index(drop=True), \n                          img_folder=TRAIN_PATH, \n                          transform=get_transforms(),\n                         )\n    loader = DataLoader(dataset, \n                        batch_size=BATCH_SIZE, \n                        shuffle=shuffle, \n                        sampler=sampler,\n                        num_workers=2,\n                        drop_last=drop_last\n                       )\n    return loader","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:12:17.695245Z","iopub.execute_input":"2023-02-02T14:12:17.695948Z","iopub.status.idle":"2023-02-02T14:12:17.703132Z","shell.execute_reply.started":"2023-02-02T14:12:17.695906Z","shell.execute_reply":"2023-02-02T14:12:17.702013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedGroupKFold\ndef create_fold_split(df, fold=0, seed=451):\n    if fold >= N_SPLIT:\n        raise ValueError(f\"Fold number {fold} is bigger then total splits count {N_SPLIT}\")\n    cv = StratifiedGroupKFold(n_splits = N_SPLIT, shuffle=True, random_state=seed)\n    train_idx, test_idx = list(cv.split(df, \n                           df['cancer'], \n                           groups=df['patient_id']))[fold]\n    return df.iloc[train_idx], df.iloc[test_idx]","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:12:17.739772Z","iopub.execute_input":"2023-02-02T14:12:17.740463Z","iopub.status.idle":"2023-02-02T14:12:17.747229Z","shell.execute_reply.started":"2023-02-02T14:12:17.740425Z","shell.execute_reply":"2023-02-02T14:12:17.746273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I create deterministic train-val-test split to controll fitting in WandB interface. \nProportion of splits: 1 fold for test, N-1 folds for train and val. Val is just 1/10 of train.\n    So if we use 3 folds, we would have 1/3 of data for test, 1/15 for val and 9/15 for train. \n    I set up stratification by target + group balancig for each patient to avoid leaks. ","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedGroupKFold, StratifiedShuffleSplit\ndef create_split(df, seed=451):\n    cv = StratifiedGroupKFold(n_splits = N_SPLIT, shuffle=True, random_state=seed)\n    splits = list(cv.split(df, \n                          df['cancer'], \n                          groups=df['patient_id']))\n    train_val_test_split = []\n    for split in splits:\n        val_cv = StratifiedGroupKFold(n_splits = 10, shuffle=True, random_state=seed)\n        slice_df = df.loc[split[0]].copy()\n        train_idx, val_idx = list(val_cv.split(slice_df, \n                          slice_df['cancer'], \n                          groups=slice_df['patient_id']))[0]\n        train_idx, val_idx = slice_df.iloc[train_idx].index, slice_df.iloc[val_idx].index\n        train_val_test_split.append((train_idx, val_idx, split[1]))\n    return train_val_test_split","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:12:17.751604Z","iopub.execute_input":"2023-02-02T14:12:17.751934Z","iopub.status.idle":"2023-02-02T14:12:17.764174Z","shell.execute_reply.started":"2023-02-02T14:12:17.751908Z","shell.execute_reply":"2023-02-02T14:12:17.763335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_splits = np.array(create_split(df))\nnp.save('dataset_splits.npy', dataset_splits)","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:12:17.765249Z","iopub.execute_input":"2023-02-02T14:12:17.765612Z","iopub.status.idle":"2023-02-02T14:12:43.5081Z","shell.execute_reply.started":"2023-02-02T14:12:17.765584Z","shell.execute_reply":"2023-02-02T14:12:43.50686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def check_split(splits):\n    control_len = None\n    for split in splits:\n        assert len(split) == 3\n        train_idx, val_idx, test_idx = split\n        if control_len:\n            assert len(train_idx) + len(val_idx) + len(test_idx) == control_len\n        else:\n            control_len = len(train_idx) + len(val_idx) + len(test_idx)\n            print('Control len is ', control_len)\n        assert len(set(train_idx) & set(val_idx)) == 0\n        assert len(set(test_idx) & set(val_idx)) == 0\n        assert len(set(train_idx) & set(test_idx)) == 0\n        \ncheck_split(dataset_splits)","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:12:43.52071Z","iopub.execute_input":"2023-02-02T14:12:43.521247Z","iopub.status.idle":"2023-02-02T14:12:43.572887Z","shell.execute_reply.started":"2023-02-02T14:12:43.521205Z","shell.execute_reply":"2023-02-02T14:12:43.571882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And here is a train-loop. \nFirst the projection of images is chosen. The key idea is to create specific model for each projection and utilise features of every projection later.\nAfter it for each split data-loaders are created. For train loader I use DynamicBalanceClassSampler. We have high class imbalance in this task, so it helps to achieve stability.\n\nAfter it I just add some unnecessay atributes to my class (it's all for logging in WandB) and predict test fold for later analysis.","metadata":{}},{"cell_type":"code","source":"import wandb\nfrom catalyst.data import DynamicBalanceClassSampler\n\npredictors = {}\n\ndef select_projection(df, proj, idx):\n    df = df.loc[idx]\n    df = df[df['view'] == proj]\n    return df\n\nfor projection in ['MLO', 'CC']:\n    for split_num, split_ in enumerate(dataset_splits):\n        train_index, val_index, test_index= split_#create_fold_split(df, fold=split)\n        train_df = select_projection(df, projection, train_index)\n        val_df = select_projection(df, projection, val_index)\n        test_df = select_projection(df, projection, test_index)\n        if debug:\n            train_df = train_df.sample(frac=0.1)\n            test_df = test_df.sample(frac=0.1)\n#         train_df, val_df = create_fold_split(train_df, fold=split)\n\n        train_labels = train_df['cancer'].values.tolist()\n        sampler = DynamicBalanceClassSampler(train_labels)\n        train_loader = create_loader(train_df.reset_index(drop=True), \n                                     sampler=sampler, \n                                     drop_last=True)\n        val_loader = create_loader(val_df.reset_index(drop=True))\n        test_loader = create_loader(test_df.reset_index(drop=True))\n\n        runner = RSNAMammoRunner(model=get_model())\n        runner.split_ = split_num\n        runner._name = 'debug' if debug else runner.get_optimizer(get_model()).__class__.__name__\n        runner._type = train_df['view'].value_counts().index[0]\n        logdir = f\"./{MODEL}_{runner._type}/{runner._name}_{split_num}\"\n        \n        predictors[runner._type] = predictors.get(runner._type, []) + [logdir]\n        loaders = {'train': train_loader,  \n                   'valid': val_loader,\n                   'infer': test_loader}\n        \n#         scheduler = torch.optim.lr_scheduler.MultiStepLR(runner.optimizer_, [2])\n        \n        runner.train(loaders = loaders,\n#                      scheduler=scheduler,\n                         logdir = logdir, \n                         verbose = False,\n                         valid_loader=\"valid\",\n                         valid_metric=\"pfbeta\",\n                         num_epochs=EPOCHS,\n                         minimize_valid_metric=False,\n    #                      fp16=True\n                        )\n        model=get_model()\n        checkpoint = torch.load(os.path.join(logdir, 'model.best.pth'))\n        model.load_state_dict(checkpoint)\n        predictions = []\n        for prediction in runner.predict_loader(loader=test_loader, \n                                                model=model,\n                                                cpu=not torch.cuda.is_available(),\n                                                ):\n            predictions.extend(prediction['logits'].detach().cpu().tolist())\n\n        test_df['fold_pred'] = predictions\n        test_df.to_csv(f'{MODEL}_{runner._type}_{split_num}.csv')","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:12:43.574278Z","iopub.execute_input":"2023-02-02T14:12:43.574633Z","iopub.status.idle":"2023-02-02T14:13:23.816575Z","shell.execute_reply.started":"2023-02-02T14:12:43.574597Z","shell.execute_reply":"2023-02-02T14:13:23.812895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"VERSION = \"1.5\"  # possible version: [\"1.5\" , \"20200325\", \"nightly\"]\n!curl https://raw.githubusercontent.com/pytorch/xla/master/contrib/scripts/env-setup.py -o pytorch-xla-env-setup.py > /dev/null","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:13:23.821349Z","iopub.status.idle":"2023-02-02T14:13:23.824426Z","shell.execute_reply.started":"2023-02-02T14:13:23.824158Z","shell.execute_reply":"2023-02-02T14:13:23.824187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/rsna-bcd-whl-ds/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:13:23.828736Z","iopub.status.idle":"2023-02-02T14:13:23.831282Z","shell.execute_reply.started":"2023-02-02T14:13:23.831025Z","shell.execute_reply":"2023-02-02T14:13:23.831051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import dicomsdl\n\ndef read_dicom(path, fix_monochrome = True):\n    dicom = dicomsdl.open(path)\n    data = dicom.pixelData(storedvalue=True)  # storedvalue = True for int16 return otherwise float32\n    data = data - np.min(data)\n    data = data / np.max(data)\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = 1.0 - data\n    return data","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:13:23.83274Z","iopub.status.idle":"2023-02-02T14:13:23.837305Z","shell.execute_reply.started":"2023-02-02T14:13:23.837056Z","shell.execute_reply":"2023-02-02T14:13:23.837081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\n\ndef create_loader(df, projection, is_test=True):\n    projection_df = df[df['view'] == projection].copy().reset_index(drop=True)\n    projection_test_dataset=RSNADataset(df=projection_df, \n                                        img_folder=TEST_PATH, \n                                        transform=None, \n                                        is_test=is_test)\n    projection_test_loader = DataLoader(projection_test_dataset, \n                                        batch_size=BATCH_SIZE, \n                                        shuffle=False)\n    return projection_test_loader\n\ndef load_model(path):\n    model = get_model()\n    model.load_state_dict(torch.load(path))\n    return model\n\ndef predict_df(df, is_test=True, mean_proba = df['cancer'].mean()):\n    df = df.copy()\n    results = {}\n    for projection in predictors:\n        loader = create_loader(df, projection, is_test=is_test)\n        fold_preds = []\n        for model_path in predictors[projection]:\n            preds = []\n            model = load_model(os.path.join(model_path, 'model.best.pth'))\n            for pred in runner.predict_loader(loader = loader, \n                                              model=model, \n                                              cpu=not torch.cuda.is_available(), \n#                                               fp16=True\n                                             ):\n                preds.extend(pred['logits'].detach().cpu().numpy().tolist())\n            fold_preds.append(preds)\n        results[projection] = fold_preds\n\n    results_proba = {}\n    for projection in results:\n        results_proba[projection] = sigmoid(np.array(results[projection])).mean(axis=0)\n    df['pred_cancer'] = mean_proba\n    \n    for projection in results_proba:\n        df.loc[df['view'] == projection, 'pred_cancer'] = results_proba[projection]\n        \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:13:23.839006Z","iopub.status.idle":"2023-02-02T14:13:23.839697Z","shell.execute_reply.started":"2023-02-02T14:13:23.83947Z","shell.execute_reply":"2023-02-02T14:13:23.839492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\ndf_test[\"img_name\"] = df_test[\"patient_id\"].astype(str) + \"/\" + df_test[\"image_id\"].astype(str) + \".png\"\ndf_test[\"dcm_path\"] = df_test[\"patient_id\"].astype(str) + \"/\" + df_test[\"image_id\"].astype(str) + \".dcm\"\n\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:13:23.841014Z","iopub.status.idle":"2023-02-02T14:13:23.841832Z","shell.execute_reply.started":"2023-02-02T14:13:23.841559Z","shell.execute_reply":"2023-02-02T14:13:23.841585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if PREDICT:\n    prediction_df = predict_df(df_test)\n\n    prediction_id = prediction_df['patient_id'].astype(str) + \"_\" + prediction_df['laterality']\n\n    data = {\"prediction_id\": np.array(list((prediction_id))),\n            \"cancer\": prediction_df['pred_cancer'].values\n    }\n\n    sub_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv\")\n\n    sub_df = pd.DataFrame(data=data)\n\n    subb = sub_df.groupby('prediction_id')['cancer'].mean().to_frame().reset_index()\n\n    subb.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-02-02T14:13:23.843643Z","iopub.status.idle":"2023-02-02T14:13:23.851185Z","shell.execute_reply.started":"2023-02-02T14:13:23.850899Z","shell.execute_reply":"2023-02-02T14:13:23.850922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}