{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Install libraries and environment needed","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:05:52.25984Z","iopub.execute_input":"2023-03-12T07:05:52.260244Z","iopub.status.idle":"2023-03-12T07:05:52.284494Z","shell.execute_reply.started":"2023-03-12T07:05:52.260137Z","shell.execute_reply":"2023-03-12T07:05:52.283653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    import pylibjpeg\nexcept:\n    # The following *.whl files were collected from these pip packages:\n    #!pip install -U \"python-gdcm\" pydicom pylibjpeg    # Required for JPEG decompression. See: https://www.kaggle.com/competitions/rsna-2022-cervical-spine-fracture-detection/discussion/341412\n    #!pip install -U torchvision                        # For EfficientNetV2\n\n    # Offline dependencies:\n    !mkdir -p /root/.cache/torch/hub/checkpoints/\n    !cp ../input/rsna-2022-whl/efficientnet_v2_s-dd5fe13b.pth  /root/.cache/torch/hub/checkpoints/\n    !pip install /kaggle/input/rsna-2022-whl/{pydicom-2.3.0-py3-none-any.whl,pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}\n    !pip install /kaggle/input/rsna-2022-whl/{torch-1.12.1-cp37-cp37m-manylinux1_x86_64.whl,torchvision-0.13.1-cp37-cp37m-manylinux1_x86_64.whl}","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:05:52.285868Z","iopub.execute_input":"2023-03-12T07:05:52.286122Z","iopub.status.idle":"2023-03-12T07:07:45.039762Z","shell.execute_reply.started":"2023-03-12T07:05:52.286098Z","shell.execute_reply":"2023-03-12T07:07:45.038538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install dicomsdl pytorch_lightning timm --no-index --find-links=../input/rbcd-downloads","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:07:45.04204Z","iopub.execute_input":"2023-03-12T07:07:45.042797Z","iopub.status.idle":"2023-03-12T07:07:56.436272Z","shell.execute_reply.started":"2023-03-12T07:07:45.042749Z","shell.execute_reply":"2023-03-12T07:07:56.435097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.path.append('/kaggle/input/timm-pytorch-image-models/pytorch-image-models-master/')\nimport timm\n\nimport numpy as np\nimport pandas as pd\nimport glob\n\nimport os\nimport cv2\nimport random\nimport gc\nimport wandb\nfrom pathlib import Path\nimport multiprocessing as mp\nfrom joblib import delayed\nfrom joblib import Parallel\nimport dicomsdl\n\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.preprocessing import LabelEncoder, normalize\n\nfrom tqdm import tqdm\n\nimport torch\nimport torchvision\nfrom torch.utils.data import Dataset,DataLoader\nimport torch.nn as nn\nfrom torch.optim import Adam\n\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:07:56.438368Z","iopub.execute_input":"2023-03-12T07:07:56.439128Z","iopub.status.idle":"2023-03-12T07:07:59.825336Z","shell.execute_reply.started":"2023-03-12T07:07:56.439078Z","shell.execute_reply":"2023-03-12T07:07:59.824259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"KAGGLE_DIR = Path(\"/\") / \"kaggle\"\n\nINPUT_DIR = KAGGLE_DIR / \"input\"\nOUTPUT_DIR = KAGGLE_DIR / \"working\"\n\nDATA_ROOT_DIR = INPUT_DIR / \"rsna-breast-cancer-detection\"\n\nTEST_IMAGES_DIR = DATA_ROOT_DIR / \"test_images\"\nTEST_CSV_PATH = DATA_ROOT_DIR / \"test.csv\"\n\nOUTPUT_TEST_IMAGES_DIR = OUTPUT_DIR / \"test_images\"\nOUTPUT_TEST_IMAGES_DIR.mkdir(exist_ok=True)\n\nACCELERATOR = \"gpu\"\nBATCH_SIZE = 16\nDEVICES = 1\nIMAGE_SIZE = 1024\nNUM_WORKERS = mp.cpu_count()\nPRECISION = 16","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:07:59.828289Z","iopub.execute_input":"2023-03-12T07:07:59.828898Z","iopub.status.idle":"2023-03-12T07:07:59.839262Z","shell.execute_reply.started":"2023-03-12T07:07:59.828857Z","shell.execute_reply":"2023-03-12T07:07:59.838111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntest = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n\nimage_paths = sorted(TEST_IMAGES_DIR.glob(\"*/*.dcm\"))\nimage_paths ","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:07:59.840875Z","iopub.execute_input":"2023-03-12T07:07:59.841334Z","iopub.status.idle":"2023-03-12T07:07:59.99706Z","shell.execute_reply.started":"2023-03-12T07:07:59.841299Z","shell.execute_reply":"2023-03-12T07:07:59.996141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_dcm_to_png(image_path, size, output_image_dir):\n    patient_id = image_path.parent.name\n    image_id = image_path.stem\n\n    dicom = dicomsdl.open(str(image_path))\n    img = dicom.pixelData()\n\n    img = (img - img.min()) / (img.max() - img.min())\n\n    if dicom.getPixelDataInfo()[\"PhotometricInterpretation\"] == \"MONOCHROME1\":\n        img = 1 - img\n\n    img = cv2.resize(img, (size, size))\n\n    output_image_path = output_image_dir / f\"{patient_id}_{image_id}.png\"\n    cv2.imwrite(str(output_image_path), (img * 255).astype(np.uint8))\n\n\n_ = Parallel(n_jobs=NUM_WORKERS)(\n    delayed(convert_dcm_to_png)(image_path, size=IMAGE_SIZE, output_image_dir=OUTPUT_TEST_IMAGES_DIR)\n    for image_path in tqdm(image_paths)\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:07:59.998389Z","iopub.execute_input":"2023-03-12T07:07:59.998827Z","iopub.status.idle":"2023-03-12T07:08:02.486822Z","shell.execute_reply.started":"2023-03-12T07:07:59.998765Z","shell.execute_reply":"2023-03-12T07:08:02.485736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\nTEST_DIR = '/kaggle/working/test_images'\n\ndef get_test_path(row):\n    patient = row['patient_id']\n    image = row['image_id']\n    return os.path.join(TEST_DIR, f\"{patient}_{image}.png\")\n\ntest['path']  = test.apply(get_test_path,axis = 1)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:08:02.488837Z","iopub.execute_input":"2023-03-12T07:08:02.489237Z","iopub.status.idle":"2023-03-12T07:08:02.506452Z","shell.execute_reply.started":"2023-03-12T07:08:02.489193Z","shell.execute_reply":"2023-03-12T07:08:02.505472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class RSNADataset():\n    def __init__(self, df,transform=None): \n        self.df = df\n        self.transform = transform\n        if 'cancer' in df.columns:\n            self.ds_type='train'\n        else:\n            self.ds_type='test'\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self,idx):\n        row = self.df.iloc[idx]\n        img_path =row['path']\n        image = Image.open(img_path)\n#         csv_columns = ['laterality', 'view', 'age', 'implant']\n#         csv_data = np.array(row[csv_columns].values, \n#                             dtype=np.float32)\n        csv_data=0\n        if self.transform is not None:\n            image = self.transform(image)\n        image=torch.cat((image,image,image),0)\n    \n        if self.ds_type=='train':\n            label=torch.tensor(row['cancer'])\n            return {'image':image,'label': label}\n        else:\n            return {'image':image}\n\n\n\n\n# test/test_images/10008/68070693.png\n\n\n\n    \nclass BreastCancerModel(torch.nn.Module):\n    def __init__(self, model_type, pretrained=True,dropout=0.):\n        super().__init__()        \n        self.model = timm.create_model(model_type, pretrained=pretrained, num_classes=0, drop_rate=dropout)\n\n        self.backbone_dim = self.model(torch.randn(1, 3, 256, 256)).shape[-1]\n        \n        self.nn_cancer = torch.nn.Sequential(\n            torch.nn.Linear(self.backbone_dim, 1),\n        )\n\n    def forward(self, x):\n        x = self.model(x)\n        cancer = self.nn_cancer(x).squeeze()\n        return cancer\n    \n    def predict(self, x):\n        preds=torch.sigmoid(self.forward(x))\n        return preds","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:08:02.509442Z","iopub.execute_input":"2023-03-12T07:08:02.509708Z","iopub.status.idle":"2023-03-12T07:08:02.707244Z","shell.execute_reply.started":"2023-03-12T07:08:02.509682Z","shell.execute_reply":"2023-03-12T07:08:02.706277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_transforms(data):\n    def transforms(img):\n        if data=='train':\n            tfm = [\n                torchvision.transforms.RandomHorizontalFlip(0.5),\n                torchvision.transforms.RandomRotation(degrees=(-5, 5)), \n#                 torchvision.transforms.RandomResizedCrop((512, 256), scale=(0.8, 1), ratio=(0.45, 0.55)) \n                                torchvision.transforms.Resize((512, 256))\n\n            ]\n        elif data=='valid':\n            tfm = [\n#                 torchvision.transforms.RandomHorizontalFlip(0.5),\n                torchvision.transforms.Resize((512, 256))\n            ]\n        img = torchvision.transforms.Compose(tfm + [            \n            torchvision.transforms.ToTensor(),\n            torchvision.transforms.Normalize(mean=0.2179, std=0.0529),\n        ])(img)\n        return img\n\n    return lambda img: transforms(img)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:08:02.70909Z","iopub.execute_input":"2023-03-12T07:08:02.709725Z","iopub.status.idle":"2023-03-12T07:08:02.719793Z","shell.execute_reply.started":"2023-03-12T07:08:02.709684Z","shell.execute_reply":"2023-03-12T07:08:02.718522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def models_predict(models):\n\n    TEST_PATH='/kaggle/input/rsna-bcd-roi-1024x-png-dataset/test_images'\n    test_ds=RSNADataset(test, get_transforms('train'))\n    \n    dl_test = torch.utils.data.DataLoader(test_ds, batch_size=16, \n                                          shuffle=False, num_workers=2)\n    predictions=torch.tensor([])\n    models_predictions={'model-0':torch.tensor([]), \n                       'model-1':torch.tensor([]),\n                       'model-2':torch.tensor([]),\n                       'model-3':torch.tensor([]),\n                       'model-4':torch.tensor([])}\n    for data in tqdm(dl_test):\n        images=data['image'].to(DEVICE)\n        pos=0\n        for model in models:\n            preds=model.predict(images)\n            models_predictions[f'model-{pos}']=torch.cat((models_predictions[f'model-{pos}'], preds.clone().detach().cpu()),0)\n            pos+=1\n    return models_predictions\n","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:08:02.723684Z","iopub.execute_input":"2023-03-12T07:08:02.724053Z","iopub.status.idle":"2023-03-12T07:08:02.734755Z","shell.execute_reply.started":"2023-03-12T07:08:02.724009Z","shell.execute_reply":"2023-03-12T07:08:02.733733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:08:02.736087Z","iopub.execute_input":"2023-03-12T07:08:02.736611Z","iopub.status.idle":"2023-03-12T07:08:02.809432Z","shell.execute_reply.started":"2023-03-12T07:08:02.736575Z","shell.execute_reply":"2023-03-12T07:08:02.808361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = []\n\nweights_path=[f'/kaggle/input/rsna-resnexr50-32x4d/weight_resnext50_32x4d/seresnext50_32x4d_epoch_{pos}_v1.pth' for pos in range(0,5)]\n\nfor weights in tqdm(weights_path):\n    model=BreastCancerModel('seresnext50_32x4d', pretrained=False)\n    model.load_state_dict(torch.load(weights,\n                                     map_location=DEVICE))\n    model = model.to(DEVICE)\n    models.append(model)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:13:02.550672Z","iopub.execute_input":"2023-03-12T07:13:02.551099Z","iopub.status.idle":"2023-03-12T07:13:11.053798Z","shell.execute_reply.started":"2023-03-12T07:13:02.551064Z","shell.execute_reply":"2023-03-12T07:13:11.052675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models_pred = models_predict(models)\npredictions=0\nMODELS_THRESH=[0.36, 0.38, 0.37, 0.00, 0.30]\nfor pos in range(0,5):\n    predictions+=(models_pred[f'model-{pos}']>MODELS_THRESH[pos]).float()\n\n# predictions=(predictions>THRES)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:13:15.696998Z","iopub.execute_input":"2023-03-12T07:13:15.69738Z","iopub.status.idle":"2023-03-12T07:13:16.431524Z","shell.execute_reply.started":"2023-03-12T07:13:15.697345Z","shell.execute_reply":"2023-03-12T07:13:16.430281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions\ntest['cancer']=predictions","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:13:18.993412Z","iopub.execute_input":"2023-03-12T07:13:18.993799Z","iopub.status.idle":"2023-03-12T07:13:19Z","shell.execute_reply.started":"2023-03-12T07:13:18.993762Z","shell.execute_reply":"2023-03-12T07:13:18.998951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub = test.groupby('prediction_id')[['cancer']].max()\ndf_sub","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:13:22.45327Z","iopub.execute_input":"2023-03-12T07:13:22.453782Z","iopub.status.idle":"2023-03-12T07:13:22.493816Z","shell.execute_reply.started":"2023-03-12T07:13:22.453735Z","shell.execute_reply":"2023-03-12T07:13:22.492792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"THRES=0.9\ndf_sub['cancer'] = (df_sub.cancer>THRES).astype(float)\ndf_sub","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:13:24.487153Z","iopub.execute_input":"2023-03-12T07:13:24.488132Z","iopub.status.idle":"2023-03-12T07:13:24.499999Z","shell.execute_reply.started":"2023-03-12T07:13:24.488094Z","shell.execute_reply":"2023-03-12T07:13:24.498652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub.to_csv('submission.csv', index=True)\n!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-03-12T07:13:27.544783Z","iopub.execute_input":"2023-03-12T07:13:27.545152Z","iopub.status.idle":"2023-03-12T07:13:28.575254Z","shell.execute_reply.started":"2023-03-12T07:13:27.545119Z","shell.execute_reply":"2023-03-12T07:13:28.573785Z"},"trusted":true},"execution_count":null,"outputs":[]}]}