{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"try:\n    import pylibjpeg\nexcept:\n    !pip install -q /kaggle/input/rsna-2022-whl/{pydicom-2.3.0-py3-none-any.whl,pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}\n    !pip install -q /kaggle/input/rsna-bcd-whl-ds/python_gdcm-3.0.20-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n    # !pip install -q /kaggle/input/rsna-bcd-whl-ds/pylibjpeg-1.4.0-py3-none-any.whl\n    !pip install -q /kaggle/input/rsna-bcd-whl-ds/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl","metadata":{"execution":{"iopub.status.busy":"2023-01-04T20:16:24.3379Z","iopub.execute_input":"2023-01-04T20:16:24.338628Z","iopub.status.idle":"2023-01-04T20:16:24.349159Z","shell.execute_reply.started":"2023-01-04T20:16:24.338593Z","shell.execute_reply":"2023-01-04T20:16:24.347961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import clear_output, display_html\nimport os\nimport warnings\nfrom pathlib import Path\n\n# Basic libraries\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom tqdm import tqdm\n\nimport scipy as sc\nfrom scipy import stats\n\n# Train Test Split\nfrom sklearn.model_selection import train_test_split\n\n# Cross Validation\nfrom sklearn.model_selection import KFold, cross_val_score, StratifiedKFold, learning_curve, train_test_split\n\n# Tensorflow\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n\n# PyTorch \nimport torch\nfrom torch import nn, optim\nfrom torch.utils.data import DataLoader\nimport torchvision\nfrom torchvision import transforms\nimport sys\nsys.path.append('../input/timm-pytorch-image-models/pytorch-image-models-master')\nfrom timm import create_model\n\nfrom PIL import Image\n\n# Plotly\nimport plotly.express as px\nfrom plotly.subplots import make_subplots\nimport plotly.figure_factory as ff\nimport plotly.offline as offline\nimport plotly.graph_objs as go\n\nwarnings.filterwarnings('ignore')\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\nclear_output()","metadata":{"_uuid":"e2aa93bb-5c3e-4cb9-a4b3-8702400debc6","_cell_guid":"e520a595-7ec9-47cc-9242-48b9253c422b","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-01-04T20:16:24.396401Z","iopub.execute_input":"2023-01-04T20:16:24.396744Z","iopub.status.idle":"2023-01-04T20:16:24.409957Z","shell.execute_reply.started":"2023-01-04T20:16:24.396698Z","shell.execute_reply":"2023-01-04T20:16:24.408496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_data():\n    '''Load each of the datasets we are given.'''\n    \n    data_dir = Path(\"../input/rsna-breast-cancer-detection\")\n    #train = pd.read_csv(data_dir / \"train.csv\")\n    test = pd.read_csv(data_dir / \"test.csv\")\n    #sample_submission = pd.read_csv(data_dir / 'sample_submission.csv')\n    return test\n\ntest = load_data()\nclear_output()","metadata":{"_uuid":"8988c968-ec1d-4e59-bbd9-cca8fa17d754","_cell_guid":"6dcd04dc-99cf-4fb4-8c16-326e3c788854","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-01-04T20:16:24.423643Z","iopub.execute_input":"2023-01-04T20:16:24.423956Z","iopub.status.idle":"2023-01-04T20:16:24.435367Z","shell.execute_reply.started":"2023-01-04T20:16:24.423903Z","shell.execute_reply":"2023-01-04T20:16:24.434187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEBUG = False","metadata":{"_uuid":"cbc03ec4-8533-4011-bd9b-dc6c1161b029","_cell_guid":"777d3b72-dca7-4946-adba-acf4b2d8655c","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-01-04T20:16:24.439861Z","iopub.execute_input":"2023-01-04T20:16:24.440688Z","iopub.status.idle":"2023-01-04T20:16:24.446998Z","shell.execute_reply.started":"2023-01-04T20:16:24.440653Z","shell.execute_reply":"2023-01-04T20:16:24.445996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Competition's Metric","metadata":{"_uuid":"b1b43e0f-aa33-4259-aa09-1fcc0e996e78","_cell_guid":"2f924ae0-f740-46fb-bfbc-1778bc94fa21","trusted":true}},{"cell_type":"code","source":"def pfbeta(labels, predictions, beta=1.):\n    y_true_count = 0\n    ctp = 0\n    cfp = 0\n\n    for idx in range(len(labels)):\n        prediction = min(max(predictions[idx], 0), 1)\n        if (labels[idx]):\n            y_true_count += 1\n            ctp += prediction\n        else:\n            cfp += prediction\n\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp)\n    c_recall = ctp / max(y_true_count, 1)  # avoid / 0\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n        print(type(np.asarray(result.cpu())[0]))\n        return np.asarray(result.cpu())[0]\n    else:\n        return 0\n\ndef pfbeta_torch(labels, preds, beta=1):\n    preds = preds.clip(0, 1)\n    y_true_count = labels.sum()\n    ctp = preds[labels==1].sum()\n    cfp = preds[labels==0].sum()\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp)\n    c_recall = ctp / y_true_count\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n        return result\n    else:\n        return 0.0","metadata":{"_uuid":"9873ec3a-593c-4a50-aa77-2d8547a57be6","_cell_guid":"65b29f4b-f7c0-44fe-894c-bef354f81d13","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-01-04T20:16:24.46028Z","iopub.execute_input":"2023-01-04T20:16:24.461174Z","iopub.status.idle":"2023-01-04T20:16:24.472069Z","shell.execute_reply.started":"2023-01-04T20:16:24.461135Z","shell.execute_reply":"2023-01-04T20:16:24.471135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Convertir Imágenes Test de .dcm a .png","metadata":{}},{"cell_type":"code","source":"!pip install /kaggle/input/rsnamodules/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl ","metadata":{"execution":{"iopub.status.busy":"2023-01-04T20:16:24.473716Z","iopub.execute_input":"2023-01-04T20:16:24.474499Z","iopub.status.idle":"2023-01-04T20:16:54.774494Z","shell.execute_reply.started":"2023-01-04T20:16:24.474464Z","shell.execute_reply":"2023-01-04T20:16:54.773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nimport glob\nfrom concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor\nimport pydicom\nimport dicomsdl as dicoml\nfrom joblib import Parallel, delayed\nimport cv2\n\nfrom concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor\n\ndef process(f, size=512, save_folder=None, dicom_process = True, extension=\"png\"):\n    \n    patient = f.split('/')[-2]\n    image = f.split('/')[-1][:-4]\n    dicom = pydicom.dcmread(f)\n    img = dicom.pixel_array\n    img = (img - img.min()) / (img.max() - img.min())\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n    img = cv2.resize(img, (size, size))\n    \n    # 2. Crop\n    X = img\n    # Some images have narrow exterior \"frames\" that complicate selection of the main data. Cutting off the frame\n    X = X[5:-5, 5:-5]\n    \n    # regions of non-empty pixels\n    output= cv2.connectedComponentsWithStats((X > 0.05).astype(np.uint8)[:, :], 8, cv2.CV_32S)\n\n    # stats.shape == (N, 5), where N is the number of regions, 5 dimensions correspond to:\n    # left, top, width, height, area_size\n    stats = output[2]\n    \n    # finding max area which always corresponds to the breast data. \n    idx = stats[1:, 4].argmax() + 1\n    x1, y1, w, h = stats[idx][:4]\n    x2 = x1 + w\n    y2 = y1 + h\n    \n    # cutting out the breast data\n    X_fit = X[y1: y2, x1: x2]\n    \n    patient_id, im_id = os.path.basename(os.path.dirname(f)), os.path.basename(f)[:-4]\n    os.makedirs(f'{PNG_TEST_IMAGES_PATH}/test_images/{patient_id}', exist_ok=True)\n    cv2.imwrite(f'{PNG_TEST_IMAGES_PATH}/test_images/{patient_id}/{im_id}.png', (X_fit[:, :] * 255).astype(np.uint8))\n    \n    \ndef fit_all_images(all_images):\n    with ThreadPoolExecutor(4) as p:\n        for i in tqdm(p.map(process, all_images), total=len(all_images)):\n            pass\n\nPNG_TEST_IMAGES_PATH = f'test'\n\nall_images = glob.glob('/kaggle/input/rsna-breast-cancer-detection/test_images/*/*') \n# all_images = glob.glob('/kaggle/input/rsna-breast-cancer-detection/train_images/10006/*')\nfit_all_images(all_images)","metadata":{"execution":{"iopub.status.busy":"2023-01-04T20:16:54.777548Z","iopub.execute_input":"2023-01-04T20:16:54.778016Z","iopub.status.idle":"2023-01-04T20:16:57.025454Z","shell.execute_reply.started":"2023-01-04T20:16:54.777968Z","shell.execute_reply":"2023-01-04T20:16:57.024475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transforms","metadata":{}},{"cell_type":"code","source":"import albumentations\nfrom albumentations.augmentations.crops.transforms import RandomResizedCrop\n\ndef get_transforms(crop=False):\n    def transforms(img, laterality):\n        \n        if laterality == 'R':\n            img = horizontal_flip(img)\n            \n            #tfm = [\n            #    torchvision.transforms.RandomHorizontalFlip(0.5),\n            #    torchvision.transforms.RandomRotation(degrees=(-5, 5)), \n            #    torchvision.transforms.RandomResizedCrop((1024, 512), scale=(0.8, 1), ratio=(0.45, 0.55)) \n            #]\n        #img = torchvision.transforms.RandomResizedCrop((1024, 512), scale=(0.8, 1), ratio=(0.45, 0.55))(torch.tensor(img))\n        #print(img.shape)\n        #img = cv2.resize(img, (512, 1024))  \n        if crop == True: \n            random_crop = RandomResizedCrop(1024, 512, scale=(0.8, 1), ratio=(0.45, 0.55))\n            img = random_crop(image=img)['image']\n        #img = torchvision.transforms.RandomResizedCrop((1024, 512), scale=(0.8, 1), ratio=(0.45, 0.55))(torch.tensor(img))\n        #print(img)\n        #img = img.resize(img.shape[1], img.shape[2], img.shape[0])\n        \n        return img\n\n    return lambda img, laterality: transforms(img, laterality)","metadata":{"execution":{"iopub.status.busy":"2023-01-04T20:16:57.026843Z","iopub.execute_input":"2023-01-04T20:16:57.027487Z","iopub.status.idle":"2023-01-04T20:16:57.035265Z","shell.execute_reply.started":"2023-01-04T20:16:57.027445Z","shell.execute_reply":"2023-01-04T20:16:57.033951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset","metadata":{"_uuid":"3ab1f220-05b4-41f3-a576-11316b5f27db","_cell_guid":"807b9abd-1e53-4663-a7c7-4283791a5fb5","trusted":true}},{"cell_type":"code","source":"import cv2\n\ndef horizontal_flip(img):\n    '''Flips horizontally the image given as parameter. '''\n    return cv2.flip(img, 1)","metadata":{"execution":{"iopub.status.busy":"2023-01-04T20:16:57.038244Z","iopub.execute_input":"2023-01-04T20:16:57.039022Z","iopub.status.idle":"2023-01-04T20:16:57.050658Z","shell.execute_reply.started":"2023-01-04T20:16:57.038974Z","shell.execute_reply":"2023-01-04T20:16:57.049604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BreastCancerDataSet(torch.utils.data.Dataset):\n    def __init__(self, df, path, transforms = False):\n        super().__init__()\n        self.df = df\n        self.path = path\n        self.transforms = transforms\n\n    def __getitem__(self, i):\n        path = f'{self.path}/{self.df.iloc[i].patient_id}/{self.df.iloc[i].image_id}.png'\n        try:\n            img = cv2.imread(path)\n            #print(img.shape)\n            #img = Image.open(path).convert('RGB')\n        except Exception as ex:\n            print(path, ex)\n            return None\n        \n        if self.transforms is not None: \n            img = self.transforms(img, self.df.iloc[i].laterality)\n\n        return img\n\n    def __len__(self):\n        return len(self.df)","metadata":{"_uuid":"ff9757d6-d3bb-457b-bbc4-e9e96b3670bf","_cell_guid":"9967c420-90c4-4e3e-94c3-281bdd9314a8","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-01-04T20:16:57.05231Z","iopub.execute_input":"2023-01-04T20:16:57.052699Z","iopub.status.idle":"2023-01-04T20:16:57.063075Z","shell.execute_reply.started":"2023-01-04T20:16:57.052663Z","shell.execute_reply":"2023-01-04T20:16:57.062128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{"_uuid":"f423ceaa-8951-40e2-b3a7-4ed198822ccd","_cell_guid":"3caff67a-e46d-4454-8da5-167e89d842fd","trusted":true}},{"cell_type":"code","source":"from torchvision.models import efficientnet\nimport cv2\nimport timm \n\nclass BreastCancerModel(nn.Module):\n    def __init__(self, model_type = 'efficientnet_b0'):\n        super().__init__()\n        #self.backbone = efficientnet.efficientnet_b0(pretrained = True)\n        self.backbone = timm.create_model(model_type, pretrained=False)\n        self.backbone_dim = self.backbone(torch.randn(1, 3, 512, 512)).shape[-1]\n        self.head = nn.Sequential(\n            nn.Linear(self.backbone_dim, 1), \n            nn.Sigmoid(), \n        )\n                \n    def forward(self, x):\n        x = self.backbone(x)\n        x = self.head(x)\n        return x\n    \n    def predict(self, x):\n        return self.forward(x)","metadata":{"_uuid":"605ad529-447d-4bf2-aa96-7661193d21e2","_cell_guid":"63e1dffc-0d5e-4b95-983b-c5d5b9e7e491","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-01-04T20:16:57.064462Z","iopub.execute_input":"2023-01-04T20:16:57.065618Z","iopub.status.idle":"2023-01-04T20:16:57.077063Z","shell.execute_reply.started":"2023-01-04T20:16:57.065584Z","shell.execute_reply":"2023-01-04T20:16:57.076109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_model(path, model_type = 'efficientnet_b4'): \n    model = BreastCancerModel(model_type)\n    model.load_state_dict(torch.load(path))\n    model.eval()\n    return model\n\npath = '../input/rsna-baseline/state_dict_model.pt'\nmodel = load_model(path, 'efficientnet_b4')","metadata":{"execution":{"iopub.status.busy":"2023-01-04T20:16:57.078481Z","iopub.execute_input":"2023-01-04T20:16:57.079514Z","iopub.status.idle":"2023-01-04T20:16:58.978788Z","shell.execute_reply.started":"2023-01-04T20:16:57.079479Z","shell.execute_reply":"2023-01-04T20:16:58.977791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"def load_model(name, dir='.'):\n    data = torch.load(os.path.join(dir, f'{name}'), map_location=DEVICE)\n    model = BreastCancerModel(data['model_type'])\n    model.load_state_dict(data['model'])\n    # print(data['threshold'], data['model_type'])\n    return model, data['threshold'], data['model_type']","metadata":{}},{"cell_type":"code","source":"# Config device\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nmodel = model.to(device)\n\nPNG_TEST_IMAGES_PATH = f'test/test_images'\n\ntesting_data = BreastCancerDataSet(test, PNG_TEST_IMAGES_PATH, get_transforms(crop=True))\ntest_dataloader = DataLoader(testing_data, batch_size = 4, shuffle=False, num_workers=os.cpu_count())","metadata":{"_uuid":"7c85254d-3666-45ff-a8a5-89298b8164fd","_cell_guid":"866b371e-6bb3-4199-85e2-8fa261681b61","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-01-04T20:16:58.980649Z","iopub.execute_input":"2023-01-04T20:16:58.981054Z","iopub.status.idle":"2023-01-04T20:16:59.029548Z","shell.execute_reply.started":"2023-01-04T20:16:58.981013Z","shell.execute_reply":"2023-01-04T20:16:59.028628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predictions\npreds = []\n\nwith torch.no_grad():\n    for imgs in tqdm(test_dataloader): \n        imgs = imgs.to(device)\n        preds.append(model.predict(imgs.permute(0,3,1,2).to(torch.float)).cpu())\n    \npreds","metadata":{"_uuid":"af22222b-1f4b-499f-89d2-c959056b16eb","_cell_guid":"bef71e11-d7db-4037-bce6-4c0a169ca1a9","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-01-04T20:16:59.03089Z","iopub.execute_input":"2023-01-04T20:16:59.031458Z","iopub.status.idle":"2023-01-04T20:16:59.388648Z","shell.execute_reply.started":"2023-01-04T20:16:59.031417Z","shell.execute_reply":"2023-01-04T20:16:59.3873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = torch.concat(preds).detach().numpy()    \npredictions","metadata":{"execution":{"iopub.status.busy":"2023-01-04T20:16:59.392902Z","iopub.execute_input":"2023-01-04T20:16:59.393288Z","iopub.status.idle":"2023-01-04T20:16:59.401383Z","shell.execute_reply.started":"2023-01-04T20:16:59.393251Z","shell.execute_reply":"2023-01-04T20:16:59.400241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['cancer'] = predictions\ndf_sub = test.groupby('prediction_id')[['cancer']].mean()\ndf_sub","metadata":{"execution":{"iopub.status.busy":"2023-01-04T20:16:59.403264Z","iopub.execute_input":"2023-01-04T20:16:59.403817Z","iopub.status.idle":"2023-01-04T20:16:59.422569Z","shell.execute_reply.started":"2023-01-04T20:16:59.403622Z","shell.execute_reply":"2023-01-04T20:16:59.421414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub = df_sub.cancer.astype(float)\ndf_sub.to_csv('submission.csv', index=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-04T20:16:59.424369Z","iopub.execute_input":"2023-01-04T20:16:59.424761Z","iopub.status.idle":"2023-01-04T20:16:59.43319Z","shell.execute_reply.started":"2023-01-04T20:16:59.424723Z","shell.execute_reply":"2023-01-04T20:16:59.431992Z"},"trusted":true},"execution_count":null,"outputs":[]}]}