{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install /kaggle/input/rsna-2022-whl/{pydicom-2.3.0-py3-none-any.whl,pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}\n!pip install -q /kaggle/input/nvidia-dali-nightly-cuda110-1230dev/nvidia_dali_nightly_cuda110-1.23.0.dev20230203-7187866-py3-none-manylinux2014_x86_64.whl\n!pip install /kaggle/input/nvidia-dali-wheel/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl\n\n\n\nuse_compile = False\n# if use_compile:\n#upgrade pytorch to 1.12\n!pip install /kaggle/input/pytorch112-cu113/{torch-1.12.1+cu113-cp37-cp37m-linux_x86_64.whl,torchvision-0.13.1+cu113-cp37-cp37m-linux_x86_64.whl}\n!pip install /kaggle/input/torch-tensorrt-pkg/nvidia_pyindex-1.0.9-py3-none-any.whl\n!mkdir -p /tmp/pip/cache/\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia-cublas-cu11-2022.4.8.xyz /tmp/pip/cache/nvidia-cublas-cu11-2022.4.8.tar.gz\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia-cuda-runtime-cu11-2022.4.25.xyz /tmp/pip/cache/nvidia-cuda-runtime-cu11-2022.4.25.tar.gz\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia-cudnn-cu11-2022.5.19.xyz /tmp/pip/cache/nvidia-cudnn-cu11-2022.5.19.tar.gz\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia_cublas_cu117-11.10.1.25-py3-none-manylinux1_x86_64.whl /tmp/pip/cache/\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia_cuda_runtime_cu117-11.7.60-py3-none-manylinux1_x86_64.whl /tmp/pip/cache/\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia_cudnn_cu116-8.4.0.27-py3-none-manylinux1_x86_64.whl /tmp/pip/cache/\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia_tensorrt-8.4.3.1-cp37-none-linux_x86_64.whl /tmp/pip/cache/\n!pip install --no-index --find-links /tmp/pip/cache/ nvidia_tensorrt\n#install torch_tensorrt\n!pip install /kaggle/input/torch-tensorrt-pkg/torch_tensorrt-1.2.0-cp37-cp37m-linux_x86_64.whl\n\nimport os\nos.environ['CUDA_MODULE_LOADING']='LAZY'\nimport torch_tensorrt\nimport tensorrt","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-13T09:43:01.043784Z","iopub.execute_input":"2023-06-13T09:43:01.044909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn.functional as F\nimport nvidia.dali.fn as fn\nimport nvidia.dali.types as types\nfrom nvidia.dali import pipeline_def\nfrom nvidia.dali.types import DALIDataType\nfrom pydicom.filebase import DicomBytesIO\nfrom nvidia.dali.plugin.pytorch import to_torch_type","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:37:40.4438Z","iopub.execute_input":"2023-06-13T09:37:40.444517Z","iopub.status.idle":"2023-06-13T09:37:41.133034Z","shell.execute_reply.started":"2023-06-13T09:37:40.444476Z","shell.execute_reply":"2023-06-13T09:37:41.132059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#we need to patch DALI for Int16 support\n\n\nfrom nvidia.dali.backend import TensorGPU, TensorListGPU\nfrom nvidia.dali.pipeline import Pipeline\nimport nvidia.dali.ops as ops\nfrom nvidia.dali import types\nfrom nvidia.dali.plugin.base_iterator import _DaliBaseIterator\nfrom nvidia.dali.plugin.base_iterator import LastBatchPolicy\nimport torch\nimport torch.utils.dlpack as torch_dlpack\nimport ctypes\nimport numpy as np\nimport torch.nn.functional as F\nimport pydicom\n\nto_torch_type = {\n    types.DALIDataType.FLOAT:   torch.float32,\n    types.DALIDataType.FLOAT64: torch.float64,\n    types.DALIDataType.FLOAT16: torch.float16,\n    types.DALIDataType.UINT8:   torch.uint8,\n    types.DALIDataType.INT8:    torch.int8,\n    types.DALIDataType.UINT16:  torch.int16,\n    types.DALIDataType.INT16:   torch.int16,\n    types.DALIDataType.INT32:   torch.int32,\n    types.DALIDataType.INT64:   torch.int64\n}\n\n\ndef feed_ndarray(dali_tensor, arr, cuda_stream=None):\n    \"\"\"\n    Copy contents of DALI tensor to PyTorch's Tensor.\n\n    Parameters\n    ----------\n    `dali_tensor` : nvidia.dali.backend.TensorCPU or nvidia.dali.backend.TensorGPU\n                    Tensor from which to copy\n    `arr` : torch.Tensor\n            Destination of the copy\n    `cuda_stream` : torch.cuda.Stream, cudaStream_t or any value that can be cast to cudaStream_t.\n                    CUDA stream to be used for the copy\n                    (if not provided, an internal user stream will be selected)\n                    In most cases, using pytorch's current stream is expected (for example,\n                    if we are copying to a tensor allocated with torch.zeros(...))\n    \"\"\"\n    dali_type = to_torch_type[dali_tensor.dtype]\n\n    assert dali_type == arr.dtype, (\"The element type of DALI Tensor/TensorList\"\n                                    \" doesn't match the element type of the target PyTorch Tensor: \"\n                                    \"{} vs {}\".format(dali_type, arr.dtype))\n    assert dali_tensor.shape() == list(arr.size()), \\\n        (\"Shapes do not match: DALI tensor has size {0}, but PyTorch Tensor has size {1}\".\n            format(dali_tensor.shape(), list(arr.size())))\n    cuda_stream = types._raw_cuda_stream(cuda_stream)\n\n    # turn raw int to a c void pointer\n    c_type_pointer = ctypes.c_void_p(arr.data_ptr())\n    if isinstance(dali_tensor, (TensorGPU, TensorListGPU)):\n        stream = None if cuda_stream is None else ctypes.c_void_p(cuda_stream)\n        dali_tensor.copy_to_external(c_type_pointer, stream, non_blocking=True)\n    else:\n        dali_tensor.copy_to_external(c_type_pointer)\n    return arr","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:37:44.359292Z","iopub.execute_input":"2023-06-13T09:37:44.359687Z","iopub.status.idle":"2023-06-13T09:37:44.377683Z","shell.execute_reply.started":"2023-06-13T09:37:44.359654Z","shell.execute_reply":"2023-06-13T09:37:44.376704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install -q --no-index --find-links=/kaggle/input/timm-wheel timm pillow\n!pip uninstall -y timm\nimport sys\nsys.path.append('/kaggle/input/timm-0-6-12-tf-effv2s-1520-912')","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:37:46.671812Z","iopub.execute_input":"2023-06-13T09:37:46.672195Z","iopub.status.idle":"2023-06-13T09:37:48.745619Z","shell.execute_reply.started":"2023-06-13T09:37:46.672162Z","shell.execute_reply":"2023-06-13T09:37:48.744381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gdcm\nimport glob\n\nimport os\nimport pickle\nimport random\nimport math\n\nfrom joblib import Parallel, delayed\nimport shutil\nimport cv2\nfrom tqdm.auto import tqdm\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import KFold, StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nimport timm\nfrom albumentations  import *\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.nn import Parameter\nfrom torch.optim import Adam, SGD, AdamW\nfrom torch.utils.data import DataLoader, Dataset\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom pydicom.pixel_data_handlers import apply_windowing\nimport dicomsdl\nimport gc\n\n%matplotlib inline\nimport matplotlib.pyplot as plt\n\nprint('torch version:', torch.__version__)\nprint('timm version:', timm.__version__)\n\ndevice = 'cuda' if torch.cuda.is_available() else 'cpu'\nprint('device:', device)\n\ntorch.backends.cuda.matmul.allow_tf32 = False","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:37:50.1573Z","iopub.execute_input":"2023-06-13T09:37:50.157686Z","iopub.status.idle":"2023-06-13T09:37:54.321086Z","shell.execute_reply.started":"2023-06-13T09:37:50.15765Z","shell.execute_reply":"2023-06-13T09:37:54.319726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Configuration","metadata":{}},{"cell_type":"code","source":"class config:\n    seed = 10\n    batch_size = 32\n    n_folds = 5\n    name = 'tf_efficientnetv2_s'\n    SIZE = (1024, 1024)\nclass config2:\n    seed = 10\n    batch_size = 32\n    n_folds = 4\n    name = 'tf_efficientnetv2_s'\n    SIZE = (912, 1520)\n    \ndef seed_everything(seed: int):    \n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\nseed_everything(config.seed)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:37:59.305121Z","iopub.execute_input":"2023-06-13T09:37:59.305816Z","iopub.status.idle":"2023-06-13T09:37:59.322896Z","shell.execute_reply.started":"2023-06-13T09:37:59.305771Z","shell.execute_reply":"2023-06-13T09:37:59.321778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ntrain_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain_df.drop(index=[16772, 16773, 16774, 16775, 16776, 53065], inplace=True)\ndisplay(test_df.head())\ndisplay(test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:38:01.55738Z","iopub.execute_input":"2023-06-13T09:38:01.557752Z","iopub.status.idle":"2023-06-13T09:38:01.727335Z","shell.execute_reply.started":"2023-06-13T09:38:01.557721Z","shell.execute_reply":"2023-06-13T09:38:01.726404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"windowing_stat_dict = dict()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:38:02.880131Z","iopub.execute_input":"2023-06-13T09:38:02.880508Z","iopub.status.idle":"2023-06-13T09:38:02.88547Z","shell.execute_reply.started":"2023-06-13T09:38:02.880475Z","shell.execute_reply":"2023-06-13T09:38:02.88438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utils","metadata":{}},{"cell_type":"code","source":"def convert_dicom_to_jpg(patient_id, img_id, save_folder=\"\"):\n    patient = patient_id\n    image = img_id\n    file = IMG_PATH + f\"{patient}/{image}.dcm\"\n    dcmfile = pydicom.dcmread(file)\n\n    if dcmfile.file_meta.TransferSyntaxUID == '1.2.840.10008.1.2.4.90':\n        with open(file, 'rb') as fp:\n            raw = DicomBytesIO(fp.read())\n            ds = pydicom.dcmread(raw)\n        offset = ds.PixelData.find(b\"\\x00\\x00\\x00\\x0C\")  #<---- the jpeg2000 header info we're looking for\n        hackedbitstream = bytearray()\n        hackedbitstream.extend(ds.PixelData[offset:])\n        with open(save_folder + f\"{patient}_{image}.jpg\", \"wb\") as binary_file:\n            binary_file.write(hackedbitstream)\n            \n    if dcmfile.file_meta.TransferSyntaxUID == '1.2.840.10008.1.2.4.70':\n        with open(file, 'rb') as fp:\n            raw = DicomBytesIO(fp.read())\n            ds = pydicom.dcmread(raw)\n        offset = ds.PixelData.find(b\"\\xff\\xd8\\xff\\xe0\")  #<---- the jpeg lossless header info we're looking for\n        hackedbitstream = bytearray()\n        hackedbitstream.extend(ds.PixelData[offset:])\n        with open(save_folder + f\"{patient}_{image}.jpg\", \"wb\") as binary_file:\n            binary_file.write(hackedbitstream)\n        \n\n@pipeline_def\ndef jpg_decode_pipeline(jpgfiles):\n    jpegs, _ = fn.readers.file(files=jpgfiles)\n    images = fn.experimental.decoders.image(jpegs, device='mixed', output_type=types.ANY_DATA, dtype=DALIDataType.UINT16)\n    return images\n\n\n\n# https://www.kaggle.com/code/hengck23/3hr-tensorrt-nextvit-example\ndef normalised_to_8bit(image, photometric_interpretation):\n    if photometric_interpretation == 'MONOCHROME1':\n        image = image.max()  - image\n        rev = 1\n    else:\n        rev = 0\n    xmin = image.min()\n    xmax = image.max() \n    norm = np.empty_like(image, dtype=np.uint8)\n    image = np.ascontiguousarray(image)\n    dicomsdl.util.convert_to_uint8(image, norm, xmin, xmax)\n\n    return norm, rev, xmin, xmax\n\n# https://www.kaggle.com/code/hengck23/3hr-tensorrt-nextvit-example\ndef dicomsdl_to_numpy_image(ds, index=0):\n    # https://stackoverflow.com/questions/44659924/returning-numpy-arrays-via-pybind11\n    info = ds.getPixelDataInfo()\n\n    shape = [info['Rows'], info['Cols']]\n    dtype = info['dtype']\n    outarr = np.empty(shape, dtype=dtype)\n    ds.copyFrameData(index, outarr)\n    return outarr\n\n\n# https://www.kaggle.com/code/raddar/convert-dicom-to-np-array-the-correct-way/notebook\ndef save_array(patient_id, img_id, images_dict, voi_lut = False, fix_monochrome = True):\n    path = IMG_PATH + f\"{patient_id}/{img_id}.dcm\"\n    \n    dicom = pydicom.dcmread(path)\n\n    if f\"{patient_id}/{img_id}.png\" in images_dict:\n        return None\n    \n    dicom = dicomsdl.open(path)      \n    data = dicomsdl_to_numpy_image(dicom)\n    data = data[5:-5, 5:-5]  \n    rev_val = np.amax(data) # for sigmoid windowing\n    data, rev, min_, max_ = normalised_to_8bit(data, dicom.getPixelDataInfo()['PhotometricInterpretation'])\n    if type(dicom.WindowWidth) == list:\n        center = dicom.WindowCenter[0]\n        width = dicom.WindowWidth[0]\n    else:\n        center = dicom.WindowCenter\n        width = dicom.WindowWidth\n    y_range = 2**dicom.BitsStored - 1\n    windowing_stat_dict[f'{img_id}'] = [min_, max_, center, width, y_range, rev, rev_val]\n    \n    data = cp.array(data)\n    data1 = ExtractBreast_cupy(data)\n    h, w = data1.shape\n    if (h == 0) or (w == 0):\n#         data = cv2.resize(data1, config2.SIZE, interpolation = cv2.INTER_AREA)\n        data2 = cv2.resize(data, config.SIZE, interpolation = cv2.INTER_AREA)\n        data = cv2.resize(data, config2.SIZE, interpolation = cv2.INTER_AREA)\n    else:\n        c_h = data1.shape[0]\n        if c_h > 2048:\n            pad = c_h - 2048\n            data2 = data1[pad//2:-pad//2, :]\n        else:\n            data2 = data1 # data1.copy()\n\n        data = cv2.resize(data1, config2.SIZE, interpolation = cv2.INTER_AREA)\n        data2 = cv2.resize(data2, config.SIZE, interpolation = cv2.INTER_AREA)\n#     cv2.imwrite(SAVE_FOLDER + f\"{patient_id}/{img_id}.png\", data)\n#     cv2.imwrite(SAVE_FOLDER + f\"{patient_id}/{img_id}-2.png\", data2)\n    \n#     dummy_images_dict = {}\n#     dummy_images_dict[f\"{patient_id}/{img_id}.png\"] = data\n#     dummy_images_dict[f\"{patient_id}/{img_id}-2.png\"] = data2\n#     return dummy_images_dict\n    return data, data2","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:38:04.486229Z","iopub.execute_input":"2023-06-13T09:38:04.486604Z","iopub.status.idle":"2023-06-13T09:38:04.509958Z","shell.execute_reply.started":"2023-06-13T09:38:04.486573Z","shell.execute_reply":"2023-06-13T09:38:04.509027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1. Since background in a image has constant values in a row- and col-wise direction,\n# the breast of interest can be extracted by detecting non-constant elements.\n# 2. Even if some objects other than breasts were seen in a image,\n# because they are usually smaller, we can extract the breast by picking up the larger one.\n# Heuristic way may not be always perfect....\n\n# Count up the continuing \"1\"\n# ex. [0,1,1,0,1,0,0,1,1,1,0] -> [-1,2,2,-1,1,-1,-1,3,3,3,-1]\n# Pytroch\ndef CountUpContinuingOnes_torch(b_arr):\n    # indice continuing zeros from left side.\n    # [0,1,1,0,1,0,0,1,1,1,0] -> [0,0,0,3,3,5,6,6,6,6,10]\n    left = torch.arange(len(b_arr))\n    left[b_arr>0] = 0\n    left = torch.cummax(left, dim=-1)[0]\n    \n    # from right side.\n    # [0,1,1,0,1,0,0,1,1,1,0] -> [0,3,3,3,5,5,6,10,10,10,10]\n    rev_arr = torch.flip(b_arr, [-1])\n    right = torch.arange(len(rev_arr))\n    right[rev_arr>0] = 0\n    right = torch.cummax(right, dim=-1)[0]\n    right = len(rev_arr) - 1 - torch.flip(right, [-1])\n    \n    return right - left - 1\n\nimport cupy as cp\n# Count up the continuing \"1\"\n# Ex. [0,1,1,0,1,0,0,1,1,1,0] -> [-1,2,2,-1,1,-1,-1,3,3,3,-1]\n# Cupy\ndef CountUpContinuingOnes_cupy(b_arr):\n    # indice continuing zeros from left side.\n    # ex: [0,1,1,0,1,0,0,1,1,1,0] -> [0,0,0,3,3,5,6,6,6,6,10]\n    left = cp.arange(len(b_arr))\n    left[b_arr>0] = 0\n    left = cupy_cumulative_max(left)\n    \n    # from right side.\n    # ex: [0,1,1,0,1,0,0,1,1,1,0] -> [0,3,3,3,5,5,6,10,10,10,10]\n    rev_arr = b_arr[::-1]\n    right = cp.arange(len(rev_arr))\n    right[rev_arr>0] = 0\n    right = cupy_cumulative_max(right)\n    right = len(rev_arr) - 1 - right[::-1]\n    \n    return right - left - 1\n\nimport cupy as cp\ndef cupy_cumulative_max(arr):\n    cumulative_max = cp.empty_like(arr)\n    current_max = arr[0]\n    cumulative_max[0] = current_max\n    for i in range(1, len(arr)):\n        current_max = cp.maximum(current_max, arr[i])\n        cumulative_max[i] = current_max\n    return cumulative_max\n\n\ndef ExtractBreast_cupy(img):\n    img_copy = img.copy()\n    img = cp.where(img <= 40, 0, img) # To detect backgrounds easily\n    height, _ = img.shape\n    \n    # whether each col is non-constant or not\n    y_a = height // 2 + int(height*0.4)\n    y_b = height // 2 - int(height*0.4)\n    b_arr = img[y_b:y_a].std(axis=0) != 0\n    continuing_ones = CountUpContinuingOnes_cupy(b_arr)\n    # longest should be the breast\n    col_ind = cp.where(continuing_ones == continuing_ones.max())[0]\n    img = img[:, col_ind]\n    \n    # whether each row is non-constant or not\n    _, width = img.shape\n    x_a = width // 2 + int(width*0.4)\n    x_b = width // 2 - int(width*0.4)\n    b_arr = img[:,x_b:x_a].std(axis=1) != 0\n    continuing_ones = CountUpContinuingOnes_cupy(b_arr)\n    # longest should be the breast\n    row_ind = cp.where(continuing_ones == continuing_ones.max())[0]\n    \n    return cp.asnumpy(img_copy[row_ind][:, col_ind])\n\n\ndef ExtractBreast_torch(img):\n    img_copy = torch.clone(img)\n    img = torch.where(img <= 40, torch.zeros_like(img), img) # To detect backgrounds easily\n    height, _ = img.shape\n    \n    # whether each col is non-constant or not\n    y_a = height // 2 + int(height*0.4)\n    y_b = height // 2 - int(height*0.4)\n    b_arr = torch.std(img[y_b:y_a], axis=0) != 0\n    continuing_ones = CountUpContinuingOnes_torch(b_arr)\n    # longest should be the breast\n    col_ind = torch.where(continuing_ones == continuing_ones.max())[0]\n    img = img[:, col_ind]\n    \n    # whether each row is non-constant or not\n    _, width = img.shape\n    x_a = width // 2 + int(width*0.4)\n    x_b = width // 2 - int(width*0.4)\n#     b_arr = img[:,x_b:x_a].std(axis=1) != 0\n    b_arr = torch.std(img[:, x_b:x_a], axis=1) != 0\n    continuing_ones = CountUpContinuingOnes_torch(b_arr)\n    # longest should be the breast\n    row_ind = torch.where(continuing_ones == continuing_ones.max())[0]\n    \n    return img_copy[row_ind][:, col_ind]\n","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:38:13.653308Z","iopub.execute_input":"2023-06-13T09:38:13.653691Z","iopub.status.idle":"2023-06-13T09:38:14.481593Z","shell.execute_reply.started":"2023-06-13T09:38:13.653656Z","shell.execute_reply":"2023-06-13T09:38:14.480624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset","metadata":{}},{"cell_type":"code","source":"mean_2 = 0.340530064\nstd_2 = 0.224438599\nclass MammoDataset2(Dataset):\n    def __init__(self, df, train=True, tfms=None):\n        self.df = df\n        self.train = train\n        self.tfms = tfms\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        data = self.df.iloc[idx]\n        img = images_dict[f\"{data['patient_id']}/{data['image_id']}-2.png\"]\n        if self.tfms:\n            augmented = self.tfms(image=img)\n            img = augmented['image']\n        img = img.astype('float32')\n        img -= img.min()\n        img /= img.max()\n        img = torch.tensor((img - mean_2)/std_2, dtype=torch.float32)\n        if self.train:\n            return img.unsqueeze(0), torch.tensor(data['cancer'], dtype=torch.long)\n        else:\n            return img.unsqueeze(0)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:38:21.908812Z","iopub.execute_input":"2023-06-13T09:38:21.909218Z","iopub.status.idle":"2023-06-13T09:38:21.920512Z","shell.execute_reply.started":"2023-06-13T09:38:21.909184Z","shell.execute_reply":"2023-06-13T09:38:21.919378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"def gem(x, p:int=3, eps:float=1e-6):\n    return F.avg_pool2d(x.clamp(min=eps).pow(p), (x.size(-2), x.size(-1))).pow(1.0 / p)\n\n\nclass GeM(nn.Module):\n    def __init__(self, p:int=3, eps:float=1e-6, p_trainable=False):\n        super(GeM, self).__init__()\n        if p_trainable:\n            self.p = Parameter(torch.ones(1) * p)\n        else:\n            self.p = p\n        self.eps = eps\n\n    def forward(self, x):\n        ret = gem(x, p=self.p, eps=self.eps)\n        return ret\n\n    def __repr__(self):\n        return (\n            self.__class__.__name__\n            + \"(\"\n            + \"p=\"\n            + \"{:.4f}\".format(self.p.data.tolist()[0])\n            + \", \"\n            + \"eps=\"\n            + str(self.eps)\n            + \")\"\n        )\n    \nclass MammoModel(nn.Module):\n    def __init__(self, name, *, pretrained=False, in_chans=1, p=3, p_trainable=False, eps=1e-6):\n        super().__init__()\n        model = timm.create_model(name, pretrained=pretrained, in_chans=in_chans)\n        clsf = model.default_cfg['classifier']\n        n_features = model._modules[clsf].in_features\n        model._modules[clsf] = nn.Identity()\n        \n        self.fc = nn.Linear(n_features, 1)\n        self.model = model\n\n        self.pool = nn.Sequential(\n            GeM(p=p, eps=eps, p_trainable=p_trainable),\n            nn.Flatten())\n    \n    def forward(self, x):\n        # x = self.model(x)\n        x = self.model.forward_features(x)\n        x = self.pool(x)\n        logits = self.fc(x)\n        return logits","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:38:22.807585Z","iopub.execute_input":"2023-06-13T09:38:22.807956Z","iopub.status.idle":"2023-06-13T09:38:22.824517Z","shell.execute_reply.started":"2023-06-13T09:38:22.807922Z","shell.execute_reply":"2023-06-13T09:38:22.823301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def inference_fn(test_loader, model, device):\n    model.eval()\n    preds = []\n    for inputs in test_loader:\n        inputs = inputs.to(device)\n        with torch.no_grad():\n            y_preds = model(inputs)\n        preds.append(y_preds.squeeze().sigmoid().to('cpu').numpy())\n    predictions = np.concatenate(preds)\n    return predictions\n\n\nimport torch.cuda.amp as amp\ndef inference_fn_compile(test_loader, model, device, batch_size):\n    model.eval()\n    preds = []\n    for inputs in test_loader:\n#         inputs = inputs.to(device).half()\n        inputs = pad_to_batch_size(inputs, batch_size)\n        inputs = inputs.to(device)\n        with torch.no_grad():\n            with amp.autocast(enabled=True):\n                y_preds = model(inputs)\n        preds.append(y_preds.squeeze().sigmoid().to('cpu').numpy())\n    predictions = np.concatenate(preds)\n    return predictions\n\n\ndef inference_fn_compile_tta(test_loader, model, device, batch_size):\n    model.eval()\n    preds = []\n    for inputs in tqdm(test_loader):\n#         inputs = inputs.to(device).half()\n        inputs = inputs.to(device)\n        inputs = pad_to_batch_size(inputs, batch_size=batch_size)\n        with torch.no_grad():\n            with amp.autocast(enabled=True):\n                y_preds = model(inputs).squeeze().sigmoid() / 2\n                # vertical flip tta\n                y_preds += model(torch.flip(inputs, [-2])).squeeze().sigmoid() / 2\n        preds.append(y_preds.to('cpu').numpy())\n    predictions = np.concatenate(preds)\n    return predictions\n\n\ndef pad_to_batch_size(image, batch_size):\n    B = len(image)\n    if B == batch_size:\n        return image\n    pad = F.pad(input=image, pad=(0, 0, 0, 0, 0, 0, 0, batch_size - B), mode='constant', value=0)\n    return pad","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:38:24.661386Z","iopub.execute_input":"2023-06-13T09:38:24.661764Z","iopub.status.idle":"2023-06-13T09:38:24.678294Z","shell.execute_reply.started":"2023-06-13T09:38:24.661732Z","shell.execute_reply":"2023-06-13T09:38:24.676734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile model_compile.py\n# https://www.kaggle.com/code/tivfrvqhs5/torch-tensorrt-infer-fp16-and-fp32-benchmarks/notebook\n# https://www.kaggle.com/code/hengck23/3hr-tensorrt-nextvit-example/notebook\n# https://www.kaggle.com/competitions/rsna-breast-cancer-detection/discussion/375881\n\nimport sys\nsys.path.append('/kaggle/input/timm-0-6-12-tf-effv2s-1520-912')\n\nimport gdcm\nimport glob\n\nimport os\nimport pickle\nimport random\nimport math\n\nfrom joblib import Parallel, delayed\nimport shutil\nimport cv2\nfrom tqdm.auto import tqdm\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import KFold, StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nimport timm\nfrom albumentations  import *\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.nn import Parameter\nfrom torch.optim import Adam, SGD, AdamW\nfrom torch.utils.data import DataLoader, Dataset\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom pydicom.pixel_data_handlers import apply_windowing\nimport dicomsdl\nimport gc\nimport torch.backends.cudnn as cudnn\n\nimport matplotlib.pyplot as plt\n\nprint('torch version:', torch.__version__)\nprint('timm version:', timm.__version__)\n\ndevice = 'cuda' if torch.cuda.is_available() else 'cpu'\nprint('device:', device)\n\nimport torch\nimport torch.nn.functional as F\nimport nvidia.dali.fn as fn\nimport nvidia.dali.types as types\nfrom nvidia.dali import pipeline_def\nfrom nvidia.dali.types import DALIDataType\nfrom pydicom.filebase import DicomBytesIO\nfrom nvidia.dali.plugin.pytorch import feed_ndarray, to_torch_type\n\n\nos.environ['CUDA_MODULE_LOADING']='LAZY'\n\nimport torch_tensorrt\nimport tensorrt\ncudnn.benchmark = True\n\nclass config:\n    seed = 10\n    batch_size = 16\n    n_folds = 5\n    name = 'tf_efficientnetv2_s'\n    SIZE = (1024, 1024)  \nclass config2:\n    seed = 10\n    batch_size = 16\n    n_folds = 4\n    name = 'tf_efficientnetv2_s'\n    SIZE = (912, 1520)\n    \n    \ndef seed_everything(seed: int):    \n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\nseed_everything(config.seed)\n\ndef gem(x, p:int=3, eps:float=1e-6):\n    return F.avg_pool2d(x.clamp(min=eps).pow(p), (x.size(-2), x.size(-1))).pow(1.0 / p)\n\n\nclass GeM(nn.Module):\n    def __init__(self, p:int=3, eps:float=1e-6, p_trainable=False):\n        super(GeM, self).__init__()\n        if p_trainable:\n            self.p = Parameter(torch.ones(1) * p)\n        else:\n            self.p = p\n        self.eps = eps\n\n    def forward(self, x):\n        ret = gem(x, p=self.p, eps=self.eps)\n        return ret\n\n    def __repr__(self):\n        return (\n            self.__class__.__name__\n            + \"(\"\n            + \"p=\"\n            + \"{:.4f}\".format(self.p.data.tolist()[0])\n            + \", \"\n            + \"eps=\"\n            + str(self.eps)\n            + \")\"\n        )\n    \nclass MammoModel(nn.Module):\n    def __init__(self, name, *, pretrained=False, in_chans=1, p=3, p_trainable=False, eps=1e-6):\n        super().__init__()\n        model = timm.create_model(name, pretrained=pretrained, in_chans=in_chans)\n        clsf = model.default_cfg['classifier']\n        n_features = model._modules[clsf].in_features\n        model._modules[clsf] = nn.Identity()\n        \n        self.fc = nn.Linear(n_features, 1)\n        self.model = model\n\n        self.pool = nn.Sequential(\n            GeM(p=p, eps=eps, p_trainable=p_trainable),\n            nn.Flatten())\n    \n    def forward(self, x):\n        # x = self.model(x)\n        x = self.model.forward_features(x)\n        x = self.pool(x)\n        logits = self.fc(x)\n        return logits\n    \ndef tensorrt_compile_v2s(path, config, name, fold):\n    model = MammoModel(config.name, pretrained=False)\n    current_params = model.state_dict()\n    trained_params = torch.load(path, map_location='cpu')['model']\n    for key in current_params.keys():\n        if key == 'model.conv_stem.conv.weight':\n            current_params[key] = trained_params['model.conv_stem.weight']\n        elif key == 'model.blocks.1.0.conv_exp.conv.weight':\n            current_params[key] = trained_params['model.blocks.1.0.conv_exp.weight']\n        elif key == 'model.blocks.2.0.conv_exp.conv.weight':\n            current_params[key] = trained_params['model.blocks.2.0.conv_exp.weight']\n        elif key == 'model.blocks.3.0.conv_dw.conv.weight':\n            current_params[key] = trained_params['model.blocks.3.0.conv_dw.weight']\n        else:\n            current_params[key] = trained_params[key]\n    model.load_state_dict(current_params)\n#     model.half()  ### should not be used !!!!!\n    model.eval().to(device)\n    # model.encoder.merge_bn()\n\n    # The compiled module will have precision as specified by \"op_precision\".\n    # with torch_tensorrt.logging.debug():\n    trt_model_fp16 = torch_tensorrt.compile(\n        model,\n        inputs=[\n            torch_tensorrt.Input(\n            [config.batch_size*2, 1, config.SIZE[1], config.SIZE[0]],\n            dtype=torch.float32\n        )],\n        enabled_precisions={torch.float32},  # Run with FP16\n        workspace_size=1 << 32,\n    #     debug=True,\n        require_full_compilation=True,\n    ) \n    torch.jit.save(trt_model_fp16, f'{name}-{fold}.ts')\n    \n    del trt_model_fp16\n    torch.cuda.empty_cache()\n    gc.collect()\n    print('ok')\n    \n    \ndef tensorrt_compile_B5(path, config, name, fold):\n    model = MammoModel(config.name, pretrained=False)\n    current_params = model.state_dict()\n    trained_params = torch.load(path, map_location='cpu')['model']\n    for key in current_params.keys():\n        if key == 'model.conv_stem.conv.weight':\n            current_params[key] = trained_params['model.conv_stem.weight']\n        elif key == 'model.blocks.1.0.conv_dw.conv.weight':\n            current_params[key] = trained_params['model.blocks.1.0.conv_dw.weight']\n        elif key == 'model.blocks.2.0.conv_dw.conv.weight':\n            current_params[key] = trained_params['model.blocks.2.0.conv_dw.weight']\n        elif key == 'model.blocks.3.0.conv_dw.conv.weight':\n            current_params[key] = trained_params['model.blocks.3.0.conv_dw.weight']\n        else:\n            current_params[key] = trained_params[key]\n    model.load_state_dict(current_params)\n#     model.half()  ### should not be used !!!!!\n    model.eval().to(device)\n    # model.encoder.merge_bn()\n\n    # The compiled module will have precision as specified by \"op_precision\".\n    # with torch_tensorrt.logging.debug():\n    trt_model_fp16 = torch_tensorrt.compile(\n        model,\n        inputs=[\n            torch_tensorrt.Input(\n            [config.batch_size*2, 1, config.SIZE[1], config.SIZE[0]],\n            dtype=torch.float32\n        )],\n        enabled_precisions={torch.float32},  # Run with FP16\n        workspace_size=1 << 32,\n    #     debug=True,\n        require_full_compilation=True,\n    ) \n    torch.jit.save(trt_model_fp16, f'{name}-{fold}.ts')\n    \n    del trt_model_fp16\n    torch.cuda.empty_cache()\n    gc.collect()\n    print('ok')\n\n\nfor fold in range(config.n_folds):\n    print('fold:', fold)\n    path = f'/kaggle/input/ver49-weight3/efficientnetv2_s_seed_10_fold{fold}_best_ver49_swa.pth'\n    name = 'ver49'\n    tensorrt_compile_v2s(path, config, name, fold)\n    \nfor fold in range(config2.n_folds):\n    print('fold:', fold)\n    path = f'/kaggle/input/rsna-efficientnetv2-s-pl-4folds/efficientnetv2_s_pl_seed_10_fold{fold}_best_score_ver084.pth'\n    name = 'ver84'\n    tensorrt_compile_v2s(path, config2, name, fold)\n    \nfor fold in range(config3.n_folds):\n    print('fold:', fold)\n    path = f'/kaggle/input/rsna-efficientnetb5-pl-4folds/efficientnetb5_pl_seed_10_fold{fold}_best_score_ver083.pth'\n    name = 'ver83'\n    tensorrt_compile_B5(path, config3, name, fold)\n\nprint('trt_fp16 ok')","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:38:27.098478Z","iopub.execute_input":"2023-06-13T09:38:27.098855Z","iopub.status.idle":"2023-06-13T09:38:27.114282Z","shell.execute_reply.started":"2023-06-13T09:38:27.098815Z","shell.execute_reply":"2023-06-13T09:38:27.112973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nif use_compile:\n    !python model_compile.py\nelse:\n    !rm model_compile.py","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:38:28.973473Z","iopub.execute_input":"2023-06-13T09:38:28.973841Z","iopub.status.idle":"2023-06-13T09:38:30.099832Z","shell.execute_reply.started":"2023-06-13T09:38:28.973808Z","shell.execute_reply":"2023-06-13T09:38:30.098274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"phase = 'test' # 'test'\n\nif phase == 'train':\n    df = train_df[:5000]\nelse:\n    if len(test_df) < 100:\n        phase = 'train'\n        df = train_df[:100]\n    else:\n        df = test_df\n    \n\nIMG_PATH = f\"/kaggle/input/rsna-breast-cancer-detection/{phase}_images/\"\n# test_images = glob.glob(f\"{IMG_PATH}*/*.dcm\")\n    \nprint(\"Number of images :\", len(df))\n\nSAVE_FOLDER = f'{phase}_images/'\n\n\nos.makedirs(SAVE_FOLDER, exist_ok=True)\nfor patient_id in df['patient_id'].unique():\n    os.makedirs(SAVE_FOLDER+str(patient_id), exist_ok=True)\n\nif len(df) > 10000:\n    N_CHUNKS = 100\nelse:\n    N_CHUNKS = 1\n\nCHUNKS = [(len(df) / N_CHUNKS * k, len(df) / N_CHUNKS * (k + 1)) for k in range(N_CHUNKS)]\nCHUNKS = np.array(CHUNKS).astype(int)\n\n# J2K_FOLDER = \"/tmp/j2k/\"\nJPG_FOLDER = \"/tmp/jpg/\"\n\nif len(df) <= 10000:\n    debug_flag = True\nelse:\n    debug_flag = False\n\nif debug_flag:\n    config.n_folds = 2","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:38:31.019176Z","iopub.execute_input":"2023-06-13T09:38:31.020331Z","iopub.status.idle":"2023-06-13T09:38:31.039692Z","shell.execute_reply.started":"2023-06-13T09:38:31.020274Z","shell.execute_reply":"2023-06-13T09:38:31.038684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generate images","metadata":{}},{"cell_type":"code","source":"csize = len(df['patient_id'].unique())//N_CHUNKS\npatients = []\nfor i in range(0, len(df['patient_id'].unique()), csize):\n    patients.append(df['patient_id'].unique()[i:i+csize])\n    \nCHUNKS = []\nfor p in patients:\n    CHUNKS.append([df[df.patient_id.isin(p)].index[0], df[df.patient_id.isin(p)].index[-1]+1])","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:38:33.054501Z","iopub.execute_input":"2023-06-13T09:38:33.054904Z","iopub.status.idle":"2023-06-13T09:38:33.068755Z","shell.execute_reply.started":"2023-06-13T09:38:33.054872Z","shell.execute_reply":"2023-06-13T09:38:33.067693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all = df.copy()\nfor chunk in tqdm(CHUNKS):\n    print(chunk)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:38:34.570101Z","iopub.execute_input":"2023-06-13T09:38:34.570827Z","iopub.status.idle":"2023-06-13T09:38:34.617563Z","shell.execute_reply.started":"2023-06-13T09:38:34.570785Z","shell.execute_reply":"2023-06-13T09:38:34.616582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_png(patient_id, img_id):\n    cv2.imwrite(SAVE_FOLDER + f\"{patient_id}/{img_id}.png\", images_dict[f\"{patient_id}/{img_id}.png\"])\ndf_all['pat_lat'] = df_all['patient_id'].astype(str) + df_all['laterality']\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:38:36.022607Z","iopub.execute_input":"2023-06-13T09:38:36.022982Z","iopub.status.idle":"2023-06-13T09:38:36.247803Z","shell.execute_reply.started":"2023-06-13T09:38:36.022948Z","shell.execute_reply":"2023-06-13T09:38:36.246645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds_array = np.array([])\nsecond_df = pd.DataFrame()\nif torch.cuda.is_available():\n# if True:\n\n    for chunk in tqdm(CHUNKS):\n        print('start')\n        df = df_all.iloc[chunk[0]: chunk[1]]\n        images_dict = {}\n        \n        os.makedirs(JPG_FOLDER, exist_ok=True)\n\n\n        _ = Parallel(n_jobs=2)(\n            delayed(convert_dicom_to_jpg)(patient_id, img_id, save_folder=JPG_FOLDER)\n            for patient_id, img_id in zip(df['patient_id'].values, df['image_id'].values)\n        )\n    \n        jpgfiles = glob.glob(JPG_FOLDER + \"*.jpg\")\n\n        if len(jpgfiles):\n            pipe = jpg_decode_pipeline(jpgfiles, batch_size=1, num_threads=2, device_id=0)\n            pipe.build()\n\n        for f in jpgfiles:\n            patient, image = f.split('/')[-1][:-4].split('_')\n            dicom = pydicom.dcmread(IMG_PATH + f\"{patient}/{image}.dcm\")\n            try:\n\n                out = pipe.run()\n\n                # Dali -> Torch\n                img = out[0][0]\n                img_torch = torch.empty(img.shape(), dtype=torch.int16, device=\"cuda\")\n                feed_ndarray(img, img_torch, cuda_stream=torch.cuda.current_stream(device=0))\n                img = img_torch.float()\n                img = img.reshape(img.shape[0], img.shape[1])\n\n\n                # read_mammography()に相当\n                img = img[5:-5, 5:-5]\n                \n                rev = 0\n                rev_val = img.max().item() # for sigmoid windowing\n                if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n                    img = img.max() - img\n                min_, max_ = img.min(), img.max()\n                img = (img - min_) / (max_ - min_)\n                img = (img * 255)\n                \n                if dicom['WindowWidth'].VM > 1:\n                    center = float(dicom.WindowCenter[0])\n                    width = float(dicom.WindowWidth[0])\n                else:\n                    center = float(dicom.WindowCenter)\n                    width = float(dicom.WindowWidth)\n                y_range = 2**dicom.BitsStored - 1\n                windowing_stat_dict[f'{image}'] = [min_.item(), max_.item(), center, width, y_range, rev, rev_val]\n\n                img1 = ExtractBreast_torch(img).cpu().numpy().astype(np.uint8)\n\n                h, w = img1.shape\n                if (h == 0) or (w == 0):\n#                     img = cv2.resize(img1, config2.SIZE, interpolation =cv2.INTER_AREA)\n                    img2 = cv2.resize(img, config.SIZE, interpolation =cv2.INTER_AREA)\n                    img = cv2.resize(img, config2.SIZE, interpolation =cv2.INTER_AREA)\n                else:\n                    c_h = img1.shape[0]\n                    if c_h > 2048:\n                        pad = c_h - 2048\n                        img2 = img1[pad//2:-pad//2, :]\n                    else:\n                        img2 = img1 # img1.copy()\n\n                    img = cv2.resize(img1, config2.SIZE, interpolation =cv2.INTER_AREA)\n                    img2 = cv2.resize(img2, config.SIZE, interpolation =cv2.INTER_AREA)\n\n    #             fig, axes = plt.subplots(figsize=(5, 5), ncols=1, nrows=1)\n    #             axes.imshow(img, cmap='bone')\n\n                images_dict[f\"{patient}/{image}.png\"] = img\n                images_dict[f\"{patient}/{image}-2.png\"] = img2\n            \n            except Exception as e:\n                print(patient, image)\n\n        shutil.rmtree(JPG_FOLDER)\n        torch.cuda.empty_cache()\n        gc.collect()\n        \n        \n        for patient_id, img_id in tqdm(zip(df['patient_id'].values, df['image_id'].values)):\n            images = save_array(patient_id, img_id, images_dict)\n            if images:\n                images_dict[f\"{patient_id}/{img_id}.png\"] = images[0] # 912*1520\n                images_dict[f\"{patient_id}/{img_id}-2.png\"] = images[1] # 1024*1024\n    \n        \n        test_preds = np.zeros(len(df), dtype='float64')\n    \n    \n        test_dataset = MammoDataset2(df, train=False)\n        test_loader = DataLoader(test_dataset,\n                                     batch_size=config.batch_size,\n                                     shuffle=False,\n                                     num_workers=2, pin_memory=True, drop_last=False)\n        \n        for fold in range(config.n_folds):\n            model = torch.jit.load(f'/kaggle/input/compile-ver49-83-83/ver49-{fold}.ts')\n            pred = inference_fn_compile(test_loader, model, device, config.batch_size)\n            pred = pred[:chunk[1] - chunk[0]]\n            test_preds += pred/config.n_folds\n\n#         test_preds = np.random.rand(len(df))\n        df['prediction'] = test_preds\n        df_agg = df[['patient_id', 'laterality', 'prediction', 'pat_lat']].groupby(['patient_id', 'laterality', 'pat_lat']).mean()\n        df_agg = df_agg[(df_agg['prediction'] > 0.01) & (df_agg['prediction'] < 0.9)].reset_index()\n        \n        df = df[df.pat_lat.isin(set(df_agg['pat_lat']))]\n        \n        del model\n        torch.cuda.empty_cache()\n        gc.collect()\n        \n#         _ = Parallel(n_jobs=2)(\n#             delayed(save_png)(patient_id, img_id)\n#             for patient_id, img_id in zip(df['patient_id'].values, df['image_id'].values)\n#         )\n        for patient_id, img_id in tqdm(zip(df['patient_id'].values, df['image_id'].values)):\n            save_png(patient_id, img_id)\n        \n        second_df = pd.concat([second_df, df])\n            \n        test_preds_array = np.concatenate([test_preds_array, test_preds])\n        \n        del images_dict\n        torch.cuda.empty_cache()\n        gc.collect()\n        print('end')\nsecond_df = second_df.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:38:37.101229Z","iopub.execute_input":"2023-06-13T09:38:37.10161Z","iopub.status.idle":"2023-06-13T09:39:19.705922Z","shell.execute_reply.started":"2023-06-13T09:38:37.101576Z","shell.execute_reply":"2023-06-13T09:39:19.704834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"second_df","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:39:25.507748Z","iopub.execute_input":"2023-06-13T09:39:25.508167Z","iopub.status.idle":"2023-06-13T09:39:25.536936Z","shell.execute_reply.started":"2023-06-13T09:39:25.508131Z","shell.execute_reply":"2023-06-13T09:39:25.535981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df_all.copy()\ndf['prediction_ver49'] = test_preds_array\n# df['prediction_ver49'] = np.random.rand(len(df))","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:39:27.493828Z","iopub.execute_input":"2023-06-13T09:39:27.494236Z","iopub.status.idle":"2023-06-13T09:39:27.502488Z","shell.execute_reply.started":"2023-06-13T09:39:27.494202Z","shell.execute_reply":"2023-06-13T09:39:27.500287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Second Stage","metadata":{}},{"cell_type":"markdown","source":"# Breast Level","metadata":{}},{"cell_type":"code","source":"class config:\n    seed = 10\n    batch_size = 16\n    n_folds = 4\n    name = 'tf_efficientnetv2_s'\n    \nif debug_flag:\n    config.n_folds = 2\n\nmean = 0.3089279\nstd = 0.25053555408335154\nclass MammoDatasetBreast(Dataset):\n    def __init__(self, df, train=True, tfms=None):\n        self.df = df\n        self.train = train\n        self.tfms = tfms\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        data = self.df.iloc[idx]\n        path = SAVE_FOLDER + f\"{data['patient_id']}/{data['image_id']}.png\"\n        img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n        if self.tfms:\n            augmented = self.tfms(image=img)\n            img = augmented['image']\n        img = img.astype('float32')\n        img -= img.min()\n        img /= img.max()\n        img = torch.tensor((img - mean)/std, dtype=torch.float32)\n        if self.train:\n            return img.unsqueeze(0), torch.tensor(data['cancer'], dtype=torch.long)\n        else:\n            return img.unsqueeze(0)\n        \n        \nmean = 0.3089279\nstd = 0.25053555408335154\nclass MammoDatasetPatLat(Dataset):\n    def __init__(self, df, train=True, tfms=None, size=1520):\n        self.df = df\n        self.train = train\n        self.tfms = tfms\n        self.size = (size, size)\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        data = self.df.iloc[idx]\n        imgs = []\n        for img_id in data['image_id']:\n            path = SAVE_FOLDER + f\"{data['patient_id']}/{img_id}.png\"\n            img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n            imgs.append(img)\n        imgs = np.concatenate(imgs, axis=1)\n        imgs = cv2.resize(imgs, self.size, interpolation=cv2.INTER_AREA)\n        \n#         path = BASE_PATH + f\"{data['patient_id']}/\" + str(data['image_id']) + '.png'\n#         path = f\"test_images/{data['patient_id']}/{data['image_id']}.png\"\n        # img = pickle.load(open(path, 'rb'))\n#         img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n        if self.tfms:\n            augmented = self.tfms(image=imgs)\n            imgs = augmented['image']\n#         img = torch.tensor((img/255. - mean)/std, dtype=torch.float32)\n        imgs = imgs.astype('float32')\n        imgs -= imgs.min()\n        imgs /= imgs.max()\n        imgs = torch.tensor((imgs - mean)/std, dtype=torch.float32)\n        if self.train:\n            return imgs.unsqueeze(0), torch.tensor(data['cancer'], dtype=torch.long)\n        else:\n            return imgs.unsqueeze(0)\n\n# sigmoid windowing; sw\nmean_sw = 0.21902776\nstd_sw = 0.2058490367876737\nclass MammoDatasetBreastSW(Dataset):\n    def __init__(self, df, train=True, tfms=None):\n        self.df = df\n        self.train = train\n        self.tfms = tfms\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        data = self.df.iloc[idx]\n        windowing_stat = windowing_stat_dict[f\"{data['image_id']}\"]\n        min_, max_, center_, width_, y_range_, rev_, rev_max_ = windowing_stat\n#         path = BASE_PATH + f\"{data['patient_id']}/\" + str(data['image_id']) + '.png'\n#         path = f\"test_images/{data['patient_id']}/{data['image_id']}.png\"\n        path = SAVE_FOLDER + f\"{data['patient_id']}/{data['image_id']}.png\"\n        # img = pickle.load(open(path, 'rb'))\n        img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n        if self.tfms:\n            augmented = self.tfms(image=img)\n            img = augmented['image']\n#         img = torch.tensor((img/255. - mean)/std, dtype=torch.float32)\n        img = img.astype('float32')\n        \n        # sigmoid windowing\n        img /= 255\n        img = img * (max_ - min_) + min_\n        if rev_ == 1:\n            img = rev_max_ - img\n        img = y_range_ / (1 + np.exp(-4 * (img - center_) / width_))\n        if rev_ == 1:\n            img = np.amax(img) - img\n        img -= img.min()\n        img /= img.max()\n        img = torch.tensor((img - mean_sw)/std_sw, dtype=torch.float32)\n        if self.train:\n            return img.unsqueeze(0), torch.tensor(data['cancer'], dtype=torch.long)\n        else:\n            return img.unsqueeze(0)\n\n# sigmoid windowing; sw\nmean_sw = 0.21902776\nstd_sw = 0.2058490367876737\nclass MammoDatasetPatLatSW(Dataset):\n    def __init__(self, df, train=True, tfms=None, size=1520):\n        self.df = df\n        self.train = train\n        self.tfms = tfms\n        self.size = (size, size)\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        data = self.df.iloc[idx]\n        imgs = []\n        for n, img_id in enumerate(data['image_id']):\n            if n == 0:\n                windowing_stat = windowing_stat_dict[f\"{img_id}\"]\n#             path = f\"test_images/{data['patient_id']}/{img_id}.png\"\n            path = SAVE_FOLDER + f\"{data['patient_id']}/{img_id}.png\"\n            img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n            imgs.append(img)\n        min_, max_, center_, width_, y_range_, rev_, rev_max_ = windowing_stat\n        imgs = np.concatenate(imgs, axis=1)\n        \n        # sigmoid windowing\n        imgs = imgs.astype('float32')\n        imgs /= 255\n        imgs = imgs * (max_ - min_) + min_\n        if rev_ == 1:\n            imgs = rev_max_ - imgs\n        imgs = y_range_ / (1 + np.exp(-4 * (imgs - center_) / width_))\n        if rev_ == 1:\n            imgs = np.amax(imgs) - imgs\n        imgs -= imgs.min()\n        imgs /= imgs.max()\n        imgs *= 255\n        \n        imgs = cv2.resize(imgs, self.size, interpolation=cv2.INTER_AREA)\n        \n        imgs = imgs.astype('float32')\n        imgs -= imgs.min()\n        imgs /= imgs.max()\n        imgs = torch.tensor((imgs - mean_sw)/std_sw, dtype=torch.float32)\n        if self.train:\n            return imgs.unsqueeze(0), torch.tensor(data['cancer'], dtype=torch.long)\n        else:\n            return imgs.unsqueeze(0)\n        \ntest_dataset = MammoDatasetBreast(second_df, train=False)\ntest_loader = DataLoader(test_dataset,\n                             batch_size=32,\n                             shuffle=False,\n                             num_workers=2, pin_memory=True, drop_last=False)\ntest_preds_1 = np.zeros(len(second_df), dtype='float64')\nfor fold in range(config.n_folds):\n    model = torch.jit.load(f'/kaggle/input/efficientnetv2s-exp084-tensorrt/efficientnetv2s_ver084-seed10-{fold}-swa.ts')\n    pred = inference_fn_compile_tta(test_loader, model, device, 32)\n    test_preds_1 += pred[:len(second_df)] / config.n_folds\n    \n    if fold == 1:\n        del model\n        torch.cuda.empty_cache()\n        gc.collect()\nif fold != 1:\n    del model\n    torch.cuda.empty_cache()\n    gc.collect()\n    \n    \ntest_preds_2 = np.zeros(len(second_df), dtype='float64')\nfor fold in range(config.n_folds):\n    model = torch.jit.load(f'/kaggle/input/efficientnetv2s-exp084-tensorrt/efficientnetv2s_ver084-seed660-{fold}-swa.ts')\n    pred = inference_fn_compile_tta(test_loader, model, device, 32)\n    test_preds_2 += pred[:len(second_df)] / config.n_folds\n#     second_df[f'prediction{fold}'] = pred\n\n    if fold == 1:\n        del model\n        torch.cuda.empty_cache()\n        gc.collect()\nif fold != 1:\n    del model\n    torch.cuda.empty_cache()\n    gc.collect()\n\n# Sigmoid Windowing\ntest_dataset = MammoDatasetBreastSW(second_df, train=False)\ntest_loader = DataLoader(test_dataset,\n                             batch_size=32,\n                             shuffle=False,\n                             num_workers=2, pin_memory=True, drop_last=False)\ntest_preds_3 = np.zeros(len(second_df), dtype='float64')\nfor fold in range(config.n_folds):\n    model = torch.jit.load(f'/kaggle/input/rsna-efficientnetv2s-pseudo-sig-tensorrt/efficientnetv2s_ver099-seed10-{fold}.ts')\n    pred = inference_fn_compile(test_loader, model, device, 32)\n    test_preds_3 += pred[:len(second_df)] / config.n_folds\n#     second_df[f'prediction{fold}'] = pred\n\n    if fold == 1:\n        del model\n        torch.cuda.empty_cache()\n        gc.collect()\nif fold != 1:\n    del model\n    torch.cuda.empty_cache()\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:39:29.350546Z","iopub.execute_input":"2023-06-13T09:39:29.350932Z","iopub.status.idle":"2023-06-13T09:40:12.978616Z","shell.execute_reply.started":"2023-06-13T09:39:29.3509Z","shell.execute_reply":"2023-06-13T09:40:12.977465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# second_df['prediction_ver84'] = np.random.rand(len(second_df))\n# p = 0.45\n# test_preds = p * test_preds + (1 - p) * test_preds_2\n# second_df['prediction_ver84'] = test_preds\n# second_df","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:40:12.981011Z","iopub.execute_input":"2023-06-13T09:40:12.981401Z","iopub.status.idle":"2023-06-13T09:40:12.986985Z","shell.execute_reply.started":"2023-06-13T09:40:12.981363Z","shell.execute_reply":"2023-06-13T09:40:12.985987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p1 = 0.55\np2 = 0.35\ntest_preds = p1 * test_preds_1 + p2 * test_preds_2 + (1 - p1 - p2) * test_preds_3\nsecond_df['prediction_ver84'] = test_preds\nsecond_df","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:40:12.988907Z","iopub.execute_input":"2023-06-13T09:40:12.989615Z","iopub.status.idle":"2023-06-13T09:40:13.031596Z","shell.execute_reply.started":"2023-06-13T09:40:12.989576Z","shell.execute_reply":"2023-06-13T09:40:13.030565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.merge(second_df, how='outer')\ndf['prediction_ver84'] = df['prediction_ver84'].fillna(0)\ndf","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:40:13.034238Z","iopub.execute_input":"2023-06-13T09:40:13.034579Z","iopub.status.idle":"2023-06-13T09:40:13.07801Z","shell.execute_reply.started":"2023-06-13T09:40:13.034547Z","shell.execute_reply":"2023-06-13T09:40:13.07699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df[['patient_id', 'laterality', 'prediction_ver49', 'prediction_ver84']].groupby(['patient_id', 'laterality']).mean()\ndf = df.reset_index()\ndf['prediction_id'] = [f'{p}_{l}' for p, l in zip(df['patient_id'].values, df['laterality'].values)]\ndf.index = df['prediction_id'].values","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:40:13.079537Z","iopub.execute_input":"2023-06-13T09:40:13.079887Z","iopub.status.idle":"2023-06-13T09:40:13.093104Z","shell.execute_reply.started":"2023-06-13T09:40:13.079852Z","shell.execute_reply":"2023-06-13T09:40:13.091951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Patient-Laterality Level","metadata":{}},{"cell_type":"code","source":"test_df_patlat = second_df.copy()\ntest_df_patlat = test_df_patlat.loc[[v in ['CC', 'MLO'] for v in test_df_patlat['view'].values]].reset_index(drop=True) # exclude minor view images\nimage_ids = test_df_patlat.groupby(['patient_id', 'laterality'])['image_id'].apply(list).reset_index()['image_id'].values\ntest_df_patlat = test_df_patlat.groupby(['patient_id', 'laterality']).first().reset_index()\ntest_df_patlat['image_id'] = image_ids","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:40:13.094784Z","iopub.execute_input":"2023-06-13T09:40:13.095375Z","iopub.status.idle":"2023-06-13T09:40:13.116425Z","shell.execute_reply.started":"2023-06-13T09:40:13.095329Z","shell.execute_reply":"2023-06-13T09:40:13.115369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = MammoDatasetPatLat(test_df_patlat, train=False)\ntest_loader = DataLoader(test_dataset,\n                             batch_size=16,\n                             shuffle=False,\n                             num_workers=2, pin_memory=True, drop_last=False)\ntest_preds_1 = np.zeros(len(test_df_patlat), dtype='float64')\nfor fold in range(config.n_folds):\n    model = torch.jit.load(f'/kaggle/input/efficientnetv2s-exp085-tensorrt/efficientnetv2s_ver085-{fold}.ts')\n    pred = inference_fn_compile(test_loader, model, device, 16)\n    test_preds_1 += pred[:len(test_df_patlat)] / config.n_folds\n    if fold == 1:\n        del model\n        torch.cuda.empty_cache()\n        gc.collect()\nif fold != 1:\n    del model\n    torch.cuda.empty_cache()\n    gc.collect()\n    \n\ntest_preds_2 = np.zeros(len(test_df_patlat), dtype='float64')\nfor fold in range(config.n_folds):\n    model = torch.jit.load(f'/kaggle/input/efficientnetv2s-exp085-tensorrt/efficientnetv2s_ver085-seed660-{fold}.ts')\n    pred = inference_fn_compile(test_loader, model, device, 16)\n    test_preds_2 += pred[:len(test_df_patlat)] / config.n_folds\n    \n    if fold == 1:\n        del model\n        torch.cuda.empty_cache()\n        gc.collect()\nif fold != 1:\n    del model\n    torch.cuda.empty_cache()\n    gc.collect()\n\ntest_dataset = MammoDatasetPatLatSW(test_df_patlat, train=False)\ntest_loader = DataLoader(test_dataset,\n                             batch_size=16,\n                             shuffle=False,\n                             num_workers=2, pin_memory=True, drop_last=False)\ntest_preds_3 = np.zeros(len(test_df_patlat), dtype='float64')\nfor fold in range(config.n_folds):\n    model = torch.jit.load(f'/kaggle/input/rsna-efficientnetv2s-pseudo-sig-tensorrt/efficientnetv2s_ver100-seed10-{fold}.ts')\n    pred = inference_fn_compile(test_loader, model, device, 16)\n    test_preds_3 += pred[:len(test_df_patlat)] / config.n_folds\n    \n    if fold == 1:\n        del model\n        torch.cuda.empty_cache()\n        gc.collect()\nif fold != 1:\n    del model\n    torch.cuda.empty_cache()\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:40:13.117879Z","iopub.execute_input":"2023-06-13T09:40:13.118251Z","iopub.status.idle":"2023-06-13T09:40:30.63432Z","shell.execute_reply.started":"2023-06-13T09:40:13.118215Z","shell.execute_reply":"2023-06-13T09:40:30.63311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# p = 0.4\n# test_preds = p * test_preds + (1 - p) * test_preds_2\n# test_df_patlat['prediction_ver85'] = test_preds\n# test_df_patlat['prediction_ver85'] = np.random.rand(len(test_df_patlat))\n\n# if not 'prediction_id' in test_df_patlat.columns:\n#     test_df_patlat['prediction_id'] = [f'{p}_{l}' for p, l in zip(test_df_patlat['patient_id'].values, test_df_patlat['laterality'].values)]\n# df = df.merge(test_df_patlat[['prediction_id', 'prediction_ver85']], how='outer')\n# df['prediction_ver85'] = df['prediction_ver85'].fillna(0)\n# df","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:40:30.63592Z","iopub.execute_input":"2023-06-13T09:40:30.636716Z","iopub.status.idle":"2023-06-13T09:40:30.642825Z","shell.execute_reply.started":"2023-06-13T09:40:30.63667Z","shell.execute_reply":"2023-06-13T09:40:30.6413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p1 = 0.45\np2 = 0.2\ntest_preds = p1 * test_preds_1 + p2 * test_preds_2 + (1 - p1 - p2) * test_preds_3\ntest_df_patlat['prediction_ver85'] = test_preds\n\nif not 'prediction_id' in test_df_patlat.columns:\n    test_df_patlat['prediction_id'] = [f'{p}_{l}' for p, l in zip(test_df_patlat['patient_id'].values, test_df_patlat['laterality'].values)]\ndf = df.merge(test_df_patlat[['prediction_id', 'prediction_ver85']], how='outer')\ndf['prediction_ver85'] = df['prediction_ver85'].fillna(0)\ndf","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:40:30.644419Z","iopub.execute_input":"2023-06-13T09:40:30.64506Z","iopub.status.idle":"2023-06-13T09:40:30.686849Z","shell.execute_reply.started":"2023-06-13T09:40:30.645002Z","shell.execute_reply":"2023-06-13T09:40:30.685693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# weight1 = 0.54\n# weight2 = 0.81\n# th = 0.185\np = 0.75\nth = 0.225\n# test_preds = weight2 * (weight1 * df['prediction_ver49'].values + (1 - weight1) * df['prediction_ver84'].values) + (1 - weight2) * df['prediction_ver85'].values\ntest_preds = p*df['prediction_ver84'].values + (1-p)*df['prediction_ver85'].values\ndf['prediction'] = test_preds\ndf['max'] = [max(i, j, k) for i,j,k in zip(df['prediction_ver49'], df['prediction_ver84'], df['prediction_ver85'])]\ndf['prediction'] = [j if j > 0.9 else i for i, j in zip(df['prediction'],df['max'])]\ntest_preds = df['prediction'].values\ntest_preds = (test_preds >= th).astype('int')","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:40:30.691171Z","iopub.execute_input":"2023-06-13T09:40:30.691478Z","iopub.status.idle":"2023-06-13T09:40:30.701408Z","shell.execute_reply.started":"2023-06-13T09:40:30.69145Z","shell.execute_reply":"2023-06-13T09:40:30.700238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'prediction_id': df['prediction_id'].values,\n                         'cancer': test_preds})\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:40:30.705009Z","iopub.execute_input":"2023-06-13T09:40:30.705468Z","iopub.status.idle":"2023-06-13T09:40:30.717319Z","shell.execute_reply.started":"2023-06-13T09:40:30.705425Z","shell.execute_reply":"2023-06-13T09:40:30.716132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T09:40:30.719208Z","iopub.execute_input":"2023-06-13T09:40:30.719653Z","iopub.status.idle":"2023-06-13T09:40:30.730825Z","shell.execute_reply.started":"2023-06-13T09:40:30.719612Z","shell.execute_reply":"2023-06-13T09:40:30.729671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}