{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install /kaggle/input/rsna-2023-whl/{pydicom-2.4.2-py3-none-any.whl,pylibjpeg-1.4.0-py3-none-any.whl,einops-0.6.1-py3-none-any.whl}","metadata":{"execution":{"iopub.status.busy":"2023-08-10T06:57:48.030591Z","iopub.execute_input":"2023-08-10T06:57:48.03138Z","iopub.status.idle":"2023-08-10T06:58:25.134163Z","shell.execute_reply.started":"2023-08-10T06:57:48.031329Z","shell.execute_reply":"2023-08-10T06:58:25.133047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport os\nimport pydicom\nimport cv2\n# import einops","metadata":{"execution":{"iopub.status.busy":"2023-08-10T06:58:25.137599Z","iopub.execute_input":"2023-08-10T06:58:25.13799Z","iopub.status.idle":"2023-08-10T06:58:25.724529Z","shell.execute_reply.started":"2023-08-10T06:58:25.13794Z","shell.execute_reply":"2023-08-10T06:58:25.723558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define a fucntion to read dicom file and return it's pixel\ndef read_dicom(dicom_file):\n    dicom = pydicom.dcmread(dicom_file) # read dciom files\n    dicom_pixel = dicom.pixel_array # read the pixel of images\n\n    dicom_pixel = (dicom_pixel - dicom_pixel.min()) / (dicom_pixel.max() - dicom_pixel.min()) # Standardie with transferig to [0,1] space\n\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        dicom_pixel = 1 - dicom_pixel # a special kind of format in dicom files, you can search it on the net\n    \n    return dicom_pixel","metadata":{"execution":{"iopub.status.busy":"2023-08-10T06:58:25.726041Z","iopub.execute_input":"2023-08-10T06:58:25.726416Z","iopub.status.idle":"2023-08-10T06:58:25.733458Z","shell.execute_reply.started":"2023-08-10T06:58:25.726381Z","shell.execute_reply":"2023-08-10T06:58:25.731563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_patient_ids = []\ntest_path = \"/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images\"\n\n\nfor patient_id in os.listdir(test_path):\n    test_patient_ids.append(patient_id)\ntest_result = np.zeros((len(test_patient_ids), 13))","metadata":{"execution":{"iopub.status.busy":"2023-08-10T06:58:25.73653Z","iopub.execute_input":"2023-08-10T06:58:25.737502Z","iopub.status.idle":"2023-08-10T06:58:25.751673Z","shell.execute_reply.started":"2023-08-10T06:58:25.737471Z","shell.execute_reply":"2023-08-10T06:58:25.7507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dcm_to_png(images, patient_id, series, i, number_of_train_images, series_batch, image_sizes, extension ='jpg', count=0):\n    if i == number_of_train_images - 2:\n        file_0 = images[i]\n        file_1 = images[i+1]\n        file_2 = images[i+1]\n    elif i == number_of_train_images - 1:\n        file_0 = images[i]\n        file_1 = images[i]\n        file_2 = images[i]\n    else:\n        file_0 = images[i]\n        file_1 = images[i+1]\n        file_2 = images[i+2]\n\n    image_id_0 = file_0.split('/')[-1][:-4]\n    image_id_1 = file_1.split('/')[-1][:-4]\n    image_id_2 = file_2.split('/')[-1][:-4]\n    dicom_pixel_0 = read_dicom(dicom_file=file_0) # read dicom image pixels\n    dicom_pixel_1 = read_dicom(dicom_file=file_1)\n    dicom_pixel_2 = read_dicom(dicom_file=file_2)\n    \n#     cropped_dicom = crop_dicom(dicom_file=dicom_pixel) # crop the dicom image\n    for j, size in enumerate(image_sizes):\n        resized_img_0 = cv2.resize(np.expand_dims(dicom_pixel_0, 2),dsize=(size,size)) # resize it to specific size\n        resized_img_1 = cv2.resize(np.expand_dims(dicom_pixel_1, 2),dsize=(size,size)) # resize it to specific size\n        resized_img_2 = cv2.resize(np.expand_dims(dicom_pixel_2, 2),dsize=(size,size)) # resize it to specific size\n        resized_img_25d = np.stack([resized_img_0, resized_img_1, resized_img_2], axis=2)\n        print('---------------------===============================',resized_img_0.shape)\n        series_batch[j][count] = np.transpose(resized_img_25d, (2,0,1))\n    count += 1","metadata":{"execution":{"iopub.status.busy":"2023-08-10T06:58:25.753293Z","iopub.execute_input":"2023-08-10T06:58:25.753648Z","iopub.status.idle":"2023-08-10T06:58:25.767279Z","shell.execute_reply.started":"2023-08-10T06:58:25.753618Z","shell.execute_reply":"2023-08-10T06:58:25.766355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.path.append(\"/kaggle/input/caformer\")\nimport timm\nfrom torch import nn\n# from model import FeatureExtractor\n\nN_EVAL = 14","metadata":{"execution":{"iopub.status.busy":"2023-08-10T06:58:25.768765Z","iopub.execute_input":"2023-08-10T06:58:25.769085Z","iopub.status.idle":"2023-08-10T06:58:33.441074Z","shell.execute_reply.started":"2023-08-10T06:58:25.769055Z","shell.execute_reply":"2023-08-10T06:58:33.440087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from functools import partial\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nfrom timm.models.layers import trunc_normal_, DropPath\nfrom timm.models.registry import register_model\nfrom timm.data import IMAGENET_DEFAULT_MEAN, IMAGENET_DEFAULT_STD\nfrom timm.layers.helpers import to_2tuple\n\nclass Downsampling(nn.Module):\n    \"\"\"\n    Downsampling implemented by a layer of convolution.\n    \"\"\"\n    def __init__(self, in_channels, out_channels, \n        kernel_size, stride=1, padding=0, \n        pre_norm=None, post_norm=None, pre_permute=False):\n        super().__init__()\n        self.pre_norm = pre_norm(in_channels) if pre_norm else nn.Identity()\n        self.pre_permute = pre_permute\n        self.conv = nn.Conv2d(in_channels, out_channels, kernel_size=kernel_size, \n                              stride=stride, padding=padding)\n        self.post_norm = post_norm(out_channels) if post_norm else nn.Identity()\n\n    def forward(self, x):\n        x = self.pre_norm(x)\n        if self.pre_permute:\n            # if take [B, H, W, C] as input, permute it to [B, C, H, W]\n            x = x.permute(0, 3, 1, 2)\n        x = self.conv(x)\n        x = x.permute(0, 2, 3, 1) # [B, C, H, W] -> [B, H, W, C]\n        x = self.post_norm(x)\n        return x\n\n\nclass Scale(nn.Module):\n    \"\"\"\n    Scale vector by element multiplications.\n    \"\"\"\n    def __init__(self, dim, init_value=1.0, trainable=True):\n        super().__init__()\n        self.scale = nn.Parameter(init_value * torch.ones(dim), requires_grad=trainable)\n\n    def forward(self, x):\n        return x * self.scale\n        \n\nclass SquaredReLU(nn.Module):\n    \"\"\"\n        Squared ReLU: https://arxiv.org/abs/2109.08668\n    \"\"\"\n    def __init__(self, inplace=False):\n        super().__init__()\n        self.relu = nn.ReLU(inplace=inplace)\n    def forward(self, x):\n        return torch.square(self.relu(x))\n\n\nclass StarReLU(nn.Module):\n    \"\"\"\n    StarReLU: s * relu(x) ** 2 + b\n    \"\"\"\n    def __init__(self, scale_value=1.0, bias_value=0.0,\n        scale_learnable=True, bias_learnable=True, \n        mode=None, inplace=False):\n        super().__init__()\n        self.inplace = inplace\n        self.relu = nn.ReLU(inplace=inplace)\n        self.scale = nn.Parameter(scale_value * torch.ones(1),\n            requires_grad=scale_learnable)\n        self.bias = nn.Parameter(bias_value * torch.ones(1),\n            requires_grad=bias_learnable)\n    def forward(self, x):\n        return self.scale * self.relu(x)**2 + self.bias\n\n\nclass Meta_Attention(nn.Module):\n    \"\"\"\n    Vanilla self-attention from Transformer: https://arxiv.org/abs/1706.03762.\n    Modified from timm.\n    \"\"\"\n    def __init__(self, dim, head_dim=32, num_heads=None, qkv_bias=False,\n        attn_drop=0., proj_drop=0., proj_bias=False, **kwargs):\n        super().__init__()\n\n        self.head_dim = head_dim\n        self.scale = head_dim ** -0.5\n\n        self.num_heads = num_heads if num_heads else dim // head_dim\n        if self.num_heads == 0:\n            self.num_heads = 1\n        \n        self.attention_dim = self.num_heads * self.head_dim\n\n        self.qkv = nn.Linear(dim, self.attention_dim * 3, bias=qkv_bias)\n        self.attn_drop = nn.Dropout(attn_drop)\n        self.proj = nn.Linear(self.attention_dim, dim, bias=proj_bias)\n        self.proj_drop = nn.Dropout(proj_drop)\n\n        \n    def forward(self, x):\n        B, H, W, C = x.shape\n        N = H * W\n        qkv = self.qkv(x).reshape(B, N, 3, self.num_heads, self.head_dim).permute(2, 0, 3, 1, 4)\n        q, k, v = qkv.unbind(0)   # make torchscript happy (cannot use tensor as tuple)\n\n        attn = (q @ k.transpose(-2, -1)) * self.scale\n        attn = attn.softmax(dim=-1)\n        attn = self.attn_drop(attn)\n\n        x = (attn @ v).transpose(1, 2).reshape(B, H, W, self.attention_dim)\n        x = self.proj(x)\n        x = self.proj_drop(x)\n        return x\n\n\nclass RandomMixing(nn.Module):\n    def __init__(self, num_tokens=196, **kwargs):\n        super().__init__()\n        self.random_matrix = nn.parameter.Parameter(\n            data=torch.softmax(torch.rand(num_tokens, num_tokens), dim=-1), \n            requires_grad=False)\n    def forward(self, x):\n        B, H, W, C = x.shape\n        x = x.reshape(B, H*W, C)\n        x = torch.einsum('mn, bnc -> bmc', self.random_matrix, x)\n        x = x.reshape(B, H, W, C)\n        return x\n\n\nclass LayerNormGeneral(nn.Module):\n    r\"\"\" General LayerNorm for different situations.\n\n    Args:\n        affine_shape (int, list or tuple): The shape of affine weight and bias.\n            Usually the affine_shape=C, but in some implementation, like torch.nn.LayerNorm,\n            the affine_shape is the same as normalized_dim by default. \n            To adapt to different situations, we offer this argument here.\n        normalized_dim (tuple or list): Which dims to compute mean and variance. \n        scale (bool): Flag indicates whether to use scale or not.\n        bias (bool): Flag indicates whether to use scale or not.\n\n        We give several examples to show how to specify the arguments.\n\n        LayerNorm (https://arxiv.org/abs/1607.06450):\n            For input shape of (B, *, C) like (B, N, C) or (B, H, W, C),\n                affine_shape=C, normalized_dim=(-1, ), scale=True, bias=True;\n            For input shape of (B, C, H, W),\n                affine_shape=(C, 1, 1), normalized_dim=(1, ), scale=True, bias=True.\n\n        Modified LayerNorm (https://arxiv.org/abs/2111.11418)\n            that is idental to partial(torch.nn.GroupNorm, num_groups=1):\n            For input shape of (B, N, C),\n                affine_shape=C, normalized_dim=(1, 2), scale=True, bias=True;\n            For input shape of (B, H, W, C),\n                affine_shape=C, normalized_dim=(1, 2, 3), scale=True, bias=True;\n            For input shape of (B, C, H, W),\n                affine_shape=(C, 1, 1), normalized_dim=(1, 2, 3), scale=True, bias=True.\n\n        For the several metaformer baslines,\n            IdentityFormer, RandFormer and PoolFormerV2 utilize Modified LayerNorm without bias (bias=False);\n            ConvFormer and CAFormer utilizes LayerNorm without bias (bias=False).\n    \"\"\"\n    def __init__(self, affine_shape=None, normalized_dim=(-1, ), scale=True, \n        bias=True, eps=1e-5):\n        super().__init__()\n        self.normalized_dim = normalized_dim\n        self.use_scale = scale\n        self.use_bias = bias\n        self.weight = nn.Parameter(torch.ones(affine_shape)) if scale else None\n        self.bias = nn.Parameter(torch.zeros(affine_shape)) if bias else None\n        self.eps = eps\n\n    def forward(self, x):\n        c = x - x.mean(self.normalized_dim, keepdim=True)\n        s = c.pow(2).mean(self.normalized_dim, keepdim=True)\n        x = c / torch.sqrt(s + self.eps)\n        if self.use_scale:\n            x = x * self.weight\n        if self.use_bias:\n            x = x + self.bias\n        return x\n\n\nclass LayerNormWithoutBias(nn.Module):\n    \"\"\"\n    Equal to partial(LayerNormGeneral, bias=False) but faster, \n    because it directly utilizes otpimized F.layer_norm\n    \"\"\"\n    def __init__(self, normalized_shape, eps=1e-5, **kwargs):\n        super().__init__()\n        self.eps = eps\n        self.bias = None\n        if isinstance(normalized_shape, int):\n            normalized_shape = (normalized_shape,)\n        self.weight = nn.Parameter(torch.ones(normalized_shape))\n        self.normalized_shape = normalized_shape\n    def forward(self, x):\n        return F.layer_norm(x, self.normalized_shape, weight=self.weight, bias=self.bias, eps=self.eps)\n\n\nclass SepConv(nn.Module):\n    r\"\"\"\n    Inverted separable convolution from MobileNetV2: https://arxiv.org/abs/1801.04381.\n    \"\"\"\n    def __init__(self, dim, expansion_ratio=2,\n        act1_layer=StarReLU, act2_layer=nn.Identity, \n        bias=False, kernel_size=7, padding=3,\n        **kwargs, ):\n        super().__init__()\n        med_channels = int(expansion_ratio * dim)\n        self.pwconv1 = nn.Linear(dim, med_channels, bias=bias)\n        self.act1 = act1_layer()\n        self.dwconv = nn.Conv2d(\n            med_channels, med_channels, kernel_size=kernel_size,\n            padding=padding, groups=med_channels, bias=bias) # depthwise conv\n        self.act2 = act2_layer()\n        self.pwconv2 = nn.Linear(med_channels, dim, bias=bias)\n\n    def forward(self, x):\n        x = self.pwconv1(x)\n        x = self.act1(x)\n        x = x.permute(0, 3, 1, 2)\n        x = self.dwconv(x)\n        x = x.permute(0, 2, 3, 1)\n        x = self.act2(x)\n        x = self.pwconv2(x)\n        return x\n\n\nclass Pooling(nn.Module):\n    \"\"\"\n    Implementation of pooling for PoolFormer: https://arxiv.org/abs/2111.11418\n    Modfiled for [B, H, W, C] input\n    \"\"\"\n    def __init__(self, pool_size=3, **kwargs):\n        super().__init__()\n        self.pool = nn.AvgPool2d(\n            pool_size, stride=1, padding=pool_size//2, count_include_pad=False)\n\n    def forward(self, x):\n        y = x.permute(0, 3, 1, 2)\n        y = self.pool(y)\n        y = y.permute(0, 2, 3, 1)\n        return y - x\n\n\nclass Mlp(nn.Module):\n    \"\"\" MLP as used in MetaFormer models, eg Transformer, MLP-Mixer, PoolFormer, MetaFormer baslines and related networks.\n    Mostly copied from timm.\n    \"\"\"\n    def __init__(self, dim, mlp_ratio=4, out_features=None, act_layer=StarReLU, drop=0., bias=False, **kwargs):\n        super().__init__()\n        in_features = dim\n        out_features = out_features or in_features\n        hidden_features = int(mlp_ratio * in_features)\n        drop_probs = to_2tuple(drop)\n\n        self.fc1 = nn.Linear(in_features, hidden_features, bias=bias)\n        self.act = act_layer()\n        self.drop1 = nn.Dropout(drop_probs[0])\n        self.fc2 = nn.Linear(hidden_features, out_features, bias=bias)\n        self.drop2 = nn.Dropout(drop_probs[1])\n\n    def forward(self, x):\n        x = self.fc1(x)\n        x = self.act(x)\n        x = self.drop1(x)\n        x = self.fc2(x)\n        x = self.drop2(x)\n        return x\n\n\nclass MlpHead(nn.Module):\n    \"\"\" MLP classification head\n    \"\"\"\n    def __init__(self, dim, num_classes=1000, mlp_ratio=4, act_layer=SquaredReLU,\n        norm_layer=nn.LayerNorm, head_dropout=0., bias=True):\n        super().__init__()\n        hidden_features = int(mlp_ratio * dim)\n        self.fc1 = nn.Linear(dim, hidden_features, bias=bias)\n        self.act = act_layer()\n        self.norm = norm_layer(hidden_features)\n        self.fc2 = nn.Linear(hidden_features, num_classes, bias=bias)\n        self.head_dropout = nn.Dropout(head_dropout)\n\n\n    def forward(self, x):\n        x = self.fc1(x)\n        x = self.act(x)\n        x = self.norm(x)\n        x = self.head_dropout(x)\n        x = self.fc2(x)\n        return x\n\n\nclass MetaFormerBlock(nn.Module):\n    \"\"\"\n    Implementation of one MetaFormer block.\n    \"\"\"\n    def __init__(self, dim,\n                 token_mixer=nn.Identity, mlp=Mlp,\n                 norm_layer=nn.LayerNorm,\n                 drop=0., drop_path=0.,\n                 layer_scale_init_value=None, res_scale_init_value=None\n                 ):\n\n        super().__init__()\n\n        self.norm1 = norm_layer(dim)\n        self.token_mixer = token_mixer(dim=dim, drop=drop)\n        self.drop_path1 = DropPath(drop_path) if drop_path > 0. else nn.Identity()\n        self.layer_scale1 = Scale(dim=dim, init_value=layer_scale_init_value) \\\n            if layer_scale_init_value else nn.Identity()\n        self.res_scale1 = Scale(dim=dim, init_value=res_scale_init_value) \\\n            if res_scale_init_value else nn.Identity()\n\n        self.norm2 = norm_layer(dim)\n        self.mlp = mlp(dim=dim, drop=drop)\n        self.drop_path2 = DropPath(drop_path) if drop_path > 0. else nn.Identity()\n        self.layer_scale2 = Scale(dim=dim, init_value=layer_scale_init_value) \\\n            if layer_scale_init_value else nn.Identity()\n        self.res_scale2 = Scale(dim=dim, init_value=res_scale_init_value) \\\n            if res_scale_init_value else nn.Identity()\n        \n    def forward(self, x):\n        x = self.res_scale1(x) + \\\n            self.layer_scale1(\n                self.drop_path1(\n                    self.token_mixer(self.norm1(x))\n                )\n            )\n        x = self.res_scale2(x) + \\\n            self.layer_scale2(\n                self.drop_path2(\n                    self.mlp(self.norm2(x))\n                )\n            )\n        return x\ndef _cfg(url='', **kwargs):\n    return {\n        'url': url,\n        'num_classes': 1000, 'input_size': (3, 224, 224), 'pool_size': None,\n        'crop_pct': 1.0, 'interpolation': 'bicubic',\n        'mean': IMAGENET_DEFAULT_MEAN, 'std': IMAGENET_DEFAULT_STD, 'classifier': 'head',\n        **kwargs\n    }    \ndefault_cfgs = {\n    'caformer_b36_384_in21ft1k': _cfg(\n        url='https://huggingface.co/sail/dl/resolve/main/caformer/caformer_b36_384_in21ft1k.pth',\n        input_size=(3, 384, 384)),\n}\nDOWNSAMPLE_LAYERS_FOUR_STAGES = [partial(Downsampling,\n            kernel_size=7, stride=4, padding=2,\n            post_norm=partial(LayerNormGeneral, bias=False, eps=1e-6)\n            )] + \\\n            [partial(Downsampling,\n                kernel_size=3, stride=2, padding=1, \n                pre_norm=partial(LayerNormGeneral, bias=False, eps=1e-6), pre_permute=True\n            )]*3\n\n\nclass MetaFormer(nn.Module):\n    r\"\"\" MetaFormer\n        A PyTorch impl of : `MetaFormer Baselines for Vision`  -\n          https://arxiv.org/abs/2210.13452\n\n    Args:\n        in_chans (int): Number of input image channels. Default: 3.\n        num_classes (int): Number of classes for classification head. Default: 1000.\n        depths (list or tuple): Number of blocks at each stage. Default: [2, 2, 6, 2].\n        dims (int): Feature dimension at each stage. Default: [64, 128, 320, 512].\n        downsample_layers: (list or tuple): Downsampling layers before each stage.\n        token_mixers (list, tuple or token_fcn): Token mixer for each stage. Default: nn.Identity.\n        mlps (list, tuple or mlp_fcn): Mlp for each stage. Default: Mlp.\n        norm_layers (list, tuple or norm_fcn): Norm layers for each stage. Default: partial(LayerNormGeneral, eps=1e-6, bias=False).\n        drop_path_rate (float): Stochastic depth rate. Default: 0.\n        head_dropout (float): dropout for MLP classifier. Default: 0.\n        layer_scale_init_values (list, tuple, float or None): Init value for Layer Scale. Default: None.\n            None means not use the layer scale. Form: https://arxiv.org/abs/2103.17239.\n        res_scale_init_values (list, tuple, float or None): Init value for Layer Scale. Default: [None, None, 1.0, 1.0].\n            None means not use the layer scale. From: https://arxiv.org/abs/2110.09456.\n        output_norm: norm before classifier head. Default: partial(nn.LayerNorm, eps=1e-6).\n        head_fn: classification head. Default: nn.Linear.\n    \"\"\"\n    def __init__(self, in_chans=3, num_classes=1000, \n                 depths=[2, 2, 6, 2],\n                 dims=[64, 128, 320, 512],\n                 downsample_layers=DOWNSAMPLE_LAYERS_FOUR_STAGES,\n                 token_mixers=nn.Identity,\n                 mlps=Mlp,\n                 norm_layers=partial(LayerNormWithoutBias, eps=1e-6), # partial(LayerNormGeneral, eps=1e-6, bias=False),\n                 drop_path_rate=0.,\n                 head_dropout=0.0, \n                 layer_scale_init_values=None,\n                 res_scale_init_values=[None, None, 1.0, 1.0],\n                 output_norm=partial(nn.LayerNorm, eps=1e-6), \n                 head_fn=nn.Linear,\n                 **kwargs,\n                 ):\n        super().__init__()\n        self.num_classes = num_classes\n\n        if not isinstance(depths, (list, tuple)):\n            depths = [depths] # it means the model has only one stage\n        if not isinstance(dims, (list, tuple)):\n            dims = [dims]\n\n        num_stage = len(depths)\n        self.num_stage = num_stage\n\n        if not isinstance(downsample_layers, (list, tuple)):\n            downsample_layers = [downsample_layers] * num_stage\n        down_dims = [in_chans] + dims\n        self.downsample_layers = nn.ModuleList(\n            [downsample_layers[i](down_dims[i], down_dims[i+1]) for i in range(num_stage)]\n        )\n        \n        if not isinstance(token_mixers, (list, tuple)):\n            token_mixers = [token_mixers] * num_stage\n\n        if not isinstance(mlps, (list, tuple)):\n            mlps = [mlps] * num_stage\n\n        if not isinstance(norm_layers, (list, tuple)):\n            norm_layers = [norm_layers] * num_stage\n        \n        dp_rates=[x.item() for x in torch.linspace(0, drop_path_rate, sum(depths))]\n\n        if not isinstance(layer_scale_init_values, (list, tuple)):\n            layer_scale_init_values = [layer_scale_init_values] * num_stage\n        if not isinstance(res_scale_init_values, (list, tuple)):\n            res_scale_init_values = [res_scale_init_values] * num_stage\n\n        self.stages = nn.ModuleList() # each stage consists of multiple metaformer blocks\n        cur = 0\n        for i in range(num_stage):\n            stage = nn.Sequential(\n                *[MetaFormerBlock(dim=dims[i],\n                token_mixer=token_mixers[i],\n                mlp=mlps[i],\n                norm_layer=norm_layers[i],\n                drop_path=dp_rates[cur + j],\n                layer_scale_init_value=layer_scale_init_values[i],\n                res_scale_init_value=res_scale_init_values[i],\n                ) for j in range(depths[i])]\n            )\n            self.stages.append(stage)\n            cur += depths[i]\n\n        self.norm = output_norm(dims[-1])\n\n        if head_dropout > 0.0:\n            self.head = head_fn(dims[-1], num_classes, head_dropout=head_dropout)\n        else:\n            self.head = head_fn(dims[-1], num_classes)\n\n        self.apply(self._init_weights)\n\n    def _init_weights(self, m):\n        if isinstance(m, (nn.Conv2d, nn.Linear)):\n            trunc_normal_(m.weight, std=.02)\n            if m.bias is not None:\n                nn.init.constant_(m.bias, 0)\n\n    @torch.jit.ignore\n    def no_weight_decay(self):\n        return {'norm'}\n\n    def forward_features(self, x):\n        for i in range(self.num_stage):\n            x = self.downsample_layers[i](x)\n            x = self.stages[i](x)\n        return self.norm(x.mean([1, 2])) # (B, H, W, C) -> (B, C)\n\n    def forward(self, x):\n        x = self.forward_features(x)\n        x = self.head(x)\n        return x\n@register_model\ndef caformer_b36_384_in21ft1k(pretrained=False, **kwargs):\n    model = MetaFormer(\n        depths=[3, 12, 18, 3],\n        dims=[128, 256, 512, 768],\n        token_mixers=[SepConv, SepConv, Meta_Attention, Meta_Attention],\n        head_fn=MlpHead,\n        **kwargs)\n    model.default_cfg = default_cfgs['caformer_b36_384_in21ft1k']\n    if pretrained:\n        state_dict = torch.hub.load_state_dict_from_url(\n            url= model.default_cfg['url'], map_location=\"\", check_hash=True)\n        model.load_state_dict(state_dict)\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-08-10T06:58:33.442689Z","iopub.execute_input":"2023-08-10T06:58:33.443033Z","iopub.status.idle":"2023-08-10T06:58:33.527295Z","shell.execute_reply.started":"2023-08-10T06:58:33.443Z","shell.execute_reply":"2023-08-10T06:58:33.526178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom torch import nn\n\nfrom einops import rearrange, repeat\nfrom einops.layers.torch import Rearrange\n\n# helpers\n\ndef pair(t):\n    return t if isinstance(t, tuple) else (t, t)\n\n# classes\n\nclass PreNorm(nn.Module):\n    def __init__(self, dim, fn):\n        super().__init__()\n        self.norm = nn.LayerNorm(dim)\n        self.fn = fn\n    def forward(self, x, **kwargs):\n        return self.fn(self.norm(x), **kwargs)\n\nclass FeedForward(nn.Module):\n    def __init__(self, dim, hidden_dim, dropout = 0.):\n        super().__init__()\n        self.net = nn.Sequential(\n            nn.Linear(dim, hidden_dim),\n            nn.GELU(),\n            nn.Dropout(dropout),\n            nn.Linear(hidden_dim, dim),\n            nn.Dropout(dropout)\n        )\n    def forward(self, x):\n        return self.net(x)\n\nclass Attention(nn.Module):\n    def __init__(self, dim, heads = 8, dim_head = 64, dropout = 0.):\n        super().__init__()\n        inner_dim = dim_head *  heads\n        project_out = not (heads == 1 and dim_head == dim)\n\n        self.heads = heads\n        self.scale = dim_head ** -0.5\n\n        self.attend = nn.Softmax(dim = -1)\n        self.dropout = nn.Dropout(dropout)\n\n        self.to_qkv = nn.Linear(dim, inner_dim * 3, bias = False)\n\n        self.to_out = nn.Sequential(\n            nn.Linear(inner_dim, dim),\n            nn.Dropout(dropout)\n        ) if project_out else nn.Identity()\n\n    def forward(self, x):\n        qkv = self.to_qkv(x).chunk(3, dim = -1)\n        q, k, v = map(lambda t: rearrange(t, 'b n (h d) -> b h n d', h = self.heads), qkv)\n\n        dots = torch.matmul(q, k.transpose(-1, -2)) * self.scale\n\n        attn = self.attend(dots)\n        attn = self.dropout(attn)\n\n        out = torch.matmul(attn, v)\n        out = rearrange(out, 'b h n d -> b n (h d)')\n        return self.to_out(out)\n\nclass Transformer(nn.Module):\n    def __init__(self, dim, depth, heads, dim_head, mlp_dim, dropout = 0.):\n        super().__init__()\n        self.layers = nn.ModuleList([])\n        for _ in range(depth):\n            self.layers.append(nn.ModuleList([\n                PreNorm(dim, Attention(dim, heads = heads, dim_head = dim_head, dropout = dropout)),\n                PreNorm(dim, FeedForward(dim, mlp_dim, dropout = dropout))\n            ]))\n    def forward(self, x):\n        for attn, ff in self.layers:\n            x = attn(x) + x\n            x = ff(x) + x\n        return x\n\nclass ViT(nn.Module):\n    def __init__(self, *,n_frames, dim, depth, heads, mlp_dim, pool = 'cls', channels = 3, dim_head = 64, dropout = 0., emb_dropout = 0.):\n        super().__init__()\n        # image_height, image_width = pair(image_size)\n        # patch_height, patch_width = pair(patch_size)\n\n        # assert image_height % patch_height == 0 and image_width % patch_width == 0, 'Image dimensions must be divisible by the patch size.'\n\n        # num_patches = (image_height // patch_height) * (image_width // patch_width)\n        # patch_dim = channels * patch_height * patch_width\n        assert pool in {'cls', 'mean'}, 'pool type must be either cls (cls token) or mean (mean pooling)'\n\n        # self.to_patch_embedding = nn.Sequential(\n        #     Rearrange('b c (h p1) (w p2) -> b (h w) (p1 p2 c)', p1 = patch_height, p2 = patch_width),\n        #     nn.LayerNorm(patch_dim),\n        #     nn.Linear(patch_dim, dim),\n        #     nn.LayerNorm(dim),\n        # )\n\n        self.pos_embedding = nn.Parameter(torch.randn(1, n_frames + 1, dim))\n        self.cls_token = nn.Parameter(torch.randn(1, 1, dim))\n        self.dropout = nn.Dropout(emb_dropout)\n\n        self.transformer = Transformer(dim, depth, heads, dim_head, mlp_dim, dropout)\n\n        self.pool = pool\n        self.to_latent = nn.Identity()\n\n        self.mlp_head_bowel = nn.Sequential(\n            nn.LayerNorm(dim),\n            nn.Linear(dim, 2)\n        )\n        self.mlp_head_extravasation = nn.Sequential(\n            nn.LayerNorm(dim),\n            nn.Linear(dim, 2)\n        )\n        self.mlp_head_kidney = nn.Sequential(\n            nn.LayerNorm(dim),\n            nn.Linear(dim, 3)\n        )\n        self.mlp_head_liver = nn.Sequential(\n            nn.LayerNorm(dim),\n            nn.Linear(dim, 3)\n        )\n        self.mlp_head_spleen = nn.Sequential(\n            nn.LayerNorm(dim),\n            nn.Linear(dim, 3)\n        )\n        \n\n    def forward(self, x):\n        # x = self.to_patch_embedding(img)\n        print(x.shape)\n        b,n, _ = x.shape\n\n        cls_tokens = repeat(self.cls_token, '1 1 d -> b 1 d', b = b)\n#         print(\"cls_token: \", cls_tokens.shape)\n#         print(\"x shape: \", x.shape)  \n        x = torch.cat([cls_tokens, x], dim=1)\n#         print(\"After concat x shape: \", x.shape)\n#         print(\"pos_embedding shape: \", self.pos_embedding.shape)\n        x += self.pos_embedding[:, :(n + 1)]\n        x = self.dropout(x)\n\n        x = self.transformer(x)\n\n        x = x.mean(dim = 1) if self.pool == 'mean' else x[:, 0]\n\n        x = self.to_latent(x)\n        \n        return  self.mlp_head_bowel(x), self.mlp_head_extravasation(x), self.mlp_head_kidney(x), self.mlp_head_liver(x), self.mlp_head_spleen(x)","metadata":{"execution":{"iopub.status.busy":"2023-08-10T06:58:33.529591Z","iopub.execute_input":"2023-08-10T06:58:33.530183Z","iopub.status.idle":"2023-08-10T06:58:33.5639Z","shell.execute_reply.started":"2023-08-10T06:58:33.530116Z","shell.execute_reply":"2023-08-10T06:58:33.562962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def caformer_b36_384_in21ft1k(pretrained=False, **kwargs):\n#     model = MetaFormer(\n#         depths=[3, 12, 18, 3],\n#         dims=[128, 256, 512, 768],\n#         token_mixers=[SepConv, SepConv, Attention, Attention],\n#         head_fn=MlpHead,\n#         **kwargs)\n#     model.default_cfg = default_cfgs['caformer_b36_384_in21ft1k']\n#     if pretrained:\n#         pretrained_state_dict = torch.load(\"/kaggle/input/caformerdino/backbone.pth\")['model']\n#         state_dict = model.state_dict()\n#         new_state_dict = {}\n#         for k, v in pretrained_state_dict.items():\n#           new_k = k.replace(\"backbone.model.\", \"\")\n#           if new_k in state_dict.keys():\n#             new_state_dict[new_k] = v\n#         model.load_state_dict(new_state_dict, strict=False)\n#     return model","metadata":{"execution":{"iopub.status.busy":"2023-08-10T06:58:33.565389Z","iopub.execute_input":"2023-08-10T06:58:33.565738Z","iopub.status.idle":"2023-08-10T06:58:33.57712Z","shell.execute_reply.started":"2023-08-10T06:58:33.565707Z","shell.execute_reply":"2023-08-10T06:58:33.576207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class FeatureExtractor(nn.Module):\n    def __init__(self, name, model_path=None):\n        super().__init__()\n        self.model = MetaFormer(\n            depths=[3, 12, 18, 3],\n            dims=[128, 256, 512, 768],\n            token_mixers=[SepConv, SepConv, Meta_Attention, Meta_Attention],\n            head_fn=MlpHead)\n        self.model.default_cfg = default_cfgs[name]\n        if model_path is not None:\n            pretrained_state_dict = torch.load(model_path)['model']\n            state_dict = self.model.state_dict()\n            new_state_dict = {}\n            for k, v in pretrained_state_dict.items():\n              new_k = k.replace(\"backbone.model.\", \"\")\n              if new_k in state_dict.keys():\n                new_state_dict[new_k] = v\n            self.model.load_state_dict(new_state_dict, strict=False)\n   \n    def forward(self,x):\n#         y = torch.zeros((x.shape[0],768)).cuda()'\n        y = self.model.forward_features(x.squeeze(0))\n#         print('-------------',x.shape)\n#         x = x.mean(dim=1)\n        return y\n   \n","metadata":{"execution":{"iopub.status.busy":"2023-08-10T06:58:33.581013Z","iopub.execute_input":"2023-08-10T06:58:33.581293Z","iopub.status.idle":"2023-08-10T06:58:33.600322Z","shell.execute_reply.started":"2023-08-10T06:58:33.581269Z","shell.execute_reply":"2023-08-10T06:58:33.599318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_feature_worker(q_in, q_out):\n    model_paths = [\"/kaggle/input/caformerdino/backbone.pth\"]\n    models = []\n    for model_path in model_paths:\n        print('huhu')\n        model = FeatureExtractor(\"caformer_b36_384_in21ft1k\", model_path).to(\"cuda:0\")\n        model.eval()\n        models.append(model)\n        \n    with torch.no_grad():\n        print('llolo')\n        while True:\n            print('kk')\n            batch_queue = q_in.get()\n            if batch_queue is None:\n                print('jiji')\n                q_out.put(None, block=True, timeout=None)\n                break\n            print('jaja')\n            patient_id = batch_queue[\"patient_id\"]\n            series = batch_queue[\"series\"]\n            image_size = batch_queue[\"image_size\"]\n            batch = torch.Tensor(np.array(batch_queue[\"batch\"]))\n            batch = batch.to(\"cuda:0\").float()\n            ensemble_features = []\n            for i in range(len(models)):\n                #dim = (14, n_dim)\n                print('help')\n                features = models[i](batch)\n                ensemble_features.append(features)\n            #dim = (14, n_dim_1 + n_dim_2 + ... + n_dim_m)\n            ensemble_features = torch.cat(ensemble_features, 1).detach().cpu().numpy()\n            features_queue = {\"patient_id\": patient_id, \"series\": series, \"image_size\": image_size, \"batch\": ensemble_features}\n            q_out.put(features_queue, block=True, timeout=None)","metadata":{"execution":{"iopub.status.busy":"2023-08-10T06:58:33.601711Z","iopub.execute_input":"2023-08-10T06:58:33.602131Z","iopub.status.idle":"2023-08-10T06:58:33.614057Z","shell.execute_reply.started":"2023-08-10T06:58:33.602097Z","shell.execute_reply":"2023-08-10T06:58:33.613132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MultiTaskNet(nn.Module):\n    def __init__(self, in_features = 768):\n        super().__init__()\n  \n        self.vit = ViT(n_frames=40, dim=768, \n                       depth=2, heads=1, mlp_dim=32, pool = 'cls', channels = 3, dim_head = 64, dropout = 0., emb_dropout = 0.)\n \n    def forward(self, input):\n        input = input.unsqueeze(0)\n        output = self.vit(input)\n       \n        prd_bowel = output[0]\n        prd_extravasation = output[1]\n        prd_kidney = output[2]\n        prd_liver = output[3]\n        prd_spleen = output[4]\n       \n        \n        return output\n    ","metadata":{"execution":{"iopub.status.busy":"2023-08-10T06:58:33.615391Z","iopub.execute_input":"2023-08-10T06:58:33.61631Z","iopub.status.idle":"2023-08-10T06:58:33.631083Z","shell.execute_reply.started":"2023-08-10T06:58:33.616254Z","shell.execute_reply":"2023-08-10T06:58:33.630099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = torch.load('/kaggle/input/rsna-caformerdino-vit/best_model_caformer_vit.pt')","metadata":{"execution":{"iopub.status.busy":"2023-08-10T06:58:33.632405Z","iopub.execute_input":"2023-08-10T06:58:33.632919Z","iopub.status.idle":"2023-08-10T06:58:33.805693Z","shell.execute_reply.started":"2023-08-10T06:58:33.632885Z","shell.execute_reply":"2023-08-10T06:58:33.804566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input = torch.randn((14,768))\nmodel(input)","metadata":{"execution":{"iopub.status.busy":"2023-08-10T06:58:33.807314Z","iopub.execute_input":"2023-08-10T06:58:33.807721Z","iopub.status.idle":"2023-08-10T06:58:34.139456Z","shell.execute_reply.started":"2023-08-10T06:58:33.807685Z","shell.execute_reply":"2023-08-10T06:58:34.138394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_worker(q_in, d):\n    model_paths = [\"\"]\n    models = []\n    for model_path in model_paths:\n        model = torch.load('/kaggle/input/rsna-caformerdino-vit/best_model_caformer_vit.pt')\n        model = model.cuda()\n#         model.load_state_dict(torch.load('/kaggle/input/ckpt-rsna/model_caformer_vit.pt'))\n        model.eval()\n        models.append(model)\n    with torch.no_grad():\n        while True:\n            batch_queue = q_in.get()\n            if batch_queue is None:\n                break\n            patient_id = batch_queue[\"patient_id\"]\n            series = batch_queue[\"series\"]\n            image_size = batch_queue[\"image_size\"]\n            batch = np.array(batch_queue[\"batch\"])\n#             print(batch)\n            batch = torch.Tensor(batch).to(\"cuda:0\").float()\n            ensemble_predict = np.zeros((13, ))\n            for i in range(len(models)):\n                #dim = (14, n_dim)\n                print(models[i](batch))\n                out = models[i](batch)\n                pred = np.zeros((1,))\n                for p in out:\n                  ce_p = F.softmax(p, dim=-1)\n                  pred = np.concatenate([pred, ce_p[0].detach().cpu().numpy()], axis=-1)\n                print(pred)\n                predict = pred\n                print(predict.shape)\n                ensemble_predict += predict[1:]\n            ensemble_predict = ensemble_predict/len(models)\n            d[patient_id] = d[patient_id] + ensemble_predict","metadata":{"execution":{"iopub.status.busy":"2023-08-10T06:58:34.140983Z","iopub.execute_input":"2023-08-10T06:58:34.141424Z","iopub.status.idle":"2023-08-10T06:58:34.151924Z","shell.execute_reply.started":"2023-08-10T06:58:34.141389Z","shell.execute_reply":"2023-08-10T06:58:34.151036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = np.zeros((1,))\nb = np.zeros((2,))\nnp.concatenate([a,b], axis=-1)","metadata":{"execution":{"iopub.status.busy":"2023-08-10T06:58:34.15331Z","iopub.execute_input":"2023-08-10T06:58:34.15432Z","iopub.status.idle":"2023-08-10T06:58:34.1721Z","shell.execute_reply.started":"2023-08-10T06:58:34.154289Z","shell.execute_reply":"2023-08-10T06:58:34.170773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import multiprocessing as mp\nfrom joblib import Parallel, delayed\nfrom multiprocessing import Process, Queue, Manager, Lock\nimport pandas as pd\nimport glob\n\nq_in_extract_feature = Queue(maxsize=10)\nq_out_extract_feature = Queue(maxsize=10)\nmanager = Manager()\nd = manager.dict()\nq = Queue()\nfor patient_id in os.listdir(test_path):\n    d[patient_id] = np.zeros((13, ))\n\nextract_feature_process = Process(target=extract_feature_worker, args=(q_in_extract_feature, q_out_extract_feature, ))\nextract_feature_process.start()\nprint('hehe')\n\npredict_process = Process(target=predict_worker, args=(q_out_extract_feature, d, ))\npredict_process.start()\n\nIMAGE_SIZES = [384]\nfor patient_id in os.listdir(test_path):\n    patient_path = os.path.join(test_path, patient_id)\n    for series in os.listdir(patient_path):\n        series_path = os.path.join(patient_path, series)\n        images = sorted(glob.glob(f'/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images/{patient_id}/{series}/*.dcm'))\n        total_images = len(images)\n        step = int(total_images // N_EVAL)\n        series_batch = []\n        for image_size in IMAGE_SIZES:\n            series_batch.append(np.zeros((N_EVAL, 3, image_size, image_size)).astype(\"float32\"))\n        if step == 0:\n            print('step 0')\n            step = int(1)\n        count = 0\n        _ = Parallel(n_jobs=4,max_nbytes='50M')(\n            delayed(dcm_to_png)(images, patient_id, series, i, total_images, series_batch, image_sizes=IMAGE_SIZES, count=count)\n            for i in range(0, total_images, step)  # you can use train_images[:100] for testing\n        )\n        for image_size in IMAGE_SIZES:\n            print(':3')\n            batch_queue = {\"patient_id\": patient_id, \"series\": series, \"image_size\": image_size, \"batch\": series_batch}\n            q_in_extract_feature.put(batch_queue, block=True, timeout=None)\nq_in_extract_feature.put(None, block=True, timeout=None)\nextract_feature_process.join()\npredict_process.join()\n\nresults = []\nprint('koko')\n\nfor patient_id in os.listdir(test_path):\n    patient_path = os.path.join(test_path, patient_id)\n    num_series = len(os.listdir(patient_path))\n    d[patient_id] = d[patient_id] / (num_series*len(IMAGE_SIZES))\nprint('poop')\nfor k, v in d.items():\n    result = []\n    result.append(k)\n    for i in range(13):\n        result.append(v[i])\n    results.append(result)\n    \nresults = np.array(results)\nprint(results)\n# print(model_ensemble_fname.shape, model_ensemble_liveness_score.shape)\ndf = pd.DataFrame(results, columns=[\"patient_id\", \"bowel_healthy\", \"bowel_injury\", \"extravasation_healthy\", \"extravasation_injury\", \"kidney_healthy\",\n                                   \"kidney_low\", \"kidney_high\", \"liver_healthy\", \"liver_low\", \"liver_high\", \"spleen_healthy\", \"spleen_low\", \"spleen_high\"])\nprint('start export')\ndf.to_csv('/kaggle/working/submission.csv', index=False)\nprint('done export')","metadata":{"execution":{"iopub.status.busy":"2023-08-10T06:59:21.264634Z","iopub.execute_input":"2023-08-10T06:59:21.265031Z","iopub.status.idle":"2023-08-10T06:59:42.218471Z","shell.execute_reply.started":"2023-08-10T06:59:21.264993Z","shell.execute_reply":"2023-08-10T06:59:42.217082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}