{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Hello fellow Kagglers,\n\nThis notebook demonstrates the inference process using an EfficientNetV2T model training on a TPU.\n\nThe inference process consists of 2 steps.\n\n1) Preprocess and save all images in parallel, images are cropped, padded to correct aspect ratio and resized to the target size.\n\n2) Prediction on preprocessed images in a simple loop, final cancer score is mean of all images.\n\n**V3**\n\n* Switch to ConvNextV2 models\n\n**V6**\n\nThis will be the final update of my notebooks for this competitions, which should achieve a LB score in the low 0.50s. I will continue participating in this competition, however, I will not share my progress anymore.\n\n* Mainly updated preprocessing procedure, see examples under Example Preprocessing.\n* Correct linear/sigmoid normalization of images, many thanks to [Bob de Graaf](https://www.kaggle.com/bobdegraaf) which shared this amazing notebook: [DicomSDL & VOI-LUT](https://www.kaggle.com/code/bobdegraaf/dicomsdl-voi-lut)\n* Inference should be faster now due to actually using dicomsdl and not just installing it...\n\n**Training Notebook:** [RSNA ConvNextV2 Training Tensorflow TPU](https://www.kaggle.com/code/markwijkhuizen/rsna-efficientnetv2-training-tensorflow-tpu/notebook)\n\n**Data Preprocessing Notebook:** [RSNA Cropped TFRecords 768x1344 Dataset](https://www.kaggle.com/code/markwijkhuizen/rsna-cropped-tfrecords-768x1344-dataset/notebook)\n\n\nGood luck to all of you in the last month of this excisting competition!","metadata":{"papermill":{"duration":0.008021,"end_time":"2023-02-21T17:43:35.450058","exception":false,"start_time":"2023-02-21T17:43:35.442037","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%capture\n# Source: https://www.kaggle.com/code/remekkinas/fast-dicom-processing-1-6-2x-faster?scriptVersionId=113360473\n!pip install /kaggle/input/rsnamodules/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl \n\ntry:\n    import pylibjpeg\nexcept:\n   !pip install /kaggle/input/rsna-2022-whl/{pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:43:35.465334Z","iopub.status.busy":"2023-02-21T17:43:35.464814Z","iopub.status.idle":"2023-02-21T17:44:35.741433Z","shell.execute_reply":"2023-02-21T17:44:35.739916Z"},"papermill":{"duration":60.287679,"end_time":"2023-02-21T17:44:35.744474","exception":false,"start_time":"2023-02-21T17:43:35.456795","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Install Keras CV Attention Model Pip Package for ConvNextV2 Models\n!pip install  --no-deps /kaggle/input/keras-cv-attention-models/keras_cv_attention_models-1.3.9-py3-none-any.whl","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:44:35.759257Z","iopub.status.busy":"2023-02-21T17:44:35.758921Z","iopub.status.idle":"2023-02-21T17:44:58.105426Z","shell.execute_reply":"2023-02-21T17:44:58.104278Z"},"papermill":{"duration":22.35648,"end_time":"2023-02-21T17:44:58.10782","exception":false,"start_time":"2023-02-21T17:44:35.75134","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pylibjpeg\nimport pydicom\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\n\nfrom joblib import Parallel, delayed\nfrom tqdm.notebook import tqdm\nfrom multiprocessing import cpu_count\nfrom keras_cv_attention_models import convnext,efficientnet\n\nimport cv2\nimport glob\nimport importlib\nimport os\nimport joblib\nimport time\nimport dicomsdl\nimport gc\n\n# Tensorflow and CV2 set number of threads to 1 for speedup in parallell function mapping\ntf.config.threading.set_inter_op_parallelism_threads(num_threads=1)\ncv2.setNumThreads(1)\n\n# Pandas DataFrame Display Options\npd.options.display.max_colwidth = 99","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:44:58.122908Z","iopub.status.busy":"2023-02-21T17:44:58.122579Z","iopub.status.idle":"2023-02-21T17:45:03.648868Z","shell.execute_reply":"2023-02-21T17:45:03.647893Z"},"papermill":{"duration":5.536486,"end_time":"2023-02-21T17:45:03.651316","exception":false,"start_time":"2023-02-21T17:44:58.11483","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Config","metadata":{"papermill":{"duration":0.007084,"end_time":"2023-02-21T17:45:03.665362","exception":false,"start_time":"2023-02-21T17:45:03.658278","status":"completed"},"tags":[]}},{"cell_type":"code","source":"IS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\n\nTARGET_HEIGHT = 1344\nTARGET_WIDTH = 768\n\nN_CHANNELS = 1\nINPUT_SHAPE = (TARGET_HEIGHT, TARGET_WIDTH, N_CHANNELS)\nTARGET_HEIGHT_WIDTH_RATIO = TARGET_HEIGHT / TARGET_WIDTH\nTHRESHOLD_BEST = 0.5\n\nCLAHE = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(32, 32))\n\nCROP_IMAGE = True\nAPPLY_CLAHE = False\nAPPLY_EQ_HIST = False\n\nIMAGE_FORMAT = 'jpg'","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:03.681373Z","iopub.status.busy":"2023-02-21T17:45:03.680837Z","iopub.status.idle":"2023-02-21T17:45:03.689624Z","shell.execute_reply":"2023-02-21T17:45:03.688773Z"},"papermill":{"duration":0.01824,"end_time":"2023-02-21T17:45:03.691583","exception":false,"start_time":"2023-02-21T17:45:03.673343","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# VOI LUT","metadata":{"papermill":{"duration":0.006398,"end_time":"2023-02-21T17:45:03.704447","exception":false,"start_time":"2023-02-21T17:45:03.698049","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Source: https://www.kaggle.com/code/bobdegraaf/dicomsdl-voi-lut\ndef voi_lut(image, dicom):\n    # Additional Checks\n    if 'WindowWidth' not in dicom.getPixelDataInfo() or 'WindowWidth' not in dicom.getPixelDataInfo():\n        return image\n    \n    # Load only the variables we need\n    center = dicom['WindowCenter']\n    width = dicom['WindowWidth']\n    bits_stored = dicom['BitsStored']\n    voi_lut_function = dicom['VOILUTFunction']\n\n    # For sigmoid it's a list, otherwise a single value\n    if isinstance(center, list):\n        center = center[0]\n    if isinstance(width, list):\n        width = width[0]\n\n    # Set y_min, max & range\n    y_min = 0\n    y_max = float(2**bits_stored - 1)\n    y_range = y_max\n\n    # Function with default LINEAR (so for Nan, it will use linear)\n    if voi_lut_function == 'SIGMOID':\n        image = y_range / (1 + np.exp(-4 * (image - center) / width)) + y_min\n    else:\n        # Checks width for < 1 (in our case not necessary, always >= 750)\n        center -= 0.5\n        width -= 1\n\n        below = image <= (center - width / 2)\n        above = image > (center + width / 2)\n        between = np.logical_and(~below, ~above)\n\n        image[below] = y_min\n        image[above] = y_max\n        if between.any():\n            image[between] = (\n                ((image[between] - center) / width + 0.5) * y_range + y_min\n            )\n\n    return image","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:03.719365Z","iopub.status.busy":"2023-02-21T17:45:03.718607Z","iopub.status.idle":"2023-02-21T17:45:03.72883Z","shell.execute_reply":"2023-02-21T17:45:03.727951Z"},"papermill":{"duration":0.019808,"end_time":"2023-02-21T17:45:03.730786","exception":false,"start_time":"2023-02-21T17:45:03.710978","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Crop Image","metadata":{"papermill":{"duration":0.006295,"end_time":"2023-02-21T17:45:03.743735","exception":false,"start_time":"2023-02-21T17:45:03.73744","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Smooth vector used to smoothen sums/stds of axes\ndef smooth(l):\n    # kernel size is 1% of vector\n    kernel_size = int(len(l) * 0.01)\n    kernel = np.ones(kernel_size) / kernel_size\n    return np.convolve(l, kernel, mode='same')\n\n# X Crop offset based on first column with sum below 5% of maximum column sums*std\ndef get_x_offset(image, max_col_sum_ratio_threshold=0.05, debug=None):\n    # Image Dimensions\n    H, W = image.shape\n    # Percentual margin added to offset\n    margin = int(image.shape[1] * 0.00)\n    # Threshold values based on smoothed sum x std to capture varying intensity columns\n    vv = smooth(image.sum(axis=0).squeeze()) * smooth(image.std(axis=0).squeeze())\n    # Find maximum sum in first 75% of columns\n    vv_argmax = vv[:int(image.shape[1] * 0.75)].argmax()\n    # Threshold value\n    vv_threshold = vv.max() * max_col_sum_ratio_threshold\n    \n    # Find first column after maximum column below threshold value\n    for offset, v in enumerate(vv):\n        # Start searching from vv_argmax\n        if offset < vv_argmax:\n            continue\n        \n        # Column below threshold value found\n        if v < vv_threshold:\n            offset = min(W, offset + margin)\n            break\n            \n    if isinstance(debug, np.ndarray):\n        debug[1].imshow(image)\n        debug[1].set_title('X Offset')\n        vv_scale = H / vv.max() * 0.90\n        # Values\n        debug[1].plot(H - vv * vv_scale , c='red', label='vv')\n        # Threshold\n        debug[1].hlines(H - vv_threshold * vv_scale, 0, W -1, colors='orange', label='threshold')\n        # Max Value\n        debug[1].scatter(vv_argmax, H - vv[vv_argmax] * vv_scale, c='blue', s=100, label='Max', zorder=np.PINF)\n        # First Column Below Threshold\n        debug[1].scatter(offset, H - vv[offset] * vv_scale, c='purple', s=100, label='Offset', zorder=np.PINF)\n        debug[1].set_ylim(H, 0)\n        debug[1].legend()\n        debug[1].axis('off')\n        \n    return offset\n\n# Y Crop offset based on first bottom and top rows with sum below 10% of maximum row sum*std\ndef get_y_offsets(image, max_row_sum_ratio_threshold=0.10, debug=None):\n    # Image Dimensions\n    H, W = image.shape\n    # Margin to add to offsets\n    margin = 0\n    # Threshold values based on smoothed sum x std to capture varying intensity columns\n    vv = smooth(image.sum(axis=1).squeeze()) * smooth(image.std(axis=1).squeeze())\n    # Find maximum sum * std row in inter quartile rows\n    vv_argmax = int(image.shape[0] * 0.25) + vv[int(image.shape[0] * 0.25):int(image.shape[0] * 0.75)].argmax()\n    # Threshold value\n    vv_threshold = vv.max() * max_row_sum_ratio_threshold\n    # Default crop offsets\n    offset_bottom = 0\n    offset_top = H\n\n    # Bottom offset, search from argmax to bottom\n    for offset in reversed(range(0, vv_argmax)):\n        v = vv[offset]\n        if v < vv_threshold:\n            offset_bottom = offset\n            break\n    \n    if isinstance(debug, np.ndarray):\n        debug[2].imshow(image)\n        debug[2].set_title('Y Bottom Offset')\n        vv_scale = W / vv.max() * 0.90\n        # Values\n        debug[2].plot(vv * vv_scale, np.arange(H), c='red', label='vv')\n        # Threshold\n        debug[2].vlines(vv_threshold * vv_scale, 0, H -1, colors='orange', label='threshold')\n        # Max Value\n        debug[2].scatter(vv[vv_argmax] * vv_scale, vv_argmax, c='blue', s=100, label='Max', zorder=np.PINF)\n        # First Column Below Threshold\n        debug[2].scatter(vv[offset_bottom] * vv_scale, offset_bottom, c='purple', s=100, label='Offset', zorder=np.PINF)\n        debug[2].set_ylim(H, 0)\n        debug[2].legend()\n        debug[2].axis('off')\n            \n    # Top offset, search from argmax to top\n    for offset in range(vv_argmax, H):\n        v = vv[offset]\n        if v < vv_threshold:\n            offset_top = offset\n            break\n            \n    if isinstance(debug, np.ndarray):\n        debug[3].imshow(image)\n        debug[3].set_title('Y Top Offset')\n        vv_scale = W / vv.max() * 0.90\n        # Values\n        debug[3].plot(vv * vv_scale, np.arange(H) , c='red', label='vv')\n        # Threshold\n        debug[3].vlines(vv_threshold * vv_scale, 0, H -1, colors='orange', label='threshold')\n        # Max Value\n        debug[3].scatter(vv[vv_argmax] * vv_scale, vv_argmax, c='blue', s=100, label='Max', zorder=np.PINF)\n        # First Column Below Threshold\n        debug[3].scatter(vv[offset_top] * vv_scale, offset_top, c='purple', s=100, label='Offset', zorder=np.PINF)\n        debug[2].set_ylim(H, 0)\n        debug[3].legend()\n        debug[3].axis('off')\n            \n    return max(0, offset_bottom - margin), min(image.shape[0], offset_top + margin)\n\n# Crop image and pad offsets to target image height/width ratio to preserve information\ndef crop(image, size=None, debug=False):\n    # Image dimensions\n    H, W = image.shape\n    # Compute x/bottom/top offsets\n    x_offset = get_x_offset(image, debug=debug)\n    offset_bottom, offset_top = get_y_offsets(image[:,:x_offset], debug=debug)\n    # Crop Height and Width\n    h_crop = offset_top - offset_bottom\n    w_crop = x_offset\n    \n    # Pad crop offsets to target aspect ratio\n    if size is not None:\n        # Height too large, pad x offset\n        if (h_crop / w_crop) > TARGET_HEIGHT_WIDTH_RATIO:\n            x_offset += int(h_crop / TARGET_HEIGHT_WIDTH_RATIO - w_crop)\n        else:\n            # Height too small, pad bottom/top offsets\n            offset_bottom -= int(0.50 * (w_crop * TARGET_HEIGHT_WIDTH_RATIO - h_crop))\n            offset_bottom_correction = max(0, -offset_bottom)\n            offset_bottom += offset_bottom_correction\n\n            offset_top += int(0.50 * (w_crop * TARGET_HEIGHT_WIDTH_RATIO - h_crop))\n            offset_top += offset_bottom_correction\n        \n    # Crop Image\n    image = image[offset_bottom:offset_top:,:x_offset]\n        \n    return image","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:03.758402Z","iopub.status.busy":"2023-02-21T17:45:03.758043Z","iopub.status.idle":"2023-02-21T17:45:03.781841Z","shell.execute_reply":"2023-02-21T17:45:03.780915Z"},"papermill":{"duration":0.033836,"end_time":"2023-02-21T17:45:03.784045","exception":false,"start_time":"2023-02-21T17:45:03.750209","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Process Image","metadata":{"papermill":{"duration":0.006306,"end_time":"2023-02-21T17:45:03.796939","exception":false,"start_time":"2023-02-21T17:45:03.790633","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def process(file_path, size=(TARGET_WIDTH, TARGET_HEIGHT), crop_image=CROP_IMAGE, apply_clahe=APPLY_CLAHE, apply_eq_hist=APPLY_EQ_HIST, debug=False, save=True):\n    # Read Dicom File\n    dicom = dicomsdl.open(file_path)\n    image = dicom.pixelData()\n    \n    # Save original image for debug purposes\n    if debug:\n        fig, axes = plt.subplots(1, 5, figsize=(20,10))\n        image0 = np.copy(image)\n        axes[0].imshow(image0)\n        axes[0].set_title('Original Image')\n        axes[0].axis('off')\n    else:\n        axes = False\n    \n    # voi_lut\n    try:\n        image = voi_lut(image, dicom)\n    except:\n        pass\n    \n    # Some images have 0 values as highest intensity and need to be inverted\n    if dicom.getPixelDataInfo()['PhotometricInterpretation'] == 'MONOCHROME1':\n        image = np.max(image) - image\n\n    # Normalize [0,1] range\n    image = (image - image.min()) / (image.max() - image.min())\n\n    # Convert to uint8 image in range [0, 255]\n    image = (image * 255).astype(np.uint8)\n    \n    # Flip T0 Left/Right Orientation\n    h0, w0 = image.shape\n    if image[:,int(-w0 * 0.10):].sum() > image[:,:int(w0 * 0.10)].sum():\n        image = np.flip(image, axis=1)\n    \n    # Crop Image\n    if crop_image:\n        image = crop(image, debug=axes)\n        \n    # Resize\n    if size is not None:\n        # Pad black pixels to make square image\n        h, w = image.shape\n        if (h / w) > TARGET_HEIGHT_WIDTH_RATIO:\n            pad = int(h / TARGET_HEIGHT_WIDTH_RATIO - w)\n            image = np.pad(image, [[0,0], [0, pad]])\n            h, w = image.shape\n        else:\n            pad = int(0.50 * (w * TARGET_HEIGHT_WIDTH_RATIO - h))\n            image = np.pad(image, [[pad, pad], [0,0]])\n            h, w = image.shape\n        # Resize\n        image = cv2.resize(image, size, interpolation=cv2.INTER_AREA)\n        \n    # Apply CLAHE contrast enhancement\n    if apply_clahe:\n        image = CLAHE.apply(image)\n        \n     # Apply Histogram Equalization\n    if apply_eq_hist:\n        image = cv2.equalizeHist(image)\n        \n    # Show Processed Image    \n    if debug:\n        axes[4].imshow(image)\n        axes[4].set_title('Processed Image')\n        axes[4].axis('off')\n        plt.show()\n        \n    # Save Only\n    if save:\n        image_id = file_path.split('/')[-1].split('.')[0]\n        if IMAGE_FORMAT == 'png':\n            cv2.imwrite(f'{image_id}.png', image)\n        else:\n            cv2.imwrite(f'{image_id}.jpg', image, [cv2.IMWRITE_JPEG_QUALITY, 95])","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:03.811771Z","iopub.status.busy":"2023-02-21T17:45:03.810889Z","iopub.status.idle":"2023-02-21T17:45:03.824801Z","shell.execute_reply":"2023-02-21T17:45:03.823889Z"},"papermill":{"duration":0.023262,"end_time":"2023-02-21T17:45:03.826695","exception":false,"start_time":"2023-02-21T17:45:03.803433","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Example Preprocessing","metadata":{"papermill":{"duration":0.00639,"end_time":"2023-02-21T17:45:03.839594","exception":false,"start_time":"2023-02-21T17:45:03.833204","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\n    \ndef get_file_path(args):\n    patient_id, image_id = args\n    return f'/kaggle/input/rsna-breast-cancer-detection/train_images/{patient_id}/{image_id}.dcm'\n    \ntrain['file_path'] = train[['patient_id', 'image_id']].apply(get_file_path, axis=1)","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:03.854675Z","iopub.status.busy":"2023-02-21T17:45:03.853749Z","iopub.status.idle":"2023-02-21T17:45:04.376529Z","shell.execute_reply":"2023-02-21T17:45:04.375548Z"},"papermill":{"duration":0.532787,"end_time":"2023-02-21T17:45:04.379013","exception":false,"start_time":"2023-02-21T17:45:03.846226","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N = 8\n\nfor fp in tqdm(train['file_path'].head(N)):\n    process(fp, crop_image=True, size=(TARGET_WIDTH, TARGET_HEIGHT), debug=False, save=False)","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:04.395162Z","iopub.status.busy":"2023-02-21T17:45:04.393686Z","iopub.status.idle":"2023-02-21T17:45:12.686106Z","shell.execute_reply":"2023-02-21T17:45:12.685101Z"},"papermill":{"duration":8.302724,"end_time":"2023-02-21T17:45:12.688593","exception":false,"start_time":"2023-02-21T17:45:04.385869","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{"papermill":{"duration":0.006401,"end_time":"2023-02-21T17:45:12.702055","exception":false,"start_time":"2023-02-21T17:45:12.695654","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def normalize(image):\n    # Repeat channels to create 3 channel images required by pretrained ConvNextV2 models\n    image = tf.repeat(image, repeats=3, axis=3)\n    # Cast to float 32\n    image = tf.cast(image, tf.float32)\n    # Normalize with respect to ImageNet mean/std\n    image = tf.keras.applications.imagenet_utils.preprocess_input(image, mode='torch')\n\n    return image","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:12.71778Z","iopub.status.busy":"2023-02-21T17:45:12.716263Z","iopub.status.idle":"2023-02-21T17:45:12.722765Z","shell.execute_reply":"2023-02-21T17:45:12.721931Z"},"papermill":{"duration":0.016125,"end_time":"2023-02-21T17:45:12.724764","exception":false,"start_time":"2023-02-21T17:45:12.708639","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def get_model_1():\n#     # Inputs, note the names are equal to the dictionary keys in the dataset\n#     image = tf.keras.layers.Input(INPUT_SHAPE, name='image', dtype=tf.uint8)\n\n#     # Normalize Input\n#     image_norm = normalize(image)\n\n#     # CNN Feature Maps\n#     x = convnext.ConvNeXtV2Tiny(\n#         input_shape=(TARGET_HEIGHT, TARGET_WIDTH, 3),\n#         pretrained=None,\n#         num_classes=0,\n#     )(image_norm)\n\n#     # Average Pooling BxHxWxC -> BxC\n#     x = tf.keras.layers.GlobalAveragePooling2D()(x)\n#     # Dropout to prevent Overfitting\n#     x = tf.keras.layers.Dropout(0.50)(x)\n#     # Output value between [0, 1] using Sigmoid function\n#     outputs = tf.keras.layers.Dense(1, activation='sigmoid')(x)\n\n#     # Define model with inputs and outputs\n#     model = tf.keras.models.Model(inputs=image, outputs=outputs)\n\n#     # Load pretrained Model Weights\n#     model.load_weights('/kaggle/input/rsna-efficientnetv2-training-tensorflow-tpu-ds/model.h5')\n\n#     # Set model non-trainable\n#     model.trainable = False\n\n#     # Compile model\n#     model.compile()\n\n#     return model","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:12.739088Z","iopub.status.busy":"2023-02-21T17:45:12.738816Z","iopub.status.idle":"2023-02-21T17:45:12.743167Z","shell.execute_reply":"2023-02-21T17:45:12.742277Z"},"papermill":{"duration":0.013716,"end_time":"2023-02-21T17:45:12.745148","exception":false,"start_time":"2023-02-21T17:45:12.731432","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def get_model_2():\n#     # Inputs, note the names are equal to the dictionary keys in the dataset\n#     image = tf.keras.layers.Input(INPUT_SHAPE, name='image', dtype=tf.uint8)\n\n#     # Normalize Input\n#     image_norm = normalize(image)\n\n#     # CNN Feature Maps\n#     x = efficientnet.EfficientNetV2M(\n#         input_shape=(TARGET_HEIGHT, TARGET_WIDTH, 3),\n#         pretrained=None,\n#         num_classes=0,\n#     )(image_norm)\n\n#     # Average Pooling BxHxWxC -> BxC\n#     x = tf.keras.layers.GlobalAveragePooling2D()(x)\n#     # Dropout to prevent Overfitting\n#     x = tf.keras.layers.Dropout(0.30)(x)\n#     # Output value between [0, 1] using Sigmoid function\n#     outputs = tf.keras.layers.Dense(1, activation='sigmoid')(x)\n\n#     # Define model with inputs and outputs\n#     model = tf.keras.models.Model(inputs=image, outputs=outputs)\n\n#     # Load pretrained Model Weights\n#     model.load_weights('/kaggle/input/rsna-efficientnet-training-tensorflow-tpu/model.h5')\n\n#     # Set model non-trainable\n#     model.trainable = False\n\n#     # Compile model\n#     model.compile()\n\n#     return model","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:12.759874Z","iopub.status.busy":"2023-02-21T17:45:12.759077Z","iopub.status.idle":"2023-02-21T17:45:12.763735Z","shell.execute_reply":"2023-02-21T17:45:12.76278Z"},"papermill":{"duration":0.013961,"end_time":"2023-02-21T17:45:12.765621","exception":false,"start_time":"2023-02-21T17:45:12.75166","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model(path):\n    # Inputs, note the names are equal to the dictionary keys in the dataset\n    image = tf.keras.layers.Input(INPUT_SHAPE, name='image', dtype=tf.uint8)\n\n    # Normalize Input\n    image_norm = normalize(image)\n\n    # CNN Feature Maps\n    x = convnext.ConvNeXtV2Tiny(\n        input_shape=(TARGET_HEIGHT, TARGET_WIDTH, 3),\n        pretrained=None,\n        num_classes=0,\n    )(image_norm)\n\n    # Average Pooling BxHxWxC -> BxC\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    # Dropout to prevent Overfitting\n    x = tf.keras.layers.Dropout(0.1)(x)\n    # Output value between [0, 1] using Sigmoid function\n    outputs = tf.keras.layers.Dense(1, activation='sigmoid')(x)\n\n    # Define model with inputs and outputs\n    model = tf.keras.models.Model(inputs=image, outputs=outputs)\n\n    # Load pretrained Model Weights\n    model.load_weights(path)\n\n    # Set model non-trainable\n    model.trainable = False\n\n    # Compile model\n    model.compile()\n\n    return model","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:12.780473Z","iopub.status.busy":"2023-02-21T17:45:12.77961Z","iopub.status.idle":"2023-02-21T17:45:12.787187Z","shell.execute_reply":"2023-02-21T17:45:12.786388Z"},"papermill":{"duration":0.016987,"end_time":"2023-02-21T17:45:12.789121","exception":false,"start_time":"2023-02-21T17:45:12.772134","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pretrained File Path: '/kaggle/input/sartorius-training-dataset/model.h5'\ntf.keras.backend.clear_session()\n# enable XLA optmizations\ntf.config.optimizer.set_jit(True)\n\n# model_1 = get_model_1()\n# model_2 = get_model_2()\nmodel_1 = get_model('/kaggle/input/5-fold-tf/model_fold_0.h5')\nmodel_2 = get_model('/kaggle/input/5-fold-tf/model_fold_1.h5')\nmodel_3 = get_model('/kaggle/input/5-fold-tf/model_fold_2.h5')\n# model_4 = get_model('/kaggle/input/5-fold-tf/model_fold_3.h5')\n# model_5 = get_model('/kaggle/input/5-fold-tf/model_fold_4.h5')","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:12.803899Z","iopub.status.busy":"2023-02-21T17:45:12.80314Z","iopub.status.idle":"2023-02-21T17:45:28.531617Z","shell.execute_reply":"2023-02-21T17:45:28.530621Z"},"papermill":{"duration":15.738271,"end_time":"2023-02-21T17:45:28.534057","exception":false,"start_time":"2023-02-21T17:45:12.795786","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Plot model summary\n# model.summary()","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:28.551186Z","iopub.status.busy":"2023-02-21T17:45:28.549605Z","iopub.status.idle":"2023-02-21T17:45:28.554916Z","shell.execute_reply":"2023-02-21T17:45:28.554081Z"},"papermill":{"duration":0.015381,"end_time":"2023-02-21T17:45:28.556931","exception":false,"start_time":"2023-02-21T17:45:28.54155","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test","metadata":{"papermill":{"duration":0.006649,"end_time":"2023-02-21T17:45:28.570356","exception":false,"start_time":"2023-02-21T17:45:28.563707","status":"completed"},"tags":[]}},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n\ndef get_file_path(args):\n    patient_id, image_id = args\n    return f'/kaggle/input/rsna-breast-cancer-detection/test_images/{patient_id}/{image_id}.dcm'\n    \ntest['file_path'] = test[['patient_id', 'image_id']].apply(get_file_path, axis=1)\n\ndisplay(test.info())\ndisplay(test.head())","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:28.586401Z","iopub.status.busy":"2023-02-21T17:45:28.5849Z","iopub.status.idle":"2023-02-21T17:45:28.62398Z","shell.execute_reply":"2023-02-21T17:45:28.62283Z"},"papermill":{"duration":0.049394,"end_time":"2023-02-21T17:45:28.626477","exception":false,"start_time":"2023-02-21T17:45:28.577083","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sample Submission","metadata":{"papermill":{"duration":0.006894,"end_time":"2023-02-21T17:45:28.641134","exception":false,"start_time":"2023-02-21T17:45:28.63424","status":"completed"},"tags":[]}},{"cell_type":"code","source":"sample_submission = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv')\n\ndisplay(sample_submission.info())\ndisplay(sample_submission.head())","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:28.656477Z","iopub.status.busy":"2023-02-21T17:45:28.656159Z","iopub.status.idle":"2023-02-21T17:45:28.683321Z","shell.execute_reply":"2023-02-21T17:45:28.68222Z"},"papermill":{"duration":0.03757,"end_time":"2023-02-21T17:45:28.685732","exception":false,"start_time":"2023-02-21T17:45:28.648162","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Parallel Image Processing","metadata":{"papermill":{"duration":0.008424,"end_time":"2023-02-21T17:45:28.703815","exception":false,"start_time":"2023-02-21T17:45:28.695391","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Preprocess a single image and saves it\ndef preprocess_and_save_image(args):\n    (patient_id, laterality), g = args\n    cancer = 0.0\n    for row_idx, row in g.iterrows():\n        process(row['file_path'])","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:28.723182Z","iopub.status.busy":"2023-02-21T17:45:28.722228Z","iopub.status.idle":"2023-02-21T17:45:28.728957Z","shell.execute_reply":"2023-02-21T17:45:28.727928Z"},"papermill":{"duration":0.018524,"end_time":"2023-02-21T17:45:28.731042","exception":false,"start_time":"2023-02-21T17:45:28.712518","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocess all images in parallel using Joblib\njobs = [joblib.delayed(preprocess_and_save_image)(args) for args in test.groupby(['patient_id', 'laterality'])]\nSUBMISSION_ROWS = joblib.Parallel(\n    n_jobs=cpu_count(),\n    verbose=IS_INTERACTIVE,\n    backend='multiprocessing',\n    prefer='threads',\n)(jobs)","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:28.750307Z","iopub.status.busy":"2023-02-21T17:45:28.749982Z","iopub.status.idle":"2023-02-21T17:45:31.40872Z","shell.execute_reply":"2023-02-21T17:45:31.407222Z"},"papermill":{"duration":2.671567,"end_time":"2023-02-21T17:45:31.411437","exception":false,"start_time":"2023-02-21T17:45:28.73987","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{"papermill":{"duration":0.007122,"end_time":"2023-02-21T17:45:31.426372","exception":false,"start_time":"2023-02-21T17:45:31.41925","status":"completed"},"tags":[]}},{"cell_type":"code","source":"SUBMISSION_ROWS = []\n# Iterate over all patient_id/laterality combinations groups\nfor idx, ((patient_id, laterality), g) in enumerate(tqdm(test.groupby(['patient_id', 'laterality']))):\n    # Cancer target is mean of predicted cancer values\n    cancer = 0\n    cancer_pred_1 = 0\n    cancer_pred_2 = 0\n    cancer_pred_3 = 0\n    cancer_pred_4 = 0\n    cancer_pred_5 = 0\n    \n    # Iterate over all scans in group\n    for row_idx, row in g.iterrows():\n        # Load Image\n        image_id = row['image_id']\n        image = cv2.imread(f'{image_id}.{IMAGE_FORMAT}', -1)\n        # Show First Few Images\n#         if idx < 16:\n#             plt.figure(figsize=(5,8))\n#             plt.imshow(image)\n#             plt.show()\n        \n        # Expand to Batch HxW -> 1xHxWx1\n        image = np.expand_dims(image, [0, 3])\n        # Make Prediction\n        cancer_pred_1 += model_1.predict_on_batch(image).squeeze() / len(g)\n        cancer_pred_2 += model_2.predict_on_batch(image).squeeze() / len(g)\n        cancer_pred_3 += model_3.predict_on_batch(image).squeeze() / len(g)\n#         cancer_pred_4 += model_4.predict_on_batch(image).squeeze() / len(g)\n#         cancer_pred_5 += model_5.predict_on_batch(image).squeeze() / len(g)\n        \n        \n        # Remove Image\n        os.remove(f'{image_id}.{IMAGE_FORMAT}')\n        \n    # Add Submission Row\n    cancer = (cancer_pred_1+ cancer_pred_2+ cancer_pred_3)/3\n    th = np.quantile(cancer,0.97935)\n\n    SUBMISSION_ROWS.append({\n        'prediction_id': f'{patient_id}_{laterality}',\n        'cancer': np.int8(cancer > th),\n    })\n    \n    if np.random.rand() > 0.99:\n        gc.collect()","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:31.444567Z","iopub.status.busy":"2023-02-21T17:45:31.442823Z","iopub.status.idle":"2023-02-21T17:45:52.809475Z","shell.execute_reply":"2023-02-21T17:45:52.808454Z"},"papermill":{"duration":21.377418,"end_time":"2023-02-21T17:45:52.811607","exception":false,"start_time":"2023-02-21T17:45:31.434189","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save Submission","metadata":{"papermill":{"duration":0.007436,"end_time":"2023-02-21T17:45:52.827055","exception":false,"start_time":"2023-02-21T17:45:52.819619","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Create DataFrame from submission rows\nsubmission_df = pd.DataFrame(SUBMISSION_ROWS)\n\ndisplay(submission_df.info())\ndisplay(submission_df.head())","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:52.843545Z","iopub.status.busy":"2023-02-21T17:45:52.843209Z","iopub.status.idle":"2023-02-21T17:45:52.863486Z","shell.execute_reply":"2023-02-21T17:45:52.862429Z"},"papermill":{"duration":0.030977,"end_time":"2023-02-21T17:45:52.86562","exception":false,"start_time":"2023-02-21T17:45:52.834643","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save submission as CSV\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:52.882942Z","iopub.status.busy":"2023-02-21T17:45:52.882652Z","iopub.status.idle":"2023-02-21T17:45:52.89012Z","shell.execute_reply":"2023-02-21T17:45:52.889222Z"},"papermill":{"duration":0.018547,"end_time":"2023-02-21T17:45:52.892233","exception":false,"start_time":"2023-02-21T17:45:52.873686","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sanity Check\ndisplay(pd.read_csv('submission.csv').head())","metadata":{"execution":{"iopub.execute_input":"2023-02-21T17:45:52.90948Z","iopub.status.busy":"2023-02-21T17:45:52.909175Z","iopub.status.idle":"2023-02-21T17:45:52.920957Z","shell.execute_reply":"2023-02-21T17:45:52.920039Z"},"papermill":{"duration":0.022906,"end_time":"2023-02-21T17:45:52.922997","exception":false,"start_time":"2023-02-21T17:45:52.900091","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}