{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%capture\n!pip install /kaggle/input/rsnamodules/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl \n\ntry:\n    import pylibjpeg\nexcept:\n   !pip install /kaggle/input/rsna-2022-whl/{pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-12T02:52:38.872017Z","iopub.execute_input":"2023-03-12T02:52:38.87281Z","iopub.status.idle":"2023-03-12T02:53:05.266606Z","shell.execute_reply.started":"2023-03-12T02:52:38.872759Z","shell.execute_reply":"2023-03-12T02:53:05.264861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pylibjpeg\nimport pydicom\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\n\nfrom joblib import Parallel, delayed\nfrom tqdm.notebook import tqdm\nfrom multiprocessing import cpu_count\n\nimport cv2\nimport glob\nimport importlib\nimport os\nimport joblib\nimport sys\nimport dicomsdl\n\nprint(f'Tensorflow Version: {tf.__version__}')\nprint(f'Python Version: {sys.version}')\n\ntf.config.threading.set_inter_op_parallelism_threads(num_threads=1)\ncv2.setNumThreads(1)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:53:05.269395Z","iopub.execute_input":"2023-03-12T02:53:05.269798Z","iopub.status.idle":"2023-03-12T02:53:15.07424Z","shell.execute_reply.started":"2023-03-12T02:53:05.269755Z","shell.execute_reply":"2023-03-12T02:53:15.072672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\n\nTARGET_HEIGHT = 1344\nTARGET_WIDTH = 768\nN_CHANNELS = 1\nTARGET_HEIGHT_WIDTH_RATIO = TARGET_HEIGHT / TARGET_WIDTH\n\nCLAHE = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(32, 32))\nAPPLY_CLAHE = False\nAPPLY_EQ_HIST = False\n\nIMAGE_FORMAT = 'JPG'\nIMAGE_QUALITY = 95\n\nSEED = 42","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:53:15.075933Z","iopub.execute_input":"2023-03-12T02:53:15.076684Z","iopub.status.idle":"2023-03-12T02:53:15.088328Z","shell.execute_reply.started":"2023-03-12T02:53:15.076636Z","shell.execute_reply":"2023-03-12T02:53:15.087186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mpl.rcParams.update(mpl.rcParamsDefault)\nmpl.rcParams['xtick.labelsize'] = 16\nmpl.rcParams['ytick.labelsize'] = 16\nmpl.rcParams['axes.labelsize'] = 18\nmpl.rcParams['axes.titlesize'] = 24","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:53:15.089835Z","iopub.execute_input":"2023-03-12T02:53:15.090463Z","iopub.status.idle":"2023-03-12T02:53:15.14038Z","shell.execute_reply.started":"2023-03-12T02:53:15.090419Z","shell.execute_reply":"2023-03-12T02:53:15.139013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if IS_INTERACTIVE:\n    train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv').head(1024)\nelse:\n    train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\n    \ndef get_file_path(args):\n    patient_id, image_id = args\n    return f'/kaggle/input/rsna-breast-cancer-detection/train_images/{patient_id}/{image_id}.dcm'\n    \ntrain['file_path'] = train[['patient_id', 'image_id']].apply(get_file_path, axis=1)\n    \ndisplay(train.info())\ndisplay(train.head())","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:53:15.144348Z","iopub.execute_input":"2023-03-12T02:53:15.144785Z","iopub.status.idle":"2023-03-12T02:53:15.334695Z","shell.execute_reply.started":"2023-03-12T02:53:15.144745Z","shell.execute_reply":"2023-03-12T02:53:15.333444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def voi_lut(image, dicom):\n    center = dicom['WindowCenter']\n    width = dicom['WindowWidth']\n    bits_stored = dicom['BitsStored']\n    voi_lut_function = dicom['VOILUTFunction']\n    \n    if isinstance(center, list):\n        center = center[0]\n    if isinstance(width, list):\n        width = width[0]\n\n    y_min = 0\n    y_max = float(2**bits_stored - 1)\n    y_range = y_max\n\n    if voi_lut_function == 'SIGMOID':\n        image = y_range / (1 + np.exp(-4 * (image - center) / width)) + y_min\n    else:\n        center -= 0.5\n        width -= 1\n\n        below = image <= (center - width / 2)\n        above = image > (center + width / 2)\n        between = np.logical_and(~below, ~above)\n\n        image[below] = y_min\n        image[above] = y_max\n        if between.any():\n            image[between] = (\n                ((image[between] - center) / width + 0.5) * y_range + y_min\n            )\n\n    #some images are inverted\n    if dicom['PhotometricInterpretation'] == 'MONOCHROME1':\n        image = np.max(image) - image\n\n    return image","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:53:15.336643Z","iopub.execute_input":"2023-03-12T02:53:15.337144Z","iopub.status.idle":"2023-03-12T02:53:15.349583Z","shell.execute_reply.started":"2023-03-12T02:53:15.337094Z","shell.execute_reply":"2023-03-12T02:53:15.348262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def smooth(l):\n    kernel_size = int(len(l) * 0.01)\n    kernel = np.ones(kernel_size) / kernel_size\n    return np.convolve(l, kernel, mode='same')\n\ndef get_x_offset(image, max_col_sum_ratio_threshold=0.05, debug=None):\n    H, W = image.shape\n    margin = int(image.shape[1] * 0.00)\n    vv = smooth(image.sum(axis=0).squeeze()) * smooth(image.std(axis=0).squeeze())\n    vv_argmax = vv[:int(image.shape[1] * 0.75)].argmax()\n    vv_threshold = vv.max() * max_col_sum_ratio_threshold\n    \n    for offset, v in enumerate(vv):\n        if offset < vv_argmax:\n            continue\n       \n        if v < vv_threshold:\n            offset = min(W, offset + margin)\n            break\n            \n    if isinstance(debug, np.ndarray):\n        debug[1].imshow(image)\n        debug[1].set_title('X Offset')\n        vv_scale = H / vv.max() * 0.90\n        debug[1].plot(H - vv * vv_scale , c='red', label='vv')\n        debug[1].hlines(H - vv_threshold * vv_scale, 0, W -1, colors='orange', label='threshold')\n        debug[1].scatter(vv_argmax, H - vv[vv_argmax] * vv_scale, c='blue', s=100, label='Max', zorder=np.PINF)\n        debug[1].scatter(offset, H - vv[offset] * vv_scale, c='purple', s=100, label='Offset', zorder=np.PINF)\n        debug[1].set_ylim(H, 0)\n        debug[1].legend()\n        debug[1].axis('off')\n        \n    return offset\n\ndef get_y_offsets(image, max_row_sum_ratio_threshold=0.10, debug=None):\n    H, W = image.shape\n    margin = 0\n    vv = smooth(image.sum(axis=1).squeeze()) * smooth(image.std(axis=1).squeeze())\n    vv_argmax = int(image.shape[0] * 0.25) + vv[int(image.shape[0] * 0.25):int(image.shape[0] * 0.75)].argmax()\n    vv_threshold = vv.max() * max_row_sum_ratio_threshold\n    offset_bottom = 0\n    offset_top = H\n\n    for offset in reversed(range(0, vv_argmax)):\n        v = vv[offset]\n        if v < vv_threshold:\n            offset_bottom = offset\n            break\n    \n    if isinstance(debug, np.ndarray):\n        debug[2].imshow(image)\n        debug[2].set_title('Y Bottom Offset')\n        vv_scale = W / vv.max() * 0.90\n        debug[2].plot(vv * vv_scale, np.arange(H), c='red', label='vv')\n        debug[2].vlines(vv_threshold * vv_scale, 0, H -1, colors='orange', label='threshold')\n        debug[2].scatter(vv[vv_argmax] * vv_scale, vv_argmax, c='blue', s=100, label='Max', zorder=np.PINF)\n        debug[2].scatter(vv[offset_bottom] * vv_scale, offset_bottom, c='purple', s=100, label='Offset', zorder=np.PINF)\n        debug[2].set_ylim(H, 0)\n        debug[2].legend()\n        debug[2].axis('off')\n            \n    for offset in range(vv_argmax, H):\n        v = vv[offset]\n        if v < vv_threshold:\n            offset_top = offset\n            break\n            \n    if isinstance(debug, np.ndarray):\n        debug[3].imshow(image)\n        debug[3].set_title('Y Top Offset')\n        vv_scale = W / vv.max() * 0.90\n        debug[3].plot(vv * vv_scale, np.arange(H) , c='red', label='vv')\n        debug[3].vlines(vv_threshold * vv_scale, 0, H -1, colors='orange', label='threshold')\n        debug[3].scatter(vv[vv_argmax] * vv_scale, vv_argmax, c='blue', s=100, label='Max', zorder=np.PINF)\n        debug[3].scatter(vv[offset_top] * vv_scale, offset_top, c='purple', s=100, label='Offset', zorder=np.PINF)\n        debug[2].set_ylim(H, 0)\n        debug[3].legend()\n        debug[3].axis('off')\n            \n    return max(0, offset_bottom - margin), min(image.shape[0], offset_top + margin)\n\ndef crop(image, size=None, debug=False):\n    H, W = image.shape\n    x_offset = get_x_offset(image, debug=debug)\n    offset_bottom, offset_top = get_y_offsets(image[:,:x_offset], debug=debug)\n    h_crop = offset_top - offset_bottom\n    w_crop = x_offset\n    \n    if size is not None:\n        if (h_crop / w_crop) > TARGET_HEIGHT_WIDTH_RATIO:\n            x_offset += int(h_crop / TARGET_HEIGHT_WIDTH_RATIO - w_crop)\n        else:\n            offset_bottom -= int(0.50 * (w_crop * TARGET_HEIGHT_WIDTH_RATIO - h_crop))\n            offset_bottom_correction = max(0, -offset_bottom)\n            offset_bottom += offset_bottom_correction\n\n            offset_top += int(0.50 * (w_crop * TARGET_HEIGHT_WIDTH_RATIO - h_crop))\n            offset_top += offset_bottom_correction\n        \n    image = image[offset_bottom:offset_top:,:x_offset]\n        \n    return image","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:53:15.352015Z","iopub.execute_input":"2023-03-12T02:53:15.352724Z","iopub.status.idle":"2023-03-12T02:53:15.384324Z","shell.execute_reply.started":"2023-03-12T02:53:15.352681Z","shell.execute_reply":"2023-03-12T02:53:15.382902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(file_path, size=None, dicom_process=True, ret_target=False, crop_image=False, apply_clahe=APPLY_CLAHE, apply_eq_hist=APPLY_EQ_HIST, debug=False):\n    dicom = dicomsdl.open(file_path)\n    image = dicom.pixelData()\n    \n    if debug:\n        fig, axes = plt.subplots(1, 5, figsize=(20,10))\n        image0 = np.copy(image)\n        axes[0].imshow(image0)\n        axes[0].set_title('Original Image')\n        axes[0].axis('off')\n    else:\n        axes = False\n        \n    image = voi_lut(image, dicom)\n\n    image = (image - image.min()) / (image.max() - image.min())\n\n    image = (image * 255).astype(np.uint8)\n    \n    h0, w0 = image.shape\n    if image[:,int(-w0 * 0.10):].sum() > image[:,:int(w0 * 0.10)].sum():\n        image = np.flip(image, axis=1)\n    \n    if crop_image:\n        image = crop(image, size=size, debug=axes)\n    \n    if size is not None:\n        h, w = image.shape\n        if (h / w) > TARGET_HEIGHT_WIDTH_RATIO:\n            pad = int(h / TARGET_HEIGHT_WIDTH_RATIO - w)\n            image = np.pad(image, [[0,0], [0, pad]])\n            h, w = image.shape\n        else:\n            pad = int(0.50 * (w * TARGET_HEIGHT_WIDTH_RATIO - h))\n            image = np.pad(image, [[pad, pad], [0,0]])\n            h, w = image.shape\n        image = cv2.resize(image, size, interpolation=cv2.INTER_AREA)\n        \n    if apply_clahe:\n        image = CLAHE.apply(image)\n        \n    if apply_eq_hist:\n        image = cv2.equalizeHist(image)\n        \n    if debug:\n        axes[4].imshow(image)\n        axes[4].set_title('Processed Image')\n        axes[4].axis('off')\n        plt.show()\n\n    if ret_target:\n        patient_id = int(file_path.split('/')[-2])\n        image_id = int(file_path.split('/')[-1].split('.')[0])\n\n        target = PATIENT_ID_IMAGE_ID2CANCER[(patient_id, image_id)]\n        \n        return image, target\n    else:\n        if debug:\n            return image0, image\n        else:\n            return image","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:53:15.38628Z","iopub.execute_input":"2023-03-12T02:53:15.387386Z","iopub.status.idle":"2023-03-12T02:53:15.405885Z","shell.execute_reply.started":"2023-03-12T02:53:15.387337Z","shell.execute_reply":"2023-03-12T02:53:15.404447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_original_processed_examples(rows=48, cols=5):\n    fig, axes = plt.subplots(rows, cols, figsize=(cols * 4, 6 * rows))\n    for r in tqdm(range(rows)):\n        for c in range(cols):\n            idx = (r * cols) + c\n            image = process(\n                    train.loc[idx, 'file_path'],\n                    crop_image=True,\n                    size=(TARGET_WIDTH, TARGET_HEIGHT),\n                    apply_clahe=APPLY_CLAHE,\n                    apply_eq_hist=APPLY_EQ_HIST,\n                    debug=False,\n                )\n            axes[r, c].imshow(image)\n            axes[r, c].set_title(f'{idx} | processed')\n\n    plt.show()\n    \nplot_original_processed_examples(rows=8 if IS_INTERACTIVE else 32)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:53:15.408264Z","iopub.execute_input":"2023-03-12T02:53:15.409125Z","iopub.status.idle":"2023-03-12T02:54:17.459045Z","shell.execute_reply.started":"2023-03-12T02:53:15.409079Z","shell.execute_reply":"2023-03-12T02:54:17.457198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not IS_INTERACTIVE:\n    for g_idx, g in tqdm(train.groupby('patient_id')):\n        if 'CC' not in g['view'].values or 'MLO' not in g['view'].values:\n            display(g)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:54:17.461735Z","iopub.execute_input":"2023-03-12T02:54:17.462718Z","iopub.status.idle":"2023-03-12T02:54:17.471971Z","shell.execute_reply.started":"2023-03-12T02:54:17.462641Z","shell.execute_reply":"2023-03-12T02:54:17.470444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATIENT_ID_IMAGE_ID2CANCER = train.set_index(['patient_id', 'image_id'])['cancer'].to_dict()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:54:17.473641Z","iopub.execute_input":"2023-03-12T02:54:17.474809Z","iopub.status.idle":"2023-03-12T02:54:17.491276Z","shell.execute_reply.started":"2023-03-12T02:54:17.47476Z","shell.execute_reply":"2023-03-12T02:54:17.490281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n\ndisplay(test.info())\ndisplay(test.head())","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:54:17.492841Z","iopub.execute_input":"2023-03-12T02:54:17.493698Z","iopub.status.idle":"2023-03-12T02:54:17.525697Z","shell.execute_reply.started":"2023-03-12T02:54:17.493655Z","shell.execute_reply":"2023-03-12T02:54:17.524462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv')\n\ndisplay(sample_submission.info())\ndisplay(sample_submission.head())","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:54:17.527394Z","iopub.execute_input":"2023-03-12T02:54:17.528031Z","iopub.status.idle":"2023-03-12T02:54:17.554563Z","shell.execute_reply.started":"2023-03-12T02:54:17.527991Z","shell.execute_reply":"2023-03-12T02:54:17.553366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 8))\ntrain['cancer'].value_counts().plot(kind='pie', autopct='%1.1f%%', title='Cancer Distribution')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:54:17.560245Z","iopub.execute_input":"2023-03-12T02:54:17.560661Z","iopub.status.idle":"2023-03-12T02:54:17.819288Z","shell.execute_reply.started":"2023-03-12T02:54:17.560619Z","shell.execute_reply":"2023-03-12T02:54:17.817444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FOLDER_PATHS = glob.glob('/kaggle/input/rsna-breast-cancer-detection/train_images/*')\nprint(f'Found {len(FOLDER_PATHS)} Train Folders')","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:54:17.821657Z","iopub.execute_input":"2023-03-12T02:54:17.823448Z","iopub.status.idle":"2023-03-12T02:54:18.070327Z","shell.execute_reply.started":"2023-03-12T02:54:17.823362Z","shell.execute_reply":"2023-03-12T02:54:18.06881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FILE_PATHS = glob.glob('/kaggle/input/rsna-breast-cancer-detection/train_images/*/*.dcm')\nprint(f'Found {len(FILE_PATHS)} Train Files')","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:54:18.072089Z","iopub.execute_input":"2023-03-12T02:54:18.072623Z","iopub.status.idle":"2023-03-12T02:55:15.202552Z","shell.execute_reply.started":"2023-03-12T02:54:18.072564Z","shell.execute_reply":"2023-03-12T02:55:15.201211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,8))\nplt.title('Number of Scans Per Patient per Laterality (Left/Right side)')\ntrain.groupby(['patient_id', 'laterality']).apply(len).value_counts().sort_index().plot(kind='bar')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:55:15.20409Z","iopub.execute_input":"2023-03-12T02:55:15.204776Z","iopub.status.idle":"2023-03-12T02:55:15.491441Z","shell.execute_reply.started":"2023-03-12T02:55:15.204732Z","shell.execute_reply":"2023-03-12T02:55:15.489883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train['view'].value_counts().to_frame('count'))","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:55:15.493756Z","iopub.execute_input":"2023-03-12T02:55:15.49428Z","iopub.status.idle":"2023-03-12T02:55:15.507727Z","shell.execute_reply.started":"2023-03-12T02:55:15.49422Z","shell.execute_reply":"2023-03-12T02:55:15.505987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train.sort_values('view').groupby(['patient_id', 'laterality'])['view'].apply(tuple).value_counts().to_frame('Count'))","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:55:15.509808Z","iopub.execute_input":"2023-03-12T02:55:15.510204Z","iopub.status.idle":"2023-03-12T02:55:15.541112Z","shell.execute_reply.started":"2023-03-12T02:55:15.510166Z","shell.execute_reply":"2023-03-12T02:55:15.540073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(42)\n\n# Get height/width statistics\nN = int(16 if IS_INTERACTIVE else 1024)\nWIDTHS = []\nHEIGHTS = []\nfor fp in tqdm(np.random.choice(FILE_PATHS, N)):\n    h, w = process(fp).shape\n    HEIGHTS.append(h)\n    WIDTHS.append(w)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:55:15.542735Z","iopub.execute_input":"2023-03-12T02:55:15.543736Z","iopub.status.idle":"2023-03-12T02:55:32.833031Z","shell.execute_reply.started":"2023-03-12T02:55:15.543677Z","shell.execute_reply":"2023-03-12T02:55:32.831507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,8))\nplt.title('Image Dimensions', size=24)\npd.Series(HEIGHTS).plot(kind='hist', alpha=0.50, label='heights')\npd.Series(WIDTHS).plot(kind='hist', alpha=0.50, label='widths')\nplt.grid()\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:55:32.835104Z","iopub.execute_input":"2023-03-12T02:55:32.835634Z","iopub.status.idle":"2023-03-12T02:55:33.20124Z","shell.execute_reply.started":"2023-03-12T02:55:32.835579Z","shell.execute_reply":"2023-03-12T02:55:33.199964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(pd.Series(np.array(HEIGHTS) / np.array(WIDTHS)).describe().to_frame('Height/Width Ratio\\'s'))\n\nplt.figure(figsize=(15,8))\npd.Series(np.array(HEIGHTS) / np.array(WIDTHS)).plot(kind='hist')\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:55:33.202814Z","iopub.execute_input":"2023-03-12T02:55:33.203274Z","iopub.status.idle":"2023-03-12T02:55:33.490048Z","shell.execute_reply.started":"2023-03-12T02:55:33.203228Z","shell.execute_reply":"2023-03-12T02:55:33.488781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(SEED)\n\nN = int(16 if IS_INTERACTIVE else 1024)\nWIDTHS_CROPPED = []\nHEIGHTS_CROPPED = []\n\nfor fp in tqdm(np.random.choice(FILE_PATHS, N)):\n    h, w = process(fp, crop_image=True).shape\n    WIDTHS_CROPPED.append(h)\n    HEIGHTS_CROPPED.append(w)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:55:33.492523Z","iopub.execute_input":"2023-03-12T02:55:33.493848Z","iopub.status.idle":"2023-03-12T02:55:50.600241Z","shell.execute_reply.started":"2023-03-12T02:55:33.493795Z","shell.execute_reply":"2023-03-12T02:55:50.598767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,8))\nplt.title('Cropped Image Dimensions', size=24)\npd.Series(WIDTHS_CROPPED).plot(kind='hist', alpha=0.50, label='cropped heights')\npd.Series(HEIGHTS_CROPPED).plot(kind='hist', alpha=0.50, label='cropped widths')\nplt.grid()\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:55:50.601888Z","iopub.execute_input":"2023-03-12T02:55:50.602346Z","iopub.status.idle":"2023-03-12T02:55:50.942687Z","shell.execute_reply.started":"2023-03-12T02:55:50.602308Z","shell.execute_reply":"2023-03-12T02:55:50.941483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(pd.Series(np.array(WIDTHS_CROPPED) / np.array(HEIGHTS_CROPPED)).describe().to_frame('Cropped Height/Width Ratio\\'s'))\nplt.figure(figsize=(15,8))\npd.Series(np.array(WIDTHS_CROPPED) / np.array(HEIGHTS_CROPPED)).plot(kind='hist')\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:55:50.944574Z","iopub.execute_input":"2023-03-12T02:55:50.945117Z","iopub.status.idle":"2023-03-12T02:55:51.256341Z","shell.execute_reply.started":"2023-03-12T02:55:50.945065Z","shell.execute_reply":"2023-03-12T02:55:51.254736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FILE_PATHS_PAIRS = []\nfor row_idx, row in tqdm(train.iterrows(), total=len(train)):\n        FILE_PATHS_PAIRS.append(row[['patient_id', 'image_id']].values)\n        \nFILE_PATHS_PAIRS = np.array(FILE_PATHS_PAIRS, dtype=object)\nprint(f'FILE_PATHS_PAIRS shape: {FILE_PATHS_PAIRS.shape}')","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:55:51.257699Z","iopub.execute_input":"2023-03-12T02:55:51.258801Z","iopub.status.idle":"2023-03-12T02:55:52.116686Z","shell.execute_reply.started":"2023-03-12T02:55:51.258749Z","shell.execute_reply":"2023-03-12T02:55:52.115348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_CHUNKS = 100\nCHUNKS = np.array_split(FILE_PATHS_PAIRS, N_CHUNKS)\n\nprint(f'N_CHUNKS: {N_CHUNKS}, CHUNK len: {len(CHUNKS[0])}, shape: {CHUNKS[0].shape}')","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:55:52.118323Z","iopub.execute_input":"2023-03-12T02:55:52.119711Z","iopub.status.idle":"2023-03-12T02:55:52.129553Z","shell.execute_reply.started":"2023-03-12T02:55:52.119657Z","shell.execute_reply":"2023-03-12T02:55:52.127576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_chunk(args):\n    patient_id, image_id = args\n    fp = f'/kaggle/input/rsna-breast-cancer-detection/train_images/{patient_id}/{image_id}.dcm'\n    image, target = process(fp, size=(TARGET_WIDTH, TARGET_HEIGHT), ret_target=True, crop_image=True)\n\n    image = np.expand_dims(image, 2)\n    \n    if IMAGE_FORMAT == 'PNG':\n        image_serialized = tf.io.encode_png(image, compression=9).numpy()\n    else:\n        image_serialized = tf.io.encode_jpeg(image, quality=IMAGE_QUALITY, optimize_size=True).numpy()\n    \n    return image_serialized, target, patient_id, image_id","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:55:52.131773Z","iopub.execute_input":"2023-03-12T02:55:52.132271Z","iopub.status.idle":"2023-03-12T02:55:52.144247Z","shell.execute_reply.started":"2023-03-12T02:55:52.132219Z","shell.execute_reply":"2023-03-12T02:55:52.143066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_tf_records(chunks):\n    for chunk_idx, chunk in enumerate(tqdm(chunks)):\n        print(f'===== GENERATING TFRECORDS {chunk_idx} =====')\n        tfrecord_name = f'batch_{chunk_idx}.tfrecords'\n        \n        options = tf.io.TFRecordOptions(compression_type='GZIP', compression_level=9)\n        with tf.io.TFRecordWriter(tfrecord_name, options=options) as file_writer:\n            jobs = [joblib.delayed(process_chunk)(args) for args in chunk]\n            chunk_processed = joblib.Parallel(\n                n_jobs=cpu_count(),\n                verbose=0,\n                backend='multiprocessing',\n                prefer='threads',\n            )(jobs)\n            \n            for image, target, patient_id, image_id in chunk_processed:\n                record_bytes = tf.train.Example(features=tf.train.Features(feature={\n                    # image\n                    'image': tf.train.Feature(bytes_list=tf.train.BytesList(value=[image])),\n\n                    # target\n                    'target': tf.train.Feature(int64_list=tf.train.Int64List(value=[target])),\n                    \n                    # patient_id\n                    'patient_id': tf.train.Feature(int64_list=tf.train.Int64List(value=[patient_id])),\n                    \n                    # image_id\n                    'image_id': tf.train.Feature(int64_list=tf.train.Int64List(value=[image_id])),\n                })).SerializeToString()\n                file_writer.write(record_bytes)\n            \nto_tf_records(CHUNKS)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T02:55:52.146199Z","iopub.execute_input":"2023-03-12T02:55:52.147484Z","iopub.status.idle":"2023-03-12T03:04:30.335921Z","shell.execute_reply.started":"2023-03-12T02:55:52.147439Z","shell.execute_reply":"2023-03-12T03:04:30.334523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N = 32\nN","metadata":{"execution":{"iopub.status.busy":"2023-03-12T03:04:30.337632Z","iopub.execute_input":"2023-03-12T03:04:30.337985Z","iopub.status.idle":"2023-03-12T03:04:30.347566Z","shell.execute_reply.started":"2023-03-12T03:04:30.337945Z","shell.execute_reply":"2023-03-12T03:04:30.346188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_tfrecord(record_bytes):\n    features = tf.io.parse_single_example(record_bytes, {\n        'image': tf.io.FixedLenFeature([], tf.string),\n        'target': tf.io.FixedLenFeature([], tf.int64),\n        'patient_id': tf.io.FixedLenFeature([], tf.int64),\n        'image_id': tf.io.FixedLenFeature([], tf.int64),\n    })\n        \n    if IMAGE_FORMAT == 'PNG':\n        image = tf.io.decode_png(features['image'], channels=N_CHANNELS)\n    else:\n        image = tf.io.decode_jpeg(features['image'], channels=N_CHANNELS)\n        \n    image = tf.reshape(image, [TARGET_HEIGHT, TARGET_WIDTH, N_CHANNELS])\n\n    target = features['target']\n    patient_id = features['patient_id']\n    image_id = features['image_id']\n    \n    return image, target, patient_id, image_id","metadata":{"execution":{"iopub.status.busy":"2023-03-12T03:04:30.348987Z","iopub.execute_input":"2023-03-12T03:04:30.349346Z","iopub.status.idle":"2023-03-12T03:04:30.359316Z","shell.execute_reply.started":"2023-03-12T03:04:30.349314Z","shell.execute_reply":"2023-03-12T03:04:30.357889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_dataset():\n    FNAMES_TRAIN_TFRECORDS = tf.io.gfile.glob('./*.tfrecords')\n    train_dataset = tf.data.TFRecordDataset(FNAMES_TRAIN_TFRECORDS, num_parallel_reads=1, compression_type='GZIP')\n    train_dataset = train_dataset.map(decode_tfrecord)\n    train_dataset = train_dataset.batch(N)\n    \n    return train_dataset","metadata":{"execution":{"iopub.status.busy":"2023-03-12T03:04:30.361057Z","iopub.execute_input":"2023-03-12T03:04:30.361428Z","iopub.status.idle":"2023-03-12T03:04:30.377262Z","shell.execute_reply.started":"2023-03-12T03:04:30.36139Z","shell.execute_reply":"2023-03-12T03:04:30.375893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_batch(dataset, rows=N, cols=1):\n    images, targets, patient_ids, image_ids = next(iter(dataset))\n    images = np.moveaxis(images, 3, 1)\n    fig, axes = plt.subplots(nrows=rows, ncols=cols, figsize=(cols*6, rows*10))\n    for r in range(rows):\n        for c in range(cols):\n            img = images[r,c]\n            axes[r].imshow(img)\n            if c == 0:\n                target = targets[r]\n                patient_id = patient_ids[r]\n                image_id = image_ids[r]\n                axes[r].set_title(f'target: {target}, patient_id: {patient_id}, image_id: {image_id}', fontsize=12, pad=16)\n        \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T03:04:30.379092Z","iopub.execute_input":"2023-03-12T03:04:30.37945Z","iopub.status.idle":"2023-03-12T03:04:30.389263Z","shell.execute_reply.started":"2023-03-12T03:04:30.379415Z","shell.execute_reply":"2023-03-12T03:04:30.388167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = get_train_dataset()\nshow_batch(train_dataset)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T03:04:30.390878Z","iopub.execute_input":"2023-03-12T03:04:30.391287Z","iopub.status.idle":"2023-03-12T03:04:36.218231Z","shell.execute_reply.started":"2023-03-12T03:04:30.391247Z","shell.execute_reply":"2023-03-12T03:04:36.216761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}