{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# RSNA: Cut Off Empty Space from Images\n\nThis notebook generates images with all empty space removed. \nThis magnifies important parts of the data and helps to train better models with the same compute. \n\nNote that extra artifacts such as labels, noise are also removed from images!\n\nIt uses pregenerated PNGs from https://www.kaggle.com/code/theoviel/dicom-resized-png-jpg as an input.\n\nSo if you want to use this preprocessing in inference:\n1. Run code from https://www.kaggle.com/code/theoviel/dicom-resized-png-jpg on test data first.\n2. Run the code from this notebook on the output of (1)\n\nSee https://www.kaggle.com/vslaykovsky/infer-effnetv2-aux-targets-weighted-loss-thres for an example use in inference. \n\n\n**UPD**\n\n\n\n**Upvote if you think this is awesome!**","metadata":{}},{"cell_type":"code","source":"from concurrent.futures import ProcessPoolExecutor\nimport cv2\nimport glob\nimport numpy as np\nfrom tqdm import tqdm\nimport os\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport re\n\n\nDEBUG = False\n ","metadata":{"execution":{"iopub.status.busy":"2022-12-18T10:19:29.581105Z","iopub.execute_input":"2022-12-18T10:19:29.581544Z","iopub.status.idle":"2022-12-18T10:19:29.619851Z","shell.execute_reply.started":"2022-12-18T10:19:29.58151Z","shell.execute_reply":"2022-12-18T10:19:29.618927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fit_image(fname):\n    X = cv2.imread(fname)\n    \n    # Some images have narrow exterior \"frames\" that complicate selection of the main data. Cutting off the frame\n    X = X[5:-5, 5:-5]\n    \n    # regions of non-empty pixels\n    output= cv2.connectedComponentsWithStats((X > 20).astype(np.uint8)[:, :, 0], 8, cv2.CV_32S)\n\n    # stats.shape == (N, 5), where N is the number of regions, 5 dimensions correspond to:\n    # left, top, width, height, area_size\n    stats = output[2]\n\n    # finding max area which always corresponds to the breast data. \n    idx = stats[1:, 4].argmax() + 1\n    x1, y1, w, h = stats[idx][:4]\n    x2 = x1 + w\n    y2 = y1 + h\n    \n    # cutting out the breast data\n    X_fit = X[y1: y2, x1: x2]\n    \n    patient_id, im_id = re.findall('(\\d+)_(\\d+).png', os.path.basename(fname))[0]\n    os.makedirs(patient_id, exist_ok=True)\n    cv2.imwrite(f'{patient_id}/{im_id}.png', X_fit[:, :, 0])\n\ndef fit_all_images(all_images):\n    with ProcessPoolExecutor(4) as p:\n        for i in tqdm(p.map(fit_image, all_images), total=len(all_images)):\n            pass\n\n\nfit_image('/kaggle/input/rsna-breast-cancer-1024-pngs/output/52316_201080742.png')\nnp.array(Image.open('52316/201080742.png'))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-18T10:19:29.679227Z","iopub.execute_input":"2022-12-18T10:19:29.679676Z","iopub.status.idle":"2022-12-18T10:19:29.826175Z","shell.execute_reply.started":"2022-12-18T10:19:29.679637Z","shell.execute_reply":"2022-12-18T10:19:29.824905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nall_images = glob.glob('/kaggle/input/rsna-breast-cancer-1024-pngs/output/*')\nif DEBUG:\n    all_images = np.random.choice(all_images, size=100)\nfit_all_images(all_images)","metadata":{"execution":{"iopub.status.busy":"2022-12-18T10:19:29.863606Z","iopub.execute_input":"2022-12-18T10:19:29.864077Z","iopub.status.idle":"2022-12-18T10:19:38.374923Z","shell.execute_reply.started":"2022-12-18T10:19:29.864041Z","shell.execute_reply":"2022-12-18T10:19:38.373156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(123)\n\nfor fname in np.random.choice(glob.glob('*/*'), size=100):\n    plt.figure(figsize=(20, 10))\n    patient_id, im_id = re.findall('(\\d+)/(\\d+).png', fname)[0]\n    plt.suptitle(f'[{fname}]')\n    im1 = Image.open(fname).convert('F')\n    plt.subplot(121).imshow(im1)\n    plt.subplot(121).set_title(f'Output image {im1.size}')\n    im2 = Image.open(f'/kaggle/input/rsna-breast-cancer-1024-pngs/output/{patient_id}_{im_id}.png').convert('F')\n    plt.subplot(122).imshow(\n        im2\n    )\n    plt.subplot(122).set_title(f'Source image {im2.size}')","metadata":{"execution":{"iopub.status.busy":"2022-12-17T11:00:24.294451Z","iopub.execute_input":"2022-12-17T11:00:24.294836Z","iopub.status.idle":"2022-12-17T11:01:46.282817Z","shell.execute_reply.started":"2022-12-17T11:00:24.294797Z","shell.execute_reply":"2022-12-17T11:01:46.281667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ndf = pd.DataFrame(dict(zip(('y', 'x', 'c'), cv2.imread(i).shape)) for i in glob.glob('*/*.png'))","metadata":{"execution":{"iopub.status.busy":"2022-12-17T11:01:46.284272Z","iopub.execute_input":"2022-12-17T11:01:46.284621Z","iopub.status.idle":"2022-12-17T11:01:46.864964Z","shell.execute_reply.started":"2022-12-17T11:01:46.284589Z","shell.execute_reply":"2022-12-17T11:01:46.863736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[['x', 'y']].plot.hist(alpha=0.7, bins=100)","metadata":{"execution":{"iopub.status.busy":"2022-12-17T11:01:46.866445Z","iopub.execute_input":"2022-12-17T11:01:46.866796Z","iopub.status.idle":"2022-12-17T11:01:47.62258Z","shell.execute_reply.started":"2022-12-17T11:01:46.866764Z","shell.execute_reply":"2022-12-17T11:01:47.621801Z"},"trusted":true},"execution_count":null,"outputs":[]}]}