{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook is mostly a combination of these two notebook:\n- [Mammography - Remove Letter Markers by David Roberts](https://www.kaggle.com/code/davidbroberts/mammography-remove-letter-markers)\n- [Dicom -> Resized PNG/JPG by Theo Viel](https://www.kaggle.com/code/theoviel/dicom-resized-png-jpg)\n\nIt  allows to extract a bounding box around the breast, remove text in images, then save the image as png; as in the original notebook, datasets with resolution of 256 -> 1024  are created.","metadata":{"execution":{"iopub.status.busy":"2022-12-03T20:35:52.665435Z","iopub.execute_input":"2022-12-03T20:35:52.665841Z","iopub.status.idle":"2022-12-03T20:35:52.674211Z","shell.execute_reply.started":"2022-12-03T20:35:52.665808Z","shell.execute_reply":"2022-12-03T20:35:52.672016Z"}}},{"cell_type":"markdown","source":"## Dataset Links :\n- [256 x 256](https://www.kaggle.com/datasets/fabiendaniel/rsna-cropped-png-256) \n- [512 x 512](https://www.kaggle.com/datasets/fabiendaniel/rsna-cropped-png-512) \n- [768 x 768](https://www.kaggle.com/datasets/fabiendaniel/rsna-cropped-png-768)\n- [1024 x 1024](https://www.kaggle.com/datasets/fabiendaniel/rsna-cropped-png-1024)","metadata":{}},{"cell_type":"markdown","source":"## Initialization","metadata":{}},{"cell_type":"code","source":"!pip install -qU python-gdcm pydicom pylibjpeg","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-04T13:19:01.686834Z","iopub.execute_input":"2022-12-04T13:19:01.687248Z","iopub.status.idle":"2022-12-04T13:19:17.805601Z","shell.execute_reply.started":"2022-12-04T13:19:01.687215Z","shell.execute_reply":"2022-12-04T13:19:17.804065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport glob\nimport gdcm\nimport pydicom\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport json\n\nfrom tqdm.notebook import tqdm\nfrom joblib import Parallel, delayed","metadata":{"execution":{"iopub.status.busy":"2022-12-04T13:19:17.809631Z","iopub.execute_input":"2022-12-04T13:19:17.810055Z","iopub.status.idle":"2022-12-04T13:19:18.901703Z","shell.execute_reply.started":"2022-12-04T13:19:17.810015Z","shell.execute_reply":"2022-12-04T13:19:18.900355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images = glob.glob(\"/kaggle/input/rsna-breast-cancer-detection/train_images/*/*.dcm\")\n\nlen(train_images)  # 54706","metadata":{"execution":{"iopub.status.busy":"2022-12-04T13:19:18.903431Z","iopub.execute_input":"2022-12-04T13:19:18.904552Z","iopub.status.idle":"2022-12-04T13:19:43.150102Z","shell.execute_reply.started":"2022-12-04T13:19:18.904508Z","shell.execute_reply":"2022-12-04T13:19:43.147783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Crop image  ","metadata":{}},{"cell_type":"code","source":"def crop_image(img, show=True):\n    # Binarize the image\n    bin_pixels = cv2.threshold(img, 20, 255, cv2.THRESH_BINARY)[1]\n   \n    # Make contours around the binarized image, keep only the largest contour\n    contours, _ = cv2.findContours(bin_pixels, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_NONE)\n    contour = max(contours, key=cv2.contourArea)\n\n    # Create a mask from the largest contour\n    mask = np.zeros(img.shape, np.uint8)\n    cv2.drawContours(mask, [contour], -1, 255, cv2.FILLED)\n   \n    # Use bitwise_and to get masked part of the original image\n    out = cv2.bitwise_and(img, mask)\n    \n    # get bounding box of contour\n    y1, y2 = np.min(contour[:, :, 1]), np.max(contour[:, :, 1])\n    x1, x2 = np.min(contour[:, :, 0]), np.max(contour[:, :, 0])\n    \n    x1 = int(0.99 * x1)\n    x2 = int(1.01 * x2)\n    y1 = int(0.99 * y1)\n    y2 = int(1.01 * y2)\n    \n    if show:\n        plt.imshow(out[y1:y2, x1:x2], cmap=\"gray\") ; \n\n    return out[y1:y2, x1:x2]","metadata":{"execution":{"iopub.status.busy":"2022-12-04T13:19:43.152808Z","iopub.execute_input":"2022-12-04T13:19:43.153416Z","iopub.status.idle":"2022-12-04T13:19:43.165772Z","shell.execute_reply.started":"2022-12-04T13:19:43.153378Z","shell.execute_reply":"2022-12-04T13:19:43.164576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for f in tqdm(train_images[5:10]):\n    print(90*\"=\")\n    patient = f.split('/')[-2]\n    image = f.split('/')[-1][:-4]\n    \n    print(f\"patient {patient}\\n\")\n    \n    dicom = pydicom.dcmread(f)\n    img = dicom.pixel_array\n\n    img = (img - img.min()) / (img.max() - img.min())    \n    img *= 255\n    img = np.uint8(img)\n\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n\n    plt.figure(figsize=(5, 5))\n    plt.imshow(img, cmap=\"gray\")\n    plt.title(f\"original image for {patient} {image}\")\n    plt.show()\n        \n    img = crop_image(img, show=False)\n    \n    plt.figure(figsize=(5, 5))\n    plt.imshow(img, cmap=\"gray\")\n    plt.title(f\"after text removal / cropping {patient} {image}\")\n    plt.show()\n    \n    crop_image(img, show=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-04T13:19:43.167957Z","iopub.execute_input":"2022-12-04T13:19:43.168761Z","iopub.status.idle":"2022-12-04T13:20:02.889589Z","shell.execute_reply.started":"2022-12-04T13:19:43.168669Z","shell.execute_reply":"2022-12-04T13:20:02.888261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save the processed data\n**Images are quite big so resizing them is necessary.**\n  - Use 256 to train your first models, or if you don't have a lot of compute\n  - use 512 to have competitive models\n  - Check if 768/1024 is better, if you have the compute power\n\n**I advise using the `png` format because the jpg compression can be annoying during inference.**","metadata":{}},{"cell_type":"code","source":"DATASET_NAME = f'RSNA-cropped-png-256-test'\nSAVE_FOLDER = f\"/kaggle/working/{DATASET_NAME}\"\nSIZE = 256\nEXTENSION = \"png\"","metadata":{"execution":{"iopub.status.busy":"2022-12-04T13:20:02.89112Z","iopub.execute_input":"2022-12-04T13:20:02.892125Z","iopub.status.idle":"2022-12-04T13:20:02.898597Z","shell.execute_reply.started":"2022-12-04T13:20:02.892082Z","shell.execute_reply":"2022-12-04T13:20:02.897243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Create Kaggle Dataset\n\nos.makedirs(SAVE_FOLDER, exist_ok=True)\nos.makedirs(SAVE_FOLDER + '/images/', exist_ok=True)\n\nwith open('../input/kaggle-secrets/kaggle.json') as f:\n    kaggle_creds = json.load(f)\n    \nos.environ['KAGGLE_USERNAME'] = kaggle_creds['username']\nos.environ['KAGGLE_KEY'] = kaggle_creds['key']\n\n!kaggle datasets init -p '{SAVE_FOLDER}'\n\nwith open(f'{SAVE_FOLDER}/dataset-metadata.json') as f:\n    dataset_meta = json.load(f)\n    \ndataset_meta['id'] = f'fabiendaniel/{DATASET_NAME}'\ndataset_meta['title'] = DATASET_NAME\n\nwith open(f'{SAVE_FOLDER}/dataset-metadata.json', \"w\") as outfile:\n    json.dump(dataset_meta, outfile)\nprint(dataset_meta)\n\n!cp '{SAVE_FOLDER}'/dataset-metadata.json '{SAVE_FOLDER}'/meta.json\n!ls '{SAVE_FOLDER}'\n\n!kaggle datasets create -u -p '{SAVE_FOLDER}'","metadata":{"execution":{"iopub.status.busy":"2022-12-04T13:20:02.900157Z","iopub.execute_input":"2022-12-04T13:20:02.90064Z","iopub.status.idle":"2022-12-04T13:20:08.447394Z","shell.execute_reply.started":"2022-12-04T13:20:02.900595Z","shell.execute_reply":"2022-12-04T13:20:08.445927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(f, size=512, save_folder=\"\", extension=\"png\"):\n    patient = f.split('/')[-2]\n    image = f.split('/')[-1][:-4]\n\n    dicom = pydicom.dcmread(f)\n    img = dicom.pixel_array\n\n    img = (img - img.min()) / (img.max() - img.min())\n    img *= 255\n    img = np.uint8(img)\n    \n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n\n    img = cv2.resize(img, (size, size))\n    img = crop_image(img, show=False)\n    \n    cv2.imwrite(save_folder + f\"/images/{patient}_{image}.{extension}\", img)","metadata":{"execution":{"iopub.status.busy":"2022-12-04T13:20:08.44956Z","iopub.execute_input":"2022-12-04T13:20:08.449982Z","iopub.status.idle":"2022-12-04T13:20:08.460132Z","shell.execute_reply.started":"2022-12-04T13:20:08.449942Z","shell.execute_reply":"2022-12-04T13:20:08.458732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = Parallel(n_jobs=4)(\n    delayed(process)(uid, size=SIZE, save_folder=SAVE_FOLDER, extension=EXTENSION)\n    for uid in tqdm(train_images[:10])\n)","metadata":{"execution":{"iopub.status.busy":"2022-12-04T13:20:08.461889Z","iopub.execute_input":"2022-12-04T13:20:08.46269Z","iopub.status.idle":"2022-12-04T13:20:18.691333Z","shell.execute_reply.started":"2022-12-04T13:20:08.462641Z","shell.execute_reply":"2022-12-04T13:20:18.689792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datetime import datetime\nversion_name = datetime.now().strftime(\"%Y%m%d-%H%M%S\")\nprint(version_name)","metadata":{"execution":{"iopub.status.busy":"2022-12-04T13:20:18.69571Z","iopub.execute_input":"2022-12-04T13:20:18.698695Z","iopub.status.idle":"2022-12-04T13:20:18.706483Z","shell.execute_reply.started":"2022-12-04T13:20:18.69863Z","shell.execute_reply":"2022-12-04T13:20:18.704846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_images = glob.glob(f\"{SAVE_FOLDER}/images/*.png\")\n\nlen(output_images)  # 54706\n","metadata":{"execution":{"iopub.status.busy":"2022-12-04T13:20:52.621143Z","iopub.execute_input":"2022-12-04T13:20:52.621598Z","iopub.status.idle":"2022-12-04T13:20:52.629777Z","shell.execute_reply.started":"2022-12-04T13:20:52.621564Z","shell.execute_reply":"2022-12-04T13:20:52.62842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!kaggle datasets version -m {version_name} -p \"{SAVE_FOLDER}\"  -r tar -r zip","metadata":{"execution":{"iopub.status.busy":"2022-12-04T13:21:04.958921Z","iopub.execute_input":"2022-12-04T13:21:04.959876Z","iopub.status.idle":"2022-12-04T13:21:08.080543Z","shell.execute_reply.started":"2022-12-04T13:21:04.959824Z","shell.execute_reply":"2022-12-04T13:21:08.079325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Done !","metadata":{}}]}