{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Improved ROI extraction\nThis notebook based on [⭐️⭐️ Breast Cancer - ROI (brest) extractor ⭐️⭐️](https://www.kaggle.com/code/remekkinas/breast-cancer-roi-brest-extractor) notebook's [fine-tuned](https://www.kaggle.com/datasets/remekkinas/rsna-breast-cancer-detection-roi-model) extraction model [yolov5](https://github.com/ultralytics/yolov5) to extract better ROI from images.\n\n<div class='alert alert-block alert-success'>\n    <h4>1:1 ROI DataSets:</h4>\n    <ul>\n        <li><a href='https://www.kaggle.com/datasets/olegbaryshnikov/rsna-roi-1024x1024-pngs'><b>ROI 1024x1024 pngs</b></a></li>\n        <li><a href='https://www.kaggle.com/datasets/olegbaryshnikov/rsna-roi-768x768-pngs'><b>ROI 768x768 pngs</b></a></li>\n        <li><a href='https://www.kaggle.com/datasets/olegbaryshnikov/rsna-roi-512x512-pngs'><b>ROI 512x512 pngs</b></a></li>\n    </ul>\n</div>    \n<div class='alert alert-block alert-success'>\n    <h4>1:2 ROI DataSets <span style=\"color: orange\">[Recommended]</span>:</h4>\n    <ul>\n        <li><a href='https://www.kaggle.com/datasets/olegbaryshnikov/rsna-roi-512x1024-pngs'><b>ROI 512x1024 pngs</b></a></li>\n        <li><a href='https://www.kaggle.com/datasets/olegbaryshnikov/rsna-roi-384x768-pngs'><b>ROI 384x768 pngs</b></a></li>\n        <li><a href='https://www.kaggle.com/datasets/olegbaryshnikov/rsna-roi-256x512-pngs'><b>ROI 256x512 pngs</b></a></li>\n    </ul>\n</div>\n\n**Main improvements:**\n1. ROIs are cropped from original dcm images and then resized (to lose less data)\n2. Image transformations applied to extract ROIs from more images\n3. ROI completed to desired aspect ratio to avoid noticeable image stretches\n4. Too small ROIs are removed\n\n**Data pipeline:**\n1. Extract ROI coordinates from [768x768 png images](https://www.kaggle.com/datasets/theoviel/rsna-breast-cancer-768-pngs) using yolov5\n2. For images with unextracted ROI coordinates apply transformations until ROI coordinates will not be extracted\n3. Resize ROI coordinates to apply on original size images (dcm)\n4. Crop ROI images, resize them to output_size and convert it to png format\n\n**Result on train DS:**\n* 54634 images - processed successfully  (was 54601)\n* 72 images - detection failed (was 105)\n\n\n<div class='alert alert-block alert-info'> \n    <b>\n        1024x1024 ROI extraction (CPU):\n        <a href='https://www.kaggle.com/code/olegbaryshnikov/rsna-improved-roi-extraction-yolov5?scriptVersionId=113409846'>[Part1]</a>\n        <a href='https://www.kaggle.com/code/olegbaryshnikov/rsna-improved-roi-extraction-yolov5?scriptVersionId=113409854'>[Part2]</a>\n        <a href='https://www.kaggle.com/code/olegbaryshnikov/rsna-improved-roi-extraction-yolov5?scriptVersionId=113409869'>[Part3]</a>\n        <a href='https://www.kaggle.com/code/olegbaryshnikov/rsna-improved-roi-extraction-yolov5?scriptVersionId=113409875'>[Part4]</a>\n    </b>\n</div>\n\n<div class='alert alert-block alert-info'> \n    <b>\n        512x1024 ROI extraction (CPU):\n        <a href='https://www.kaggle.com/code/olegbaryshnikov/rsna-improved-roi-extraction-yolov5?scriptVersionId=113822294'>[Part1]</a>\n        <a href='https://www.kaggle.com/code/olegbaryshnikov/rsna-improved-roi-extraction-yolov5?scriptVersionId=113822320'>[Part2]</a>\n    </b>\n</div>","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"!pip install -qU python-gdcm pydicom pylibjpeg","metadata":{"execution":{"iopub.status.busy":"2022-12-14T14:51:30.186363Z","iopub.execute_input":"2022-12-14T14:51:30.186911Z","iopub.status.idle":"2022-12-14T14:51:47.500661Z","shell.execute_reply.started":"2022-12-14T14:51:30.186733Z","shell.execute_reply":"2022-12-14T14:51:47.498935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport math\nimport numpy as np\nimport pandas as pd\nfrom IPython.display import display\n\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport cv2\n\nfrom tqdm.notebook import tqdm\nimport gc\n\nimport glob\n\n#for dcm files\nfrom joblib import Parallel, delayed\nimport pydicom\n\n#for ROI model\nimport torch","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-14T14:51:47.503501Z","iopub.execute_input":"2022-12-14T14:51:47.505067Z","iopub.status.idle":"2022-12-14T14:51:50.592369Z","shell.execute_reply.started":"2022-12-14T14:51:47.505005Z","shell.execute_reply":"2022-12-14T14:51:50.591041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Config","metadata":{}},{"cell_type":"markdown","source":"* set 'skip_small_ROIs' to True to skip small size ROIs\n* set 'skip_unrecognized' to True to skip images with no ROI extracted\n* set 'avoid_stretches' to True to avoid noticeable stretches when crop ROIs","metadata":{}},{"cell_type":"code","source":"Config = {\n    'output_dim_x' : 512,\n    'output_dim_y' : 1024,\n    'output_extension' : 'png',\n    'skip_small_ROIs': True,\n    'skip_unrecognized': True,\n    'avoid_stretches': True\n}","metadata":{"execution":{"iopub.status.busy":"2022-12-14T14:51:50.593644Z","iopub.execute_input":"2022-12-14T14:51:50.594345Z","iopub.status.idle":"2022-12-14T14:51:50.599984Z","shell.execute_reply.started":"2022-12-14T14:51:50.594301Z","shell.execute_reply":"2022-12-14T14:51:50.598588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv_path = '/kaggle/input/rsna-breast-cancer-detection/train.csv'\ntrain_images_folder = '/kaggle/input/rsna-breast-cancer-detection/train_images'\ntrain_images_folder_png_768 = '/kaggle/input/rsna-breast-cancer-768-pngs/output'","metadata":{"execution":{"iopub.status.busy":"2022-12-14T14:51:50.603545Z","iopub.execute_input":"2022-12-14T14:51:50.604031Z","iopub.status.idle":"2022-12-14T14:51:50.615561Z","shell.execute_reply.started":"2022-12-14T14:51:50.603977Z","shell.execute_reply":"2022-12-14T14:51:50.614382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data","metadata":{}},{"cell_type":"markdown","source":"<span class=\"alert alert-block alert-danger\"><b>Don't forget to remove/change train_csv slicing size</b></span>","metadata":{}},{"cell_type":"code","source":"train_csv = pd.read_csv(train_csv_path)[['patient_id','image_id']]\ntrain_csv = train_csv.iloc[20000:25000,:]\ntrain_csv","metadata":{"execution":{"iopub.status.busy":"2022-12-14T14:51:50.617032Z","iopub.execute_input":"2022-12-14T14:51:50.617482Z","iopub.status.idle":"2022-12-14T14:51:50.779688Z","shell.execute_reply.started":"2022-12-14T14:51:50.617435Z","shell.execute_reply":"2022-12-14T14:51:50.778434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_image(img):\n    fig=plt.figure(figsize=(5, 5))\n    plt.imshow(img, cmap='bone')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-14T14:51:50.781504Z","iopub.execute_input":"2022-12-14T14:51:50.781929Z","iopub.status.idle":"2022-12-14T14:51:50.789327Z","shell.execute_reply.started":"2022-12-14T14:51:50.78189Z","shell.execute_reply":"2022-12-14T14:51:50.788525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_image_and_ROI(img,ROI):\n    fig=plt.figure(figsize=(5, 5))\n    \n    rect = cv2.rectangle(img, (int(ROI['xmin']), int(ROI['ymin'])), (int(ROI['xmax']), int(ROI['ymax'])), (255,0,0), 4)\n    \n    plt.imshow(rect, cmap='bone')\n    \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-14T14:51:50.790372Z","iopub.execute_input":"2022-12-14T14:51:50.791306Z","iopub.status.idle":"2022-12-14T14:51:50.805013Z","shell.execute_reply.started":"2022-12-14T14:51:50.791272Z","shell.execute_reply":"2022-12-14T14:51:50.803565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_png_img(images_folder, patient_id, image_id):\n    img_path = os.path.join(images_folder,f'{patient_id}_{image_id}.png')\n    img = cv2.imread(img_path)\n    \n    return img","metadata":{"execution":{"iopub.status.busy":"2022-12-14T14:51:50.806498Z","iopub.execute_input":"2022-12-14T14:51:50.806843Z","iopub.status.idle":"2022-12-14T14:51:50.8161Z","shell.execute_reply.started":"2022-12-14T14:51:50.806812Z","shell.execute_reply":"2022-12-14T14:51:50.815132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://www.kaggle.com/code/theoviel/dicom-resized-png-jpg\ndef read_dcm_img(images_folder, patient_id, image_id):\n    img_path = os.path.join(images_folder, patient_id, f'{image_id}.dcm')\n    \n    dicom = pydicom.dcmread(img_path)\n    img = dicom.pixel_array\n    \n    img = (img - img.min()) / (img.max() - img.min())\n    \n    if dicom.PhotometricInterpretation == 'MONOCHROME1':\n        img = 1 - img\n    \n    img = (img * 255).astype(np.uint8)\n    \n    return img","metadata":{"execution":{"iopub.status.busy":"2022-12-14T14:51:50.817594Z","iopub.execute_input":"2022-12-14T14:51:50.818248Z","iopub.status.idle":"2022-12-14T14:51:50.828952Z","shell.execute_reply.started":"2022-12-14T14:51:50.818213Z","shell.execute_reply":"2022-12-14T14:51:50.827992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ROIs Extraction","metadata":{}},{"cell_type":"markdown","source":"#### Model preparation","metadata":{}},{"cell_type":"code","source":"%%capture\n#clone model in hidden folder\nos.makedirs('../kaggle/yolov5', exist_ok = True)\n# Clone yolov5 repository\n!git clone https://github.com/ultralytics/yolov5 /kaggle/yolov5","metadata":{"execution":{"iopub.status.busy":"2022-12-14T14:51:50.832972Z","iopub.execute_input":"2022-12-14T14:51:50.833827Z","iopub.status.idle":"2022-12-14T14:51:53.674736Z","shell.execute_reply.started":"2022-12-14T14:51:50.833765Z","shell.execute_reply":"2022-12-14T14:51:53.673102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check hidden folder\n!ls /kaggle/yolov5","metadata":{"execution":{"iopub.status.busy":"2022-12-14T14:51:53.676636Z","iopub.execute_input":"2022-12-14T14:51:53.677065Z","iopub.status.idle":"2022-12-14T14:51:54.775673Z","shell.execute_reply.started":"2022-12-14T14:51:53.677025Z","shell.execute_reply":"2022-12-14T14:51:54.774445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get yolov5 and preserve plt backend\ndef get_yolo():\n    b = plt.get_backend()\n    model = torch.hub.load('/kaggle/yolov5/', 'custom', path='/kaggle/input/rsna-breast-cancer-detection-roi-model/rsna-roi-003.pt', source='local')\n    matplotlib.use(b)\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-12-14T14:51:54.778404Z","iopub.execute_input":"2022-12-14T14:51:54.779347Z","iopub.status.idle":"2022-12-14T14:51:54.787104Z","shell.execute_reply.started":"2022-12-14T14:51:54.779292Z","shell.execute_reply":"2022-12-14T14:51:54.78624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROI_model = get_yolo()","metadata":{"execution":{"iopub.status.busy":"2022-12-14T14:51:54.788651Z","iopub.execute_input":"2022-12-14T14:51:54.789463Z","iopub.status.idle":"2022-12-14T14:51:56.31614Z","shell.execute_reply.started":"2022-12-14T14:51:54.789428Z","shell.execute_reply":"2022-12-14T14:51:56.314762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ROI extraction function\n#input: image\n#output: (xmin,xmax,ymin,ymax)\ndef ROI_extraction(model, img,transform=None):\n    if(transform):\n        img = transform(img)\n    \n    #select only best prediction\n    prediction = model(img).pandas().xyxy[0].to_dict(orient='records')\n    \n    if(len(prediction)==0):\n        return None\n        \n    prediction = prediction[0]\n    \n    result = {key:prediction[key] for key in ['xmin','xmax','ymin','ymax']}\n    \n    if(transform):\n        result = transform.inverse(result)\n    \n    return result","metadata":{"execution":{"iopub.status.busy":"2022-12-14T14:51:56.318047Z","iopub.execute_input":"2022-12-14T14:51:56.319019Z","iopub.status.idle":"2022-12-14T14:51:56.327317Z","shell.execute_reply.started":"2022-12-14T14:51:56.318972Z","shell.execute_reply":"2022-12-14T14:51:56.325943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Image transformations","metadata":{}},{"cell_type":"code","source":"class Transformation:\n    name = 'No transformations'\n    def __init__(self):\n        super().__init__()\n        \n    def __call__(self, img):\n        return img\n    \n    def inverse(self, coords):\n        return coords\n\nclass Vertical_Flip(Transformation):\n    name = 'Vertical_Flip'\n    def __call__(self, img):\n        img = cv2.flip(img, 0)\n        return img\n    \n    def inverse(self, coords):\n        coords['ymin'] = 767-coords['ymax']\n        coords['ymax'] = 767-coords['ymin']\n        \n        return coords\n    \nclass Horizontal_Flip(Transformation):\n    name = 'Horizontal_Flip'\n    def __call__(self, img):\n        img = cv2.flip(img, 1)\n        return img\n    \n    def inverse(self, coords):\n        coords['xmin'] = 767-coords['xmax']\n        coords['xmax'] = 767-coords['xmin']\n        \n        return coords\n\nclass Change_Contrast(Transformation):\n    def __init__(self, clipLimit=2.0, tileGridSize=(8,8)):\n        super().__init__()\n        \n        self.name = f'Change_Contrast_{clipLimit}_{tileGridSize}'\n        \n        self.clipLimit = clipLimit\n        self.tileGridSize = tileGridSize\n        \n    #https://stackoverflow.com/questions/39308030/how-do-i-increase-the-contrast-of-an-image-in-python-opencv\n    def __call__(self, img):\n        lab= cv2.cvtColor(img, cv2.COLOR_BGR2LAB)\n        l_channel, a, b = cv2.split(lab)\n        \n        # Applying CLAHE to L-channel\n        # feel free to try different values for the limit and grid size:\n        clahe = cv2.createCLAHE(clipLimit=self.clipLimit, tileGridSize=self.tileGridSize)\n        cl = clahe.apply(l_channel)\n        \n        # merge the CLAHE enhanced L-channel with the a and b channel\n        limg = cv2.merge((cl,a,b))\n\n        # Converting image from LAB Color model to BGR color spcae\n        img = cv2.cvtColor(limg, cv2.COLOR_LAB2BGR)\n\n        return img\n    \n    def inverse(self, coords):\n        \n        return coords\n    \nclass Translation(Transformation):\n    def __init__(self, x = 30,y=30):\n        super().__init__()\n        \n        self.name = f'Translation_{x}_{y}'\n        \n        self.x = x\n        self.y = y\n        self.M = np.float32([[1,0,x],[0,1,y]])\n        \n    #https://stackoverflow.com/questions/32609098/how-to-fast-change-image-brightness-with-python-opencv\n    def __call__(self, img):\n        rows,cols,_ = img.shape\n        \n        img = cv2.warpAffine(img,self.M,(cols,rows))\n        \n        return img\n    \n    def inverse(self, coords):\n        coords['xmin'] = coords['xmax']+self.x\n        coords['xmax'] = coords['xmin']+self.x\n        coords['ymin'] = coords['ymax']+self.y\n        coords['ymax'] = coords['ymin']+self.y\n        \n        coords = {key:np.clip(coords[key],0,767) for key in ['xmin','xmax','ymin','ymax']}\n        return coords","metadata":{"execution":{"iopub.status.busy":"2022-12-14T14:51:56.329659Z","iopub.execute_input":"2022-12-14T14:51:56.330197Z","iopub.status.idle":"2022-12-14T14:51:56.350435Z","shell.execute_reply.started":"2022-12-14T14:51:56.330157Z","shell.execute_reply":"2022-12-14T14:51:56.349419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#first transformation should return the same image\n#apply transformations from less to more significant\ntransformations = [\n    Transformation(),\n    Vertical_Flip(),Horizontal_Flip(),\n    Change_Contrast(clipLimit=2.0),Change_Contrast(clipLimit=15.0),\n    Change_Contrast(clipLimit=2.0, tileGridSize=(128,128)),\n    Change_Contrast(clipLimit=30.0),\n    Change_Contrast(clipLimit=15.0, tileGridSize=(128,128)),\n    \n    Translation(x=0,y=30),Translation(x=0,y=-30),\n    Translation(x=-5,y=0),Translation(x=5,y=0),\n    Translation(x=0,y=50),Translation(x=0,y=-50),\n    Translation(x=-15,y=0),Translation(x=15,y=0),\n    \n    Translation(x=0,y=-100),Translation(x=0,y=-150),\n    Translation(x=-30,y=0),Translation(x=30,y=0)\n]\n\nfor data in tqdm(train_csv.iloc[1400:1403,:].itertuples(), total=3):\n    img = read_png_img(train_images_folder_png_768, str(data.patient_id), str(data.image_id))\n\n    fig=plt.figure(figsize=(100, 40))\n    \n    print('patient_id: ',str(data.patient_id),' image_id: ',str(data.image_id))\n    for tr_i,transform in enumerate(transformations):\n        img_transformed = transform(img)\n        \n        ax = fig.add_subplot(3, math.ceil(len(transformations)/3), tr_i+1)\n        ax.title.set_text(transform.name)\n        ax.title.set_fontsize(50)\n        ax.imshow(img_transformed, cmap='bone')\n    \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-14T14:51:56.351754Z","iopub.execute_input":"2022-12-14T14:51:56.352414Z","iopub.status.idle":"2022-12-14T14:52:22.442726Z","shell.execute_reply.started":"2022-12-14T14:51:56.352374Z","shell.execute_reply":"2022-12-14T14:52:22.439296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### ROI extraction","metadata":{}},{"cell_type":"code","source":"train_csv['ROI'] = None\ntrain_csv","metadata":{"execution":{"iopub.status.busy":"2022-12-14T14:52:22.445211Z","iopub.execute_input":"2022-12-14T14:52:22.445605Z","iopub.status.idle":"2022-12-14T14:52:22.462043Z","shell.execute_reply.started":"2022-12-14T14:52:22.44557Z","shell.execute_reply":"2022-12-14T14:52:22.46073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for transform in transformations:\n    print('\\nCurrent transformation: ',transform.name)\n    \n    train_csv_to_extract = train_csv[train_csv['ROI'].isnull()]\n\n    for data in tqdm(train_csv_to_extract.itertuples(), total=len(train_csv_to_extract)):\n        img = read_png_img(train_images_folder_png_768, str(data.patient_id), str(data.image_id))\n        extracted_ROI = ROI_extraction(ROI_model, img, transform)\n        train_csv.loc[(train_csv.patient_id == data.patient_id)&(train_csv.image_id == data.image_id),'ROI'] = [extracted_ROI]\n    \n    ROI = train_csv[~train_csv['ROI'].isnull()]\n    no_ROI = train_csv[train_csv['ROI'].isnull()]\n    \n    print('len(ROIs): ',len(ROI))\n    print('len(no_ROI_imgs): ',len(no_ROI))\n    \n    if(len(no_ROI)==0):\n        break","metadata":{"execution":{"iopub.status.busy":"2022-12-14T14:52:22.463836Z","iopub.execute_input":"2022-12-14T14:52:22.46419Z","iopub.status.idle":"2022-12-14T15:12:43.806202Z","shell.execute_reply.started":"2022-12-14T14:52:22.464157Z","shell.execute_reply":"2022-12-14T15:12:43.804963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check extracted ROIs","metadata":{}},{"cell_type":"code","source":"train_csv","metadata":{"execution":{"iopub.status.busy":"2022-12-14T15:12:43.807462Z","iopub.execute_input":"2022-12-14T15:12:43.807885Z","iopub.status.idle":"2022-12-14T15:12:43.826452Z","shell.execute_reply.started":"2022-12-14T15:12:43.807854Z","shell.execute_reply":"2022-12-14T15:12:43.825166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check some data\nrecognized = train_csv[~train_csv['ROI'].isnull()].iloc[40:45,:]\n\nfor data in tqdm(recognized.itertuples(), total=len(recognized)):\n    img = read_png_img(train_images_folder_png_768, str(data.patient_id), str(data.image_id))\n    show_image_and_ROI(img,data.ROI)","metadata":{"execution":{"iopub.status.busy":"2022-12-14T15:12:43.828485Z","iopub.execute_input":"2022-12-14T15:12:43.829912Z","iopub.status.idle":"2022-12-14T15:12:45.031505Z","shell.execute_reply.started":"2022-12-14T15:12:43.829856Z","shell.execute_reply":"2022-12-14T15:12:45.030345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check unrecognized data\nunrecognized = train_csv[train_csv['ROI'].isnull()]\ndisplay(unrecognized)\n\nfor data in tqdm(unrecognized.itertuples(), total=len(unrecognized)):\n    img = read_png_img(train_images_folder_png_768, str(data.patient_id), str(data.image_id))\n    show_image(img)","metadata":{"execution":{"iopub.status.busy":"2022-12-14T15:12:45.03306Z","iopub.execute_input":"2022-12-14T15:12:45.034423Z","iopub.status.idle":"2022-12-14T15:12:45.959588Z","shell.execute_reply.started":"2022-12-14T15:12:45.034376Z","shell.execute_reply":"2022-12-14T15:12:45.958733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Extract ROIs","metadata":{}},{"cell_type":"code","source":"#Crop ROIs from some dcms and check the results\n\nto_crop = train_csv\nif(Config['skip_unrecognized']):\n    to_crop = to_crop[~to_crop['ROI'].isnull()]\nelse:\n    to_crop.loc[to_crop['ROI'].isnull(),'ROI'] = {'xmin':0,'xmax':np.Inf,'ymin':0,'ymax':np.Inf}\n    \nfor data in tqdm(to_crop.iloc[10:11,:].itertuples(), total=len(to_crop.iloc[10:11,:])):\n    img = read_dcm_img(train_images_folder, str(data.patient_id), str(data.image_id))\n    show_image(img)\n    print(img.shape)\n    \n    y_orig, x_orig = img.shape\n    x_orig, y_orig = x_orig-1, y_orig-1\n    k_x, k_y = float(x_orig)/767.0, float(y_orig)/767.0\n    \n    print(data.ROI)\n    #resize ROI coordinates\n    ROI_resized={}\n    ROI_resized['xmin'] = np.clip(round(k_x * data.ROI['xmin']), 0, x_orig)\n    ROI_resized['xmax'] = np.clip(round(k_x * data.ROI['xmax']), 0, x_orig)\n    ROI_resized['ymin'] = np.clip(round(k_y * data.ROI['ymin']), 0, y_orig)\n    ROI_resized['ymax'] = np.clip(round(k_y * data.ROI['ymax']), 0, y_orig)\n    \n    \n    if(Config['avoid_stretches']):\n        #how many times one size can be less then other\n        size_threshold = Config['output_dim_y'] / Config['output_dim_x']\n        \n        x_distance = ROI_resized['xmax']-ROI_resized['xmin']\n        y_distance = ROI_resized['ymax']-ROI_resized['ymin']\n        \n        #params to calculate y coords\n        y_distance_threshold = 3\n        \n        if((x_distance) < (y_distance)/size_threshold):\n            right_distance = x_orig - ROI_resized['xmax']\n            left_distance = ROI_resized['xmin'] - 0\n            min_x_range_value = int((y_distance)/size_threshold)\n            \n            #find the side where breast located\n            #and increase x ROI value on the opposite side\n            if(left_distance<right_distance):\n                print(\"xmax increased!!!\")\n                ROI_resized['xmax'] = ROI_resized['xmin'] + min_x_range_value\n                ROI_resized['xmax'] = np.clip(ROI_resized['xmax'], 0, x_orig)\n            elif(right_distance<left_distance):\n                print(\"xmin increased!!!\")\n                ROI_resized['xmin'] = ROI_resized['xmax'] - min_x_range_value\n                ROI_resized['xmin'] = np.clip(ROI_resized['xmin'], 0, x_orig)\n                \n        elif((y_distance) < (x_distance)*size_threshold):\n            top_distance = y_orig - ROI_resized['ymax']\n            bottom_distance = ROI_resized['ymin'] - 0\n            min_y_range_value = int((x_distance)*size_threshold)\n            \n            #if one of the y coordinates is much closer to img border increase opposite side\n            if(bottom_distance<top_distance/y_distance_threshold):\n                print(\"ymax increased!!!\")\n                ROI_resized['ymax'] = ROI_resized['ymin'] + min_y_range_value\n                ROI_resized['ymax'] = np.clip(ROI_resized['ymax'], 0, y_orig)\n            elif(top_distance<bottom_distance/y_distance_threshold):\n                print(\"ymin increased!!!\")\n                ROI_resized['ymin'] = ROI_resized['ymax'] - min_y_range_value\n                ROI_resized['ymin'] = np.clip(ROI_resized['ymin'], 0, y_orig)\n            #else change both y and clip\n            else:\n                print(\"both y increased!!!\")\n                ymin, ymax = ROI_resized['ymin'], ROI_resized['ymax']\n                y_dist = ymax - ymin\n                y_dist_add = (min_y_range_value - y_dist)//2\n                \n                ROI_resized['ymin'] = ymin - y_dist_add\n                ROI_resized['ymax'] = ymax + y_dist_add\n                ROI_resized['ymin'] = np.clip(ROI_resized['ymin'], 0, y_orig)\n                ROI_resized['ymax'] = np.clip(ROI_resized['ymax'], 0, y_orig)\n    \n    print(ROI_resized)\n    \n    ROI_img = img[ROI_resized['ymin']:ROI_resized['ymax']+1,ROI_resized['xmin']:ROI_resized['xmax']+1]\n    show_image(ROI_img)\n    print(ROI_img.shape)\n    \n    if(Config['skip_small_ROIs']):\n        size_threshold = 16\n        min_size_x = Config['output_dim_x']/size_threshold\n        min_size_y = Config['output_dim_y']/size_threshold\n        \n        if(ROI_img.shape[0]<min_size_y or ROI_img.shape[1]<min_size_x):\n            print('Skipped: ',data.patient_id,data.image_id)\n            show_image(ROI_img)\n            continue\n    \n    #ROI resizing\n    ROI_img = cv2.resize(ROI_img, (Config['output_dim_x'], Config['output_dim_y']))\n    show_image(ROI_img)\n    print(ROI_img.shape)","metadata":{"execution":{"iopub.status.busy":"2022-12-14T15:12:45.961306Z","iopub.execute_input":"2022-12-14T15:12:45.961908Z","iopub.status.idle":"2022-12-14T15:12:48.226726Z","shell.execute_reply.started":"2022-12-14T15:12:45.961871Z","shell.execute_reply":"2022-12-14T15:12:48.22552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(data):\n    img = read_dcm_img(train_images_folder, str(data.patient_id), str(data.image_id))\n    y_orig, x_orig = img.shape\n    x_orig, y_orig = x_orig-1, y_orig-1\n    k_x, k_y = float(x_orig)/767.0, float(y_orig)/767.0\n    \n    #resize ROI coordinates\n    ROI_resized={}\n    ROI_resized['xmin'] = np.clip(round(k_x * data.ROI['xmin']), 0, x_orig)\n    ROI_resized['xmax'] = np.clip(round(k_x * data.ROI['xmax']), 0, x_orig)\n    ROI_resized['ymin'] = np.clip(round(k_y * data.ROI['ymin']), 0, y_orig)\n    ROI_resized['ymax'] = np.clip(round(k_y * data.ROI['ymax']), 0, y_orig)\n    \n    \n    if(Config['avoid_stretches']):\n        #how many times one size can be less then other\n        size_threshold = Config['output_dim_y'] / Config['output_dim_x']\n        \n        x_distance = ROI_resized['xmax']-ROI_resized['xmin']\n        y_distance = ROI_resized['ymax']-ROI_resized['ymin']\n        \n        #params to calculate y coords\n        y_distance_threshold = 3\n        \n        if((x_distance) < (y_distance)/size_threshold):\n            right_distance = x_orig - ROI_resized['xmax']\n            left_distance = ROI_resized['xmin'] - 0\n            min_x_range_value = int((y_distance)/size_threshold)\n            \n            #find the side where breast located\n            #and increase x ROI value on the opposite side\n            if(left_distance<right_distance):\n                ROI_resized['xmax'] = ROI_resized['xmin'] + min_x_range_value\n                ROI_resized['xmax'] = np.clip(ROI_resized['xmax'], 0, x_orig)\n            elif(right_distance<left_distance):\n                ROI_resized['xmin'] = ROI_resized['xmax'] - min_x_range_value\n                ROI_resized['xmin'] = np.clip(ROI_resized['xmin'], 0, x_orig)\n                \n        elif((y_distance) < (x_distance)*size_threshold):\n            top_distance = y_orig - ROI_resized['ymax']\n            bottom_distance = ROI_resized['ymin'] - 0\n            min_y_range_value = int((x_distance)*size_threshold)\n            \n            #if one of the y coordinates is much closer to img border increase opposite side\n            if(bottom_distance<top_distance/y_distance_threshold):\n                ROI_resized['ymax'] = ROI_resized['ymin'] + min_y_range_value\n                ROI_resized['ymax'] = np.clip(ROI_resized['ymax'], 0, y_orig)\n            elif(top_distance<bottom_distance/y_distance_threshold):\n                ROI_resized['ymin'] = ROI_resized['ymax'] - min_y_range_value\n                ROI_resized['ymin'] = np.clip(ROI_resized['ymin'], 0, y_orig)\n            #else change both y and clip\n            else:\n                ymin, ymax = ROI_resized['ymin'], ROI_resized['ymax']\n                y_dist = ymax - ymin\n                y_dist_add = (min_y_range_value - y_dist)//2\n                \n                ROI_resized['ymin'] = ymin - y_dist_add\n                ROI_resized['ymax'] = ymax + y_dist_add\n                ROI_resized['ymin'] = np.clip(ROI_resized['ymin'], 0, y_orig)\n                ROI_resized['ymax'] = np.clip(ROI_resized['ymax'], 0, y_orig)\n    \n    \n    ROI_img = img[ROI_resized['ymin']:ROI_resized['ymax']+1,ROI_resized['xmin']:ROI_resized['xmax']+1]\n    \n    if(Config['skip_small_ROIs']):\n        size_threshold = 16\n        min_size_x = Config['output_dim_x']/size_threshold\n        min_size_y = Config['output_dim_y']/size_threshold\n        \n        if(ROI_img.shape[0]<min_size_y or ROI_img.shape[1]<min_size_x):\n            return {'patient_id':data.patient_id,'image_id':data.image_id}\n    \n    #ROI resizing\n    ROI_img = cv2.resize(ROI_img, (Config['output_dim_x'], Config['output_dim_y']))\n    cv2.imwrite(f'{data.patient_id}_{data.image_id}.{Config[\"output_extension\"]}', ROI_img)\n    return None","metadata":{"execution":{"iopub.status.busy":"2022-12-14T15:12:48.228635Z","iopub.execute_input":"2022-12-14T15:12:48.22901Z","iopub.status.idle":"2022-12-14T15:12:48.249423Z","shell.execute_reply.started":"2022-12-14T15:12:48.228977Z","shell.execute_reply":"2022-12-14T15:12:48.248262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#save ROIs\n\nto_save = train_csv\nif(Config['skip_unrecognized']):\n    to_save = to_save[~to_save['ROI'].isnull()]\nelse:\n    to_save.loc[to_save['ROI'].isnull(),'ROI'] = {'xmin':0,'xmax':np.Inf,'ymin':0,'ymax':np.Inf}\n    \nsmall_ROIs = Parallel(n_jobs=-1)(\n    delayed(process)(data)\n    for data in tqdm(to_save.itertuples(), total=len(to_save))\n)","metadata":{"execution":{"iopub.status.busy":"2022-12-14T15:12:48.251169Z","iopub.execute_input":"2022-12-14T15:12:48.25154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"small_ROIs = [x for x in small_ROIs if x is not None]\nsmall_ROIs = pd.DataFrame(small_ROIs)\nsmall_ROIs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#save all unrecognized data to csv\nunrecognized = train_csv[train_csv['ROI'].isnull()][['patient_id','image_id']]\ndisplay(unrecognized)\n\nunrecognized_images = pd.concat([small_ROIs,unrecognized]).reset_index(drop=True)\nif(len(unrecognized_images)>0):\n    unrecognized_images.to_csv('unrecognized_images.csv')\nunrecognized_images","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}