{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"},{"sourceId":9119496,"sourceType":"datasetVersion","datasetId":5504968}],"dockerImageVersionId":30746,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -qU python-gdcm pydicom pylibjpeg\n","metadata":{"execution":{"iopub.status.busy":"2024-08-06T16:57:52.764066Z","iopub.execute_input":"2024-08-06T16:57:52.764666Z","iopub.status.idle":"2024-08-06T16:58:11.273803Z","shell.execute_reply.started":"2024-08-06T16:57:52.764619Z","shell.execute_reply":"2024-08-06T16:58:11.271518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport math\nimport numpy as np\nimport pandas as pd\nfrom IPython.display import display\n\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport cv2\n\nfrom tqdm.notebook import tqdm\nimport gc\n\nimport glob\n\n#for dcm files\nfrom joblib import Parallel, delayed\nimport pydicom\n\n#for ROI model\nimport torch","metadata":{"execution":{"iopub.status.busy":"2024-08-06T16:58:11.276235Z","iopub.execute_input":"2024-08-06T16:58:11.276685Z","iopub.status.idle":"2024-08-06T16:58:11.285308Z","shell.execute_reply.started":"2024-08-06T16:58:11.276647Z","shell.execute_reply":"2024-08-06T16:58:11.283792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Config = {\n    'output_dim_x' : 512,\n    'output_dim_y' : 1024,\n    'output_extension' : 'png',\n    'skip_small_ROIs': True,\n    'skip_unrecognized': True,\n    'avoid_stretches': True\n}","metadata":{"execution":{"iopub.status.busy":"2024-08-06T16:58:11.287603Z","iopub.execute_input":"2024-08-06T16:58:11.288149Z","iopub.status.idle":"2024-08-06T16:58:11.300392Z","shell.execute_reply.started":"2024-08-06T16:58:11.288082Z","shell.execute_reply":"2024-08-06T16:58:11.299059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv_path = '/kaggle/input/rsna-breast-cancer-detection/train.csv'\ntrain_images_folder = '/kaggle/input/rsna-breast-cancer-detection/train_images'\n# train_images_folder_png_768 = '/kaggle/input/rsna-breast-cancer-768-pngs/output'\ntrain_images_folder_png_768 ='/kaggle/working'\n","metadata":{"execution":{"iopub.status.busy":"2024-08-06T17:17:52.395915Z","iopub.execute_input":"2024-08-06T17:17:52.396593Z","iopub.status.idle":"2024-08-06T17:17:52.405972Z","shell.execute_reply.started":"2024-08-06T17:17:52.396548Z","shell.execute_reply":"2024-08-06T17:17:52.404642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv(train_csv_path)[['patient_id','image_id']]\ntrain_csv = train_csv.iloc[20000:25000,:]\ntrain_csv","metadata":{"execution":{"iopub.status.busy":"2024-08-06T17:17:52.783346Z","iopub.execute_input":"2024-08-06T17:17:52.78378Z","iopub.status.idle":"2024-08-06T17:17:52.908579Z","shell.execute_reply.started":"2024-08-06T17:17:52.783747Z","shell.execute_reply":"2024-08-06T17:17:52.907199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_image(img):\n    fig=plt.figure(figsize=(5, 5))\n    plt.imshow(img, cmap='bone')\n    plt.show()\n    \ndef show_image_and_ROI(img,ROI):\n    fig=plt.figure(figsize=(5, 5))\n    \n    rect = cv2.rectangle(img, (int(ROI['xmin']), int(ROI['ymin'])), (int(ROI['xmax']), int(ROI['ymax'])), (255,0,0), 4)\n    \n    plt.imshow(rect, cmap='bone')\n    \n    plt.show()\n    \ndef read_png_img(images_folder, patient_id, image_id):\n    img_path = os.path.join(images_folder,f'{patient_id}_{image_id}.png')\n    img = cv2.imread(img_path)\n    \n    return img\n\n#https://www.kaggle.com/code/theoviel/dicom-resized-png-jpg\ndef read_dcm_img(images_folder, patient_id, image_id):\n    img_path = os.path.join(images_folder, patient_id, f'{image_id}.dcm')\n    \n    dicom = pydicom.dcmread(img_path)\n    img = dicom.pixel_array\n    \n    img = (img - img.min()) / (img.max() - img.min())\n    \n    if dicom.PhotometricInterpretation == 'MONOCHROME1':\n        img = 1 - img\n    \n    img = (img * 255).astype(np.uint8)\n    \n    return img","metadata":{"execution":{"iopub.status.busy":"2024-08-06T17:17:53.165696Z","iopub.execute_input":"2024-08-06T17:17:53.166122Z","iopub.status.idle":"2024-08-06T17:17:53.179432Z","shell.execute_reply.started":"2024-08-06T17:17:53.16607Z","shell.execute_reply":"2024-08-06T17:17:53.17795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Extraction (preparation)","metadata":{}},{"cell_type":"markdown","source":"## Yolov5","metadata":{}},{"cell_type":"code","source":"%%capture\n#clone model in hidden folder\nos.makedirs('../kaggle/yolov5', exist_ok = True)\n# Clone yolov5 repository\n!git clone https://github.com/ultralytics/yolov5 /kaggle/yolov5","metadata":{"execution":{"iopub.status.busy":"2024-08-06T17:17:54.385907Z","iopub.execute_input":"2024-08-06T17:17:54.386345Z","iopub.status.idle":"2024-08-06T17:17:55.572198Z","shell.execute_reply.started":"2024-08-06T17:17:54.386311Z","shell.execute_reply":"2024-08-06T17:17:55.570604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check hidden folder\n!ls /kaggle/yolov5","metadata":{"execution":{"iopub.status.busy":"2024-08-06T17:17:55.574945Z","iopub.execute_input":"2024-08-06T17:17:55.575432Z","iopub.status.idle":"2024-08-06T17:17:56.755334Z","shell.execute_reply.started":"2024-08-06T17:17:55.575392Z","shell.execute_reply":"2024-08-06T17:17:56.753691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get yolov5 and preserve plt backend\ndef get_yolo():\n    b = plt.get_backend()\n    model = torch.hub.load('/kaggle/yolov5/', 'custom', path='/kaggle/input/model-yolov5-crop-image-roi-breast-cancer/archive/rsna-roi-003.pt', source='local')\n    matplotlib.use(b)\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-08-06T17:49:05.432965Z","iopub.execute_input":"2024-08-06T17:49:05.433567Z","iopub.status.idle":"2024-08-06T17:49:05.440607Z","shell.execute_reply.started":"2024-08-06T17:49:05.433515Z","shell.execute_reply":"2024-08-06T17:49:05.439191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROI_model = get_yolo()\n","metadata":{"execution":{"iopub.status.busy":"2024-08-06T17:49:13.54061Z","iopub.execute_input":"2024-08-06T17:49:13.541058Z","iopub.status.idle":"2024-08-06T17:50:11.056346Z","shell.execute_reply.started":"2024-08-06T17:49:13.541021Z","shell.execute_reply":"2024-08-06T17:50:11.054832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv['ROI'] = None\ntrain_csv","metadata":{"execution":{"iopub.status.busy":"2024-08-06T17:52:13.270687Z","iopub.execute_input":"2024-08-06T17:52:13.271159Z","iopub.status.idle":"2024-08-06T17:52:13.289779Z","shell.execute_reply.started":"2024-08-06T17:52:13.271121Z","shell.execute_reply":"2024-08-06T17:52:13.288385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ROI_extraction(model, img,transform=None):\n    if(transform):\n        img = transform(img)\n    \n    #select only best prediction\n    prediction = model(img).pandas().xyxy[0].to_dict(orient='records')\n    \n    if(len(prediction)==0):\n        return None\n        \n    prediction = prediction[0]\n    \n    result = {key:prediction[key] for key in ['xmin','xmax','ymin','ymax']}\n    \n    if(transform):\n        result = transform.inverse(result)\n    \n    return result","metadata":{"execution":{"iopub.status.busy":"2024-08-06T18:00:54.466404Z","iopub.execute_input":"2024-08-06T18:00:54.466813Z","iopub.status.idle":"2024-08-06T18:00:54.475474Z","shell.execute_reply.started":"2024-08-06T18:00:54.466783Z","shell.execute_reply":"2024-08-06T18:00:54.474147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Transformation:\n    name = 'No transformations'\n    def __init__(self):\n        super().__init__()\n        \n    def __call__(self, img):\n        return img\n    \n    def inverse(self, coords):\n        return coords\n\nclass Vertical_Flip(Transformation):\n    name = 'Vertical_Flip'\n    def __call__(self, img):\n        img = cv2.flip(img, 0)\n        return img\n    \n    def inverse(self, coords):\n        coords['ymin'] = 767-coords['ymax']\n        coords['ymax'] = 767-coords['ymin']\n        \n        return coords\n    \nclass Horizontal_Flip(Transformation):\n    name = 'Horizontal_Flip'\n    def __call__(self, img):\n        img = cv2.flip(img, 1)\n        return img\n    \n    def inverse(self, coords):\n        coords['xmin'] = 767-coords['xmax']\n        coords['xmax'] = 767-coords['xmin']\n        \n        return coords\n\nclass Change_Contrast(Transformation):\n    def __init__(self, clipLimit=2.0, tileGridSize=(8,8)):\n        super().__init__()\n        \n        self.name = f'Change_Contrast_{clipLimit}_{tileGridSize}'\n        \n        self.clipLimit = clipLimit\n        self.tileGridSize = tileGridSize\n        \n    #https://stackoverflow.com/questions/39308030/how-do-i-increase-the-contrast-of-an-image-in-python-opencv\n    def __call__(self, img):\n        lab= cv2.cvtColor(img, cv2.COLOR_BGR2LAB)\n        l_channel, a, b = cv2.split(lab)\n        \n        # Applying CLAHE to L-channel\n        # feel free to try different values for the limit and grid size:\n        clahe = cv2.createCLAHE(clipLimit=self.clipLimit, tileGridSize=self.tileGridSize)\n        cl = clahe.apply(l_channel)\n        \n        # merge the CLAHE enhanced L-channel with the a and b channel\n        limg = cv2.merge((cl,a,b))\n\n        # Converting image from LAB Color model to BGR color spcae\n        img = cv2.cvtColor(limg, cv2.COLOR_LAB2BGR)\n\n        return img\n    \n    def inverse(self, coords):\n        \n        return coords\n    \nclass Translation(Transformation):\n    def __init__(self, x = 30,y=30):\n        super().__init__()\n        \n        self.name = f'Translation_{x}_{y}'\n        \n        self.x = x\n        self.y = y\n        self.M = np.float32([[1,0,x],[0,1,y]])\n        \n    #https://stackoverflow.com/questions/32609098/how-to-fast-change-image-brightness-with-python-opencv\n    def __call__(self, img):\n        rows,cols,_ = img.shape\n        \n        img = cv2.warpAffine(img,self.M,(cols,rows))\n        \n        return img\n    \n    def inverse(self, coords):\n        coords['xmin'] = coords['xmax']+self.x\n        coords['xmax'] = coords['xmin']+self.x\n        coords['ymin'] = coords['ymax']+self.y\n        coords['ymax'] = coords['ymin']+self.y\n        \n        coords = {key:np.clip(coords[key],0,767) for key in ['xmin','xmax','ymin','ymax']}\n        return coords","metadata":{"execution":{"iopub.status.busy":"2024-08-06T18:00:55.15918Z","iopub.execute_input":"2024-08-06T18:00:55.159619Z","iopub.status.idle":"2024-08-06T18:00:55.183262Z","shell.execute_reply.started":"2024-08-06T18:00:55.159584Z","shell.execute_reply":"2024-08-06T18:00:55.181836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transformations = [\n    Transformation(),\n    Vertical_Flip(),Horizontal_Flip(),\n    Change_Contrast(clipLimit=2.0),Change_Contrast(clipLimit=15.0),\n    Change_Contrast(clipLimit=2.0, tileGridSize=(128,128)),\n    Change_Contrast(clipLimit=30.0),\n    Change_Contrast(clipLimit=15.0, tileGridSize=(128,128)),\n    \n    Translation(x=0,y=30),Translation(x=0,y=-30),\n    Translation(x=-5,y=0),Translation(x=5,y=0),\n    Translation(x=0,y=50),Translation(x=0,y=-50),\n    Translation(x=-15,y=0),Translation(x=15,y=0),\n    \n    Translation(x=0,y=-100),Translation(x=0,y=-150),\n    Translation(x=-30,y=0),Translation(x=30,y=0)\n]\n","metadata":{"execution":{"iopub.status.busy":"2024-08-06T18:00:55.840805Z","iopub.execute_input":"2024-08-06T18:00:55.84126Z","iopub.status.idle":"2024-08-06T18:00:55.852535Z","shell.execute_reply.started":"2024-08-06T18:00:55.841224Z","shell.execute_reply":"2024-08-06T18:00:55.851375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv['ROI'] = None\ntrain_csv","metadata":{"execution":{"iopub.status.busy":"2024-08-06T18:00:56.576028Z","iopub.execute_input":"2024-08-06T18:00:56.576512Z","iopub.status.idle":"2024-08-06T18:00:56.592316Z","shell.execute_reply.started":"2024-08-06T18:00:56.576478Z","shell.execute_reply":"2024-08-06T18:00:56.590723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for transform in transformations:\n    print('\\nCurrent transformation: ',transform.name)\n    \n    train_csv_to_extract = train_csv[train_csv['ROI'].isnull()]\n\n    for data in tqdm(train_csv_to_extract.itertuples(), total=len(train_csv_to_extract)):\n        img = read_png_img(train_images_folder_png_768, str(data.patient_id), str(data.image_id))\n        try:\n            extracted_ROI = ROI_extraction(ROI_model, img, transform)\n        except Exception as e:\n            print(e)\n#         train_csv.loc[(train_csv.patient_id == data.patient_id)&(train_csv.image_id == data.image_id),'ROI'] = [extracted_ROI]\n    \n#     ROI = train_csv[~train_csv['ROI'].isnull()]\n#     no_ROI = train_csv[train_csv['ROI'].isnull()]\n    \n#     print('len(ROIs): ',len(ROI))\n#     print('len(no_ROI_imgs): ',len(no_ROI))\n    \n    if(len(no_ROI)==0):\n        break","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nto_crop = train_csv\nif(Config['skip_unrecognized']):\n    to_crop = to_crop[~to_crop['ROI'].isnull()]\nelse:\n    to_crop.loc[to_crop['ROI'].isnull(),'ROI'] = {'xmin':0,'xmax':np.Inf,'ymin':0,'ymax':np.Inf}\n    \nfor data in tqdm(to_crop.iloc[10:11,:].itertuples(), total=len(to_crop.iloc[10:11,:])):\n    img = read_dcm_img(train_images_folder, str(data.patient_id), str(data.image_id))\n    show_image(img)\n    print(img.shape)\n    \n    y_orig, x_orig = img.shape\n    x_orig, y_orig = x_orig-1, y_orig-1\n    k_x, k_y = float(x_orig)/767.0, float(y_orig)/767.0\n    \n    print(data.ROI)\n    #resize ROI coordinates\n    ROI_resized={}\n    ROI_resized['xmin'] = np.clip(round(k_x * data.ROI['xmin']), 0, x_orig)\n    ROI_resized['xmax'] = np.clip(round(k_x * data.ROI['xmax']), 0, x_orig)\n    ROI_resized['ymin'] = np.clip(round(k_y * data.ROI['ymin']), 0, y_orig)\n    ROI_resized['ymax'] = np.clip(round(k_y * data.ROI['ymax']), 0, y_orig)\n    \n    \n    if(Config['avoid_stretches']):\n        #how many times one size can be less then other\n        size_threshold = Config['output_dim_y'] / Config['output_dim_x']\n        \n        x_distance = ROI_resized['xmax']-ROI_resized['xmin']\n        y_distance = ROI_resized['ymax']-ROI_resized['ymin']\n        \n        #params to calculate y coords\n        y_distance_threshold = 3\n        \n        if((x_distance) < (y_distance)/size_threshold):\n            right_distance = x_orig - ROI_resized['xmax']\n            left_distance = ROI_resized['xmin'] - 0\n            min_x_range_value = int((y_distance)/size_threshold)\n            \n            #find the side where breast located\n            #and increase x ROI value on the opposite side\n            if(left_distance<right_distance):\n                print(\"xmax increased!!!\")\n                ROI_resized['xmax'] = ROI_resized['xmin'] + min_x_range_value\n                ROI_resized['xmax'] = np.clip(ROI_resized['xmax'], 0, x_orig)\n            elif(right_distance<left_distance):\n                print(\"xmin increased!!!\")\n                ROI_resized['xmin'] = ROI_resized['xmax'] - min_x_range_value\n                ROI_resized['xmin'] = np.clip(ROI_resized['xmin'], 0, x_orig)\n                \n        elif((y_distance) < (x_distance)*size_threshold):\n            top_distance = y_orig - ROI_resized['ymax']\n            bottom_distance = ROI_resized['ymin'] - 0\n            min_y_range_value = int((x_distance)*size_threshold)\n            \n            #if one of the y coordinates is much closer to img border increase opposite side\n            if(bottom_distance<top_distance/y_distance_threshold):\n                print(\"ymax increased!!!\")\n                ROI_resized['ymax'] = ROI_resized['ymin'] + min_y_range_value\n                ROI_resized['ymax'] = np.clip(ROI_resized['ymax'], 0, y_orig)\n            elif(top_distance<bottom_distance/y_distance_threshold):\n                print(\"ymin increased!!!\")\n                ROI_resized['ymin'] = ROI_resized['ymax'] - min_y_range_value\n                ROI_resized['ymin'] = np.clip(ROI_resized['ymin'], 0, y_orig)\n            #else change both y and clip\n            else:\n                print(\"both y increased!!!\")\n                ymin, ymax = ROI_resized['ymin'], ROI_resized['ymax']\n                y_dist = ymax - ymin\n                y_dist_add = (min_y_range_value - y_dist)//2\n                \n                ROI_resized['ymin'] = ymin - y_dist_add\n                ROI_resized['ymax'] = ymax + y_dist_add\n                ROI_resized['ymin'] = np.clip(ROI_resized['ymin'], 0, y_orig)\n                ROI_resized['ymax'] = np.clip(ROI_resized['ymax'], 0, y_orig)\n    \n    print(ROI_resized)\n    \n    ROI_img = img[ROI_resized['ymin']:ROI_resized['ymax']+1,ROI_resized['xmin']:ROI_resized['xmax']+1]\n    show_image(ROI_img)\n    print(ROI_img.shape)\n    \n    if(Config['skip_small_ROIs']):\n        size_threshold = 16\n        min_size_x = Config['output_dim_x']/size_threshold\n        min_size_y = Config['output_dim_y']/size_threshold\n        \n        if(ROI_img.shape[0]<min_size_y or ROI_img.shape[1]<min_size_x):\n            print('Skipped: ',data.patient_id,data.image_id)\n            show_image(ROI_img)\n            continue\n    \n    #ROI resizing\n    ROI_img = cv2.resize(ROI_img, (Config['output_dim_x'], Config['output_dim_y']))\n    show_image(ROI_img)\n    print(ROI_img.shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-06T18:10:42.391779Z","iopub.execute_input":"2024-08-06T18:10:42.392262Z","iopub.status.idle":"2024-08-06T18:10:42.440639Z","shell.execute_reply.started":"2024-08-06T18:10:42.392217Z","shell.execute_reply":"2024-08-06T18:10:42.439194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(data):\n    img = read_dcm_img(train_images_folder, str(data.patient_id), str(data.image_id))\n    y_orig, x_orig = img.shape\n    x_orig, y_orig = x_orig-1, y_orig-1\n    k_x, k_y = float(x_orig)/767.0, float(y_orig)/767.0\n    \n    #resize ROI coordinates\n    ROI_resized={}\n    ROI_resized['xmin'] = np.clip(round(k_x * data.ROI['xmin']), 0, x_orig)\n    ROI_resized['xmax'] = np.clip(round(k_x * data.ROI['xmax']), 0, x_orig)\n    ROI_resized['ymin'] = np.clip(round(k_y * data.ROI['ymin']), 0, y_orig)\n    ROI_resized['ymax'] = np.clip(round(k_y * data.ROI['ymax']), 0, y_orig)\n    \n    \n    if(Config['avoid_stretches']):\n        #how many times one size can be less then other\n        size_threshold = Config['output_dim_y'] / Config['output_dim_x']\n        \n        x_distance = ROI_resized['xmax']-ROI_resized['xmin']\n        y_distance = ROI_resized['ymax']-ROI_resized['ymin']\n        \n        #params to calculate y coords\n        y_distance_threshold = 3\n        \n        if((x_distance) < (y_distance)/size_threshold):\n            right_distance = x_orig - ROI_resized['xmax']\n            left_distance = ROI_resized['xmin'] - 0\n            min_x_range_value = int((y_distance)/size_threshold)\n            \n            #find the side where breast located\n            #and increase x ROI value on the opposite side\n            if(left_distance<right_distance):\n                ROI_resized['xmax'] = ROI_resized['xmin'] + min_x_range_value\n                ROI_resized['xmax'] = np.clip(ROI_resized['xmax'], 0, x_orig)\n            elif(right_distance<left_distance):\n                ROI_resized['xmin'] = ROI_resized['xmax'] - min_x_range_value\n                ROI_resized['xmin'] = np.clip(ROI_resized['xmin'], 0, x_orig)\n                \n        elif((y_distance) < (x_distance)*size_threshold):\n            top_distance = y_orig - ROI_resized['ymax']\n            bottom_distance = ROI_resized['ymin'] - 0\n            min_y_range_value = int((x_distance)*size_threshold)\n            \n            #if one of the y coordinates is much closer to img border increase opposite side\n            if(bottom_distance<top_distance/y_distance_threshold):\n                ROI_resized['ymax'] = ROI_resized['ymin'] + min_y_range_value\n                ROI_resized['ymax'] = np.clip(ROI_resized['ymax'], 0, y_orig)\n            elif(top_distance<bottom_distance/y_distance_threshold):\n                ROI_resized['ymin'] = ROI_resized['ymax'] - min_y_range_value\n                ROI_resized['ymin'] = np.clip(ROI_resized['ymin'], 0, y_orig)\n            #else change both y and clip\n            else:\n                ymin, ymax = ROI_resized['ymin'], ROI_resized['ymax']\n                y_dist = ymax - ymin\n                y_dist_add = (min_y_range_value - y_dist)//2\n                \n                ROI_resized['ymin'] = ymin - y_dist_add\n                ROI_resized['ymax'] = ymax + y_dist_add\n                ROI_resized['ymin'] = np.clip(ROI_resized['ymin'], 0, y_orig)\n                ROI_resized['ymax'] = np.clip(ROI_resized['ymax'], 0, y_orig)\n    \n    \n    ROI_img = img[ROI_resized['ymin']:ROI_resized['ymax']+1,ROI_resized['xmin']:ROI_resized['xmax']+1]\n    \n    if(Config['skip_small_ROIs']):\n        size_threshold = 16\n        min_size_x = Config['output_dim_x']/size_threshold\n        min_size_y = Config['output_dim_y']/size_threshold\n        \n        if(ROI_img.shape[0]<min_size_y or ROI_img.shape[1]<min_size_x):\n            return {'patient_id':data.patient_id,'image_id':data.image_id}\n    \n    #ROI resizing\n    ROI_img = cv2.resize(ROI_img, (Config['output_dim_x'], Config['output_dim_y']))\n    cv2.imwrite(f'{data.patient_id}_{data.image_id}.{Config[\"output_extension\"]}', ROI_img)\n    return None","metadata":{"execution":{"iopub.status.busy":"2024-08-06T18:11:25.741189Z","iopub.execute_input":"2024-08-06T18:11:25.741707Z","iopub.status.idle":"2024-08-06T18:11:25.766538Z","shell.execute_reply.started":"2024-08-06T18:11:25.741669Z","shell.execute_reply":"2024-08-06T18:11:25.765123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to_save = train_csv\nif(Config['skip_unrecognized']):\n    to_save = to_save[~to_save['ROI'].isnull()]\nelse:\n    to_save.loc[to_save['ROI'].isnull(),'ROI'] = {'xmin':0,'xmax':np.Inf,'ymin':0,'ymax':np.Inf}\n    \nsmall_ROIs = Parallel(n_jobs=-1)(\n    delayed(process)(data)\n    for data in tqdm(to_save.itertuples(), total=len(to_save))\n)","metadata":{"execution":{"iopub.status.busy":"2024-08-06T18:11:43.253439Z","iopub.execute_input":"2024-08-06T18:11:43.254421Z","iopub.status.idle":"2024-08-06T18:11:43.286465Z","shell.execute_reply.started":"2024-08-06T18:11:43.25437Z","shell.execute_reply":"2024-08-06T18:11:43.285126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"small_ROIs = [x for x in small_ROIs if x is not None]\nsmall_ROIs = pd.DataFrame(small_ROIs)\nsmall_ROIs","metadata":{"execution":{"iopub.status.busy":"2024-08-06T18:11:50.327562Z","iopub.execute_input":"2024-08-06T18:11:50.328034Z","iopub.status.idle":"2024-08-06T18:11:50.342292Z","shell.execute_reply.started":"2024-08-06T18:11:50.327995Z","shell.execute_reply":"2024-08-06T18:11:50.341052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#save all unrecognized data to csv\nunrecognized = train_csv[train_csv['ROI'].isnull()][['patient_id','image_id']]\ndisplay(unrecognized)\n\nunrecognized_images = pd.concat([small_ROIs,unrecognized]).reset_index(drop=True)\nif(len(unrecognized_images)>0):\n    unrecognized_images.to_csv('unrecognized_images.csv')\nunrecognized_images","metadata":{"execution":{"iopub.status.busy":"2024-08-06T18:11:59.322922Z","iopub.execute_input":"2024-08-06T18:11:59.323361Z","iopub.status.idle":"2024-08-06T18:11:59.367229Z","shell.execute_reply.started":"2024-08-06T18:11:59.323327Z","shell.execute_reply":"2024-08-06T18:11:59.36594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}