{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from glob import glob\nimport os\nimport pandas as pd\nimport numpy as np\nimport re\nfrom PIL import Image\nimport seaborn as sns\nfrom random import randrange\n\nfrom albumentations import (\n    HorizontalFlip, IAAPerspective, ShiftScaleRotate, CLAHE, RandomRotate90,\n    Transpose, ShiftScaleRotate, Blur, OpticalDistortion, GridDistortion, HueSaturationValue,\n    IAAAdditiveGaussianNoise, GaussNoise, MotionBlur, MedianBlur, RandomBrightnessContrast, IAAPiecewiseAffine,\n    IAASharpen, IAAEmboss, Flip, OneOf, Compose, PadIfNeeded, RandomGamma\n)\n#checnking the input files\nprint(os.listdir(\"/kaggle/input/rsna-breast-cancer-detection\"))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-30T10:52:28.542354Z","iopub.execute_input":"2022-11-30T10:52:28.542851Z","iopub.status.idle":"2022-11-30T10:52:29.630325Z","shell.execute_reply.started":"2022-11-30T10:52:28.542748Z","shell.execute_reply":"2022-11-30T10:52:29.628966Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Data","metadata":{}},{"cell_type":"code","source":"#reading all dcm files into train and test\ntrain = sorted(glob(\"/kaggle/input/rsna-breast-cancer-detection/train_images/*/*.dcm\"))\ntest = sorted(glob(\"/kaggle/input/rsna-breast-cancer-detection/test_images/*/*.dcm\"))\nprint(\"train files: \", len(train))\nprint(\"test files: \", len(test))\n\npd.reset_option('max_colwidth')","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:52:33.620346Z","iopub.execute_input":"2022-11-30T10:52:33.620953Z","iopub.status.idle":"2022-11-30T10:52:42.79323Z","shell.execute_reply.started":"2022-11-30T10:52:33.620895Z","shell.execute_reply":"2022-11-30T10:52:42.791919Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[:10]","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:52:42.795481Z","iopub.execute_input":"2022-11-30T10:52:42.795888Z","iopub.status.idle":"2022-11-30T10:52:42.806827Z","shell.execute_reply.started":"2022-11-30T10:52:42.79585Z","shell.execute_reply":"2022-11-30T10:52:42.805829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ndisplay(train_df.head(2))","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:52:42.808226Z","iopub.execute_input":"2022-11-30T10:52:42.809101Z","iopub.status.idle":"2022-11-30T10:52:42.914639Z","shell.execute_reply.started":"2022-11-30T10:52:42.80906Z","shell.execute_reply":"2022-11-30T10:52:42.9133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ndisplay(test_df.head(2))","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:52:42.918059Z","iopub.execute_input":"2022-11-30T10:52:42.918619Z","iopub.status.idle":"2022-11-30T10:52:42.936229Z","shell.execute_reply.started":"2022-11-30T10:52:42.918566Z","shell.execute_reply":"2022-11-30T10:52:42.934743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv') \ndisplay(sample_submission.head(2))","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:52:42.937491Z","iopub.execute_input":"2022-11-30T10:52:42.938494Z","iopub.status.idle":"2022-11-30T10:52:42.953084Z","shell.execute_reply.started":"2022-11-30T10:52:42.938449Z","shell.execute_reply":"2022-11-30T10:52:42.951456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dicom exploration\n\n**DICOM** (Digital Imaging and Communications in Medicine),  developed  by  National  Electrical  Manufacturers Association, allows you to create, store, send and print single image,  a  series  of  images,  patient  information,  research, equipment, facilities, medical personnel providing inspection, etc.\n\nDICOM  Standard  defines  two  informational  levels:  \n\n- File Level  -  *DICOM File*  object is  a  file  with  tags  that  contain information about the organization for the representation of the image frame (or series of frames) and accompanying / control information (in the form of DICOM tags);\n- Network  (Communication)  Level  —  *DICOM  Network Protocols*  —  to  transfer  DICOM  files,  DICOM  and  control commands across networks with TCP / IP support.\n\nThe file layer of the 2008 DICOM 3.0 standard describes:\n\n- patient attributes and demographics;\n- model and company of the manufacturer of the device on which the survey was carried out;\n- attributes of the medical institution where the survey was carried out;\n- attributes of the personnel who examined the patient;\n- type of examination and time of its conduct;\n- conditions and parameters of the patient's examination;\n- parameters of an image or a series of images recorded in a DICOM file;\n- unique identification keys *Unique Identifier (UID)* of data groups described in the DICOM file.\n- image, series or set of series obtained during the examination of the patient.\n- presentation, first of all, of PDF documents in a DICOM file.\n- representation of DICOM-recording on optical media, including DVD-format.\n- DICOM protocol for transmission and reception over TCP / IP.\n\nDICOM files contain metadata in addition to the image. For example: age, gender, study modality, name, body part and body position.\n\n**For example: BitsAllocated and PhotometricInterpretation.**\n\nThe Photometric Interpretation value determines the intended interpretation of the pixel data.\n\nHere we are interested in the main two values for the X-ray:\n\n- MONOCHROME1 - Pixel data represent a single monochrome image plane. The minimum sample value is intended to be displayed as white after any VOI gray scale transformations have been performed.\n- MONOCHROME2 - Pixel data represent a single monochrome image plane. The minimum sample value is intended to be displayed as black after any VOI gray scale transformations have been performed.\n\nBits Allocated - Number of bits allocated for each pixel sample. Each sample shall have the same number of bits allocated. Bits Allocated shall be either 1, or a multiple of 8. \n\n\n**If you want to know more about medical image analysis, you can purchase my course:**\n- Medical Image Analysis In Python (https://alimbekovkz.gumroad.com/l/medical_image_analysis_in_python)\n- Medical Image Analysis in Python on Educative (https://www.educative.io/courses/medical-image-analysis-python)\n- [RU] Анализ медицинских изображений в Python (https://startdatajourney.com/ru/course/medical-image-analysis-in-python)","metadata":{}},{"cell_type":"code","source":"import pydicom\nimport matplotlib.pyplot as plt\n\n#displaying the image\nimg = pydicom.read_file(train[0]).pixel_array\nplt.imshow(img, cmap=plt.cm.bone)\nplt.grid(False)\n\n#displaying metadata\ndata = pydicom.dcmread(train[0])\nprint(data)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-11-30T10:52:49.375546Z","iopub.execute_input":"2022-11-30T10:52:49.376018Z","iopub.status.idle":"2022-11-30T10:52:53.131263Z","shell.execute_reply.started":"2022-11-30T10:52:49.375978Z","shell.execute_reply":"2022-11-30T10:52:53.130207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Explore Data\n","metadata":{}},{"cell_type":"code","source":"train_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:52:53.133479Z","iopub.execute_input":"2022-11-30T10:52:53.133873Z","iopub.status.idle":"2022-11-30T10:52:53.151928Z","shell.execute_reply.started":"2022-11-30T10:52:53.133838Z","shell.execute_reply":"2022-11-30T10:52:53.150915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Explore Data\n\nCheck data and target distribution","metadata":{}},{"cell_type":"code","source":"temp = train_df[\"cancer\"].value_counts()\ndf = pd.DataFrame({'labels': temp.index,\n                   'values': temp.values\n                  })\nplt.figure(figsize = (6,6))\nplt.title('Cancer distribution')\nsns.set_color_codes(\"pastel\")\nsns.barplot(x = 'labels', y=\"values\", data=df)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-11-30T10:53:05.326414Z","iopub.execute_input":"2022-11-30T10:53:05.327069Z","iopub.status.idle":"2022-11-30T10:53:05.52111Z","shell.execute_reply.started":"2022-11-30T10:53:05.327021Z","shell.execute_reply":"2022-11-30T10:53:05.519865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset is so imbalanced.","metadata":{}},{"cell_type":"code","source":"import numpy as np\n\n#convert he age column to int\ntrain_df[\"age\"] = pd.to_numeric(train_df[\"age\"])\n\nsorted_ages = np.sort(train_df[\"age\"].values)\nprint(sorted_ages)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-11-30T10:53:06.219357Z","iopub.execute_input":"2022-11-30T10:53:06.220134Z","iopub.status.idle":"2022-11-30T10:53:06.231209Z","shell.execute_reply.started":"2022-11-30T10:53:06.220094Z","shell.execute_reply":"2022-11-30T10:53:06.229631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nplt.style.use('seaborn-whitegrid')\nplt.figure(figsize=(17, 5))\nplt.hist(sorted_ages[:-2], bins=[i for i in range(100)])\nplt.title(\"All patients age histogram\", fontsize=18, pad=10)\nplt.xlabel(\"age\", labelpad=10)\nplt.xticks([i*10 for i in range(11)])\nplt.ylabel(\"count\", labelpad=10)\nplt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-11-30T10:53:06.46057Z","iopub.execute_input":"2022-11-30T10:53:06.462141Z","iopub.status.idle":"2022-11-30T10:53:06.873036Z","shell.execute_reply.started":"2022-11-30T10:53:06.462077Z","shell.execute_reply":"2022-11-30T10:53:06.871784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#calculating all and ill histograms\nbins = [i for i in range(100)]\nplt.style.use('seaborn-whitegrid')\n\nalldata = np.histogram(train_df[\"age\"].values, bins=bins)[0]\n\nilldata = np.histogram(train_df[train_df[\"cancer\"] ==1][\"age\"].values, bins=bins)[0]\n\nfig, axes = plt.subplots(figsize=(17, 16)) #ncols=1, sharey=True, figsize=(17, 16))\n\nfig.suptitle(\"Ages and Cancer\", fontsize=22, y=0.96)\n\naxes.margins(x=0.1, y=0.01)\nm1 = axes.barh(bins[:-1], alldata, color='#90CAF9')\nm2 = axes.barh(bins[:-1], illdata, color='#0D47A1')\naxes.set_title('Age', fontsize=18, pad=15)\naxes.invert_xaxis()\naxes.set(yticks=[i*5 for i in range(20)])\naxes.tick_params(axis=\"y\", labelsize=14)\naxes.yaxis.tick_right()\naxes.xaxis.tick_top()\naxes.legend((m1[0], m2[0]), ('healthy', 'with Cancer'), loc=2, prop={'size': 16})\n\nlocs = axes.get_xticks()\n\n# axes[1].margins(y=0.01)\n# w1 = axes[1].barh(bins[:-1], all_women, color='#EF9A9A')\n# w2 = axes[1].barh(bins[:-1], ill_women, color='#B71C1C')\n# axes[1].set_title('Women', fontsize=18, pad=15)\n# axes[1].xaxis.tick_top()\n# axes[1].set_xticks(locs)\n# axes[1].legend((w1[0], w2[0]), ('healthy', 'with Pneumothorax'), prop={'size': 17})\n\n# #for i, v in enumerate(depos[\"ItemViewCount\"].values):\n#    #print(i, v)\n#     #axes[1].text(int(v) + 3, int(i)-0.25, str(v))\nplt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-11-30T10:53:06.961361Z","iopub.execute_input":"2022-11-30T10:53:06.962614Z","iopub.status.idle":"2022-11-30T10:53:07.93693Z","shell.execute_reply.started":"2022-11-30T10:53:06.962569Z","shell.execute_reply":"2022-11-30T10:53:07.935456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['laterality'].value_counts().plot(kind='bar',figsize = (10, 5));\nplt.title('laterality');","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:53:07.939482Z","iopub.execute_input":"2022-11-30T10:53:07.940455Z","iopub.status.idle":"2022-11-30T10:53:08.16202Z","shell.execute_reply.started":"2022-11-30T10:53:07.94039Z","shell.execute_reply":"2022-11-30T10:53:08.160712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['view'].value_counts().plot(kind='bar',figsize = (10, 5));\nplt.title('view');","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:53:08.942813Z","iopub.execute_input":"2022-11-30T10:53:08.944046Z","iopub.status.idle":"2022-11-30T10:53:09.182834Z","shell.execute_reply.started":"2022-11-30T10:53:08.944001Z","shell.execute_reply":"2022-11-30T10:53:09.181474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's look at the ratio of target to non-target by features:","metadata":{}},{"cell_type":"code","source":"def crosstab(col):\n    CrosstabResult=pd.crosstab(index=train_df[col],columns=train_df['cancer'])\n    crossplot = CrosstabResult.plot(figsize=(10,10), rot=0, kind=\"bar\", stacked=True)\n    crossplot.legend(title=col, bbox_to_anchor=(1, 1.02), loc='upper left')","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-11-30T10:53:10.61322Z","iopub.execute_input":"2022-11-30T10:53:10.61367Z","iopub.status.idle":"2022-11-30T10:53:10.621384Z","shell.execute_reply.started":"2022-11-30T10:53:10.613635Z","shell.execute_reply":"2022-11-30T10:53:10.620149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"binarycol = ['biopsy', 'invasive', 'implant', 'difficult_negative_case', 'density', 'BIRADS']\n\nfor col in binarycol:\n    crosstab(col)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-11-30T10:53:11.061403Z","iopub.execute_input":"2022-11-30T10:53:11.06188Z","iopub.status.idle":"2022-11-30T10:53:12.73772Z","shell.execute_reply.started":"2022-11-30T10:53:11.061838Z","shell.execute_reply":"2022-11-30T10:53:12.736288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preproc - CROP\n\nAs an idea for model training, we can try cropping the breast,because most of the image does not make sense.","metadata":{}},{"cell_type":"code","source":"import scipy","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:54:19.269938Z","iopub.execute_input":"2022-11-30T10:54:19.270355Z","iopub.status.idle":"2022-11-30T10:54:19.275876Z","shell.execute_reply.started":"2022-11-30T10:54:19.270322Z","shell.execute_reply":"2022-11-30T10:54:19.274906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#displaying the image\nimg = pydicom.read_file(train[1]).pixel_array\nplt.imshow(img, cmap=plt.cm.bone)\nplt.grid(False)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:54:20.18648Z","iopub.execute_input":"2022-11-30T10:54:20.187277Z","iopub.status.idle":"2022-11-30T10:54:23.693229Z","shell.execute_reply.started":"2022-11-30T10:54:20.187235Z","shell.execute_reply":"2022-11-30T10:54:23.69195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(img.ravel())","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:54:23.694918Z","iopub.execute_input":"2022-11-30T10:54:23.695757Z","iopub.status.idle":"2022-11-30T10:54:24.533536Z","shell.execute_reply.started":"2022-11-30T10:54:23.695719Z","shell.execute_reply":"2022-11-30T10:54:24.532353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Copyright (C) 2019 Nan Wu, Jason Phang, Jungkyu Park, Yiqiu Shen, Zhe Huang, Masha Zorin, \n#   Stanisław Jastrzębski, Thibault Févry, Joe Katsnelson, Eric Kim, Stacey Wolfson, Ujas Parikh, \n#   Sushma Gaddam, Leng Leng Young Lin, Kara Ho, Joshua D. Weinstein, Beatriu Reig, Yiming Gao, \n#   Hildegard Toth, Kristine Pysarenko, Alana Lewin, Jiyon Lee, Krystal Airola, Eralda Mema, \n#   Stephanie Chung, Esther Hwang, Naziya Samreen, S. Gene Kim, Laura Heacock, Linda Moy, \n#   Kyunghyun Cho, Krzysztof J. Geras\n#\n# This file is part of breast_cancer_classifier.\n#\n# breast_cancer_classifier is free software: you can redistribute it and/or modify\n# it under the terms of the GNU Affero General Public License as\n# published by the Free Software Foundation, either version 3 of the\n# License, or (at your option) any later version.\n#\n# breast_cancer_classifier is distributed in the hope that it will be useful,\n# but WITHOUT ANY WARRANTY; without even the implied warranty of\n# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the\n# GNU Affero General Public License for more details.\n#\n# You should have received a copy of the GNU Affero General Public License\n# along with breast_cancer_classifier.  If not, see <http://www.gnu.org/licenses/>.\n# ==============================================================================","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:54:24.53757Z","iopub.execute_input":"2022-11-30T10:54:24.537986Z","iopub.status.idle":"2022-11-30T10:54:24.5441Z","shell.execute_reply.started":"2022-11-30T10:54:24.537952Z","shell.execute_reply":"2022-11-30T10:54:24.542958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_masks_and_sizes_of_connected_components(img_mask):\n    \"\"\"\n    Finds the connected components from the mask of the image\n    \"\"\"\n    mask, num_labels = scipy.ndimage.label(img_mask)\n\n    mask_pixels_dict = {}\n    for i in range(num_labels+1):\n        this_mask = (mask == i)\n        if img_mask[this_mask][0] != 0:\n            # Exclude the 0-valued mask\n            mask_pixels_dict[i] = np.sum(this_mask)\n        \n    return mask, mask_pixels_dict\n\n\ndef get_mask_of_largest_connected_component(img_mask):\n    \"\"\"\n    Finds the largest connected component from the mask of the image\n    \"\"\"\n    mask, mask_pixels_dict = get_masks_and_sizes_of_connected_components(img_mask)\n    largest_mask_index = pd.Series(mask_pixels_dict).idxmax()\n    largest_mask = mask == largest_mask_index\n    return largest_mask\n\n\ndef get_edge_values(img, largest_mask, axis):\n    \"\"\"\n    Finds the bounding box for the largest connected component\n    \"\"\"\n    assert axis in [\"x\", \"y\"]\n    has_value = np.any(largest_mask, axis=int(axis == \"y\"))\n    edge_start = np.arange(img.shape[int(axis == \"x\")])[has_value][0]\n    edge_end = np.arange(img.shape[int(axis == \"x\")])[has_value][-1] + 1\n    return edge_start, edge_end\n\n\ndef get_bottommost_pixels(img, largest_mask, y_edge_bottom):\n    \"\"\"\n    Gets the bottommost nonzero pixels of dilated mask before cropping. \n    \"\"\"\n    bottommost_nonzero_y = y_edge_bottom - 1\n    bottommost_nonzero_x = np.arange(img.shape[1])[largest_mask[bottommost_nonzero_y, :] > 0]\n    return bottommost_nonzero_y, bottommost_nonzero_x\n\n\ndef get_distance_from_starting_side(img, mode, x_edge_left, x_edge_right):\n    \"\"\"\n    If we fail to recover the original shape as a result of erosion-dilation \n    on the side where the breast starts to appear in the image, \n    we record this information.\n    \"\"\"\n    if mode == \"left\":\n        return img.shape[1] - x_edge_right\n    else:\n        return x_edge_left\n\n\ndef include_buffer_y_axis(img, y_edge_top, y_edge_bottom, buffer_size):\n    \"\"\"\n    Includes buffer in all sides of the image in y-direction\n    \"\"\"\n    if y_edge_top > 0:\n        y_edge_top -= min(y_edge_top, buffer_size)\n    if y_edge_bottom < img.shape[0]:\n        y_edge_bottom += min(img.shape[0] - y_edge_bottom, buffer_size)\n    return y_edge_top, y_edge_bottom     \n\n\ndef include_buffer_x_axis(img, mode, x_edge_left, x_edge_right, buffer_size):\n    \"\"\"\n    Includes buffer in only one side of the image in x-direction\n    \"\"\"\n    if mode == \"left\":\n        if x_edge_left > 0:\n            x_edge_left -= min(x_edge_left, buffer_size)\n    else:\n        if x_edge_right < img.shape[1]:\n            x_edge_right += min(img.shape[1] - x_edge_right, buffer_size)\n    return x_edge_left, x_edge_right\n\n\ndef convert_bottommost_pixels_wrt_cropped_image(mode, bottommost_nonzero_y, bottommost_nonzero_x,\n                                                y_edge_top, x_edge_right, x_edge_left):\n    \"\"\"\n    Once the image is cropped, adjusts the bottommost pixel values which was originally w.r.t. the original image\n    \"\"\"\n    bottommost_nonzero_y -= y_edge_top\n    if mode == \"left\":\n        bottommost_nonzero_x = x_edge_right - bottommost_nonzero_x  # in this case, not in sorted order anymore.\n        bottommost_nonzero_x = np.flip(bottommost_nonzero_x, 0)\n    else:\n        bottommost_nonzero_x -= x_edge_left\n    return bottommost_nonzero_y, bottommost_nonzero_x\n\n\ndef get_rightmost_pixels_wrt_cropped_image(mode, largest_mask_cropped, find_rightmost_from_ratio):\n    \"\"\"\n    Ignores top find_rightmost_from_ratio of the image and searches the rightmost nonzero pixels\n    of the dilated mask from the bottom portion of the image.\n    \"\"\"\n    ignore_height = int(largest_mask_cropped.shape[0] * find_rightmost_from_ratio)\n    rightmost_pixel_search_area = largest_mask_cropped[ignore_height:, :]\n    rightmost_pixel_search_area_has_value = np.any(rightmost_pixel_search_area, axis=0)\n    rightmost_nonzero_x = np.arange(rightmost_pixel_search_area.shape[1])[\n        rightmost_pixel_search_area_has_value][-1 if mode == 'right' else 0]\n    rightmost_nonzero_y = np.arange(rightmost_pixel_search_area.shape[0])[\n        rightmost_pixel_search_area[:, rightmost_nonzero_x] > 0] + ignore_height\n\n    # rightmost pixels are already found w.r.t. newly cropped image, except that we still need to\n    #   reflect horizontal_flip\n    if mode == \"left\":\n        rightmost_nonzero_x = largest_mask_cropped.shape[1] - rightmost_nonzero_x\n        \n    return rightmost_nonzero_y, rightmost_nonzero_x\n\n\ndef crop_img_from_largest_connected(img, mode, erode_dialate=True, iterations=100, \n                                    buffer_size=50, find_rightmost_from_ratio=1/3):\n    \"\"\"\n    Performs erosion on the mask of the image, selects largest connected component,\n    dialates the largest connected component, and draws a bounding box for the result\n    with buffers\n    input:\n        - img:   2D numpy array\n        - mode:  breast pointing left or right\n    output: a tuple of (window_location, rightmost_points, \n                        bottommost_points, distance_from_starting_side)\n        - window_location: location of cropping window w.r.t. original dicom image so that segmentation\n           map can be cropped in the same way for training.\n        - rightmost_points: rightmost nonzero pixels after correctly being flipped in the format of \n                            ((y_start, y_end), x)\n        - bottommost_points: bottommost nonzero pixels after correctly being flipped in the format of\n                             (y, (x_start, x_end))\n        - distance_from_starting_side: number of zero columns between the start of the image and start of\n           the largest connected component w.r.t. original dicom image.\n    \"\"\"\n    assert mode in (\"left\", \"right\")\n\n    img_mask = img < 200\n    \n    # Erosion in order to remove thin lines in the background\n    if erode_dialate:\n        img_mask = scipy.ndimage.morphology.binary_erosion(img_mask, iterations=iterations)\n    \n    # Select mask for largest connected component\n    largest_mask = get_mask_of_largest_connected_component(img_mask)\n\n    # Dilation to recover the original mask, excluding the thin lines\n    if erode_dialate:\n        largest_mask = scipy.ndimage.morphology.binary_dilation(largest_mask, iterations=iterations)\n    \n    # figure out where to crop\n    y_edge_top, y_edge_bottom = get_edge_values(img, largest_mask, \"y\")\n    x_edge_left, x_edge_right = get_edge_values(img, largest_mask, \"x\")\n\n    # extract bottommost pixel info\n    bottommost_nonzero_y, bottommost_nonzero_x = get_bottommost_pixels(img, largest_mask, y_edge_bottom)\n\n    # include maximum 'buffer_size' more pixels on both sides just to make sure we don't miss anything\n    y_edge_top, y_edge_bottom = include_buffer_y_axis(img, y_edge_top, y_edge_bottom, buffer_size)\n    \n    # If cropped image not starting from corresponding edge, they are wrong. Record the distance, will reject if not 0.\n    distance_from_starting_side = get_distance_from_starting_side(img, mode, x_edge_left, x_edge_right)\n\n    # include more pixels on either side just to make sure we don't miss anything, if the next column\n    #   contains non-zero value and isn't noise\n    x_edge_left, x_edge_right = include_buffer_x_axis(img, mode, x_edge_left, x_edge_right, buffer_size)\n\n    # convert bottommost pixel locations w.r.t. newly cropped image. Flip if necessary.\n    bottommost_nonzero_y, bottommost_nonzero_x = convert_bottommost_pixels_wrt_cropped_image(\n        mode,\n        bottommost_nonzero_y,\n        bottommost_nonzero_x,\n        y_edge_top,\n        x_edge_right,\n        x_edge_left\n    )\n\n    # calculate rightmost point from bottom portion of the image w.r.t. cropped image. Flip if necessary.\n    rightmost_nonzero_y, rightmost_nonzero_x = get_rightmost_pixels_wrt_cropped_image(\n        mode,\n        largest_mask[y_edge_top: y_edge_bottom, x_edge_left: x_edge_right],\n        find_rightmost_from_ratio\n    )\n\n    # save window location in medical mode, but everything else in training mode\n    return (y_edge_top, y_edge_bottom, x_edge_left, x_edge_right), \\\n        ((rightmost_nonzero_y[0], rightmost_nonzero_y[-1]), rightmost_nonzero_x), \\\n        (bottommost_nonzero_y, (bottommost_nonzero_x[0], bottommost_nonzero_x[-1])), \\\n        distance_from_starting_side\n\n\ndef image_orientation(horizontal_flip, side):\n    \"\"\"\n    Returns the direction where the breast should be facing in the original image\n    This information is used in cropping.crop_img_horizontally_from_largest_connected\n    \"\"\"\n    assert horizontal_flip in ['YES', 'NO'], \"Wrong horizontal flip\"\n    assert side in ['L', 'R'], \"Wrong side\"\n    if horizontal_flip == 'YES':\n        if side == 'R':\n            return 'right'\n        else:\n            return 'left'\n    else:\n        if side == 'R':\n            return 'left'\n        else:\n            return 'right'\n\n\ndef crop_mammogram(input_data_folder, exam_list_path, cropped_exam_list_path, output_data_folder,\n                   num_processes, num_iterations, buffer_size):\n    \"\"\"\n    In parallel, crops mammograms in DICOM format found in input_data_folder and save as png format in\n    output_data_folder and saves new image list in cropped_image_list_path\n    \"\"\"\n    exam_list = pickling.unpickle_from_file(exam_list_path)\n    \n    image_list = data_handling.unpack_exam_into_images(exam_list)\n    \n    if os.path.exists(output_data_folder):\n        # Prevent overwriting to an existing directory\n        print(\"Error: the directory to save cropped images already exists.\")\n        return\n    else:\n        os.makedirs(output_data_folder)\n\n    crop_mammogram_one_image_func = partial(\n        crop_mammogram_one_image_short_path,\n        input_data_folder=input_data_folder, \n        output_data_folder=output_data_folder,\n        num_iterations=num_iterations,\n        buffer_size=buffer_size,\n    )\n    with Pool(num_processes) as pool:\n        cropped_image_info = pool.map(crop_mammogram_one_image_func, image_list)\n    \n    window_location_dict = dict([x[0] for x in cropped_image_info])\n    rightmost_points_dict = dict([x[1] for x in cropped_image_info])\n    bottommost_points_dict = dict([x[2] for x in cropped_image_info])\n    distance_from_starting_side_dict = dict([x[3] for x in cropped_image_info])\n\n    data_handling.add_metadata(exam_list, \"window_location\", window_location_dict)\n    data_handling.add_metadata(exam_list, \"rightmost_points\", rightmost_points_dict)\n    data_handling.add_metadata(exam_list, \"bottommost_points\", bottommost_points_dict)\n    data_handling.add_metadata(exam_list, \"distance_from_starting_side\", distance_from_starting_side_dict)\n    \n    pickle_to_file(cropped_exam_list_path, exam_list)\n    \n\ndef crop_mammogram_one_image(scan, input_file_path, output_file_path, num_iterations, buffer_size):\n    \"\"\"\n    Crops a mammogram, saves as png file, includes the following additional information:\n        - window_location: location of cropping window w.r.t. original dicom image so that segmentation\n           map can be cropped in the same way for training.\n        - rightmost_points: rightmost nonzero pixels after correctly being flipped\n        - bottommost_points: bottommost nonzero pixels after correctly being flipped\n        - distance_from_starting_side: number of zero columns between the start of the image and start of\n           the largest connected component w.r.t. original dicom image.\n    \"\"\"\n\n    image = read_image_png(input_file_path)\n    try:\n        # error detection using erosion. Also get cropping information for this image.\n        cropping_info = crop_img_from_largest_connected(\n            image, \n            image_orientation(scan['horizontal_flip'], scan['side']), \n            True, \n            num_iterations, \n            buffer_size, \n            1/3\n        )\n    except Exception as error:\n        print(input_file_path, \"\\n\\tFailed to crop image because image is invalid.\", str(error))\n    else:\n        \n        top, bottom, left, right = cropping_info[0]\n\n        target_parent_dir = os.path.split(output_file_path)[0]\n        if not os.path.exists(target_parent_dir):\n            os.makedirs(target_parent_dir)\n        \n        try:\n            save_image_as_png(image[top:bottom, left:right], output_file_path)\n        except Exception as error:\n            print(input_file_path, \"\\n\\tError while saving image.\", str(error))\n\n        return cropping_info\n\n\ndef crop_mammogram_one_image_short_path(scan, input_data_folder, output_data_folder,\n                                        num_iterations, buffer_size):\n    \"\"\"\n    Crops a mammogram from a short_file_path\n    See: crop_mammogram_one_image\n    \"\"\"\n    full_input_file_path = os.path.join(input_data_folder, scan['short_file_path']+'.png')\n    full_output_file_path = os.path.join(output_data_folder, scan['short_file_path'] + '.png')\n    cropping_info = crop_mammogram_one_image(\n        scan=scan,\n        input_file_path=full_input_file_path,\n        output_file_path=full_output_file_path,\n        num_iterations=num_iterations,\n        buffer_size=buffer_size,\n    )\n    return list(zip([scan['short_file_path']] * 4, cropping_info))","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-11-30T10:54:24.54746Z","iopub.execute_input":"2022-11-30T10:54:24.547937Z","iopub.status.idle":"2022-11-30T10:54:24.59546Z","shell.execute_reply.started":"2022-11-30T10:54:24.547884Z","shell.execute_reply":"2022-11-30T10:54:24.594131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cropping_info_df = pd.DataFrame(columns=['PID','side', 'orientation', 'top','bottom','left','right','view'])","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:54:24.597512Z","iopub.execute_input":"2022-11-30T10:54:24.598048Z","iopub.status.idle":"2022-11-30T10:54:24.608609Z","shell.execute_reply.started":"2022-11-30T10:54:24.597997Z","shell.execute_reply":"2022-11-30T10:54:24.607254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import imageio\n\ndef save_image_as_png(image, filename):\n    imageio.imwrite(filename, image)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:54:24.610365Z","iopub.execute_input":"2022-11-30T10:54:24.610765Z","iopub.status.idle":"2022-11-30T10:54:24.621902Z","shell.execute_reply.started":"2022-11-30T10:54:24.610728Z","shell.execute_reply":"2022-11-30T10:54:24.62039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_mammogram_one_image(PID, side, orientation, scan, input_file_path, output_file_path, num_iterations, buffer_size):\n    #image = read_image_png(input_file_path)\n    global cropping_info_df\n    image = input_file_path\n    try:\n        # error detection using erosion. Also get cropping information for this image.\n        cropping_info = crop_img_from_largest_connected(\n            image, \n            image_orientation(scan['horizontal_flip'], scan['side']), \n            True, \n            num_iterations, \n            buffer_size, \n            1/3\n        )\n    except Exception as error:\n        print(input_file_path, \"\\n\\tFailed to crop image because image is invalid.\", str(error))\n    else:\n        \n        top, bottom, left, right = cropping_info[0]\n\n        target_parent_dir = os.path.split(output_file_path)[0]\n        if not os.path.exists(target_parent_dir):\n            os.makedirs(target_parent_dir)\n        \n        image = image[top:bottom, left:right]\n#         height = 1024\n#         width = round(image.shape[1]*1024/image.shape[0])\n#         image = cv2.resize(image,(width,height),3)\n        \n        \n#         pad_col = 1024-width\n#         if np.mean(image[:,0:10]) > np.mean(image[:,width-10:width]):\n#             image = np.pad(image, ((0, 0), (0, pad_col)), mode='constant', constant_values=0)\n#             views = 1\n#         else:\n#             image = np.pad(image, ((0, 0), (pad_col, 0 )), mode='constant', constant_values=0)       \n#             views = 0\n\n        views = 1\n        \n        cropping_info_df = cropping_info_df.append({'PID': PID, 'side': side, 'orientation': orientation, \\\n                                            'top': top, 'bottom': bottom, 'left':left, 'right': right, 'view':  views}, ignore_index=True)\n                   \n        #print(image[:,0:10].shape, np.sum(image[:,0:10]),np.mean(image[:,0:10]))\n        #print(image[:,width-10:width].shape, np.sum(image[:,width-10:width]),np.mean(image[:,width-10:width]))\n        \n        ''' \n        pad_col = 1024-width\n        if side == 'L':\n            image = np.pad(image, ((0, 0), (0, pad_col)), mode='constant', constant_values=0)\n        else:\n            image = np.pad(image, ((0, 0), (pad_col, 0 )), mode='constant', constant_values=0)\n        '''\n        \n        try:\n            save_image_as_png(image, output_file_path)\n        except Exception as error:\n            print(input_file_path, \"\\n\\tError while saving image.\", str(error))\n\n        return cropping_info","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-11-30T10:54:24.828073Z","iopub.execute_input":"2022-11-30T10:54:24.828518Z","iopub.status.idle":"2022-11-30T10:54:24.841644Z","shell.execute_reply.started":"2022-11-30T10:54:24.828483Z","shell.execute_reply":"2022-11-30T10:54:24.839984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_single_mammogram(PID, mammogram_path, view,\n                          cropped_mammogram_path, metadata_path,\n                          num_iterations=100, buffer_size=50, horizontal_flip='NO'):\n    \"\"\"\n    Crop a single mammogram image\n    \"\"\"\n    metadata_dict = dict(\n        short_file_path=None,\n        horizontal_flip=horizontal_flip,\n        full_view=view,\n        side=view[0],\n        view=view[2:],\n    )\n    cropped_image_info = crop_mammogram_one_image(\n        PID = PID,\n        side = view[0],\n        orientation = view[2:],\n        scan=metadata_dict,\n        input_file_path=mammogram_path,\n        output_file_path=cropped_mammogram_path,\n        num_iterations=num_iterations,\n        buffer_size=buffer_size,\n    )","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-11-30T10:54:25.021437Z","iopub.execute_input":"2022-11-30T10:54:25.021887Z","iopub.status.idle":"2022-11-30T10:54:25.030653Z","shell.execute_reply.started":"2022-11-30T10:54:25.02185Z","shell.execute_reply":"2022-11-30T10:54:25.029382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[1]","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:54:25.20855Z","iopub.execute_input":"2022-11-30T10:54:25.209048Z","iopub.status.idle":"2022-11-30T10:54:25.217242Z","shell.execute_reply.started":"2022-11-30T10:54:25.209004Z","shell.execute_reply":"2022-11-30T10:54:25.216077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = pydicom.read_file(train[1]).pixel_array\nimg = img - img.min()\nimg = img /(img.max() - img.min())\nimg *= 255\nimg = np.uint8(img)\n\n\ndata = pydicom.dcmread(train[1])\nprint(data)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:54:26.373105Z","iopub.execute_input":"2022-11-30T10:54:26.373514Z","iopub.status.idle":"2022-11-30T10:54:28.169927Z","shell.execute_reply.started":"2022-11-30T10:54:26.373481Z","shell.execute_reply":"2022-11-30T10:54:28.168429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(img.ravel())","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:54:30.580919Z","iopub.execute_input":"2022-11-30T10:54:30.58134Z","iopub.status.idle":"2022-11-30T10:54:31.381816Z","shell.execute_reply.started":"2022-11-30T10:54:30.581305Z","shell.execute_reply":"2022-11-30T10:54:31.38045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:54:31.3839Z","iopub.execute_input":"2022-11-30T10:54:31.38445Z","iopub.status.idle":"2022-11-30T10:54:31.402538Z","shell.execute_reply.started":"2022-11-30T10:54:31.384415Z","shell.execute_reply":"2022-11-30T10:54:31.40098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crop_single_mammogram('10006', img, 'L-MLO', '/kaggle/working/test.png' ,'')","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:54:32.957606Z","iopub.execute_input":"2022-11-30T10:54:32.958058Z","iopub.status.idle":"2022-11-30T10:54:34.613915Z","shell.execute_reply.started":"2022-11-30T10:54:32.958021Z","shell.execute_reply":"2022-11-30T10:54:34.612756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cropping_info_df","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:54:58.839076Z","iopub.execute_input":"2022-11-30T10:54:58.840004Z","iopub.status.idle":"2022-11-30T10:54:58.856324Z","shell.execute_reply.started":"2022-11-30T10:54:58.839961Z","shell.execute_reply":"2022-11-30T10:54:58.855105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:54:59.885243Z","iopub.execute_input":"2022-11-30T10:54:59.88573Z","iopub.status.idle":"2022-11-30T10:54:59.891995Z","shell.execute_reply.started":"2022-11-30T10:54:59.885667Z","shell.execute_reply":"2022-11-30T10:54:59.890244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = cv2.imread('/kaggle/working/test.png')","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:00.744844Z","iopub.execute_input":"2022-11-30T10:55:00.745296Z","iopub.status.idle":"2022-11-30T10:55:00.831118Z","shell.execute_reply.started":"2022-11-30T10:55:00.74523Z","shell.execute_reply":"2022-11-30T10:55:00.829796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(1, figsize=(12, 12))\nax.imshow(image[:,:,0], cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:01.683245Z","iopub.execute_input":"2022-11-30T10:55:01.683747Z","iopub.status.idle":"2022-11-30T10:55:02.701266Z","shell.execute_reply.started":"2022-11-30T10:55:01.683704Z","shell.execute_reply":"2022-11-30T10:55:02.70006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Crop by felzenszwalb","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom skimage.exposure import equalize_hist\nfrom skimage.filters.rank import median\nfrom skimage.measure import regionprops\nfrom skimage.morphology import disk\nfrom skimage.segmentation import felzenszwalb\nfrom skimage.transform import rescale\nfrom scipy.ndimage import binary_fill_holes\n#from scipy.misc import imresize\nfrom PIL import Image\n\n\ndef breast_segment(im, scale_factor=0.25, threshold=3900, felzenzwalb_scale=0.15):\n    \"\"\"\n    Fully automated breast segmentation in mammographies.\n    https://github.com/olieidel/breast_segment\n    :param im: Image\n    :param scale_factor: Scale Factor\n    :param threshold: Threshold\n    :param felzenzwalb_scale: Felzenzwalb Scale\n    :return: (im_mask, bbox) where im_mask is the segmentation mask and\n    bbox is the bounding box (rectangular) of the segmentation.\n    \"\"\"\n\n    # set threshold to remove artifacts around edges\n    im_thres = im.copy()\n    im_thres[im_thres > threshold] = 0\n    \n    # determine breast side\n    col_sums_split = np.array_split(np.sum(im_thres, axis=0), 2)\n    left_col_sum = np.sum(col_sums_split[0])\n    right_col_sum = np.sum(col_sums_split[1])\n    \n    if left_col_sum > right_col_sum:\n        breast_side = 'l'\n    else:\n        breast_side = 'r'\n\n    # rescale and filter aggressively, normalize\n    im_small = rescale(im_thres, scale_factor)\n    im_small_filt = median(im_small, disk(50))   #50\n    # this might not be helping, actually sometimes it is\n    im_small_filt = equalize_hist(im_small_filt)\n   \n    # run mr. felzenzwalb\n    segments = felzenszwalb(im_small_filt, scale=felzenzwalb_scale)\n    segments += 1  # otherwise, labels() would ignore segment with segment=0\n    \n    props = regionprops(segments)\n    \n    # Sort Props by area, descending\n    props_sorted = sorted(props, key=lambda x: x.area, reverse=True)\n    #for region in props_sorted:\n    #    print(region.area, region.bbox)\n        \n    expected_bg_index = 0 #0\n    bg_index = expected_bg_index\n\n    bg_region = props_sorted[bg_index]\n    minr, minc, maxr, maxc = bg_region.bbox\n    filled_mask = bg_region.filled_image\n    \n    #print(bg_region)\n    #print(minr, minc, maxr, maxc)\n    #print(filled_mask)\n\n    im_small_fill = np.zeros((im_small_filt.shape[0]+2, im_small_filt.shape[1]+1), dtype=int)\n    #print(im_small_fill)\n    #print(im_small_fill.shape)\n    #print(type(filled_mask))\n    \n    #imgg = Image.fromarray(filled_mask)\n    #imgg.save(\"outfile.jpeg\")\n    \n    #print(breast_side)\n\n    if breast_side == 'l':\n        # breast expected to be on left side,\n        # pad on right and bottom side\n        im_small_fill[minr+1:maxr+1, minc:maxc] = filled_mask\n        im_small_fill[0, :] = 1  # top\n        im_small_fill[-1, :] = 1  # bottom\n        im_small_fill[:, -1] = 1  # right\n    elif breast_side == 'r':\n        # breast expected to be on right side,\n        # pad on left and bottom side\n        im_small_fill[minr+1:maxr+1, minc+1:maxc+1] = filled_mask  # shift mask to right side\n        im_small_fill[0, :] = 1  # top\n        im_small_fill[-1, :] = 1  # bottom\n        im_small_fill[:, 0] = 1  # left\n\n    #print(\"------\")\n    #print(type(im_small_fill))\n    \n   \n    im_small_fill = binary_fill_holes(im_small_fill)\n\n    im_small_mask = im_small_fill[1:-1, :-1] if breast_side == 'l' \\\n                  else im_small_fill[1:-1, 1:]\n\n    # rescale mask\n    #im_mask = imresize(im_small_mask, im.shape).astype(bool)\n    #print(im.shape, type(im.shape))\n    shape = (im.shape[1],im.shape[0])\n    im_mask = np.array(Image.fromarray(im_small_mask).resize(shape)).astype(bool)\n    \n    # invert!\n    im_mask = ~im_mask\n    \n    # determine side of breast in mask and compare\n    col_sums_split = np.array_split(np.sum(im_mask, axis=0), 2)\n    left_col_sum = np.sum(col_sums_split[0])\n    right_col_sum = np.sum(col_sums_split[1])\n\n    if left_col_sum > right_col_sum:\n        breast_side_mask = 'l'\n    else:\n        breast_side_mask = 'r'\n\n    if breast_side_mask != breast_side:\n        # breast mask is not on expected side\n        # we might have segmented bg instead of breast\n        # so invert again\n        print('breast and mask side mismatch. inverting!')\n        im_mask = ~im_mask\n\n    # exclude thresholded area (artifacts) in mask, too\n    # im_mask[im > threshold] = False\n\n    # fill holes again, just in case there was a high-intensity region\n    # in the breast\n#    im_mask = binary_fill_holes(im_mask)\n\n    # if no region found, abort early and return mask of complete image\n    if im_mask.ravel().sum() == 0:\n        all_mask = np.ones_like(im).astype(bool)\n        bbox = (0, 0, im.shape[0], im.shape[1])\n        print('Couldn\\'t find any segment')\n        return all_mask, bbox\n\n    # get bbox\n    minr = np.argwhere(im_mask.any(axis=1)).ravel()[0]\n    maxr = np.argwhere(im_mask.any(axis=1)).ravel()[-1]\n    minc = np.argwhere(im_mask.any(axis=0)).ravel()[0]\n    maxc = np.argwhere(im_mask.any(axis=0)).ravel()[-1]\n\n    bbox = (minr, minc, maxr, maxc)\n\n    return im_mask, bbox","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-11-30T10:55:03.698168Z","iopub.execute_input":"2022-11-30T10:55:03.698864Z","iopub.status.idle":"2022-11-30T10:55:03.745274Z","shell.execute_reply.started":"2022-11-30T10:55:03.698824Z","shell.execute_reply":"2022-11-30T10:55:03.7443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"im_file = '/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs/train_images_processed/10006/1459541791.png'","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:03.930126Z","iopub.execute_input":"2022-11-30T10:55:03.930971Z","iopub.status.idle":"2022-11-30T10:55:03.936033Z","shell.execute_reply.started":"2022-11-30T10:55:03.930931Z","shell.execute_reply":"2022-11-30T10:55:03.934587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = cv2.imread(im_file)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:04.120426Z","iopub.execute_input":"2022-11-30T10:55:04.120905Z","iopub.status.idle":"2022-11-30T10:55:04.13293Z","shell.execute_reply.started":"2022-11-30T10:55:04.120867Z","shell.execute_reply":"2022-11-30T10:55:04.131912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(1, figsize=(12, 12))\nax.imshow(image[:,:,0], cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:04.310747Z","iopub.execute_input":"2022-11-30T10:55:04.311563Z","iopub.status.idle":"2022-11-30T10:55:04.729671Z","shell.execute_reply.started":"2022-11-30T10:55:04.31151Z","shell.execute_reply":"2022-11-30T10:55:04.728437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"im = image[:,:,0]","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:04.731932Z","iopub.execute_input":"2022-11-30T10:55:04.733238Z","iopub.status.idle":"2022-11-30T10:55:04.738986Z","shell.execute_reply.started":"2022-11-30T10:55:04.733187Z","shell.execute_reply":"2022-11-30T10:55:04.738011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(im.ravel())","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:04.740362Z","iopub.execute_input":"2022-11-30T10:55:04.741545Z","iopub.status.idle":"2022-11-30T10:55:05.229455Z","shell.execute_reply.started":"2022-11-30T10:55:04.741502Z","shell.execute_reply":"2022-11-30T10:55:05.228088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"im.shape","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:05.231724Z","iopub.execute_input":"2022-11-30T10:55:05.232222Z","iopub.status.idle":"2022-11-30T10:55:05.240567Z","shell.execute_reply.started":"2022-11-30T10:55:05.232187Z","shell.execute_reply":"2022-11-30T10:55:05.238916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask, bbox = breast_segment(im, scale_factor=1, threshold=30000)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:05.242388Z","iopub.execute_input":"2022-11-30T10:55:05.243158Z","iopub.status.idle":"2022-11-30T10:55:05.6818Z","shell.execute_reply.started":"2022-11-30T10:55:05.243069Z","shell.execute_reply":"2022-11-30T10:55:05.68011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(1, figsize=(12, 12))\nax.imshow(mask, cmap='inferno')","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:05.684847Z","iopub.execute_input":"2022-11-30T10:55:05.685357Z","iopub.status.idle":"2022-11-30T10:55:06.081832Z","shell.execute_reply.started":"2022-11-30T10:55:05.685319Z","shell.execute_reply":"2022-11-30T10:55:06.0805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bbox","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:06.083223Z","iopub.execute_input":"2022-11-30T10:55:06.083605Z","iopub.status.idle":"2022-11-30T10:55:06.091408Z","shell.execute_reply.started":"2022-11-30T10:55:06.083568Z","shell.execute_reply":"2022-11-30T10:55:06.089768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_segment_crop(img,tol=0, mask=None):\n    if mask is None:\n        mask = img > tol\n    return img[np.ix_(mask.any(1), mask.any(0))]","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:06.093248Z","iopub.execute_input":"2022-11-30T10:55:06.093946Z","iopub.status.idle":"2022-11-30T10:55:06.103234Z","shell.execute_reply.started":"2022-11-30T10:55:06.09389Z","shell.execute_reply":"2022-11-30T10:55:06.101613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tt = get_segment_crop(im, mask=mask) ","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:06.106186Z","iopub.execute_input":"2022-11-30T10:55:06.1066Z","iopub.status.idle":"2022-11-30T10:55:06.117192Z","shell.execute_reply.started":"2022-11-30T10:55:06.106565Z","shell.execute_reply":"2022-11-30T10:55:06.115395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f = plt.figure(figsize=(12, 12))\nax = plt.subplot(111)\nax.imshow(tt, cmap='gray') ","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:06.120757Z","iopub.execute_input":"2022-11-30T10:55:06.12118Z","iopub.status.idle":"2022-11-30T10:55:06.407928Z","shell.execute_reply.started":"2022-11-30T10:55:06.121143Z","shell.execute_reply":"2022-11-30T10:55:06.406476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we're actually done, but just for some fun, let's\n# have a look how we can better visualize this\n\n# prepare the bbox for visalization\ndef bbox_lines(im, bbox):\n    # destructure the bbox as into columns and rows, see above\n    minr, minc, maxr, maxc = bbox\n    line_left = mlines.Line2D([minc, minc], [0, im.shape[0]], color='y', lw=20, alpha=.5)\n    line_right = mlines.Line2D([maxc, maxc], [0, im.shape[0]], color='y', lw=20, alpha=.5)\n    line_top = mlines.Line2D([0, im.shape[1]], [minr, minr], color='y', lw=20, alpha=.5)\n    line_bot = mlines.Line2D([0, im.shape[1]], [maxr, maxr], color='y', lw=20, alpha=.5)\n    \n    return line_left, line_right, line_top, line_bot","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:06.41041Z","iopub.execute_input":"2022-11-30T10:55:06.410819Z","iopub.status.idle":"2022-11-30T10:55:06.420332Z","shell.execute_reply.started":"2022-11-30T10:55:06.410771Z","shell.execute_reply":"2022-11-30T10:55:06.418767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's plot it! .. in yellow\n\nimport matplotlib.lines as mlines\n\nf = plt.figure(figsize=(12, 12))\nax = plt.subplot(111)\nax.imshow(im, cmap='gray')\nlines = bbox_lines(im, bbox)\n[ax.add_line(l) for l in lines]","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:07.039219Z","iopub.execute_input":"2022-11-30T10:55:07.040401Z","iopub.status.idle":"2022-11-30T10:55:07.469317Z","shell.execute_reply.started":"2022-11-30T10:55:07.040356Z","shell.execute_reply":"2022-11-30T10:55:07.467773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Augmentations by albumentations","metadata":{}},{"cell_type":"code","source":"# get some random image indices from the training set\nrand_indices = [randrange(len(train_df)) for x in range(0,10)]\nrand_indices","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:07.471372Z","iopub.execute_input":"2022-11-30T10:55:07.472511Z","iopub.status.idle":"2022-11-30T10:55:07.480828Z","shell.execute_reply.started":"2022-11-30T10:55:07.472466Z","shell.execute_reply":"2022-11-30T10:55:07.479577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_IMG_PATH = \"/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs/train_images_processed/\"","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:07.632673Z","iopub.execute_input":"2022-11-30T10:55:07.633551Z","iopub.status.idle":"2022-11-30T10:55:07.638733Z","shell.execute_reply.started":"2022-11-30T10:55:07.633505Z","shell.execute_reply":"2022-11-30T10:55:07.637437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:08.029871Z","iopub.execute_input":"2022-11-30T10:55:08.030835Z","iopub.status.idle":"2022-11-30T10:55:08.063184Z","shell.execute_reply.started":"2022-11-30T10:55:08.030785Z","shell.execute_reply":"2022-11-30T10:55:08.061774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def view_aug_images(train, rand_indices, aug = None, title = ''):\n    width = 5\n    height = 2\n    counter = 0\n    fig, axs = plt.subplots(height, width, figsize=(15,5))\n    \n    for im in rand_indices:\n        image = cv2.imread(os.path.join(TRAIN_IMG_PATH,str(train.iloc[im].patient_id),str(train.iloc[im].image_id)+ '.png'))\n        if aug is not None:\n            image = aug(image=np.array(image))['image']\n        \n        i = counter // width\n        j = counter % width\n        axs[i,j].imshow(image, cmap=plt.cm.bone) #plot the data\n        axs[i,j].axis('off')\n        \n        diagnosis = train[train['image_id'] == train.iloc[im].image_id].cancer.values[0]\n        \n        axs[i,j].set_title(diagnosis)\n        counter += 1\n\n    plt.suptitle(title)\n    plt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-11-30T10:55:08.247835Z","iopub.execute_input":"2022-11-30T10:55:08.24914Z","iopub.status.idle":"2022-11-30T10:55:08.261316Z","shell.execute_reply.started":"2022-11-30T10:55:08.249078Z","shell.execute_reply":"2022-11-30T10:55:08.259641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_aug_images(train_df, rand_indices, title = 'Original images')","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:08.437029Z","iopub.execute_input":"2022-11-30T10:55:08.437504Z","iopub.status.idle":"2022-11-30T10:55:09.061715Z","shell.execute_reply.started":"2022-11-30T10:55:08.437456Z","shell.execute_reply":"2022-11-30T10:55:09.060644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aug = RandomBrightnessContrast(brightness_limit=1, contrast_limit=1, p = 1)\nview_aug_images(train_df, rand_indices, aug, title = 'RandomBrightnessContrast')","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:09.063435Z","iopub.execute_input":"2022-11-30T10:55:09.064052Z","iopub.status.idle":"2022-11-30T10:55:09.59628Z","shell.execute_reply.started":"2022-11-30T10:55:09.064015Z","shell.execute_reply":"2022-11-30T10:55:09.595023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aug = RandomGamma(gamma_limit=[180,200], p = 1)\nview_aug_images(train_df, rand_indices, aug, title = 'RandomGamma')","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:09.597995Z","iopub.execute_input":"2022-11-30T10:55:09.598417Z","iopub.status.idle":"2022-11-30T10:55:10.129075Z","shell.execute_reply.started":"2022-11-30T10:55:09.598379Z","shell.execute_reply":"2022-11-30T10:55:10.127563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aug = GridDistortion(num_steps =5, distort_limit=[-0.3,0.3], interpolation=1, border_mode= 4, p = 1)\nview_aug_images(train_df, rand_indices, aug, title = 'GridDistortion')","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:10.131467Z","iopub.execute_input":"2022-11-30T10:55:10.131911Z","iopub.status.idle":"2022-11-30T10:55:10.926936Z","shell.execute_reply.started":"2022-11-30T10:55:10.131863Z","shell.execute_reply":"2022-11-30T10:55:10.925358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aug = OpticalDistortion(shift_limit =[-0.5, 0.5], distort_limit=[-2,2], interpolation=1, border_mode= 4, p = 1)\nview_aug_images(train_df, rand_indices, aug, title = 'OpticalDistortion')","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:55:10.929415Z","iopub.execute_input":"2022-11-30T10:55:10.929947Z","iopub.status.idle":"2022-11-30T10:55:11.477117Z","shell.execute_reply.started":"2022-11-30T10:55:10.9299Z","shell.execute_reply":"2022-11-30T10:55:11.47559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Conclusion\n\n* The dataset is imbalanced.\n* Need to play with data augmentation\n\nbaseline model is ongoing ...","metadata":{}}]}