{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -qU python-gdcm pydicom pylibjpeg","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-13T13:24:30.087991Z","iopub.execute_input":"2023-01-13T13:24:30.088411Z","iopub.status.idle":"2023-01-13T13:24:41.263228Z","shell.execute_reply.started":"2023-01-13T13:24:30.088377Z","shell.execute_reply":"2023-01-13T13:24:41.262129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport gdcm\nimport pylibjpeg\nimport matplotlib.pyplot as plt\nfrom matplotlib.patches import Rectangle\nimport seaborn as sns\n\nimport os\nfrom os import listdir\n\nfrom scipy.stats import skew\nimport glob\nfrom tqdm.notebook import tqdm\nfrom joblib import Parallel, delayed\nimport cv2","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:24:41.26487Z","iopub.execute_input":"2023-01-13T13:24:41.265235Z","iopub.status.idle":"2023-01-13T13:24:41.273568Z","shell.execute_reply.started":"2023-01-13T13:24:41.265199Z","shell.execute_reply":"2023-01-13T13:24:41.271447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Custom colors for charts\n\nc_0 = np.array([2, 48, 71,256])/256\nc_1 = np.array([251, 133, 0,256])/256\nc_2 = np.array([255, 183, 3,256])/256\nc_3 = np.array([33, 158, 188,256])/256\nc_4 = np.array([142, 202, 230,256])/256\n\n# Creating a custom colormap\n# The code comes from https://matplotlib.org/3.1.1/tutorials/colors/colormap-manipulation.html\nfrom matplotlib.colors import ListedColormap\n\nN = 256\nvals = np.ones((N, 4))\nvals[:, 0] = np.linspace(c_0[0], c_1[0], N)\nvals[:, 1] = np.linspace(c_0[1], c_1[1], N)\nvals[:, 2] = np.linspace(c_0[2], c_1[2], N)\ncustom_cmp1 = ListedColormap(vals)\n\n# A colormap with white in the middle\nN = 256\nvals = np.ones((N, 4))\nvals[:, 0] = np.concatenate((np.linspace(c_0[0], 1, 128), np.linspace(1, c_1[0], 128)), axis = None)\nvals[:, 1] = np.concatenate((np.linspace(c_0[1], 1, 128), np.linspace(1, c_1[1], 128)), axis = None)\nvals[:, 2] = np.concatenate((np.linspace(c_0[2], 1, 128), np.linspace(1, c_1[2], 128)), axis = None)\ncustom_cmp2 = ListedColormap(vals)\n\n# Allow value-dependant colorbar\n\ndef color_bar(y, cmap=custom_cmp1):\n    \n    # Create the matrix for colors\n    color_list = []\n    # Normalise the values\n    y = (y-min(y))/(max(y)-min(y))\n    \n    for i in range(len(y)):\n        color_list.append(cmap(int(N*y[i])))\n    \n    return color_list","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:24:41.274773Z","iopub.execute_input":"2023-01-13T13:24:41.275076Z","iopub.status.idle":"2023-01-13T13:24:41.299624Z","shell.execute_reply.started":"2023-01-13T13:24:41.275046Z","shell.execute_reply":"2023-01-13T13:24:41.298637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# RSNA Breast Cancer Detection Project (Part 2)","metadata":{}},{"cell_type":"markdown","source":"As in the previous Notebook, the input data (metadata and images) are imported.","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ndf_test = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n\ntrain_path = '/kaggle/input/rsna-breast-cancer-detection/train_images/'\ntest_path = '/kaggle/input/rsna-breast-cancer-detection/test_images/'","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:24:41.302324Z","iopub.execute_input":"2023-01-13T13:24:41.302643Z","iopub.status.idle":"2023-01-13T13:24:41.384165Z","shell.execute_reply.started":"2023-01-13T13:24:41.302606Z","shell.execute_reply":"2023-01-13T13:24:41.383278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The function created in the previous notebook for importing the scans as png files is recreated here. The possibility to import HD files (size 1024 along the long axis avoiding distorsion) have been added as a parameter.","metadata":{}},{"cell_type":"code","source":"def import_png(patient_id, HD = False):\n    if HD:\n        train_png_path = '/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_vl_asp_1024/train_images_processed_cv2_vl_asp_1024/'\n    else:\n        train_png_path = '/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_vl_512/train_images_processed_cv2_vl_512/'\n    path = train_png_path + str(patient_id)\n    slice_path_list = [path + '/' + img_fn for img_fn in os.listdir(path)]\n    slices = [cv2.imread(slice_path, cv2.IMREAD_GRAYSCALE) for slice_path in slice_path_list]\n    slices_names = os.listdir(path)\n    \n    return slices, slices_names","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:24:41.385707Z","iopub.execute_input":"2023-01-13T13:24:41.38642Z","iopub.status.idle":"2023-01-13T13:24:41.394445Z","shell.execute_reply.started":"2023-01-13T13:24:41.386378Z","shell.execute_reply":"2023-01-13T13:24:41.393397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Data cleaning and preprocessing","metadata":{}},{"cell_type":"markdown","source":"## 1.1. Image Cropping","metadata":{}},{"cell_type":"code","source":"df_patient_id = df_train['patient_id'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:24:41.395529Z","iopub.execute_input":"2023-01-13T13:24:41.395819Z","iopub.status.idle":"2023-01-13T13:24:41.405366Z","shell.execute_reply.started":"2023-01-13T13:24:41.395791Z","shell.execute_reply":"2023-01-13T13:24:41.404455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_bondaries(data, threshold = 0.008):\n    '''\n    Return the lower and upper bondaries as a list\n    [lower, upper]\n    given a threshold (default = 0.8%)\n    '''\n    s_data = np.cumsum(data)/data.sum()\n    lower = len(s_data[s_data<= threshold])\n    upper = len(s_data[s_data<=(1- threshold)])\n    \n    return [lower, upper]","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:24:41.40668Z","iopub.execute_input":"2023-01-13T13:24:41.406967Z","iopub.status.idle":"2023-01-13T13:24:41.413166Z","shell.execute_reply.started":"2023-01-13T13:24:41.406939Z","shell.execute_reply":"2023-01-13T13:24:41.412261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(3,4, figsize = (20, 20))\n\nfor i, patient_id in enumerate(np.random.choice(df_patient_id, 3, replace = False)):\n    img = import_png(patient_id, HD=True)[0][0]\n    \n    axes[i, 0].imshow(img)\n    axes[i, 1].imshow(img)\n    lb_x, ub_x = get_bondaries(img.sum(axis = 0))\n    lb_y, ub_y = get_bondaries(img.sum(axis = 1), threshold = 0.01)\n    \n    axes[i, 1].add_patch(Rectangle((lb_x, lb_y),\n                                    ub_x - lb_x,\n                                    ub_y - lb_y,\n                                    linewidth=1,\n                                    edgecolor='r',\n                                    facecolor='none'))\n    \n    axes[i, 2].bar(height = img.sum(axis = 0),\n                   x = range(img.sum(axis = 0).shape[0]),\n                   color = color_bar(img.sum(axis = 0)),\n                   width = 1)\n    for bound in get_bondaries(img.sum(axis = 0)):\n        axes[i, 2].vlines(bound, 0, img.sum(axis = 0).max(), color ='r')\n    \n    axes[i, 3].barh(width = img.sum(axis = 1),\n                    y = range(img.sum(axis = 1).shape[0]),\n                    color = color_bar(img.sum(axis = 1)),\n                    height = 1)\n    for bound in get_bondaries(img.sum(axis = 1), threshold = 0.01):\n        axes[i, 3].hlines(bound, 0, img.sum(axis = 1).max(), color = 'r')\n    \n    axes[i, 3].invert_yaxis()","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:27:54.890337Z","iopub.execute_input":"2023-01-13T13:27:54.890766Z","iopub.status.idle":"2023-01-13T13:28:08.518163Z","shell.execute_reply.started":"2023-01-13T13:27:54.890728Z","shell.execute_reply":"2023-01-13T13:28:08.517058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A function for cropping the image is created","metadata":{}},{"cell_type":"code","source":"def img_crop(img):\n    lb_x, ub_x = get_bondaries(img.sum(axis = 0))\n    lb_y, ub_y = get_bondaries(img.sum(axis = 1), threshold = 0.01)\n    return img[lb_y:ub_y,lb_x:ub_x]","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:24:54.007791Z","iopub.execute_input":"2023-01-13T13:24:54.008375Z","iopub.status.idle":"2023-01-13T13:24:54.013877Z","shell.execute_reply.started":"2023-01-13T13:24:54.008342Z","shell.execute_reply":"2023-01-13T13:24:54.012961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.loc[df_train['patient_id'] == 10130, 'cancer'].sum()","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:24:54.018464Z","iopub.execute_input":"2023-01-13T13:24:54.018805Z","iopub.status.idle":"2023-01-13T13:24:54.029854Z","shell.execute_reply.started":"2023-01-13T13:24:54.018775Z","shell.execute_reply":"2023-01-13T13:24:54.028733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(patient_id):\n    save_path = (f\"/kaggle/working/cropped_dataset/\" + str(patient_id))\n    \n    if not os.path.isdir(save_path):\n        os.mkdir(save_path)\n        \n    img_list, name_list = import_png(patient_id, HD = True)\n    \n    for index, img in enumerate(img_list):\n        img = img_crop(img)\n        cv2.imwrite(save_path + '/' + str(name_list[index]), img)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:24:54.031315Z","iopub.execute_input":"2023-01-13T13:24:54.031712Z","iopub.status.idle":"2023-01-13T13:24:54.039483Z","shell.execute_reply.started":"2023-01-13T13:24:54.031624Z","shell.execute_reply":"2023-01-13T13:24:54.038522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not os.path.isdir('/kaggle/working/cropped_dataset'):\n    os.mkdir('/kaggle/working/cropped_dataset')\n\n    _ = Parallel(n_jobs=4)(\n    delayed(process)(uid)\n    for uid in tqdm(df_train['patient_id'][:5].unique())\n    )","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:24:54.04099Z","iopub.execute_input":"2023-01-13T13:24:54.041639Z","iopub.status.idle":"2023-01-13T13:24:54.049484Z","shell.execute_reply.started":"2023-01-13T13:24:54.041598Z","shell.execute_reply":"2023-01-13T13:24:54.04845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def import_cropped_png(patient_id, image_id):\n    train_png_path = '/kaggle/input/rsnamammographycroppedpng1024/cropped_dataset/'\n    path = train_png_path + str(patient_id) + '_' + str(image_id) + '.png'\n    print(path)\n    img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n    \n    return img","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:24:54.050772Z","iopub.execute_input":"2023-01-13T13:24:54.051076Z","iopub.status.idle":"2023-01-13T13:24:54.062542Z","shell.execute_reply.started":"2023-01-13T13:24:54.051047Z","shell.execute_reply":"2023-01-13T13:24:54.061722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.2 Image cleaning","metadata":{}},{"cell_type":"code","source":"# mean = []\n# std = []\n\n# for patient_id in df_train['patient_id'].unique()[:100]:\n#     for image_id in df_train.loc[df_train['patient_id']==patient_id, 'image_id']:\n#         img = import_cropped_png(patient_id, image_id)\n#         print(img)\n#         mean.append(np.mean(img))\n#         median.append(np.median(img))\n#         skew.append(skew(img))\n#         std.append(np.std(img))\n        ","metadata":{"execution":{"iopub.status.busy":"2023-01-13T13:24:54.063652Z","iopub.execute_input":"2023-01-13T13:24:54.064067Z","iopub.status.idle":"2023-01-13T13:24:54.074039Z","shell.execute_reply.started":"2023-01-13T13:24:54.064014Z","shell.execute_reply":"2023-01-13T13:24:54.072915Z"},"trusted":true},"execution_count":null,"outputs":[]}]}