{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Importing the libraries","metadata":{}},{"cell_type":"code","source":"import sys\n!{sys.executable} -m pip install dicomsdl\n#!pip install /kaggle/input/rsnapacks/dicomsdl-0.109.1-cp36-cp36m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl\n","metadata":{"execution":{"iopub.status.busy":"2023-02-25T18:00:36.849277Z","iopub.execute_input":"2023-02-25T18:00:36.84995Z","iopub.status.idle":"2023-02-25T18:00:49.714564Z","shell.execute_reply.started":"2023-02-25T18:00:36.849888Z","shell.execute_reply":"2023-02-25T18:00:49.712835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport shutil\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pickle\n#import pydicom as dicom\nimport dicomsdl as dicom\nimport tensorflow as tf\nimport tensorflow_io as tfio\nfrom PIL import Image as im\nimport os\nfrom tqdm import tqdm\nimport cv2\nfrom PIL import Image\nimport joblib\nfrom joblib import Parallel, delayed\nimport gc\nfrom multiprocessing import cpu_count\nimport random\nimport scipy\nimport warnings\nfrom scipy.ndimage import morphology\nwarnings.filterwarnings('ignore')\nfrom multiprocessing import Pool, cpu_count\nfrom pympler import muppy\nfrom pympler import summary\n\n","metadata":{"execution":{"iopub.status.busy":"2023-02-25T18:00:49.71785Z","iopub.execute_input":"2023-02-25T18:00:49.718327Z","iopub.status.idle":"2023-02-25T18:00:58.636782Z","shell.execute_reply.started":"2023-02-25T18:00:49.718273Z","shell.execute_reply":"2023-02-25T18:00:58.635587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Defining the image processing pipeline","metadata":{}},{"cell_type":"code","source":"\n\ndef transform_to_hu(medical_image, image):\n    intercept = medical_image.RescaleIntercept\n    slope = medical_image.RescaleSlope\n    image = image * slope + intercept\n    \n    del intercept, slope\n    gc.collect()\n    \n    return image\n   \n\ndef window_image(image, window_center, window_width):\n    img_min = window_center - window_width // 2\n    img_max = window_center + window_width // 2\n    image = image.copy()\n    image[image < img_min] = img_min\n    image[image > img_max] = img_max\n    \n    del img_min, img_max\n    gc.collect()\n    \n    return image\n    \ndef mask(image):\n    seg = morphology.binary_dilation(image, np.ones((1, 1)))\n    labels, label_nb = scipy.ndimage.label(seg)\n    label_count = np.bincount(labels.ravel().astype(np.int))\n    label_count[0] = 0\n    mask = labels == label_count.argmax()\n    mask = morphology.binary_dilation(mask, np.ones((1, 1)))\n    mask = scipy.ndimage.morphology.binary_fill_holes(mask)\n    mask = morphology.binary_dilation(mask, np.ones((3, 3)))\n    image = np.multiply(mask, image)\n    \n    del seg, mask, labels, label_count, label_nb\n    gc.collect()\n    \n    return image\n    \n\ndef remove_noise(medical_image,image):\n    if isinstance(medical_image['WindowCenter'], list):\n        window_center = medical_image['WindowCenter'][0]\n    else:\n        window_center = medical_image['WindowCenter']   \n    if isinstance(medical_image['WindowWidth'], list):\n            window_width = medical_image['WindowWidth'][0]\n    else:\n        window_width = medical_image['WindowWidth']\n\n    \n    try:\n        intercept = medical_image.RescaleIntercept\n        slope = medical_image.RescaleSlope\n        image = image * slope + intercept\n\n        del intercept, slope\n        gc.collect()\n\n    except:\n        image = image\n        \n    try:   \n        img_min = window_center - window_width // 2\n        img_max = window_center + window_width // 2\n        image = image.copy()\n        image[image < img_min] = img_min\n        image[image > img_max] = img_max\n\n        del img_min, img_max\n        gc.collect()\n    except:\n        image=image\n\n    try:\n        seg = morphology.binary_dilation(image, np.ones((1, 1)))\n        labels, label_nb = scipy.ndimage.label(seg)\n        label_count = np.bincount(labels.ravel().astype(np.int))\n        label_count[0] = 0\n        mask = labels == label_count.argmax()\n        mask = morphology.binary_dilation(mask, np.ones((1, 1)))\n        mask = scipy.ndimage.morphology.binary_fill_holes(mask)\n        mask = morphology.binary_dilation(mask, np.ones((3, 3)))\n        image = np.multiply(mask, image)\n\n        del seg, mask, labels, label_count, label_nb\n        gc.collect()\n    except:\n        image=image\n    \n    \n    gc.collect()\n    return image\n    \n\n\ndef crop(image,window):\n    height,width=image.shape\n\n    image = image[0:int(height/window)*window,0:int(width/window)*window ]\n    height,width=image.shape\n    \n    imageX = image.sum(axis=0)/np.max(image.sum(axis=0))\n    #imageX[imageX<0.05]=0\n    imageX = imageX.reshape((int(len(imageX)/window), window))\n    \n\n    for i,j in enumerate(imageX):\n        \n        if (j!=np.zeros(window)).all():\n            index2=i\n            break\n    for i,j in enumerate(imageX):\n        if i>index2 and (j==np.zeros(window)).all():\n            index1=i\n            break\n        else:\n            index1=(width/window)-1\n    \n    imageY = image.sum(axis=1)/np.max(image.sum(axis=1))\n    imageY[imageY<0.1]=0\n    imageY = imageY.reshape((int(len(imageY)/window), window))\n\n    for i,j in enumerate(imageY):\n        if (j!=np.zeros(window)).all():\n            index4=i\n            break\n    for i,j in enumerate(imageY):\n        if i>index4 and (j==np.zeros(window)).all():\n            index3=i\n            break\n        else:\n            index3=(height/window)-1\n            \n    image = image[int((index4+1)*window):int((index3+1)*window),int((index2+1)*window):int((index1+1)*window)]\n    del imageX\n    del imageY\n    gc.collect()\n    return image\n    \n\n        \n\ndef process(filename,imageID):\n    \n    medical_image = dicom.open(filename)\n    image = medical_image.pixelData()\n    \n   \n    \n    #image=remove_noise(medical_image,image)\n    if isinstance(medical_image['WindowCenter'], list):\n        window_center = medical_image['WindowCenter'][0]\n    else:\n        window_center = medical_image['WindowCenter']   \n    if isinstance(medical_image['WindowWidth'], list):\n            window_width = medical_image['WindowWidth'][0]\n    else:\n        window_width = medical_image['WindowWidth']\n\n    \n    try:\n        intercept = medical_image.RescaleIntercept\n        slope = medical_image.RescaleSlope\n        image = image * slope + intercept\n\n        del intercept, slope\n        gc.collect()\n\n    except:\n        image = image\n        \n    try:   \n        img_min = window_center - window_width // 2\n        img_max = window_center + window_width // 2\n        image = image.copy()\n        image[image < img_min] = img_min\n        image[image > img_max] = img_max\n\n        del img_min, img_max\n        gc.collect()\n    except:\n        image=image\n\n    try:\n        seg = morphology.binary_dilation(image, np.ones((1, 1)))\n        labels, label_nb = scipy.ndimage.label(seg)\n        label_count = np.bincount(labels.ravel().astype(np.int))\n        label_count[0] = 0\n        mask = labels == label_count.argmax()\n        mask = morphology.binary_dilation(mask, np.ones((1, 1)))\n        mask = scipy.ndimage.morphology.binary_fill_holes(mask)\n        mask = morphology.binary_dilation(mask, np.ones((3, 3)))\n        image = np.multiply(mask, image)\n\n        del seg, mask, labels, label_count, label_nb\n        gc.collect()\n    except:\n        image=image\n    \n    \n    gc.collect()\n    \n    \n    \n    \n    if medical_image.PhotometricInterpretation == \"MONOCHROME1\":  \n        image = 1 - image\n\n    image = (image - image.min()) / (image.max() - image.min())\n    image = (image  * 255).astype(np.uint8)\n\n    image = crop(image,40)\n\n    image = cv2.resize(image, (512,512), interpolation=cv2.INTER_AREA)\n\n    gc.collect()\n    #CLAHE = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(32, 32))\n    #image = CLAHE.apply(image)\n    #image = cv2.equalizeHist(image)\n    \n    cv2.imwrite(f'./trainImages/{imageID}.jpg', image, [cv2.IMWRITE_JPEG_QUALITY, 95])\n    \n    del image\n    gc.collect()\n    #return image\n        \n#image = process(filename,imageID)\n\n\n        \n\n\n\n\n\n\n\n\n\ndef dcmToJpg(imgData, image_ids):\n    \n    \"\"\"\n    This function is used to process all the images which will be used for the training/test dataset. It takes two arguments.\n    \n    Input:-\n    :datasetPath = Path of the cancer dataset\n    :imgPath = Path of the image dataset\n   \n    Output:-\n    :return = Array of all the processed images\n    \n    \"\"\"\n    \n    Parallel(n_jobs=128)(\\\n    delayed(process)(fname, ids) for fname, ids in zip(imgData,tqdm(image_ids))\\\n    )\n    \n    #jobs = [joblib.delayed(process)(fname, ids) for fname, ids in zip(imgData,tqdm(image_ids))]\n    #SUBMISSION_ROWS = joblib.Parallel(\n    #    n_jobs=100,\n    #    verbose=True,\n    #    backend='multiprocessing',\n    #    prefer='threads',\n    #)(jobs)\n\n \n    ","metadata":{"execution":{"iopub.status.busy":"2023-02-25T18:00:58.6383Z","iopub.execute_input":"2023-02-25T18:00:58.63901Z","iopub.status.idle":"2023-02-25T18:00:58.668853Z","shell.execute_reply.started":"2023-02-25T18:00:58.63897Z","shell.execute_reply":"2023-02-25T18:00:58.666853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datasetPath = '/kaggle/input/rsna-breast-cancer-detection/train.csv'\nimgPath = '/kaggle/input/rsna-breast-cancer-detection/train_images/'\n\n\ndataset = pd.read_csv(datasetPath)\n    \npatient_ids = dataset['patient_id']\nimage_ids = dataset['image_id']\nsides  = dataset['laterality']\n\nimgData = []\nfor pi, ii, leng in zip(patient_ids, image_ids, range(len(patient_ids))):\n    imgData.append(imgPath + str(pi) + '/' + str(ii) + '.dcm')\n\ngc.collect()\n\nsizechunk=60000\nchunks = np.arange(0, len(imgData), sizechunk)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-25T18:00:58.670277Z","iopub.execute_input":"2023-02-25T18:00:58.67073Z","iopub.status.idle":"2023-02-25T18:00:59.0671Z","shell.execute_reply.started":"2023-02-25T18:00:58.670678Z","shell.execute_reply":"2023-02-25T18:00:59.066113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filenames_and_ids=[]\nfor i,j in zip(imgData,image_ids):\n    filenames_and_ids.append((i, j))","metadata":{"execution":{"iopub.status.busy":"2023-02-25T18:00:59.069499Z","iopub.execute_input":"2023-02-25T18:00:59.069849Z","iopub.status.idle":"2023-02-25T18:00:59.103581Z","shell.execute_reply.started":"2023-02-25T18:00:59.069816Z","shell.execute_reply":"2023-02-25T18:00:59.102624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nif os.path.isdir('/kaggle/working/trainImages/'):\n    shutil.rmtree('/kaggle/working/trainImages/')\nos.mkdir('/kaggle/working/trainImages/')\n","metadata":{"execution":{"iopub.status.busy":"2023-02-25T18:00:59.105123Z","iopub.execute_input":"2023-02-25T18:00:59.105994Z","iopub.status.idle":"2023-02-25T18:00:59.110295Z","shell.execute_reply.started":"2023-02-25T18:00:59.105949Z","shell.execute_reply":"2023-02-25T18:00:59.109497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n","metadata":{"execution":{"iopub.status.busy":"2023-02-25T18:00:59.111553Z","iopub.execute_input":"2023-02-25T18:00:59.111891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(chunks)):\n    print(i)\n    \n    #allObjects = muppy.get_objects()\n    #sums = summary.summarize(allObjects)\n    #summary.print_(sums)\n    \n    if chunks[i] == (len(imgData)//sizechunk)*sizechunk :\n        dcmToJpg(imgData[chunks[i]:], image_ids[chunks[i]:])\n    else:\n        dcmToJpg(imgData[chunks[i]:chunks[i+1]], image_ids[chunks[i]:chunks[i+1]])\n    gc.collect()\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}