{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -qU python-gdcm pydicom pylibjpeg # For better performance in turning dicom to png files","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport pydicom\nfrom matplotlib import pyplot as plt\nfrom tqdm.notebook import tqdm\nimport cv2\nimport glob\nfrom joblib import Parallel, delayed\nimport os,time\nimport gdcm","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Before run this code add rsna-breast-cancer dataset\ndata_set_address = '/kaggle/input/rsna-2023-abdominal-trauma-detection/'\npatent_ids = os.listdir('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images')[2500:]\nprint(patent_ids)\ntrain_images = {}\nnumber_of_train_images = 0\nfor patent_id in patent_ids:\n    series_list = os.listdir(f'{data_set_address}train_images/{patent_id}/')\n    train_images[patent_id] = {}\n    for series in series_list:\n        images = os.listdir(f'{data_set_address}train_images/{patent_id}/{series}')\n        images = [imagename.replace('.dcm', '') for imagename in images]\n        images = list(map(int, images))\n        images = sorted(images)\n        image_paths = []\n        for imagename in images:\n            image_path = os.path.join(f'{data_set_address}train_images/{patent_id}/{series}', str(imagename) + \".dcm\")\n            image_paths.append(image_path)\n        train_images[patent_id][series] = image_paths\n        number_of_train_images += len(image_paths)\n# Use glob to work easier on files in directories\nprint('Number of train_images is :',number_of_train_images) # xepceted:54706\n\n# define a fucntion to read dicom file and return it's pixel\ndef read_dicom(dicom_file):\n    dcm = pydicom.dcmread(dicom_file) # read dciom files\n\n    pixel_array = dcm.pixel_array\n    \n    if dcm.PixelRepresentation == 1:\n        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n        dtype = pixel_array.dtype \n        pixel_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n#         pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dcm)\n\n    intercept = float(dcm.RescaleIntercept)\n    slope = float(dcm.RescaleSlope)\n    center = int(dcm.WindowCenter)\n    width = int(dcm.WindowWidth)\n    low = center - width / 2\n    high = center + width / 2    \n    \n    pixel_array = (pixel_array * slope) + intercept\n    pixel_array = np.clip(pixel_array, low, high)\n    pixel_array = (pixel_array - pixel_array.min()) / (pixel_array.max() - pixel_array.min() + 1e-6)\n\n    if dcm.PhotometricInterpretation == \"MONOCHROME1\":\n        pixel_array = 1 - pixel_array\n\n    return (pixel_array * 255).astype(np.uint8)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_dicom(dicom_file):\n    \n    result= cv2.connectedComponentsWithStats((dicom_file > 0.005).astype(np.uint8)[:, :], 8, cv2.CV_32S)\n    # result has 4 diffrenet outputs. You can see different outputs here.\n    # But for reconizing the black color(0) from others(1), we just need result[2] , called stat matrix\n    stat_matrix = result[2] # include left, top, width, height, area_size columns\n    # the second row shows the boxes of pixels which are different from background.Also, we don't need are_size cloumn. So:\n    second_row = stat_matrix[1:, 4].argmax() + 1\n    x1, y1, w, h = stat_matrix[second_row][:4]\n    x2 = x1 + w\n    y2 = y1 + h\n    cropped_dicom = dicom_file[y1: y2, x1: x2]\n    return cropped_dicom","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAVE_FOLDER = \"/kaggle/working/images/\"\nos.makedirs(SAVE_FOLDER, exist_ok=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dcm_to_png(images, patient_id, series, i, step, number_of_train_images, save_folder, size=384, extension ='png'):\n    flag = True\n    for j in range(3*step):\n        if i == number_of_train_images - (j+1):\n            if j < step:\n                delta_1 = 0\n            else:\n                delta_1 = step\n            \n            if j < step:\n                delta_2 = 0\n                delta_3 = 0\n            elif j < 2*step:\n                delta_2 = step\n                delta_3 = step\n            else:\n                delta_2 = step*2\n                delta_3 = step*2\n            \n            file_0 = images[i]\n            file_1 = images[i+delta_1]\n            file_2 = images[i+delta_2]\n            file_3 = images[i+delta_3]\n            flag = False\n    if flag == True:\n        file_0 = images[i]\n        file_1 = images[i+step]\n        file_2 = images[i+step*2]\n        file_3 = images[i+step*3]\n        \n    image_id_0 = file_0.split('/')[-1][:-4]\n    image_id_1 = file_1.split('/')[-1][:-4]\n    image_id_2 = file_2.split('/')[-1][:-4]\n    image_id_3 = file_3.split('/')[-1][:-4]\n    dicom_pixel_0 = read_dicom(dicom_file=file_0) # read dicom image pixels\n    dicom_pixel_1 = read_dicom(dicom_file=file_1)\n    dicom_pixel_2 = read_dicom(dicom_file=file_2)\n    dicom_pixel_3 = read_dicom(dicom_file=file_3)\n    \n#     cropped_dicom = crop_dicom(dicom_file=dicom_pixel) # crop the dicom image\n    resized_img_0 = cv2.resize(np.expand_dims(dicom_pixel_0, 2),dsize=(size,size)) # resize it to specific size\n    resized_img_1 = cv2.resize(np.expand_dims(dicom_pixel_1, 2),dsize=(size,size)) # resize it to specific size\n    resized_img_2 = cv2.resize(np.expand_dims(dicom_pixel_2, 2),dsize=(size,size)) # resize it to specific size\n    resized_img_3 = cv2.resize(np.expand_dims(dicom_pixel_3, 2),dsize=(size,size)) # resize it to specific size\n    resized_img_25d = np.stack([resized_img_0, resized_img_1, resized_img_2, resized_img_3], axis=2)\n    \n    cv2.imwrite(save_folder + f\"{patient_id}/{series}/{image_id_0}_{image_id_1}_{image_id_2}_{image_id_3}.{extension}\", resized_img_25d.astype(np.uint8))\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use parallel computing to spend less time on turning dcm to png file\n# if you want to test just some files, please use train_images[:10] instead of train_images, including the whole data\nN_EVAL = 28\nstart = time.time()\nfor patient_id in train_images.keys():\n    os.makedirs(SAVE_FOLDER +f\"{patient_id}/\", exist_ok=True)\n    for series in train_images[patient_id].keys():\n        os.makedirs(SAVE_FOLDER +f\"{patient_id}/{series}/\", exist_ok=True)\n        flag = True\n        if len(train_images[patient_id][series]) < N_EVAL*(1+1)*4*2:\n            _ = Parallel(n_jobs=4)(\n                delayed(dcm_to_png)(train_images[patient_id][series], patient_id, series, i, 1, len(train_images[patient_id][series]), SAVE_FOLDER, size=384)\n                for i in range(0, len(train_images[patient_id][series]), 4)  # you can use train_images[:100] for testing\n            )\n            flag = False\n        for step in [2, 3, 4]:\n            if len(train_images[patient_id][series]) >= N_EVAL*step*4*2 and len(train_images[patient_id][series]) < N_EVAL*(step+1)*4*2:\n                _ = Parallel(n_jobs=4)(\n                    delayed(dcm_to_png)(train_images[patient_id][series], patient_id, series, i, step, len(train_images[patient_id][series]), SAVE_FOLDER, size=384)\n                    for i in range(0, len(train_images[patient_id][series]), 4*step)  # you can use train_images[:100] for testing\n                )\n                flag = False\n\n        if flag == True:\n            _ = Parallel(n_jobs=4)(\n                delayed(dcm_to_png)(train_images[patient_id][series], patient_id, series, i, 4, len(train_images[patient_id][series]), SAVE_FOLDER, size=384)\n                for i in range(0, len(train_images[patient_id][series]), 4*4)  # you can use train_images[:100] for testing\n            )\nstop = time.time()\nprint('This process took you:',stop-start)\nprint('Now, you can have access to cropped files wtih png format')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport zipfile\n    \ndef zipdir(path, ziph):\n    # ziph is zipfile handle\n    for root, dirs, files in os.walk(path):\n        for file in files:\n            ziph.write(os.path.join(root, file), \n                       os.path.relpath(os.path.join(root, file), \n                                       os.path.join(path, '..')))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with zipfile.ZipFile('train.zip', 'w', zipfile.ZIP_DEFLATED) as zipf:\n    zipdir(SAVE_FOLDER, zipf)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf /kaggle/working/images/","metadata":{},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}