{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"e73d786e-cb6f-4f11-9762-14259aa16cf6","_cell_guid":"66486459-91e5-435d-96a5-07b456dffebb","jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nimport cv2\n#dcm格式数据读取的包\nimport pydicom\nfrom joblib import Parallel, delayed\n\nJ = os.path.join\noriginal_dataset_root = \"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images\"\nout_dataset_root = os.path.join(\"/kaggle/working/\")\n\nN_JOBS = 2\n\n# Perform only the first iteration\nBREAK = True #  <- #!!Set to False if you want to process the entire dataset)!!\n\n# Resize images\nRESIZE = False\nSIZE = 512\n# Reduce Slice Tickness to 5 millimeters\nTICK = 5   # mm","metadata":{"_uuid":"e8b46b8e-0bdf-4049-999e-b936a5c6e0f1","_cell_guid":"ee788328-3277-42b1-903b-57a3f34d4706","execution":{"iopub.status.busy":"2023-09-12T02:54:36.585641Z","iopub.execute_input":"2023-09-12T02:54:36.586471Z","iopub.status.idle":"2023-09-12T02:54:37.512801Z","shell.execute_reply.started":"2023-09-12T02:54:36.586437Z","shell.execute_reply":"2023-09-12T02:54:37.511397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n# shutil.rmtree(\"/kaggle/working/reduced_512_tickness_5\")","metadata":{"_uuid":"11eeabda-9231-4db7-9359-e7904bf2cda5","_cell_guid":"26008272-abe8-45b1-a9a7-fa0780ea0b2d","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-09-12T02:54:37.515911Z","iopub.execute_input":"2023-09-12T02:54:37.516587Z","iopub.status.idle":"2023-09-12T02:54:37.522699Z","shell.execute_reply.started":"2023-09-12T02:54:37.516539Z","shell.execute_reply":"2023-09-12T02:54:37.520991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# os.remove(\"/kaggle/working/test.tar.gz\")","metadata":{"_uuid":"b4af8a8e-b25c-4a04-acff-05df7adf9500","_cell_guid":"892a8fb9-4312-4128-afe5-064e161e64c5","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-09-12T02:54:37.524347Z","iopub.execute_input":"2023-09-12T02:54:37.52486Z","iopub.status.idle":"2023-09-12T02:54:37.536083Z","shell.execute_reply.started":"2023-09-12T02:54:37.524814Z","shell.execute_reply":"2023-09-12T02:54:37.535116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get DICOM meta-data\ndf_dicom = pd.read_parquet(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_dicom_tags.parquet\")\n# Gest series folder\ndf_dicom[\"PatientID\"] = df_dicom[\"PatientID\"].astype(str)\ndf_dicom[\"serie\"] = df_dicom[\"SeriesInstanceUID\"].apply(lambda x: x.split(\".\")[-1])\ndf_dicom = df_dicom.set_index([\"PatientID\", \"serie\"]).sort_index()","metadata":{"_uuid":"7b1567be-7f4d-4ee5-b286-120dcbd26754","_cell_guid":"53420eba-52e8-48d1-8db9-b65ba00b90ae","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-09-12T02:54:37.537328Z","iopub.execute_input":"2023-09-12T02:54:37.537808Z","iopub.status.idle":"2023-09-12T02:54:48.651219Z","shell.execute_reply.started":"2023-09-12T02:54:37.53778Z","shell.execute_reply":"2023-09-12T02:54:48.650026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dicom_to_image(dicom_image):\n    \"\"\"\n    读dcm格式的图片\n    \"\"\"\n    pixel_array = dicom_image.pixel_array\n    \n    if dicom_image.PixelRepresentation == 1:\n        bit_shift = dicom_image.BitsAllocated - dicom_image.BitsStored\n        dtype = pixel_array.dtype \n        new_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n        pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dicom_image)\n    \n    if dicom_image.PhotometricInterpretation == \"MONOCHROME1\":\n        pixel_array = 1 - pixel_array\n    \n    # transform to hounsfield units\n    intercept = dicom_image.RescaleIntercept\n    slope = dicom_image.RescaleSlope\n    pixel_array = pixel_array * slope + intercept\n    \n    # windowing\n    window_center = int(dicom_image.WindowCenter)\n    window_width = int(dicom_image.WindowWidth)\n    img_min = window_center - window_width // 2\n    img_max = window_center + window_width // 2\n    pixel_array = pixel_array.copy()\n    pixel_array[pixel_array < img_min] = img_min\n    pixel_array[pixel_array > img_max] = img_max\n    \n    # normalization\n    pixel_array = (pixel_array - pixel_array.min())/(pixel_array.max() - pixel_array.min())\n    \n    return (pixel_array * 255).astype(np.uint8)","metadata":{"_uuid":"28e9bce6-e762-4d71-b9fa-8f8f02838ae2","_cell_guid":"73d27949-cdb0-4f9c-91c8-1afbe97f7d2e","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-09-12T02:54:48.653464Z","iopub.execute_input":"2023-09-12T02:54:48.653795Z","iopub.status.idle":"2023-09-12T02:54:48.668535Z","shell.execute_reply.started":"2023-09-12T02:54:48.653767Z","shell.execute_reply":"2023-09-12T02:54:48.666697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not RESIZE: SIZE=512\n\nout_folder = os.path.join(out_dataset_root, f\"reduced_{SIZE}_tickness_{TICK}\")\n\ndef process(dcm_path, p, s):\n    dicom_image = pydicom.read_file(dcm_path)\n    dcm_file = os.path.basename(dcm_path)\n    # read DICOM\n    img = dicom_to_image(dicom_image)\n\n    # Resize\n    if RESIZE:\n        img = cv2.resize(img, (SIZE, SIZE))\n\n    out_path = J(out_folder, p, s, dcm_file.replace(\".dcm\",\".jpeg\"))\n    cv2.imwrite(out_path, img)\n\n\n\np_ids = os.listdir(original_dataset_root)","metadata":{"_uuid":"f69445cc-9771-4ee5-866f-8375a7d23276","_cell_guid":"85a5572a-4e0b-4622-aa5e-a575a8ab26ab","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-09-12T02:54:48.67049Z","iopub.execute_input":"2023-09-12T02:54:48.671072Z","iopub.status.idle":"2023-09-12T02:54:48.842573Z","shell.execute_reply.started":"2023-09-12T02:54:48.671022Z","shell.execute_reply":"2023-09-12T02:54:48.841642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for p in tqdm(p_ids[0:600]):\n    p_folder = J(original_dataset_root, p)\n    s_ids = os.listdir(p_folder)\n    for s in s_ids:\n        s_folder = J(p_folder, s)\n        os.makedirs(J(out_folder, p, s), exist_ok=True)\n\n        # sort dcm files\n        dcms = os.listdir(s_folder)\n\n        dcms_d = {int(x.replace(\".dcm\",\"\")): x for x in dcms}\n        dcms = sorted(dcms_d)\n        dcms = [str(x)+\".dcm\" for x in dcms]\n        \n        n_d = len(dcms)\n        \n        curr_tick = df_dicom.loc[(p, s), \"SliceThickness\"].mean()\n        step = round(TICK/curr_tick)\n\n        dcm_paths = []\n        for i in range(0, n_d, step):\n            dcm_paths.append(J(s_folder, dcms[i]))\n        \n        _ = Parallel(n_jobs=N_JOBS)(\n            delayed(process)(path, p, s) for path in dcm_paths\n        )\n\n\n    #if BREAK: break","metadata":{"_uuid":"4c469933-1cbd-4e8f-9032-27297fa046e7","_cell_guid":"54bb7dc9-3473-4069-b966-54657f9265d2","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-09-12T02:54:55.609415Z","iopub.execute_input":"2023-09-12T02:54:55.610566Z","iopub.status.idle":"2023-09-12T02:55:23.413098Z","shell.execute_reply.started":"2023-09-12T02:54:55.610525Z","shell.execute_reply":"2023-09-12T02:55:23.412175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#os.system('zip -rm test.zip /kaggle/working/')\n# os.system('tar czf test.tar.gz /kaggle/working/')","metadata":{"execution":{"iopub.status.busy":"2023-09-08T14:11:32.635348Z","iopub.execute_input":"2023-09-08T14:11:32.635803Z","iopub.status.idle":"2023-09-08T14:20:44.395721Z","shell.execute_reply.started":"2023-09-08T14:11:32.635771Z","shell.execute_reply":"2023-09-08T14:20:44.394556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n# import zipfile\n# import datetime\n\n# def file2zip(packagePath, zipPath):\n#     '''\n#   :param packagePath: 文件夹路径\n#   :param zipPath: 压缩包路径\n#   :return:\n#   '''\n#     zip = zipfile.ZipFile(zipPath, 'w', zipfile.ZIP_DEFLATED)\n#     for path, dirNames, fileNames in os.walk(packagePath):\n#         fpath = path.replace(packagePath, '')\n#         for name in fileNames:\n#             fullName = os.path.join(path, name)\n#             name = fpath + '\\\\' + name\n#             zip.write(fullName, name)\n#     zip.close()\n\n\n# if __name__ == \"__main__\":\n#     # 文件夹路径\n#     packagePath = '/kaggle/working/'#需要压缩的文件路径\n#     zipPath = '/kaggle/working/output.zip'#保存压缩包的路径\n#     if os.path.exists(zipPath):\n#         os.remove(zipPath)\n#     file2zip(packagePath, zipPath)\n#     print(\"打包完成\")\n#     print(datetime.datetime.utcnow())","metadata":{"_uuid":"257f2252-eb14-4283-a7ae-e99b5a8c63a4","_cell_guid":"ab7b8740-d5f1-4002-bb67-3c0763e503e4","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n# #切换工作路径\n# os.chdir('/kaggle/working')#改变工作路径至‘’\n# print(os.getcwd())#查看当前工作路径是否切换\n\n# #选择需要下载的文件名\n# print(os.listdir(\"/kaggle/working\"))#列出‘’路径中的文件夹的名字，这里要与工作路径一致\n# from IPython.display import FileLink\n# FileLink('output.zip')#‘’中选择os.listdir()列出的文件中的一个","metadata":{"_uuid":"15e046f2-09e4-4e50-ba42-80a8188bfd16","_cell_guid":"8375c0ed-7201-46e9-8403-fb5677126def","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"9d708e18-bc0a-4ee3-a0b3-d8fb93762f7c","_cell_guid":"9b58e579-323e-4d92-8e7e-0565ffb37c1c","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}