{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pydicom\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\nimport os\nfrom os import path","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-04T04:20:02.56821Z","iopub.execute_input":"2023-10-04T04:20:02.568733Z","iopub.status.idle":"2023-10-04T04:20:02.577409Z","shell.execute_reply.started":"2023-10-04T04:20:02.568687Z","shell.execute_reply":"2023-10-04T04:20:02.575686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_IMAGES_PATH = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images'\nTEST_IMAGES_PATH = '/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images'","metadata":{"execution":{"iopub.status.busy":"2023-10-04T04:20:02.580234Z","iopub.execute_input":"2023-10-04T04:20:02.580866Z","iopub.status.idle":"2023-10-04T04:20:02.595602Z","shell.execute_reply.started":"2023-10-04T04:20:02.580829Z","shell.execute_reply":"2023-10-04T04:20:02.594637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"No Personal identifiable information was left in the DICOM metadata :+1:\n\nIdea for Data Augmentation: Patient Position (Categorical variable), HFS, FFS etc\n\nDo more false scans show up when scanning at a particular time of day?\n\nWere more true positives found with a particular patient position?\nHow does image orientation affect the model? Perhaps rotate it back to the original position?\n\nAnnotate the dataset with slice thickness","metadata":{}},{"cell_type":"markdown","source":"N.B:  Patient Position (0018,5100) and Image Orientation (Patient)\n(0020,0037) are not related together: the Image Orientation Patient is\nrelated to the patient body, regardless how she's placed in the\nmachine, while Patient Position specifies the position of the patient\nrelative to the imaging equipment (let me say relative to the gantry).\nThe machine manufacturer has to know the Patient Position (that is,\nthe way the patient has been put into the machine) to be able to\ncalculate the Image Orientation (Patient) from the image orientation\ncosines in the frame of reference of the machine.\n\nAfter that, you do not have to take care anymore of the Patient\nPosition, it is just an annotation for understanding the way the exam\nwas done, not to find the orientation of the image you are looking at.","metadata":{}},{"cell_type":"code","source":"from io import StringIO\nimport csv\n\ndef extract_dicom_metadata(path) -> dict:\n    dcm = pydicom.read_file(path)\n    meta = {}\n    for i, itm in enumerate(dcm):\n        if itm.name == \"Pixel Data\":\n            pass\n        else:\n            meta[itm.name.lower()] = itm.value\n    return meta\n\ndef process_directory(CSV_HEADERS, output_dir, directory_path):\n    def _update_csv_headers(global_headers, row_headers):\n        for header in row_headers:\n            if header not in global_headers:\n                global_headers.append(header)\n                \n    print(f\"Processing: {directory_path}\\n\")\n    patient_id = directory_path.stem\n    csv_data = StringIO()\n    csv_writer = csv.writer(csv_data)\n    \n    for series in directory_path.iterdir():    \n        images = list(sorted(series.iterdir(), key=lambda f: int(\"\".join([c for c in f.name if c.isdigit()]))))\n        for image_index, image_path in enumerate(images):\n            metadata = extract_dicom_metadata(image_path)\n            metadata.update({\"_patient_id\": patient_id, \"_series_id\":series.stem, \"_scan_id\":image_path.stem, \"_image_index\": image_index})\n            \n            _update_csv_headers(CSV_HEADERS, metadata.keys())\n            csv_writer.writerow(\n                [metadata.get(header, None) for header in CSV_HEADERS]        \n            )\n    \n        with open(f'{output_dir}/{patient_id}.csv', 'w') as f:\n            f.write(csv_data.getvalue())","metadata":{"execution":{"iopub.status.busy":"2023-10-04T04:20:02.598143Z","iopub.execute_input":"2023-10-04T04:20:02.598601Z","iopub.status.idle":"2023-10-04T04:20:02.616171Z","shell.execute_reply.started":"2023-10-04T04:20:02.59853Z","shell.execute_reply":"2023-10-04T04:20:02.614753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def combine_split_csv(split_csv_dir, outpath, headers=\"\"):\n    with open(outpath, \"w\") as fout:\n        fout.write(headers + \"\\n\")\n        for i, split_csv in enumerate(Path(split_csv_dir).iterdir()):\n            with open(split_csv, \"r\") as fsplit:\n                fout.write(fsplit.read())","metadata":{"execution":{"iopub.status.busy":"2023-10-04T04:20:02.620192Z","iopub.execute_input":"2023-10-04T04:20:02.621869Z","iopub.status.idle":"2023-10-04T04:20:02.644775Z","shell.execute_reply.started":"2023-10-04T04:20:02.621745Z","shell.execute_reply":"2023-10-04T04:20:02.643179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import multiprocessing as mp\nfrom multiprocessing import Process, Manager\nfrom functools import partial\n\nmanager = Manager()\nCSV_HEADERS = manager.list([])\n\nCSV_TRAIN_TEMP_DIR = 'train_split_csv'\n!rm -rf {CSV_TRAIN_TEMP_DIR}\n!mkdir -p {CSV_TRAIN_TEMP_DIR}\n\nCSV_TEST_TEMP_DIR = 'test_split_csv'\n!rm -rf {CSV_TEST_TEMP_DIR}\n!mkdir -p {CSV_TEST_TEMP_DIR}\n\nCSV_TRAIN_PATH = 'train_images_dicom_meta.csv'\nCSV_TEST_PATH = 'test_images_dicom_meta.csv'\n\n# Train\ntrain_directories = list(Path(TRAIN_IMAGES_PATH).iterdir())\n# train_directories = train_directories[:5]\nwith mp.Pool(4) as p:\n    p.map(partial(process_directory, CSV_HEADERS, CSV_TRAIN_TEMP_DIR), train_directories)\n\n# Test\ntest_directories = list(Path(TEST_IMAGES_PATH).iterdir())\n# test_directories = test_directories[:5]\nwith mp.Pool(64) as p:\n    p.map(partial(process_directory, CSV_HEADERS, CSV_TEST_TEMP_DIR), test_directories)    ","metadata":{"execution":{"iopub.status.busy":"2023-10-04T04:20:02.647377Z","iopub.execute_input":"2023-10-04T04:20:02.647792Z","iopub.status.idle":"2023-10-04T04:20:34.535942Z","shell.execute_reply.started":"2023-10-04T04:20:02.647756Z","shell.execute_reply":"2023-10-04T04:20:34.533243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combine_split_csv(CSV_TRAIN_TEMP_DIR, CSV_TRAIN_PATH, \",\".join(CSV_HEADERS))\ncombine_split_csv(CSV_TEST_TEMP_DIR, CSV_TEST_PATH, \",\".join(CSV_HEADERS))","metadata":{"execution":{"iopub.status.busy":"2023-10-04T04:20:34.539303Z","iopub.execute_input":"2023-10-04T04:20:34.539841Z","iopub.status.idle":"2023-10-04T04:20:34.56759Z","shell.execute_reply.started":"2023-10-04T04:20:34.539789Z","shell.execute_reply":"2023-10-04T04:20:34.566037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}