{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":52254,"databundleVersionId":8756537,"sourceType":"competition"}],"dockerImageVersionId":30747,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-16T14:52:51.139374Z","iopub.execute_input":"2024-07-16T14:52:51.139802Z","iopub.status.idle":"2024-07-16T14:52:51.590367Z","shell.execute_reply.started":"2024-07-16T14:52:51.13977Z","shell.execute_reply":"2024-07-16T14:52:51.589095Z"},"trusted":true},"outputs":[],"execution_count":2},{"cell_type":"code","source":"path='/kaggle/input/rsna-2023-abdominal-trauma-detection'\n# List all files in the directory\nprint(\"Files in the directory:\\n\")\nfor f in os.listdir(path):\n    print(f)","metadata":{"execution":{"iopub.status.busy":"2024-07-16T14:53:54.681308Z","iopub.execute_input":"2024-07-16T14:53:54.68182Z","iopub.status.idle":"2024-07-16T14:53:54.699173Z","shell.execute_reply.started":"2024-07-16T14:53:54.681747Z","shell.execute_reply":"2024-07-16T14:53:54.697752Z"},"trusted":true},"outputs":[{"name":"stdout","text":"Files in the directory:\n\ntrain_dicom_tags.parquet\nsample_submission.csv\ntrain_series_meta.csv\ntrain_images\ntest_dicom_tags.parquet\nsegmentations\ntrain_2024.csv\ntrain_demographics_2024.csv\nimage_level_labels_2024.csv\ndeprecated_files\ntest_series_meta.csv\ntest_images\n","output_type":"stream"}],"execution_count":4},{"cell_type":"code","source":"import os\nimport zipfile\nimport shutil\nimport pydicom\nimport numpy as np\nimport cv2\n\n# Define paths\nsegmentation_dir = '/kaggle/input/rsna-2023-abdominal-trauma-detection/segmentations'\ntrain_images_dir = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images'\noutput_dir = '/kaggle/working'\nfiltered_train_dir = '/kaggle/working/filtered_train_images'","metadata":{"execution":{"iopub.status.busy":"2024-07-16T19:00:46.022992Z","iopub.execute_input":"2024-07-16T19:00:46.02341Z","iopub.status.idle":"2024-07-16T19:00:46.029389Z","shell.execute_reply.started":"2024-07-16T19:00:46.023378Z","shell.execute_reply":"2024-07-16T19:00:46.028171Z"},"trusted":true},"outputs":[],"execution_count":6},{"cell_type":"code","source":"\ndef convert_dcm_to_grayscale_png(src_path, dest_path, patient_id, series_id, image_id):\n    try:\n        # Read DICOM file\n        img = pydicom.dcmread(src_path)\n        \n        # Check Photometric Interpretation (PIP)\n        pip = img.PhotometricInterpretation\n        \n        # Extract pixel data\n        data = img.pixel_array\n        \n        # Normalize pixel values (0-1 range)\n        data = data - np.min(data)\n        if np.max(data) != 0:\n            data = data / np.max(data)\n        \n        # Convert to grayscale (0-255 range)\n        data = (data * 255).astype(np.uint8)\n        \n        # Resize to 512x512\n        data = cv2.resize(data, dsize=(512, 512))\n        \n        # Adjust pixel values based on Photometric Interpretation\n        if pip == 'MONOCHROME1':\n            data = 255 - data  # Invert pixel values\n        \n        # Construct filename and save as grayscale PNG\n        new_filename = f\"{patient_id}_{series_id}_{str(image_id).zfill(4)}.png\"\n        new_filepath = os.path.join(dest_path, new_filename)\n        \n        cv2.imwrite(new_filepath, data, [cv2.IMWRITE_PNG_COMPRESSION, 0])  # Save as grayscale PNG\n        \n        return new_filename, True\n    except Exception as e:\n        print(f\"\\nError processing {src_path}: {e}\")\n        return None, False\n\ndef process_series_ids(start_index, end_index):\n    try:\n        # Step 1: Extract all series IDs from segmentation folder\n        all_series_ids = [f.split('.')[0] for f in os.listdir(segmentation_dir) if f.endswith('.nii')]\n        print(f\"\\nAll series IDs: {all_series_ids}\")\n        print(f\"\\nTotal number of series IDs: {len(all_series_ids)}\")\n        \n        # Select series IDs from start_index to end_index\n        selected_series_ids = all_series_ids[start_index:end_index]\n        print(f\"\\nSelected series IDs: {selected_series_ids}\")\n        print(f\"\\nNumber of selected series IDs: {len(selected_series_ids)}\")\n\n        # Ensure the filtered train directory exists\n        if not os.path.exists(filtered_train_dir):\n            os.makedirs(filtered_train_dir)\n\n        # Step 2: Filter, copy and convert train images based on selected series IDs\n        for patient_id in os.listdir(train_images_dir):\n            patient_dir = os.path.join(train_images_dir, patient_id)\n            if os.path.isdir(patient_dir):\n                for series_id in os.listdir(patient_dir):\n                    if series_id in selected_series_ids:\n                        series_dir = os.path.join(patient_dir, series_id)\n                        dest_series_dir = os.path.join(filtered_train_dir, patient_id, series_id)\n                        os.makedirs(dest_series_dir, exist_ok=True)\n                        for image_file in os.listdir(series_dir):\n                            if image_file.endswith('.dcm'):\n                                image_id = os.path.splitext(image_file)[0]\n                                src_image_path = os.path.join(series_dir, image_file)\n                                convert_dcm_to_grayscale_png(src_image_path, dest_series_dir, patient_id, series_id, image_id)\n\n        # Step 3: Zip the filtered and converted train images\n        zip_filename = os.path.join(output_dir, 'filtered_train_images.zip')\n        with zipfile.ZipFile(zip_filename, 'w') as zipf:\n            for root, dirs, files in os.walk(filtered_train_dir):\n                for file in files:\n                    file_path = os.path.join(root, file)\n                    zipf.write(file_path, arcname=os.path.relpath(file_path, filtered_train_dir))\n\n        # Provide download link\n        print(f\"\\nDownload your zipped folder: [Download filtered_train_images.zip](./filtered_train_images.zip)\")\n\n        # Clean up the filtered train images directory\n        if os.path.exists(filtered_train_dir):\n            shutil.rmtree(filtered_train_dir)\n            print(f\"\\n{filtered_train_dir} has been deleted.\")\n        else:\n            print(f\"\\n{filtered_train_dir} does not exist.\")\n    \n    except Exception as e:\n        print(f\"\\nAn error occurred during processing: {e}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-07-16T19:04:08.146942Z","iopub.execute_input":"2024-07-16T19:04:08.14765Z","iopub.status.idle":"2024-07-16T19:04:08.16865Z","shell.execute_reply.started":"2024-07-16T19:04:08.147613Z","shell.execute_reply":"2024-07-16T19:04:08.167523Z"},"trusted":true},"outputs":[],"execution_count":11},{"cell_type":"code","source":"# Example usage: Process series IDs from index 0 to 67 (first 68 series IDs)\nprocess_series_ids(0, 68)","metadata":{"execution":{"iopub.status.busy":"2024-07-16T19:04:11.235812Z","iopub.execute_input":"2024-07-16T19:04:11.236182Z","iopub.status.idle":"2024-07-16T19:14:52.234624Z","shell.execute_reply.started":"2024-07-16T19:04:11.236152Z","shell.execute_reply":"2024-07-16T19:14:52.233466Z"},"trusted":true},"outputs":[{"name":"stdout","text":"\nAll series IDs: ['39222', '52961', '7334', '15415', '51141', '47775', '4929', '5260', '50434', '397', '4759', '41976', '62680', '47856', '13307', '24149', '46912', '42973', '45406', '24774', '22397', '37032', '22232', '26980', '527', '61670', '6305', '52940', '137', '13041', '26942', '61747', '17577', '47438', '17225', '59325', '15271', '28079', '10180', '19657', '53000', '50875', '13774', '13666', '33355', '8236', '55449', '62556', '60755', '42173', '31146', '48324', '63205', '24134', '44612', '34774', '24442', '23644', '16066', '4622', '30902', '61569', '15539', '62307', '40781', '58697', '10000', '31085', '36753', '24645', '5118', '16097', '1201', '12402', '525', '18207', '19468', '58548', '44712', '39864', '5103', '52970', '778', '25349', '18624', '6344', '34232', '32670', '12114', '7818', '12674', '10109', '13925', '40186', '12102', '41288', '33526', '16503', '40471', '57769', '54917', '49753', '43233', '6172', '26214', '31021', '55965', '63146', '49547', '28122', '51136', '8413', '60307', '5104', '64520', '32243', '14286', '6631', '26906', '41663', '11748', '6575', '50212', '20664', '43416', '64117', '4890', '7397', '24373', '22479', '52279', '5218', '40430', '55928', '44758', '10385', '21282', '21057', '19360', '39205', '55694', '30952', '19927', '29167', '23837', '53843', '43088', '56245', '29053', '25359', '12900', '60961', '29832', '30843', '5425', '28432', '6130', '31200', '7384', '30522', '63418', '61403', '17605', '63843', '60881', '13848', '53345', '32991', '58027', '10494', '12840', '4123', '39628', '54830', '20684', '58391', '63701', '47155', '47305', '26840', '10252', '30064', '31571', '55583', '8340', '12039', '34224', '51033', '38633', '36257', '48901', '40496', '26343', '42680', '5176', '15748', '62573', '35190', '47610', '60302', '55515', '31852', '35661', '39013', '22730', '8320']\n\nTotal number of series IDs: 206\n\nSelected series IDs: ['39222', '52961', '7334', '15415', '51141', '47775', '4929', '5260', '50434', '397', '4759', '41976', '62680', '47856', '13307', '24149', '46912', '42973', '45406', '24774', '22397', '37032', '22232', '26980', '527', '61670', '6305', '52940', '137', '13041', '26942', '61747', '17577', '47438', '17225', '59325', '15271', '28079', '10180', '19657', '53000', '50875', '13774', '13666', '33355', '8236', '55449', '62556', '60755', '42173', '31146', '48324', '63205', '24134', '44612', '34774', '24442', '23644', '16066', '4622', '30902', '61569', '15539', '62307', '40781', '58697', '10000', '31085']\n\nNumber of selected series IDs: 68\n\nDownload your zipped folder: [Download filtered_train_images.zip](./filtered_train_images.zip)\n\n/kaggle/working/filtered_train_images has been deleted.\n","output_type":"stream"}],"execution_count":12},{"cell_type":"code","source":"# Example usage: Process series IDs from index 0 to 67 (first 68 series IDs)\nprocess_series_ids(0, 68)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\n\n# Path to the directory you want to delete\ndir_path = '/kaggle/working/filtered_train_images'\n\n# Check if directory exists and delete it\nif os.path.exists(dir_path):\n    shutil.rmtree(dir_path)\n    print(f\"{dir_path} has been deleted.\")\nelse:\n    print(f\"{dir_path} does not exist.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-07-16T19:34:44.914592Z","iopub.execute_input":"2024-07-16T19:34:44.914975Z","iopub.status.idle":"2024-07-16T19:34:44.921726Z","shell.execute_reply.started":"2024-07-16T19:34:44.914946Z","shell.execute_reply":"2024-07-16T19:34:44.920466Z"},"trusted":true},"outputs":[{"name":"stdout","text":"/kaggle/working/filtered_train_images does not exist.\n","output_type":"stream"}],"execution_count":20},{"cell_type":"code","source":"# List files in the working directory to confirm the zip file is there\nworking_dir = '/kaggle/working'\nfiles = os.listdir(working_dir)\nprint(f\"Files in {working_dir}: {files}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-07-16T21:06:27.011467Z","iopub.execute_input":"2024-07-16T21:06:27.011852Z","iopub.status.idle":"2024-07-16T21:06:27.01776Z","shell.execute_reply.started":"2024-07-16T21:06:27.011822Z","shell.execute_reply":"2024-07-16T21:06:27.016739Z"},"trusted":true},"outputs":[{"name":"stdout","text":"Files in /kaggle/working: ['filtered_train_images.zip']\n","output_type":"stream"}],"execution_count":29},{"cell_type":"code","source":"import os\nfrom IPython.display import FileLink, display\n\n# Specify the path to the file\nfile_path = '/kaggle/working/filtered_train_images.zip'\n\n# Check if the file exists\nif os.path.exists(file_path):\n    # Create a download link\n    link = FileLink(file_path)\n    # Display the download link\n    display(link)\nelse:\n    print(f\"The file {file_path} does not exist.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-07-16T21:06:32.819168Z","iopub.execute_input":"2024-07-16T21:06:32.819586Z","iopub.status.idle":"2024-07-16T21:06:32.828382Z","shell.execute_reply.started":"2024-07-16T21:06:32.819556Z","shell.execute_reply":"2024-07-16T21:06:32.827201Z"},"trusted":true},"outputs":[{"output_type":"display_data","data":{"text/plain":"/kaggle/working/filtered_train_images.zip","text/html":"<a href='/kaggle/working/filtered_train_images.zip' target='_blank'>/kaggle/working/filtered_train_images.zip</a><br>"},"metadata":{}}],"execution_count":30},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}