{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport pydicom\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-01T19:30:41.782068Z","iopub.execute_input":"2023-09-01T19:30:41.782664Z","iopub.status.idle":"2023-09-01T19:30:43.352401Z","shell.execute_reply.started":"2023-09-01T19:30:41.782606Z","shell.execute_reply":"2023-09-01T19:30:43.351315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" # Files","metadata":{}},{"cell_type":"markdown","source":"### First, let's see all files and folders in project directory","metadata":{}},{"cell_type":"code","source":"main_folder = '/kaggle/input/rsna-2023-abdominal-trauma-detection'\n\nfor i in os.listdir(main_folder):\n    if os.path.isdir(os.path.join(main_folder, i)):\n        print(f'{i} - directory')\n    else:\n        print(f'{i} - file')","metadata":{"execution":{"iopub.status.busy":"2023-09-01T19:30:43.354264Z","iopub.execute_input":"2023-09-01T19:30:43.354583Z","iopub.status.idle":"2023-09-01T19:30:43.367742Z","shell.execute_reply.started":"2023-09-01T19:30:43.354555Z","shell.execute_reply":"2023-09-01T19:30:43.366562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Then a quick glance at each file/directory","metadata":{}},{"cell_type":"markdown","source":"### train.csv","metadata":{}},{"cell_type":"code","source":"pd.read_csv(os.path.join(main_folder, \"train.csv\"))","metadata":{"execution":{"iopub.status.busy":"2023-09-01T19:30:43.369287Z","iopub.execute_input":"2023-09-01T19:30:43.369697Z","iopub.status.idle":"2023-09-01T19:30:43.419412Z","shell.execute_reply.started":"2023-09-01T19:30:43.369661Z","shell.execute_reply":"2023-09-01T19:30:43.418245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Contains information about abdominal traumas for each patient, this is our target - we will analyze it in more detail in the next section\n","metadata":{}},{"cell_type":"markdown","source":"__________________________________________________________________________","metadata":{}},{"cell_type":"markdown","source":"### sample_submission.csv","metadata":{}},{"cell_type":"code","source":"pd.read_csv(os.path.join(main_folder, 'sample_submission.csv'))","metadata":{"execution":{"iopub.status.busy":"2023-09-01T19:30:43.42207Z","iopub.execute_input":"2023-09-01T19:30:43.422581Z","iopub.status.idle":"2023-09-01T19:30:43.449167Z","shell.execute_reply.started":"2023-09-01T19:30:43.422539Z","shell.execute_reply":"2023-09-01T19:30:43.44792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is how our submission should look like - for each patient we must predict probabilities of each injury.\n","metadata":{}},{"cell_type":"markdown","source":"___________________________________________________________","metadata":{}},{"cell_type":"markdown","source":"### train_images","metadata":{}},{"cell_type":"code","source":"train_images = os.listdir(os.path.join(main_folder, \"train_images\"))\nprint('train_images')\nprint(f'  -> {train_images[0]}')\nfor dir in os.listdir(os.path.join(main_folder, f'train_images/{train_images[0]}')):\n    print(f'    -> {dir}')\n    for img in os.listdir(os.path.join(main_folder, f'train_images/{train_images[0]}/{dir}'))[:10]:\n        print(f'       -> {img}')\nprint(' ... ')","metadata":{"execution":{"iopub.status.busy":"2023-09-01T19:30:43.450692Z","iopub.execute_input":"2023-09-01T19:30:43.451953Z","iopub.status.idle":"2023-09-01T19:30:43.826354Z","shell.execute_reply.started":"2023-09-01T19:30:43.451911Z","shell.execute_reply":"2023-09-01T19:30:43.825274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Directory contains computed tomography images of all patients, they have been grouped into folders named by patient id, each patient has been scanned once or twice, so in each patient's folder we have one or two folders named by scan id, these folders contain images in dcm format.","metadata":{}},{"cell_type":"markdown","source":"_____","metadata":{}},{"cell_type":"markdown","source":"### test_images","metadata":{}},{"cell_type":"markdown","source":"Same as train_images above, but patients here are not labeled, set as this will be used to evaluate our model.","metadata":{}},{"cell_type":"markdown","source":"___","metadata":{}},{"cell_type":"markdown","source":"### train_series_meta.csv","metadata":{}},{"cell_type":"markdown","source":"Each patient was scaned once or twice, these file connects patient id with scan id as well as volume of the aorta in hounsfield units on image, and information if there are any incomplete organs (organs croped on image)","metadata":{}},{"cell_type":"code","source":"tsm = pd.read_csv(os.path.join(main_folder, f'train_series_meta.csv'))\ntsm","metadata":{"execution":{"iopub.status.busy":"2023-09-01T19:30:43.828234Z","iopub.execute_input":"2023-09-01T19:30:43.828581Z","iopub.status.idle":"2023-09-01T19:30:43.862914Z","shell.execute_reply.started":"2023-09-01T19:30:43.828545Z","shell.execute_reply":"2023-09-01T19:30:43.861714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sizes = [tsm['incomplete_organ'].sum(), tsm['incomplete_organ'].count() - tsm['incomplete_organ'].sum()]\nlabels = ['Incomplete organ', 'No incomplete organ']\n\nplt.figure(figsize=(6, 6))\nplt.pie(sizes, labels=labels, autopct='%1.1f%%', startangle=140)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T19:30:43.864607Z","iopub.execute_input":"2023-09-01T19:30:43.865636Z","iopub.status.idle":"2023-09-01T19:30:44.062191Z","shell.execute_reply.started":"2023-09-01T19:30:43.86559Z","shell.execute_reply":"2023-09-01T19:30:44.061033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"___","metadata":{}},{"cell_type":"markdown","source":"### test_series_meta.csv","metadata":{}},{"cell_type":"code","source":"pd.read_csv(os.path.join(main_folder, f'test_series_meta.csv'))","metadata":{"execution":{"iopub.status.busy":"2023-09-01T19:30:44.063988Z","iopub.execute_input":"2023-09-01T19:30:44.064749Z","iopub.status.idle":"2023-09-01T19:30:44.086901Z","shell.execute_reply.started":"2023-09-01T19:30:44.064702Z","shell.execute_reply":"2023-09-01T19:30:44.085685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Same as previous but for test set and no info about incomplete organs ","metadata":{}},{"cell_type":"markdown","source":"___","metadata":{}},{"cell_type":"markdown","source":"### segmentations","metadata":{}},{"cell_type":"markdown","source":"Model generated pixel-level annotations of the relevant organs and some major bones for a subset of the scans in the training set. ","metadata":{}},{"cell_type":"markdown","source":"___","metadata":{}},{"cell_type":"markdown","source":"### image_level_labels.csv","metadata":{}},{"cell_type":"code","source":"pd.read_csv(os.path.join(main_folder, f'image_level_labels.csv'))","metadata":{"execution":{"iopub.status.busy":"2023-09-01T19:30:44.088684Z","iopub.execute_input":"2023-09-01T19:30:44.089413Z","iopub.status.idle":"2023-09-01T19:30:44.124961Z","shell.execute_reply.started":"2023-09-01T19:30:44.089365Z","shell.execute_reply":"2023-09-01T19:30:44.123827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Identifies specific images that contain either bowel or extravasation injuries.","metadata":{}},{"cell_type":"markdown","source":"___","metadata":{}},{"cell_type":"markdown","source":"### train_dicom_tags.parquet | test_dicom_tags.parquet","metadata":{}},{"cell_type":"markdown","source":"DICOM tags from every image, extracted with Pydicom.","metadata":{}},{"cell_type":"markdown","source":"___","metadata":{}},{"cell_type":"markdown","source":"# Target","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(os.path.join(main_folder, f'train.csv'))","metadata":{"execution":{"iopub.status.busy":"2023-09-01T19:30:44.129051Z","iopub.execute_input":"2023-09-01T19:30:44.12939Z","iopub.status.idle":"2023-09-01T19:30:44.140832Z","shell.execute_reply.started":"2023-09-01T19:30:44.129362Z","shell.execute_reply":"2023-09-01T19:30:44.139515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Target contains information about injuries for each patient, total of 16 columns, bowel and extravasation traumas are split into two categories injury/healthy, whereas for kidney, liver and spleen ones there are three columns: healthy/low/high. The last column shows if the patient has any other injuries.\n","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T19:30:44.142671Z","iopub.execute_input":"2023-09-01T19:30:44.143149Z","iopub.status.idle":"2023-09-01T19:30:44.15876Z","shell.execute_reply.started":"2023-09-01T19:30:44.143108Z","shell.execute_reply":"2023-09-01T19:30:44.157432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T19:30:44.160573Z","iopub.execute_input":"2023-09-01T19:30:44.161229Z","iopub.status.idle":"2023-09-01T19:30:44.183965Z","shell.execute_reply.started":"2023-09-01T19:30:44.161187Z","shell.execute_reply":"2023-09-01T19:30:44.183075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is 3147 patients in total, fortunately no null values","metadata":{}},{"cell_type":"markdown","source":"#### How target is distributed? ","metadata":{}},{"cell_type":"code","source":"categories = ['bowel', 'extravastion', 'kidney', 'liver', 'spleen', 'any injury']\nx = np.arange(len(categories))\nfig, ax = plt.subplots()\n\nax.bar(x, [train['bowel_healthy'].count(), train['extravasation_healthy'].count(), 0, 0, 0, train['any_injury'].count()], width=0.6, label='healthy', color='green')\nax.bar(x, [train['bowel_injury'].sum(), train['extravasation_injury'].sum(), 0, 0, 0, 0], width=0.6, label='injury', color='red')\nax.bar(x, [0, 0, train['kidney_healthy'].count(), train['liver_healthy'].count(), train['spleen_healthy'].count(), 0], width=0.6, color='green')\nax.bar(x, [0, 0, train['kidney_low'].sum() + train['kidney_high'].sum(), train['liver_low'].sum() + train['liver_high'].sum(), train['spleen_low'].sum() + train['spleen_high'].sum(), 0], width=0.6, color='yellow', label='low')\nax.bar(x, [0, 0, train['kidney_high'].sum(), train['liver_high'].sum(), train['spleen_high'].sum(), 0], width=0.6, color='orange', label='high')\nax.bar(x, [0, 0, 0, 0, 0, train['any_injury'].sum()], width=0.6, color='red')\n\nax.set_xticks(x)\nax.set_xticklabels(categories)\nax.set_ylabel('Count')\nax.set_title('')\nax.legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T19:30:44.184841Z","iopub.execute_input":"2023-09-01T19:30:44.18545Z","iopub.status.idle":"2023-09-01T19:30:44.565325Z","shell.execute_reply.started":"2023-09-01T19:30:44.185413Z","shell.execute_reply":"2023-09-01T19:30:44.564324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see, there is a huge class imbalance, most of images shows healthy organs","metadata":{}},{"cell_type":"markdown","source":"#### Correlation between each injuries","metadata":{}},{"cell_type":"code","source":"correlation_matrix = train.drop(columns=['patient_id']).corr()\n\nplt.figure(figsize=(10, 8))\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm', linewidths=.5)\nplt.title('abdominal traumas correlation matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T19:30:44.566784Z","iopub.execute_input":"2023-09-01T19:30:44.567284Z","iopub.status.idle":"2023-09-01T19:30:45.498208Z","shell.execute_reply.started":"2023-09-01T19:30:44.567243Z","shell.execute_reply":"2023-09-01T19:30:45.497257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Images","metadata":{}},{"cell_type":"markdown","source":"First, let's see how sample images look like","metadata":{}},{"cell_type":"code","source":"dicom_images = [pydicom.dcmread(os.path.join(main_folder, f'train_images/10004/21057/{dcm}')) for dcm in os.listdir(os.path.join(main_folder, 'train_images/10004/21057'))[:6]]\n\nfig, axes = plt.subplots(1, len(dicom_images), figsize=(20, 10))\n\nfor i in range(6):\n    axes[i].imshow(dicom_images[i].pixel_array, cmap='gray')\n    axes[i].axis('off')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T19:30:45.499425Z","iopub.execute_input":"2023-09-01T19:30:45.499936Z","iopub.status.idle":"2023-09-01T19:30:46.358417Z","shell.execute_reply.started":"2023-09-01T19:30:45.499908Z","shell.execute_reply":"2023-09-01T19:30:46.357289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Each patient was scaned once or twice, what is proportion of 'one scan patients' to 'two scans patients'?","metadata":{}},{"cell_type":"code","source":"patient_ids = os.listdir(os.path.join(main_folder, 'train_images'))\none_scan = 0\ntwo_scans = 0\nfor patient_id in patient_ids:\n    num_scans = len(os.listdir(os.path.join(main_folder, f'train_images/{patient_id}')))\n    if num_scans == 1:\n        one_scan += 1\n    elif num_scans == 2:\n        two_scans += 1\n\ncategories = ['one scan', 'two scans']\nx = np.arange(len(categories))\nfig, ax = plt.subplots()\n\nax.bar(x, [one_scan, two_scans], width=0.6)\n\nax.set_xticks(x)\nax.set_xticklabels(categories)\nax.set_ylabel('Count')\nax.set_title('Number of scans per patient')\nax.legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T19:30:46.359784Z","iopub.execute_input":"2023-09-01T19:30:46.360247Z","iopub.status.idle":"2023-09-01T19:30:54.896348Z","shell.execute_reply.started":"2023-09-01T19:30:46.360198Z","shell.execute_reply":"2023-09-01T19:30:54.895511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_images = 0\nfor patient_id in patient_ids:\n    path = os.path.join(main_folder, f'train_images/{patient_id}')\n    scan_ids = os.listdir(path)\n    for scan_id in scan_ids:\n        num_images += len(os.listdir(os.path.join(path, scan_id)))\nnum_images","metadata":{"execution":{"iopub.status.busy":"2023-09-01T19:30:54.897557Z","iopub.execute_input":"2023-09-01T19:30:54.898075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is total of 1500653 computed tomography images","metadata":{}}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}