{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-28T18:53:19.366416Z","iopub.execute_input":"2023-07-28T18:53:19.366894Z","iopub.status.idle":"2023-07-28T18:53:19.391862Z","shell.execute_reply.started":"2023-07-28T18:53:19.366833Z","shell.execute_reply":"2023-07-28T18:53:19.391009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" #### https://www.kaggle.com/code/theoviel/dicom-resized-png-jpg\n \n#### some code below is inspired by the notebook with link above! Thank you [Theo Viel](https://www.kaggle.com/theoviel)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pydicom\nimport os\nfrom glob import glob\nfrom tqdm.notebook import tqdm\npd.set_option('display.max_rows', None)\npd.set_option('display.max_columns', None)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:19.393313Z","iopub.execute_input":"2023-07-28T18:53:19.394364Z","iopub.status.idle":"2023-07-28T18:53:20.380391Z","shell.execute_reply.started":"2023-07-28T18:53:19.394332Z","shell.execute_reply":"2023-07-28T18:53:20.379246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### loading all the files","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train.csv')\nseries_meta_df = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_series_meta.csv')\nimage_labels_df = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/image_level_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:04:54.793816Z","iopub.execute_input":"2023-07-28T19:04:54.794249Z","iopub.status.idle":"2023-07-28T19:04:54.835102Z","shell.execute_reply.started":"2023-07-28T19:04:54.794215Z","shell.execute_reply":"2023-07-28T19:04:54.834144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"series_meta_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:09:17.151363Z","iopub.execute_input":"2023-07-28T19:09:17.152497Z","iopub.status.idle":"2023-07-28T19:09:17.159055Z","shell.execute_reply.started":"2023-07-28T19:09:17.152455Z","shell.execute_reply":"2023-07-28T19:09:17.158249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"series_meta_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:20.460593Z","iopub.execute_input":"2023-07-28T18:53:20.460967Z","iopub.status.idle":"2023-07-28T18:53:20.490339Z","shell.execute_reply.started":"2023-07-28T18:53:20.46092Z","shell.execute_reply":"2023-07-28T18:53:20.489065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"series_meta_df['series_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:21:57.704767Z","iopub.execute_input":"2023-07-28T19:21:57.705215Z","iopub.status.idle":"2023-07-28T19:21:57.713553Z","shell.execute_reply.started":"2023-07-28T19:21:57.705181Z","shell.execute_reply":"2023-07-28T19:21:57.712404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['patient_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:16:15.961398Z","iopub.execute_input":"2023-07-28T19:16:15.961855Z","iopub.status.idle":"2023-07-28T19:16:15.96971Z","shell.execute_reply.started":"2023-07-28T19:16:15.961806Z","shell.execute_reply":"2023-07-28T19:16:15.968476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### checking to see the distibution of Injury","metadata":{}},{"cell_type":"code","source":"from collections import Counter\n# Count the occurrence of each value\ncounts = Counter(train_df.iloc[:, -1])\n# Make the histogram using plt.hist, bins must be integer\nplt.hist(train_df.iloc[:, -1], bins=[-0.5, 0.5, 1.5], edgecolor='black')\n\nplt.title('Distribution of Injuries')\nplt.xlabel('Values')\nplt.ylabel('Counts')\n\n# Display counts above bars\nfor x, y in counts.items():\n    plt.text(x, y, str(y), ha='center', va='bottom')\n\nplt.xticks([0, 1])\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-28T20:02:16.402701Z","iopub.execute_input":"2023-07-28T20:02:16.404154Z","iopub.status.idle":"2023-07-28T20:02:16.58493Z","shell.execute_reply.started":"2023-07-28T20:02:16.404111Z","shell.execute_reply":"2023-07-28T20:02:16.583878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_labels_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:20:13.966233Z","iopub.execute_input":"2023-07-28T19:20:13.966913Z","iopub.status.idle":"2023-07-28T19:20:13.974586Z","shell.execute_reply.started":"2023-07-28T19:20:13.966868Z","shell.execute_reply":"2023-07-28T19:20:13.973586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_labels_df['injury_name'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:20.533423Z","iopub.execute_input":"2023-07-28T18:53:20.53389Z","iopub.status.idle":"2023-07-28T18:53:20.548759Z","shell.execute_reply.started":"2023-07-28T18:53:20.533818Z","shell.execute_reply":"2023-07-28T18:53:20.547997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_labels_df['patient_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:37:50.02838Z","iopub.execute_input":"2023-07-28T19:37:50.028802Z","iopub.status.idle":"2023-07-28T19:37:50.035797Z","shell.execute_reply.started":"2023-07-28T19:37:50.028768Z","shell.execute_reply":"2023-07-28T19:37:50.035048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_labels_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:20:27.915429Z","iopub.execute_input":"2023-07-28T19:20:27.915872Z","iopub.status.idle":"2023-07-28T19:20:27.926904Z","shell.execute_reply.started":"2023-07-28T19:20:27.915826Z","shell.execute_reply":"2023-07-28T19:20:27.925699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,8))\ntrain_df.iloc[:, 1:-1].sum().plot.bar()\nplt.title('Distribution of Injury Types')\nplt.xlabel('Injury Type')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:20.550207Z","iopub.execute_input":"2023-07-28T18:53:20.550769Z","iopub.status.idle":"2023-07-28T18:53:21.005136Z","shell.execute_reply.started":"2023-07-28T18:53:20.550739Z","shell.execute_reply":"2023-07-28T18:53:21.00425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,8))\nsns.histplot(series_meta_df['aortic_hu'], bins=30, kde=True)\nplt.title('Distribution of aortic_hu')\nplt.xlabel('Aortic Hounsfield Units')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:21.007075Z","iopub.execute_input":"2023-07-28T18:53:21.008158Z","iopub.status.idle":"2023-07-28T18:53:21.435996Z","shell.execute_reply.started":"2023-07-28T18:53:21.008116Z","shell.execute_reply":"2023-07-28T18:53:21.434907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sample_scan_path = glob('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/10004/21057/1000.dcm')[0] # get a sample scan\n# ds = pydicom.dcmread(sample_scan_path)\n# plt.imshow(ds.pixel_array, cmap=plt.cm.bone)\n# plt.title('Sample CT scan')\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:21.441613Z","iopub.execute_input":"2023-07-28T18:53:21.44213Z","iopub.status.idle":"2023-07-28T18:53:21.446942Z","shell.execute_reply.started":"2023-07-28T18:53:21.442096Z","shell.execute_reply":"2023-07-28T18:53:21.445833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### image labels","metadata":{}},{"cell_type":"code","source":"image_labels_df['injury_name'].value_counts().plot(kind='bar', figsize=(10,8))\nplt.title('Distribution of Injury Types at Image Level')\nplt.xlabel('Injury Type')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:21.448602Z","iopub.execute_input":"2023-07-28T18:53:21.44972Z","iopub.status.idle":"2023-07-28T18:53:21.761541Z","shell.execute_reply.started":"2023-07-28T18:53:21.449679Z","shell.execute_reply":"2023-07-28T18:53:21.760429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Quick animation to visually see what we are looking at","metadata":{}},{"cell_type":"code","source":"import matplotlib.animation as animation\nfrom IPython.display import HTML\n\n# Grab every 10th DICOM file in the series\ndicom_files = sorted(glob('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/10004/21057/*.dcm'))[::6]\ndicom_images = [pydicom.dcmread(f).pixel_array for f in dicom_files]\n\nfig, ax = plt.subplots()\n\n# plot the first frame and keep the plot object for later\nim = ax.imshow(dicom_images[0], cmap=plt.cm.bone)\n\n# create update function that iterates over your images\ndef update(i):\n    im.set_array(dicom_images[i])\n\nani = animation.FuncAnimation(fig, update, frames=range(len(dicom_images)), repeat=True)\n\n# Show the animation\nHTML(ani.to_jshtml())","metadata":{"execution":{"iopub.status.busy":"2023-07-28T20:01:07.857197Z","iopub.execute_input":"2023-07-28T20:01:07.858081Z","iopub.status.idle":"2023-07-28T20:01:34.676129Z","shell.execute_reply.started":"2023-07-28T20:01:07.858034Z","shell.execute_reply":"2023-07-28T20:01:34.675145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## load all the tags on the images","metadata":{}},{"cell_type":"code","source":"dicom_tags_df = pd.read_parquet('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_dicom_tags.parquet')\n# num_rows = int(len(dicom_tags_df) * 0.01)\n# dicom_tags_df = dicom_tags_df.head(num_rows)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:24:47.416962Z","iopub.execute_input":"2023-07-28T19:24:47.41812Z","iopub.status.idle":"2023-07-28T19:24:52.351229Z","shell.execute_reply.started":"2023-07-28T19:24:47.418067Z","shell.execute_reply":"2023-07-28T19:24:52.350268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"dicom_tags_df has no missing values","metadata":{}},{"cell_type":"code","source":"# Extract patient_id and series_id from the path in dicom_tags_df\ndicom_tags_df['patient_id'] = dicom_tags_df['path'].str.split('/').str[1].astype(int)\ndicom_tags_df['series_id'] = dicom_tags_df['path'].str.split('/').str[2].astype(int)\ndicom_tags_df['instance_number'] = dicom_tags_df['path'].str.split('/').str[3].str.split('.').str[0].astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:30:35.648828Z","iopub.execute_input":"2023-07-28T19:30:35.649311Z","iopub.status.idle":"2023-07-28T19:31:01.583112Z","shell.execute_reply.started":"2023-07-28T19:30:35.649271Z","shell.execute_reply":"2023-07-28T19:31:01.582048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# no missing values\ndicom_tags_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:32:33.195407Z","iopub.execute_input":"2023-07-28T19:32:33.195788Z","iopub.status.idle":"2023-07-28T19:32:43.971856Z","shell.execute_reply.started":"2023-07-28T19:32:33.195758Z","shell.execute_reply":"2023-07-28T19:32:43.970179Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicom_tags_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:31:55.942964Z","iopub.execute_input":"2023-07-28T19:31:55.943749Z","iopub.status.idle":"2023-07-28T19:31:55.976994Z","shell.execute_reply.started":"2023-07-28T19:31:55.943703Z","shell.execute_reply":"2023-07-28T19:31:55.975644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"\n# Merge dicom_tags_df and train_labels_df on patient_id\ncombined_df = pd.merge(dicom_tags_df, train_df, how='left', on='patient_id')\n\n# del dicom_tags_df, train_df\n\n# Merge combined_df and series_meta_df on ['patient_id', 'series_id']\ncombined_df = pd.merge(combined_df, series_meta_df, how='left',on=['patient_id', 'series_id'])\n\n# del series_meta_df\n\n# Merge combined_df and image_level_labels_df on ['patient_id', 'series_id']\ncombined_df = pd.merge(combined_df, image_labels_df, how = 'left', on=['patient_id', 'series_id','instance_number'])\n\n# del image_labels_df\ncombined_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:03:26.527045Z","iopub.execute_input":"2023-07-28T19:03:26.527472Z","iopub.status.idle":"2023-07-28T19:03:26.76281Z","shell.execute_reply.started":"2023-07-28T19:03:26.527437Z","shell.execute_reply":"2023-07-28T19:03:26.761687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_labels_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:04:29.452752Z","iopub.execute_input":"2023-07-28T19:04:29.453291Z","iopub.status.idle":"2023-07-28T19:04:29.460975Z","shell.execute_reply.started":"2023-07-28T19:04:29.453248Z","shell.execute_reply":"2023-07-28T19:04:29.459787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_labels_df[image_labels_df['patient_id'] == 49954]","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:04:21.663687Z","iopub.execute_input":"2023-07-28T19:04:21.664152Z","iopub.status.idle":"2023-07-28T19:04:21.67537Z","shell.execute_reply.started":"2023-07-28T19:04:21.664112Z","shell.execute_reply":"2023-07-28T19:04:21.674246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"49954\t41479\t532","metadata":{}},{"cell_type":"markdown","source":"missing values here because we only have 1% of dicom_tags_df","metadata":{}},{"cell_type":"code","source":"# dicom_tags_df[(dicom_tags_df['patient_id'] == 51078) & (dicom_tags_df['series_id'] == 53398)].head(1)\n# \t\t161","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:28.648484Z","iopub.execute_input":"2023-07-28T18:53:28.648811Z","iopub.status.idle":"2023-07-28T18:53:28.653876Z","shell.execute_reply.started":"2023-07-28T18:53:28.648783Z","shell.execute_reply":"2023-07-28T18:53:28.652544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the full paths to the DICOM files\n# dicom_files = sorted(glob('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/*/*/*.dcm', recursive=True))","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:28.655444Z","iopub.execute_input":"2023-07-28T18:53:28.655803Z","iopub.status.idle":"2023-07-28T18:53:28.66627Z","shell.execute_reply.started":"2023-07-28T18:53:28.655764Z","shell.execute_reply":"2023-07-28T18:53:28.665081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Extract the patient_id and series_id from each file path\n# dicom_file_df = pd.DataFrame({\n#     'path': dicom_files,\n#     'patient_id': [path.split('/')[5] for path in dicom_files],\n#     'series_id': [path.split('/')[6] for path in dicom_files],\n#     'instance_number': [path.split('/')[7].split('.')[0] for path in dicom_files],\n# })","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:28.668207Z","iopub.execute_input":"2023-07-28T18:53:28.668543Z","iopub.status.idle":"2023-07-28T18:53:28.678413Z","shell.execute_reply.started":"2023-07-28T18:53:28.668515Z","shell.execute_reply":"2023-07-28T18:53:28.67746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the DataFrame to a csv file\n# dicom_file_df.to_csv('/kaggle/working/dicom_file_df.csv', index=False)\n\n# Load the DataFrame from the csv file\ndicom_file_df = pd.read_csv('/kaggle/input/rsna-abdominal-trauma-detect-eda-animation/dicom_file_df.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:28.68021Z","iopub.execute_input":"2023-07-28T18:53:28.680504Z","iopub.status.idle":"2023-07-28T18:53:32.310662Z","shell.execute_reply.started":"2023-07-28T18:53:28.680478Z","shell.execute_reply":"2023-07-28T18:53:32.309742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicom_file_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:32.312301Z","iopub.execute_input":"2023-07-28T18:53:32.312929Z","iopub.status.idle":"2023-07-28T18:53:32.318829Z","shell.execute_reply.started":"2023-07-28T18:53:32.312888Z","shell.execute_reply":"2023-07-28T18:53:32.317894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert the id columns to integer type\ndicom_file_df['patient_id'] = dicom_file_df['patient_id'].astype(int)\ndicom_file_df['series_id'] = dicom_file_df['series_id'].astype(int)\ndicom_file_df['instance_number'] = dicom_file_df['instance_number'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:32.320314Z","iopub.execute_input":"2023-07-28T18:53:32.320619Z","iopub.status.idle":"2023-07-28T18:53:32.363489Z","shell.execute_reply.started":"2023-07-28T18:53:32.320591Z","shell.execute_reply":"2023-07-28T18:53:32.362535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicom_file_df.tail(2)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:39:54.432338Z","iopub.execute_input":"2023-07-28T19:39:54.432699Z","iopub.status.idle":"2023-07-28T19:39:54.445646Z","shell.execute_reply.started":"2023-07-28T19:39:54.432668Z","shell.execute_reply":"2023-07-28T19:39:54.444423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicom_tags_df.tail(2)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:40:19.119006Z","iopub.execute_input":"2023-07-28T19:40:19.119364Z","iopub.status.idle":"2023-07-28T19:40:19.150155Z","shell.execute_reply.started":"2023-07-28T19:40:19.119334Z","shell.execute_reply":"2023-07-28T19:40:19.149073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"####  I noticed some images described in train_dicom_tags, but not in the train_images. So I try to figure out what patient_id, series_id, and instance_number mutually exclusive to each other.","metadata":{}},{"cell_type":"code","source":"merged_df = pd.merge(dicom_file_df, dicom_tags_df, how='outer', indicator=True, on=['patient_id', 'series_id', 'instance_number'])\n\nonly_in_file_df = merged_df[merged_df['_merge'] == 'left_only'][['patient_id', 'series_id', 'instance_number']]\nonly_in_tags_df = merged_df[merged_df['_merge'] == 'right_only'][['patient_id', 'series_id', 'instance_number']]\n","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:41:41.967553Z","iopub.execute_input":"2023-07-28T19:41:41.96814Z","iopub.status.idle":"2023-07-28T19:41:49.791366Z","shell.execute_reply.started":"2023-07-28T19:41:41.968107Z","shell.execute_reply":"2023-07-28T19:41:49.79024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"only_in_file_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:45:27.038538Z","iopub.execute_input":"2023-07-28T19:45:27.038953Z","iopub.status.idle":"2023-07-28T19:45:27.045755Z","shell.execute_reply.started":"2023-07-28T19:45:27.038919Z","shell.execute_reply":"2023-07-28T19:45:27.044679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"only_in_tags_df.sample(2)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T20:06:35.118226Z","iopub.execute_input":"2023-07-28T20:06:35.118747Z","iopub.status.idle":"2023-07-28T20:06:35.135369Z","shell.execute_reply.started":"2023-07-28T20:06:35.118699Z","shell.execute_reply":"2023-07-28T20:06:35.13399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"only_in_tags_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:45:25.517546Z","iopub.execute_input":"2023-07-28T19:45:25.518006Z","iopub.status.idle":"2023-07-28T19:45:25.525282Z","shell.execute_reply.started":"2023-07-28T19:45:25.517971Z","shell.execute_reply":"2023-07-28T19:45:25.524009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# combined_df.loc[combined_df['path_y'].isna(), 'path_y'] = '/kaggle/input/rsna-2023-abdominal-trauma-detection/' + combined_df.loc[combined_df['path_y'].isna(), 'path_x']","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:32.364886Z","iopub.execute_input":"2023-07-28T18:53:32.365191Z","iopub.status.idle":"2023-07-28T18:53:32.369727Z","shell.execute_reply.started":"2023-07-28T18:53:32.365164Z","shell.execute_reply":"2023-07-28T18:53:32.368374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Merge the dataframes\ncombined_df = pd.merge(combined_df, dicom_file_df, how='left',on=['patient_id', 'series_id', 'instance_number'])","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:32.37133Z","iopub.execute_input":"2023-07-28T18:53:32.371951Z","iopub.status.idle":"2023-07-28T18:53:32.780433Z","shell.execute_reply.started":"2023-07-28T18:53:32.371909Z","shell.execute_reply":"2023-07-28T18:53:32.7794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get indices where 'path_y' is NaN\nindices = combined_df[combined_df['path_y'].isna()].index\n\n# Drop these indices from the DataFrame\ncombined_df = combined_df.drop(indices)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:32.790726Z","iopub.execute_input":"2023-07-28T18:53:32.791022Z","iopub.status.idle":"2023-07-28T18:53:32.818857Z","shell.execute_reply.started":"2023-07-28T18:53:32.790996Z","shell.execute_reply":"2023-07-28T18:53:32.817896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def standardize_pixel_array(dcm: pydicom.dataset.FileDataset) -> np.ndarray:\n    \"\"\"\n    Source : https://www.kaggle.com/code/huiminglin/rsna-2023-abdomen-dicom-standardization !\n    \"\"\"\n    # Correct DICOM pixel_array if PixelRepresentation == 1.\n    pixel_array = dcm.pixel_array\n    if dcm.PixelRepresentation == 1:\n        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n        dtype = pixel_array.dtype \n        new_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n        pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dcm)\n    return pixel_array","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:32.820152Z","iopub.execute_input":"2023-07-28T18:53:32.820487Z","iopub.status.idle":"2023-07-28T18:53:32.827918Z","shell.execute_reply.started":"2023-07-28T18:53:32.820459Z","shell.execute_reply":"2023-07-28T18:53:32.82662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"\n\nimport cv2\ndef load_dicom_image(path,size=64):\n    ds = pydicom.dcmread(path)\n#     pos_z = ds[(0x20, 0x32)].value[-1]  # to retrieve the order of frames\n    img = standardize_pixel_array(ds)\n    img = (img - img.min()) / (img.max() - img.min())\n    if ds.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n\n#     img[pos_z] = img\n    img = cv2.resize(img, (size, size))\n    \n    return img\n\n# Test the function\nimage_data = load_dicom_image(combined_df.iloc[0]['path_y'])\nplt.imshow(image_data, cmap=plt.cm.bone)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:01:10.770017Z","iopub.status.idle":"2023-07-28T19:01:10.770438Z","shell.execute_reply.started":"2023-07-28T19:01:10.770237Z","shell.execute_reply":"2023-07-28T19:01:10.770257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create an empty list to store the pixel arrays\npixel_arrays = []\n\n# Iterate over the rows of the DataFrame\nfor _, row in tqdm(combined_df.iterrows()):\n    # Load the DICOM image and flatten the pixel array\n    pixel_array = load_dicom_image(row['path_y'], size=64).flatten()\n    \n    # Append the flattened pixel array to the list\n    pixel_arrays.append(pixel_array)\n\n# Create a new DataFrame from the list of pixel arrays\npixel_df = pd.DataFrame(pixel_arrays)\n\n# Concatenate the original DataFrame and the new DataFrame along the column axis\ncombined_df = pd.concat([combined_df, pixel_df], axis=1)\ndel pixel_df","metadata":{"execution":{"iopub.status.busy":"2023-07-28T18:53:33.373069Z","iopub.execute_input":"2023-07-28T18:53:33.373625Z","iopub.status.idle":"2023-07-28T19:01:09.581097Z","shell.execute_reply.started":"2023-07-28T18:53:33.373592Z","shell.execute_reply":"2023-07-28T19:01:09.579904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combined_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:06:20.915817Z","iopub.execute_input":"2023-07-28T19:06:20.916225Z","iopub.status.idle":"2023-07-28T19:06:20.923108Z","shell.execute_reply.started":"2023-07-28T19:06:20.916196Z","shell.execute_reply":"2023-07-28T19:06:20.921871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data = pd.DataFrame()\n\nmissing_data['total_missing'] = combined_df.isnull().sum()\n\n# Find the percentage of missing values\nmissing_data['percent_missing'] = combined_df.isnull().sum() * 100 / len(combined_df)\n\n# Display the result\nmissing_data.loc[missing_data['total_missing']>0].sort_values(by='total_missing',ascending = False)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T19:01:37.543467Z","iopub.execute_input":"2023-07-28T19:01:37.544608Z","iopub.status.idle":"2023-07-28T19:01:38.448628Z","shell.execute_reply.started":"2023-07-28T19:01:37.544556Z","shell.execute_reply":"2023-07-28T19:01:38.444254Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Check for any missing values in the DataFrame\n# na_values = combined_df.isna().sum()\n\n# # Print the result\n# print(na_values)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-28T20:07:49.611678Z","iopub.execute_input":"2023-07-28T20:07:49.612116Z","iopub.status.idle":"2023-07-28T20:07:49.617986Z","shell.execute_reply.started":"2023-07-28T20:07:49.612082Z","shell.execute_reply":"2023-07-28T20:07:49.616579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"next step is fit into neural net work","metadata":{}}]}