{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sb\nimport matplotlib.pyplot as plt\nimport nibabel as nib\n\nimport os\nimport pydicom\nfrom glob import glob\nfrom tqdm import tqdm, trange\nimport cv2\nimport re\nfrom scipy.ndimage import zoom\n\nPATH = '/kaggle/input/rsna-2023-abdominal-trauma-detection'\n\nnew_dir_path = '/kaggle/working/rsna-same-shape_19_29'\nos.mkdir(new_dir_path)\n\nnew_dir_path = '/kaggle/working/rsna-same-shape_19_29/input'\nos.mkdir(new_dir_path)\n\nnew_dir_path = '/kaggle/working/rsna-same-shape_19_29/mask'\nos.mkdir(new_dir_path)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-21T14:55:54.891848Z","iopub.execute_input":"2023-08-21T14:55:54.892325Z","iopub.status.idle":"2023-08-21T14:55:56.371262Z","shell.execute_reply.started":"2023-08-21T14:55:54.892289Z","shell.execute_reply":"2023-08-21T14:55:56.369853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This notebook makes 3D numpy data(Shape is unified as 512x512x512).","metadata":{}},{"cell_type":"code","source":"#https://www.kaggle.com/code/parhammostame/construct-3d-arrays-from-dcm-nii-3-view-angles\ndef create_3D_scans(folder, downsample_rate=1): \n    filenames = os.listdir(folder)\n    filenames = [int(filename.split('.')[0]) for filename in filenames]\n    filenames = sorted(filenames)\n    filenames = [str(filename) + '.dcm' for filename in filenames]\n        \n    volume = []\n    for filename in tqdm(filenames[::downsample_rate]):\n        filepath = os.path.join(folder, filename)\n        ds = pydicom.dcmread(filepath)\n        #image = dicom_to_image(ds)\n        image = ds.pixel_array\n        \n        # find rescale params\n        if (\"RescaleIntercept\" in ds) and (\"RescaleSlope\" in ds):\n            intercept = float(ds.RescaleIntercept)\n            slope = float(ds.RescaleSlope)\n    \n        # find clipping params\n        center = int(ds.WindowCenter)\n        width = int(ds.WindowWidth)\n        low = center - width / 2\n        high = center + width / 2    \n        image = (image * slope) + intercept\n        image = np.clip(image, low, high)\n\n        image = (image / np.max(image) * 255).astype(np.int16)\n        image = image[::downsample_rate, ::downsample_rate]\n        volume.append( image )\n    \n    volume = np.stack(volume, axis=0)\n    return volume\n\ndef create_3D_segmentations(filepath, downsample_rate=1):\n    img = nib.load(filepath).get_fdata()\n    img = np.transpose(img, [2, 1, 0])\n    img = np.rot90(img, -1, (1,2))\n    img = img[::-1,:,:]\n    img = np.transpose(img, [2, 1, 0])\n    img = img[::downsample_rate, ::downsample_rate, ::downsample_rate]\n    return img\n\ndef dicom_to_image(dicom_image):\n    \"\"\"\n    Read the dicom file and preprocess appropriately.\n    \"\"\"\n    pixel_array = dicom_image.pixel_array\n    \n    if dicom_image.PixelRepresentation == 1:\n        bit_shift = dicom_image.BitsAllocated - dicom_image.BitsStored\n        dtype = pixel_array.dtype \n        new_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n        pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dicom_image)\n    \n    if dicom_image.PhotometricInterpretation == \"MONOCHROME1\":\n        pixel_array = 1 - pixel_array\n    \n    # transform to hounsfield units\n    intercept = dicom_image.RescaleIntercept\n    slope = dicom_image.RescaleSlope\n    pixel_array = pixel_array * slope + intercept\n    \n    # windowing\n    window_center = int(dicom_image.WindowCenter)\n    window_width = int(dicom_image.WindowWidth)\n    img_min = window_center - window_width // 2\n    img_max = window_center + window_width // 2\n    pixel_array = pixel_array.copy()\n    pixel_array[pixel_array < img_min] = img_min\n    pixel_array[pixel_array > img_max] = img_max\n    \n    # normalization\n    pixel_array = (pixel_array - pixel_array.min())/(pixel_array.max() - pixel_array.min())\n    \n    return (pixel_array * 255).astype(np.uint8)","metadata":{"execution":{"iopub.status.busy":"2023-08-21T14:55:56.374186Z","iopub.execute_input":"2023-08-21T14:55:56.374699Z","iopub.status.idle":"2023-08-21T14:55:56.395592Z","shell.execute_reply.started":"2023-08-21T14:55:56.374653Z","shell.execute_reply":"2023-08-21T14:55:56.393899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seg = glob(f'{PATH}/segmentations/*.nii')\ntrain = pd.read_csv(f'{PATH}/train.csv')\ntrain_series_meta = pd.read_csv(f'{PATH}/train_series_meta.csv')\ndf = pd.merge(train_series_meta, train, how='inner', on='patient_id')\n# newshape\nimage_size = 512\nnew_shape = (image_size, image_size, 100)\nfor i,_ in enumerate(seg[0:]):\n    num = seg[i][-10:]\n    series_id = int(re.sub(r\"\\D\", \"\", num))\n    print('series_id:', series_id)\n    seg_filepath = f'{PATH}/segmentations/{series_id}.nii'\n\n    patient_id = df[df['series_id'] == series_id]['patient_id'].iloc[0]\n    filepath = f'{PATH}/train_images/{patient_id}/{series_id}'\n    volume = create_3D_scans(filepath)\n    volume = volume.transpose(1, 2, 0)\n    volume_seg = create_3D_segmentations(seg_filepath)\n    volume_seg = np.where(volume_seg>0,1,0)\n    \n    print('Shape before processed:', volume.shape)\n    old_shape = volume.shape\n    zoom_factor = [new_shape[i] / old_shape[i] for i in range(len(old_shape))]\n\n    # zoom(volume.shape become 512x512x512)\n    volume = zoom(volume, zoom_factor, order=3)\n    volume_seg = zoom(volume_seg, zoom_factor, order=3)\n    print('Shape after processed:', volume.shape)\n    np.savez_compressed('/kaggle/working/rsna-same-shape_19_29/input/{0}'.format(series_id), volume)\n    np.savez_compressed('/kaggle/working/rsna-same-shape_19_29/mask/{0}'.format(series_id), volume_seg)","metadata":{"_kg_hide-output":false,"execution":{"iopub.status.busy":"2023-08-21T14:55:56.731262Z","iopub.execute_input":"2023-08-21T14:55:56.732606Z","iopub.status.idle":"2023-08-21T15:43:47.157403Z","shell.execute_reply.started":"2023-08-21T14:55:56.732546Z","shell.execute_reply":"2023-08-21T15:43:47.154941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n\nshutil.make_archive('/kaggle/working/rsna-same-shape_19_29', format='zip', root_dir='/kaggle/working/rsna-same-shape_19_29')","metadata":{"execution":{"iopub.status.busy":"2023-08-21T07:32:13.387388Z","iopub.execute_input":"2023-08-21T07:32:13.387818Z","iopub.status.idle":"2023-08-21T07:32:21.842307Z","shell.execute_reply.started":"2023-08-21T07:32:13.387784Z","shell.execute_reply":"2023-08-21T07:32:21.84104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_data = np.load('/kaggle/working/rsna-same-shape_19_29/input/42173.npz')['arr_0']\nmask_data = np.load('/kaggle/working/rsna-same-shape_19_29/mask/42173.npz')['arr_0']\n\nz = 20\nprint(raw_data.shape, mask_data.shape)\nfig = plt.figure()\n\nax1 = fig.add_subplot(121)\nax1.imshow(raw_data[:,:,z], cmap='gray')\n\nax2 = fig.add_subplot(122)\nax2.imshow(mask_data[:,:,z], cmap='gray')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-21T08:23:06.519387Z","iopub.execute_input":"2023-08-21T08:23:06.519826Z","iopub.status.idle":"2023-08-21T08:23:07.891054Z","shell.execute_reply.started":"2023-08-21T08:23:06.519793Z","shell.execute_reply":"2023-08-21T08:23:07.890238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"images_dir ='/kaggle/working/rsna-same-shape_19_29/input'\nmasks_dir = '/kaggle/working/rsna-same-shape_19_29/mask' # left kidney\norgans=['bowel','liver','spleen','right kidney','left kidney']\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-08-21T09:52:51.054523Z","iopub.execute_input":"2023-08-21T09:52:51.055008Z","iopub.status.idle":"2023-08-21T09:52:51.063544Z","shell.execute_reply.started":"2023-08-21T09:52:51.054931Z","shell.execute_reply":"2023-08-21T09:52:51.061427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"images_listdir0 = sorted(os.listdir(images_dir))\nimages_listdir=[]\nfor i in range(len(images_listdir0)):\n    images_listdir+=[images_listdir0[i]]\nimages_listdir= images_listdir[0:100]\nmasks_listdir = sorted(os.listdir(masks_dir))\nN=list(range(9))\nrandom_N = N\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-08-21T09:52:51.065213Z","iopub.execute_input":"2023-08-21T09:52:51.065654Z","iopub.status.idle":"2023-08-21T09:52:51.08082Z","shell.execute_reply.started":"2023-08-21T09:52:51.065614Z","shell.execute_reply":"2023-08-21T09:52:51.07977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"image_size = 512\nMASKS=np.zeros((1,image_size, image_size, 100), dtype=np.uint8)\nIMAGES=np.zeros((1,image_size, image_size, 100),dtype=np.uint8)\n\nfor j,file in enumerate(images_listdir):   ##the smaller, the faster\n    try:\n        image = np.load(f\"{images_dir}/{file}\")['arr_0']\n        image_ex = np.expand_dims(image, axis=0)\n        print(image_ex.shape)\n        IMAGES = np.vstack([IMAGES, image_ex])\n        mask = np.load(f\"{masks_dir}/{masks_listdir[j]}\")['arr_0'] \n        #mask = cv2.cvtColor(mask, cv2.COLOR_BGR2GRAY)\n        #mask = mask.reshape(512,512,1)\n        mask_ex = np.expand_dims(mask, axis=0)    \n        MASKS = np.vstack([MASKS, mask_ex])\n    except:\n        print(file)\n        continue\nnumber=120\nimages=np.array(IMAGES)[1:number+1]\nmasks=np.array(MASKS)[1:number+1]\nimages = images[:,:,:,:,np.newaxis]\nmasks = masks[:,:,:,:,np.newaxis]\nprint(images.shape,masks.shape)\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-08-21T09:52:51.083025Z","iopub.execute_input":"2023-08-21T09:52:51.083618Z","iopub.status.idle":"2023-08-21T09:57:36.86373Z","shell.execute_reply.started":"2023-08-21T09:52:51.083585Z","shell.execute_reply":"2023-08-21T09:57:36.858294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from sklearn.model_selection import train_test_split\n#images_train, images_test, masks_train, masks_test = train_test_split(images, masks, test_size=0.4, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-08-21T09:59:15.842513Z","iopub.execute_input":"2023-08-21T09:59:15.84301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}