{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":52254,"databundleVersionId":6863140,"sourceType":"competition"},{"sourceId":6720593,"sourceType":"datasetVersion","datasetId":3871992},{"sourceId":6945297,"sourceType":"datasetVersion","datasetId":3988720},{"sourceId":7026096,"sourceType":"datasetVersion","datasetId":4040781},{"sourceId":7041459,"sourceType":"datasetVersion","datasetId":4051279},{"sourceId":7278556,"sourceType":"datasetVersion","datasetId":4220041}],"dockerImageVersionId":30559,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import library","metadata":{}},{"cell_type":"code","source":"import os\nimport math\nimport gc\n# You can use `tensorflow`, `pytorch`, `jax` here\n# KerasCore makes the notebook backend agnostic :)\n# os.environ[\"KERAS_BACKEND\"] = \"tensorflow\"\n\n# import keras_cv\n# import keras_core as keras\n# from keras_core import layers\n\nimport torch\nfrom torch import nn\nfrom torch.utils.data import DataLoader\nfrom torch.utils.data import Dataset\nfrom PIL import Image\nimport torchvision\nfrom torchvision import transforms\n\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\n#import tensorflow as tf\nfrom matplotlib import pyplot as plt\nfrom sklearn.model_selection import train_test_split\nimport nibabel as niba\nimport cv2\nfrom tqdm.notebook import tqdm\nfrom glob import glob\nimport pydicom\nfrom joblib import Parallel, delayed\nimport pydicom\nimport xarray as xr\nimport matplotlib.patches as patches","metadata":{"execution":{"iopub.status.busy":"2023-12-27T02:12:21.760995Z","iopub.execute_input":"2023-12-27T02:12:21.76137Z","iopub.status.idle":"2023-12-27T02:12:21.85986Z","shell.execute_reply.started":"2023-12-27T02:12:21.761333Z","shell.execute_reply":"2023-12-27T02:12:21.858891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(os.listdir('/kaggle/input/oknice/kidney/train'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare to train","metadata":{}},{"cell_type":"markdown","source":"## Dataset","metadata":{}},{"cell_type":"markdown","source":"## Create png from dicom","metadata":{}},{"cell_type":"markdown","source":"Convert dicom to png","metadata":{}},{"cell_type":"code","source":"def standardize_pixel_array(dcm):\n    # Correct DICOM pixel_array if PixelRepresentation == 1.\n    pixel_array = dcm.pixel_array\n    if dcm.PixelRepresentation == 1:\n        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n        dtype = pixel_array.dtype \n        new_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n        pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dcm)\n    return pixel_array\n\ndef read_xray(path, fix_monochrome=True):\n    dicom = pydicom.dcmread(path)\n    data = standardize_pixel_array(dicom)\n    data = data - np.min(data)\n    data = data / (np.max(data) + 1e-5)\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = 1.0 - data\n    return data\n\ndef resize_and_save(df, idx):\n    path = df['image_path'][idx].split('/')\n    set_type = path[4]\n    patient_id = path[5]\n    series_id = path[6]\n\n    kidney = 0\n    liver = 0\n    spleen = 0\n    \n    if set_type == 'train_images':\n        DF = patient_df[patient_df['patient_id'] == int(patient_id)]\n        #For kidney\n        if DF['kidney_low'].values[0] == 1:\n            kidney = 1\n        elif DF['kidney_high'].values[0] == 1:\n            kidney = 2\n\n        #For liver\n        if DF['liver_low'].values[0] == 1:\n            liver = 1\n        elif DF['liver_high'].values[0] == 1:\n            liver = 2\n\n        #For spleen\n        if DF['spleen_low'].values[0] == 1:\n            spleen = 1\n        elif DF['spleen_high'].values[0] == 1:\n            spleen = 2\n\n        save_folder = f'{BASE_PATH}/{set_type}/{patient_id}_{kidney}_{liver}_{spleen}/{series_id}'\n    else:\n        save_folder = f'{BASE_PATH}/{set_type}/{patient_id}/{series_id}'\n    os.makedirs(save_folder, exist_ok = True)\n#     print(df.instance_number[idx])\n#     print(save_folder)\n    if os.path.exists(f'{save_folder}/{df.instance_number[idx]}.png'):\n        return\n    img = read_xray(df.dicom_path[idx])\n    #h, w = img.shape[:2]  # orig hw\n    img = cv2.resize(img, config.IMAGE_SIZE, cv2.INTER_LINEAR)\n    img = (img * 255).astype(np.uint8)\n    cv2.imwrite(f'{save_folder}/{df.instance_number[idx]}.png', img)\n    return","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p1 = Parallel(n_jobs=2, backend=\"threading\")(\n    delayed(resize_and_save)(test_df, test_idx) for test_idx in tqdm(range(len(test_df)), leave=True, position=0)\n    \n)\n#!zip -m test.zip /kaggle/working/Group_Project/test_images\np2 = Parallel(n_jobs=2, backend=\"threading\")(\n    delayed(resize_and_save)(train_df, train_idx) for train_idx in tqdm(range(len(train_df)), leave=True, position=0)\n)\n#!zip -m train_up.zip /kaggle/working/Group_Project/train_images\n\ndel p1, p2\ngc.collect()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install kaggle","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!touch kaggle.json","metadata":{"execution":{"iopub.status.busy":"2023-12-26T16:00:30.585631Z","iopub.execute_input":"2023-12-26T16:00:30.58604Z","iopub.status.idle":"2023-12-26T16:00:31.6014Z","shell.execute_reply.started":"2023-12-26T16:00:30.586005Z","shell.execute_reply":"2023-12-26T16:00:31.599966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %load kaggle.json\n","metadata":{"execution":{"iopub.status.busy":"2023-12-25T15:45:17.217701Z","iopub.execute_input":"2023-12-25T15:45:17.218102Z","iopub.status.idle":"2023-12-25T15:45:17.224845Z","shell.execute_reply.started":"2023-12-25T15:45:17.218067Z","shell.execute_reply":"2023-12-25T15:45:17.223854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile kaggle.json\n{\"username\":\"alonelymess\",\"key\":\"d3d07dd9250b1493af76bd923d0f3484\"}","metadata":{"execution":{"iopub.status.busy":"2023-12-26T16:00:32.528873Z","iopub.execute_input":"2023-12-26T16:00:32.530222Z","iopub.status.idle":"2023-12-26T16:00:32.540289Z","shell.execute_reply.started":"2023-12-26T16:00:32.530179Z","shell.execute_reply":"2023-12-26T16:00:32.538752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir '/kaggle/working/right kidney'","metadata":{"execution":{"iopub.status.busy":"2023-12-26T11:02:44.813121Z","iopub.execute_input":"2023-12-26T11:02:44.813466Z","iopub.status.idle":"2023-12-26T11:02:45.76618Z","shell.execute_reply.started":"2023-12-26T11:02:44.813433Z","shell.execute_reply":"2023-12-26T11:02:45.765227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!kaggle datasets init -p '/kaggle/working/right kidney'","metadata":{"execution":{"iopub.status.busy":"2023-12-27T04:40:44.301512Z","iopub.execute_input":"2023-12-27T04:40:44.30181Z","iopub.status.idle":"2023-12-27T04:40:45.768552Z","shell.execute_reply.started":"2023-12-27T04:40:44.301785Z","shell.execute_reply":"2023-12-27T04:40:45.767371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %load '/kaggle/working/liver/dataset-metadata.json'\n{\n  \"title\": \"INSERT_TITLE_HERE\",\n  \"id\": \"alonelymess/INSERT_SLUG_HERE\",\n  \"licenses\": [\n    {\n      \"name\": \"CC0-1.0\"\n    }\n  ]\n}","metadata":{"execution":{"iopub.status.busy":"2023-12-25T15:46:56.191132Z","iopub.execute_input":"2023-12-25T15:46:56.192006Z","iopub.status.idle":"2023-12-25T15:46:56.19916Z","shell.execute_reply.started":"2023-12-25T15:46:56.191962Z","shell.execute_reply":"2023-12-25T15:46:56.198078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile '/kaggle/working/right kidney/dataset-metadata.json'\n{\n  \"title\": \"rightkidney1\",\n  \"id\": \"alonelymess/rightkidney1\",\n  \"licenses\": [\n    {\n      \"name\": \"CC0-1.0\"\n    }\n  ]\n}","metadata":{"execution":{"iopub.status.busy":"2023-12-27T04:40:45.769984Z","iopub.execute_input":"2023-12-27T04:40:45.770295Z","iopub.status.idle":"2023-12-27T04:40:45.778143Z","shell.execute_reply.started":"2023-12-27T04:40:45.770267Z","shell.execute_reply":"2023-12-27T04:40:45.777158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!kaggle datasets create -p '/kaggle/working/right kidney' -r zip","metadata":{"execution":{"iopub.status.busy":"2023-12-27T04:40:45.780072Z","iopub.execute_input":"2023-12-27T04:40:45.780704Z","iopub.status.idle":"2023-12-27T04:40:47.06288Z","shell.execute_reply.started":"2023-12-27T04:40:45.780674Z","shell.execute_reply":"2023-12-27T04:40:47.06166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"","metadata":{}},{"cell_type":"markdown","source":"# Prepare dataset (╬▔皿▔)╯","metadata":{}},{"cell_type":"code","source":"organ = 'kidney'\nos.makedirs(f'/kaggle/working/data/{organ}_healthy', exist_ok = True)\nos.makedirs(f'/kaggle/working/data/{organ}_low', exist_ok = True)\nos.makedirs(f'/kaggle/working/data/{organ}_high', exist_ok = True)\nos.makedirs(f'/kaggle/working/data/train', exist_ok = True)\nos.makedirs(f'/kaggle/working/data/test', exist_ok = True)\nos.makedirs(f'/kaggle/working/data/val', exist_ok = True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\np = Path('/kaggle/input/blahblah/train_images')\nimages_path = list(p.rglob('*.png'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"str(images_path[0])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\nfor image_path in images_path:\n    path = str(image_path).split('/')\n    info = path[5]\n    info = info.split('_')\n    patient_id = info[0]\n    kidney = info[1]\n    series_id = path[6]\n    instance = path[7]\n    \n    if kidney == '0':\n        shutil.copy(image_path, f'/kaggle/working/data/{organ}_healthy/{patient_id}_{series_id}_{instance}')\n    elif kidney == '1':\n        shutil.copy(image_path, f'/kaggle/working/data/{organ}_low/{patient_id}_{series_id}_{instance}')\n    elif kidney == '2':\n        shutil.copy(image_path, f'/kaggle/working/data/{organ}_high/{patient_id}_{series_id}_{instance}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Handle the split for each group \\(￣︶￣*\\))   \n\nnp.random.seed(config.SEED)\nnp.random.shuffle(np.array(images_path))\ntrain_images, test_images = images_path[80:], images_path[:80]\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def move_to_folder(imgs, set_type):\n    for image_path in tqdm(imgs):\n        path = str(image_path).split('/')\n        info = path[5]\n        info = info.split('_')\n        patient_id = info[0]\n        kidney = info[1]\n        series_id = path[6]\n        instance = path[7]\n        \n        os.makedirs(f'/kaggle/working/data/{set_type}/{organ}_healthy', exist_ok = True)\n        os.makedirs(f'/kaggle/working/data/{set_type}/{organ}_low', exist_ok = True)\n        os.makedirs(f'/kaggle/working/data/{set_type}/{organ}_high', exist_ok = True)\n        if kidney == '0':\n            shutil.move(f'/kaggle/working/data/{organ}_healthy/{patient_id}_{series_id}_{instance}',\n                        f'/kaggle/working/data/{set_type}/{organ}_healthy/{patient_id}_{kidney}_{series_id}_{instance}')\n        elif kidney == '1':\n            shutil.move(f'/kaggle/working/data/{organ}_low/{patient_id}_{series_id}_{instance}',\n                        f'/kaggle/working/data/{set_type}/{organ}_low/{patient_id}_{kidney}_{series_id}_{instance}')\n        elif kidney == '2':\n            shutil.move(f'/kaggle/working/data/{organ}_high/{patient_id}_{series_id}_{instance}',\n                        f'/kaggle/working/data/{set_type}/{organ}_high/{patient_id}_{kidney}_{series_id}_{instance}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"move_to_folder(test_images, 'test')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparing Object detection dataset","metadata":{}},{"cell_type":"code","source":"paths=[]\nids=[]\nfor dirname, _, filenames in os.walk('/kaggle/input/rsna-2023-abdominal-trauma-detection/segmentations'):\n    for filename in filenames:\n        ids+=[filename[0:-4]]\n        paths+=[(os.path.join(dirname, filename))]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_bound(image, colors):\n    list_col = []\n    benchmark_colors=[1, 2, 3, 4]   \n#   organs=['liver','spleen', 'left kidney','right kidney']\n    for color in colors:\n        if color in benchmark_colors:\n            x_list = [i[1] for i in np.argwhere(image==color)]\n            y_list = [i[0] for i in np.argwhere(image==color)]\n            xmin = min(x_list)\n            xmax = max(x_list)\n            ymin = min(y_list)\n            ymax = max(y_list)\n            xcenter = (xmin + xmax)/2\n            xcenter = xcenter/image.shape[1]\n            ycenter = (ymin + ymax)/2\n            ycenter = ycenter/image.shape[0]\n            width = ymax - ymin\n            width = width/image.shape[1]\n            height = xmax - xmin\n            height = height/image.shape[0]\n            label_org = benchmark_colors.index(color)\n            list_col.append((label_org, xcenter, ycenter, width, height))\n#         list_col.append((xmin, xmax, ymin, ymax))\n    return list_col","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.makedirs('/kaggle/working/train', exist_ok = True)\nos.makedirs('/kaggle/working/test', exist_ok = True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use number in segmentation to find imgs in dicom\ndicom_path = list((src/ 'train_images').glob(f'*/10494'))[0]\npatient_id, series_id = dicom_path.parent.name, dicom_path.name\nprint('patient_id is ' + patient_id, 'series_id is ' +series_id + ' in Segmentation folder')\n\nimgs = []\nfiles = sorted(\n    list((src / 'train_images' / patient_id / series_id).glob('*.dcm')),\n    key = lambda x : int(x.stem)\n)\nfor f in files:\n    imgs.append(pydicom.dcmread(f).pixel_array)\nimgs = np.stack(imgs)\n\nmasks = niba.load('/kaggle/input/rsna-2023-abdominal-trauma-detection/segmentations/10494.nii').get_fdata()\nmasks = masks.transpose((2, 1, 0))[::-1, ::-1, :] # we change the order of the dimensions. Flip horizontally and vertically.\n\nxarr_imgs = xr.DataArray(\n    imgs,\n    dims = ['file', 'height', 'width'],\n    coords = [\n        [f.name for f in files],\n        [i for i in range(512)],\n        [i for i in range(512)],\n    ]\n)\n\nxarr_masks = xr.DataArray(\n    masks,\n    dims = ['file', 'height', 'width'],\n    coords = [\n        [f.name for f in files],\n        [i for i in range(512)],\n        [i for i in range(512)],\n    ]\n)\n\nfor j in tqdm(range(xarr_masks.shape[0])):\n    # Read segment\n    seg_img = (xarr_masks.isel(file = j).values).astype(np.uint8)\n    colors_exist = np.unique(seg_img)\n\n    if len(colors_exist)>2: # If more than 1 organ, >2 because we ignore bowel\n        bboxes = get_bound(seg_img, colors_exist)\n        fig, ax = plt.subplots(1, 2, figsize = (12, 6))\n        ax[0].imshow(seg_img)\n        ax[1].imshow(xarr_imgs.isel(file=j).values, cmap = 'gray')\n        plt.axis('off')\n        for bbox in bboxes:\n            xcenter = bbox[1]*img.shape[1]\n            ycenter = bbox[2]*img.shape[0]\n            width = bbox[3]*img.shape[1]\n            height = bbox[4]*img.shape[0]\n        \n            # Calculate the top-left and bottom-right coordinates of the bounding box\n            x1 = int(xcenter - width / 2)\n            y1 = int(ycenter - height / 2)\n            x2 = int(xcenter + width / 2)\n            y2 = int(ycenter + height / 2)\n\n            # Draw the bounding box on the image\n            cv2.rectangle(image, (x1, y1), (x2, y2), (0, 255, 0), 2)\n        plt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(masks)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"src = Path('/kaggle/input/rsna-2023-abdominal-trauma-detection')\ncolors=[1, 2, 4, 3]\norgans=['liver','spleen','right kidney','left kidney']\nfor i in tqdm(range(len(ids))):\n    # use number in segmentation to find imgs in dicom\n    dicom_path = list((src/ 'train_images').glob(f'*/{ids[i]}'))[0]\n    patient_id, series_id = dicom_path.parent.name, dicom_path.name\n    print('patient_id is ' + patient_id, 'series_id is ' +series_id + ' in Segmentation folder')\n    \n    imgs = []\n    files = sorted(\n        list((src / 'train_images' / patient_id / series_id).glob('*.dcm')),\n        key = lambda x : int(x.stem)\n    )\n    for f in files:\n        imgs.append(pydicom.dcmread(f).pixel_array)\n    imgs = np.stack(imgs)\n    \n    masks = niba.load(paths[i]).get_fdata()\n    masks = masks.transpose((2, 1, 0))[::-1, ::-1, :] # we change the order of the dimensions. Flip horizontally and vertically.\n    \n    xarr_imgs = xr.DataArray(\n        imgs,\n        dims = ['file', 'height', 'width'],\n        coords = [\n            [f.name for f in files],\n            [i for i in range(512)],\n            [i for i in range(512)],\n        ]\n    )\n\n    xarr_masks = xr.DataArray(\n        masks,\n        dims = ['file', 'height', 'width'],\n        coords = [\n            [f.name for f in files],\n            [i for i in range(512)],\n            [i for i in range(512)],\n        ]\n    )\n    \n    for j in tqdm(range(xarr_masks.shape[0])):\n        # Read segment\n        seg_img = (xarr_masks.isel(file = j).values).astype(np.uint8)\n        colors_exist = np.unique(seg_img)\n                \n        if len(colors_exist)>2 and 1 not in colors_exist: # If more than 1 organ\n            fig, ax = plt.subplots(1, 2, figsize = (12, 6))\n            ax[0].imshow(seg_img)\n            ax[1].imshow(xarr_imgs.isel(file=j).values, cmap = 'gray')\n            plt.axis('off')\n            plt.show()\n            print(colors_exist)\n#             label_box = get_bound(seg_img, colors_exist)\n#             print(label_box)\n#             focus = xarr_imgs.isel(file=j).values\n#             newfile= '/kaggle/working/sg/image/'+slice_id+'.jpg'\n#             #cv2.imwrite(newfile,focus)\n#             label_box = get_bound(hsv, color_exist)\n#             with open(f'/kaggle/working/sg/label/{slice_id}.txt', 'w',encoding='UTF-8') as file:\n#                 for box in label_box:\n#                     file.write(f'{box[0]} {box[1]} {box[2]} {box[3]} {box[4]}\\n')\n            \n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images=[]\ncolors=[(51, 51, 51), (255, 255, 255), (102, 102, 102), (204, 204, 204), (153, 153, 153)]\norgans=['bowel','liver','spleen','right kidney','left kidney']\nfor path in paths:\n    if path.split('/')[-1]=='51136.nii':\n        nii_img = niba.load(path)\n        data = nii_img.get_fdata()\n        print(data.shape)\n        imagesi=[]\n        for i in range(data.shape[2]):\n            \n            #Read segment\n            seg_img=(data[:,:,i]*51).astype(np.uint8)\n            #image = cv2.imread(path0, cv2.IMREAD_COLOR)\n            seg_image = cv2.cvtColor(seg_img, cv2.COLOR_BGR2RGB)\n            hsv = cv2.cvtColor(seg_image, cv2.COLOR_BGR2HSV)\n            pixels = np.argwhere(hsv>0)\n            all_colors = {tuple(image[pixel[0], pixel[1]]) for pixel in pixels}\n            \n            #if img.max()>0 and len(all_colors)>=4:\n            fig, axs = plt.subplots(1, 2, figsize = (12, 6))\n            #print(img.max())\n            img=np.rot90(img)\n            axs[0].imshow(img)\n            axs[1].imshow(read_xray(test_path + '/' + str(i+sorted(test)[0])+'.dcm'), cmap = 'gray')\n            plt.axis('off')\n            plt.show()\n            newfile='10494_seg_'+str(i).zfill(4)+'.png'\n            print(test_path + '/' + str(i+sorted(test)[0])+'.dcm')\n            print(newfile)\n            #cv2.imwrite(newfile,img)\n            imagesi+=[img]\n        print(len(imagesi))\n        images+=[imagesi]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os \nfor file in os.listdir('/kaggle/input/sggggg/sg/image')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training time ☆*: .｡. o(≧▽≦)o .｡.:*☆","metadata":{}},{"cell_type":"markdown","source":"## Training object detection","metadata":{}},{"cell_type":"code","source":"!pip install ultralytics --q\n\nfrom IPython import display\ndisplay.clear_output()\n\nimport ultralytics\nultralytics.checks()\nfrom ultralytics import YOLO","metadata":{"execution":{"iopub.status.busy":"2023-12-27T02:12:21.862186Z","iopub.execute_input":"2023-12-27T02:12:21.862574Z","iopub.status.idle":"2023-12-27T02:12:38.711086Z","shell.execute_reply.started":"2023-12-27T02:12:21.862537Z","shell.execute_reply":"2023-12-27T02:12:38.710126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Splitting data\nfrom sklearn.model_selection import train_test_split\nimport shutil\n\nbase_img_path = '/kaggle/input/objectdetect/images'\nsegment_path = '/kaggle/input/objectdetect/labels'\n\nimg_file = os.listdir(base_img_path)\nsegment_file = os.listdir(segment_path)\n\n# Sort the list based on the numeric value\nsorted_img = sorted(img_file, key=lambda x: x[:-4])\nsorted_segment = sorted(segment_file, key=lambda x: x[:-4])\n\nX_train, x_test, y_train, y_test = train_test_split(sorted_img, sorted_segment, test_size=0.2, random_state=42)\n\nx_train, x_val, y_train, y_val = train_test_split(X_train, y_train, test_size=1/8, random_state=42)\n\ndef create_folder(X, Y, set_type):\n    img_path = f'/kaggle/working/{set_type}/images'\n    label_path = f'/kaggle/working/{set_type}/labels'\n    os.makedirs(img_path,  exist_ok = True)\n    os.makedirs(label_path,  exist_ok = True)\n    for x, y in zip(tqdm(X), Y):\n        old_x_path = base_img_path+'/'+x\n        new_x_path = img_path+'/' + x\n        if os.path.exists(new_x_path):\n            continue\n        shutil.copy(old_x_path, new_x_path)\n\n        old_y_path = segment_path+'/'+y\n        new_y_path = label_path+'/' + y\n        shutil.copy(old_y_path, new_y_path)\n\n\ncreate_folder(x_train, y_train, 'train')\ncreate_folder(x_test, y_test, 'test')\ncreate_folder(x_val, y_val, 'val')","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:13:11.915189Z","iopub.execute_input":"2023-12-25T14:13:11.915631Z","iopub.status.idle":"2023-12-25T14:21:13.450919Z","shell.execute_reply.started":"2023-12-25T14:13:11.91559Z","shell.execute_reply":"2023-12-25T14:21:13.449904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('/kaggle/working/data.yaml', 'w') as data:\n    data.write(\n    \"\"\"\n    #Classes\n    nc: 4 #num of classes\n    names: ['liver','spleen', 'left kidney','right kidney']\n    \n    test: /kaggle/working/test/images # test images\n    train: /kaggle/working/train/images  # train images\n    val: /kaggle/working/val/images # val images\n    \"\"\")","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:22:00.773485Z","iopub.execute_input":"2023-12-25T14:22:00.773894Z","iopub.status.idle":"2023-12-25T14:22:00.779645Z","shell.execute_reply.started":"2023-12-25T14:22:00.773856Z","shell.execute_reply":"2023-12-25T14:22:00.778744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#yolov8n(nano version for testing)\n!yolo task=detect mode=train model=yolov8n.pt data='/kaggle/working/data.yaml' epochs=50","metadata":{"execution":{"iopub.status.busy":"2023-12-25T12:54:09.662714Z","iopub.status.idle":"2023-12-25T12:54:09.663075Z","shell.execute_reply.started":"2023-12-25T12:54:09.662914Z","shell.execute_reply":"2023-12-25T12:54:09.66293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!yolo task=detect mode=val model='/kaggle/input/yolov8n-50-epochs-with-default-params/best_50_23_11_2023.pt' data='/kaggle/working/data.yaml' plots = True save = True","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:22:02.558761Z","iopub.execute_input":"2023-12-25T14:22:02.559636Z","iopub.status.idle":"2023-12-25T14:22:41.528264Z","shell.execute_reply.started":"2023-12-25T14:22:02.559602Z","shell.execute_reply":"2023-12-25T14:22:41.526479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import display, Image","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:22:44.922806Z","iopub.execute_input":"2023-12-25T14:22:44.923692Z","iopub.status.idle":"2023-12-25T14:22:44.928248Z","shell.execute_reply.started":"2023-12-25T14:22:44.923656Z","shell.execute_reply":"2023-12-25T14:22:44.927217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Validation","metadata":{}},{"cell_type":"code","source":"val_path = '/kaggle/working/runs/detect/val2/'","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:23:02.439976Z","iopub.execute_input":"2023-12-25T14:23:02.440355Z","iopub.status.idle":"2023-12-25T14:23:02.44465Z","shell.execute_reply.started":"2023-12-25T14:23:02.440324Z","shell.execute_reply":"2023-12-25T14:23:02.44369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image(filename=val_path+'R_curve.png', width=600)\nplt.imshow(cv2.imread(val_path+'R_curve.png'))\nplt.axis('off')\nplt.savefig('/kaggle/working/R_curve.png')","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:28:04.190627Z","iopub.execute_input":"2023-12-25T14:28:04.191347Z","iopub.status.idle":"2023-12-25T14:28:05.177075Z","shell.execute_reply.started":"2023-12-25T14:28:04.191312Z","shell.execute_reply":"2023-12-25T14:28:05.176088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(cv2.imread(val_path+'P_curve.png'))\nplt.axis('off')\nplt.savefig('/kaggle/working/P_curve.png')","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:28:19.449646Z","iopub.execute_input":"2023-12-25T14:28:19.450487Z","iopub.status.idle":"2023-12-25T14:28:20.431449Z","shell.execute_reply.started":"2023-12-25T14:28:19.450453Z","shell.execute_reply":"2023-12-25T14:28:20.430476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(cv2.imread(val_path+'PR_curve.png'))\nplt.axis('off')\nplt.savefig('/kaggle/working/PR_curve.png')","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:28:27.319679Z","iopub.execute_input":"2023-12-25T14:28:27.320165Z","iopub.status.idle":"2023-12-25T14:28:28.292912Z","shell.execute_reply.started":"2023-12-25T14:28:27.320119Z","shell.execute_reply":"2023-12-25T14:28:28.292001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(cv2.imread(val_path+'F1_curve.png'))\nplt.axis('off')\nplt.savefig('/kaggle/working/F1_curve.png')","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:28:34.967222Z","iopub.execute_input":"2023-12-25T14:28:34.967985Z","iopub.status.idle":"2023-12-25T14:28:35.894598Z","shell.execute_reply.started":"2023-12-25T14:28:34.967952Z","shell.execute_reply":"2023-12-25T14:28:35.893626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(cv2.imread(val_path+'confusion_matrix_normalized.png'))\nplt.axis('off')\nplt.savefig('/kaggle/working/confusion_matrix_normalized.png')","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:28:52.303695Z","iopub.execute_input":"2023-12-25T14:28:52.304408Z","iopub.status.idle":"2023-12-25T14:28:54.026954Z","shell.execute_reply.started":"2023-12-25T14:28:52.304377Z","shell.execute_reply":"2023-12-25T14:28:54.02602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Image(filename=val_path+'val_batch1_pred.jpg', width=600)","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:23:03.224655Z","iopub.execute_input":"2023-12-25T14:23:03.225392Z","iopub.status.idle":"2023-12-25T14:23:03.247568Z","shell.execute_reply.started":"2023-12-25T14:23:03.225362Z","shell.execute_reply":"2023-12-25T14:23:03.246692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{}},{"cell_type":"code","source":"# !yolo task=detect mode=predict model='/kaggle/working/runs/detect/train/weights/best.pt' conf=0.45 source='/kaggle/working/test/images/*.jpg' save=True save_txt=True line_thickness=1 #hide_labels=True\nmodel = YOLO('/kaggle/working/runs/detect/train/weights/best.pt')\nresults = model.predict(source='/kaggle/working/test/images/*.jpg', conf=0.45, save=True, save_txt=True, verbose=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T08:06:22.133803Z","iopub.execute_input":"2023-11-23T08:06:22.134703Z","iopub.status.idle":"2023-11-23T08:08:25.455786Z","shell.execute_reply.started":"2023-11-23T08:06:22.134669Z","shell.execute_reply":"2023-11-23T08:08:25.454795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_path = '/kaggle/working/runs/detect/runs/detect/predict/labels'\nlabels = os.listdir(labels_path)\nlabel = labels[len(labels)//2 + 200]\nimg_name = label[:-4]\nimg = cv2.imread(f'/kaggle/input/objectdetect/images/{img_name}.jpg')\nnames =  ['liver','spleen', 'left kidney','right kidney']\n# Define the class labels and corresponding colors\nclass_colors = {\n    'liver': (255, 0, 0),    # Red\n    'spleen': (0, 255, 0),    # Green\n    'left kidney': (0, 0, 255),    # Blue\n    'right kidney': (255, 255, 0)   # Yellow\n}\n\nwith open(labels_path + '/' + label) as label_data:\n    bboxes = label_data.readlines()\n    for bbox in bboxes:\n        bbox = bbox.split(' ')\n        \n        # Define the text to be drawn\n        organ = int(bbox[0])\n        name = names[organ]\n        \n        # Calculate the top-left and bottom-right coordinates of the bounding box\n        xcenter = float(bbox[1])*img.shape[1]\n        ycenter = float(bbox[2])*img.shape[0]\n        width = float(bbox[3])*img.shape[1]\n        height = float(bbox[4])*img.shape[0]\n        \n        x1 = int(xcenter - width / 2)\n        y1 = int(ycenter - height / 2)\n        x2 = int(xcenter + width / 2)\n        y2 = int(ycenter + height / 2)\n\n        # Draw the bounding box on the image\n        cv2.rectangle(img, (x1, y1), (x2, y2), class_colors[name], 2)\n    \nplt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-23T08:41:08.783565Z","iopub.execute_input":"2023-11-23T08:41:08.784014Z","iopub.status.idle":"2023-11-23T08:41:09.039022Z","shell.execute_reply.started":"2023-11-23T08:41:08.783984Z","shell.execute_reply":"2023-11-23T08:41:09.038095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label","metadata":{"execution":{"iopub.status.busy":"2023-11-23T08:26:42.804491Z","iopub.execute_input":"2023-11-23T08:26:42.80513Z","iopub.status.idle":"2023-11-23T08:26:42.810867Z","shell.execute_reply.started":"2023-11-23T08:26:42.8051Z","shell.execute_reply":"2023-11-23T08:26:42.809905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(cv2.imread('/kaggle/working/runs/detect/runs/detect/predict/47263_22397_469.jpg'))","metadata":{"execution":{"iopub.status.busy":"2023-11-23T08:20:24.869993Z","iopub.execute_input":"2023-11-23T08:20:24.870416Z","iopub.status.idle":"2023-11-23T08:20:25.171016Z","shell.execute_reply.started":"2023-11-23T08:20:24.870388Z","shell.execute_reply":"2023-11-23T08:20:25.170028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2","metadata":{"execution":{"iopub.status.busy":"2023-12-25T10:08:59.815019Z","iopub.execute_input":"2023-12-25T10:08:59.815735Z","iopub.status.idle":"2023-12-25T10:08:59.819855Z","shell.execute_reply.started":"2023-12-25T10:08:59.8157Z","shell.execute_reply":"2023-12-25T10:08:59.818854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = YOLO('/kaggle/input/yolov8n-50-epochs-with-default-params/best_50_23_11_2023.pt')","metadata":{"execution":{"iopub.status.busy":"2023-12-26T01:01:28.012181Z","iopub.execute_input":"2023-12-26T01:01:28.012888Z","iopub.status.idle":"2023-12-26T01:01:28.176996Z","shell.execute_reply.started":"2023-12-26T01:01:28.012858Z","shell.execute_reply":"2023-12-26T01:01:28.176036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\nimport numpy as np\nnames =  ['liver','spleen', 'left kidney','right kidney']\ntarget_organ = 'kidney'\nif target_organ == 'kidney':\n    target = [names.index('left kidney'), names.index('right kidney')]\n\n# # Load the DICOM file\n# dcm_file = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/10004/21057/1086.dcm'\n# dcm_data = pydicom.dcmread(dcm_file)\nnames =  ['liver','spleen', 'left kidney','right kidney']\n# Extract the pixel data\n# image = np.moveaxis(np.stack([dcm_data.pixel_array.astype(np.uint8)]*3),0, 2)\nimages, labels, patient_ids, series_ids, paths = next(iter(dataloader))\nresult = model(images.tolist(), classes = target)\n# result = model(image, conf=0.25, save=False, save_txt=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images.numpy().shape","metadata":{"execution":{"iopub.status.busy":"2023-12-27T02:50:04.169699Z","iopub.execute_input":"2023-12-27T02:50:04.170343Z","iopub.status.idle":"2023-12-27T02:50:04.176664Z","shell.execute_reply.started":"2023-12-27T02:50:04.170309Z","shell.execute_reply":"2023-12-27T02:50:04.175645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for r in result[0]:\n    print(r)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result[0].names[3.0]","metadata":{"execution":{"iopub.status.busy":"2023-12-23T08:45:20.908412Z","iopub.execute_input":"2023-12-23T08:45:20.909239Z","iopub.status.idle":"2023-12-23T08:45:20.915035Z","shell.execute_reply.started":"2023-12-23T08:45:20.909206Z","shell.execute_reply":"2023-12-23T08:45:20.914112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1","metadata":{"execution":{"iopub.status.busy":"2023-12-23T08:30:11.469536Z","iopub.execute_input":"2023-12-23T08:30:11.469859Z","iopub.status.idle":"2023-12-23T08:30:11.477439Z","shell.execute_reply.started":"2023-12-23T08:30:11.469836Z","shell.execute_reply":"2023-12-23T08:30:11.476556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crop.shape","metadata":{"execution":{"iopub.status.busy":"2023-12-23T08:31:12.668858Z","iopub.execute_input":"2023-12-23T08:31:12.669706Z","iopub.status.idle":"2023-12-23T08:31:12.676693Z","shell.execute_reply.started":"2023-12-23T08:31:12.669668Z","shell.execute_reply":"2023-12-23T08:31:12.675626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crop=image[y1:y2, x1:x2]\nplt.imshow(crop)","metadata":{"execution":{"iopub.status.busy":"2023-12-23T08:31:01.195738Z","iopub.execute_input":"2023-12-23T08:31:01.196174Z","iopub.status.idle":"2023-12-23T08:31:01.483254Z","shell.execute_reply.started":"2023-12-23T08:31:01.196139Z","shell.execute_reply":"2023-12-23T08:31:01.482312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/55081/29350')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nlen(os.listdir('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/55081/29350'))","metadata":{"execution":{"iopub.status.busy":"2023-12-25T10:18:24.159979Z","iopub.execute_input":"2023-12-25T10:18:24.160266Z","iopub.status.idle":"2023-12-25T10:18:24.166944Z","shell.execute_reply.started":"2023-12-25T10:18:24.160243Z","shell.execute_reply":"2023-12-25T10:18:24.166028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv2.imread('/kaggle/input/blahblah/train_images/10004_1_0_2/21057/1000.png')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.moveaxis(np.stack([image]*3),0, 2).shape","metadata":{"execution":{"iopub.status.busy":"2023-12-25T13:02:40.062709Z","iopub.execute_input":"2023-12-25T13:02:40.063025Z","iopub.status.idle":"2023-12-25T13:02:40.069663Z","shell.execute_reply.started":"2023-12-25T13:02:40.062997Z","shell.execute_reply":"2023-12-25T13:02:40.068687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('/kaggle/input/balance-dataset/final_liver.csv')['image_path'][0].split('/')[-1][:-4]","metadata":{"execution":{"iopub.status.busy":"2023-12-25T15:17:25.552104Z","iopub.execute_input":"2023-12-25T15:17:25.55246Z","iopub.status.idle":"2023-12-25T15:17:25.974824Z","shell.execute_reply.started":"2023-12-25T15:17:25.55243Z","shell.execute_reply":"2023-12-25T15:17:25.97384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.makedirs(f'{save_path}/{target_organ}/{df[\"patient_id\"][i]}_{df[\"series_id\"][i]}', exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-25T15:31:15.963949Z","iopub.execute_input":"2023-12-25T15:31:15.964676Z","iopub.status.idle":"2023-12-25T15:31:15.970472Z","shell.execute_reply.started":"2023-12-25T15:31:15.964637Z","shell.execute_reply":"2023-12-25T15:31:15.969531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-12-26T00:57:29.530911Z","iopub.execute_input":"2023-12-26T00:57:29.531278Z","iopub.status.idle":"2023-12-26T00:57:29.539113Z","shell.execute_reply.started":"2023-12-26T00:57:29.531243Z","shell.execute_reply":"2023-12-26T00:57:29.537883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\nclass Data(Dataset):\n    def __init__(self, csv_file, organ):\n        self.data = pd.read_csv(csv_file)\n        self.labels = self.data[organ]\n        self.img_paths = self.data['image_path']\n    def __len__(self):\n        return len(self.labels)\n    def __getitem__(self, idx):\n        dcm_data = pydicom.dcmread(self.img_paths[idx])\n        image = dcm_data.pixel_array.astype(np.uint8)\n        image = cv2.resize(image, (512, 512), interpolation=cv2.INTER_AREA)\n        image = np.moveaxis([image]*3, 0, 2)\n        \n        return image, self.labels[idx], self.data['patient_id'][idx], self.data['series_id'][idx], self.img_paths[idx]\ndata = Data('/kaggle/input/balance-dataset/final_kidney.csv', 'kidney')\ndataloader = DataLoader(data, batch_size=512)","metadata":{"execution":{"iopub.status.busy":"2023-12-27T03:31:48.149657Z","iopub.execute_input":"2023-12-27T03:31:48.150632Z","iopub.status.idle":"2023-12-27T03:31:48.404729Z","shell.execute_reply.started":"2023-12-27T03:31:48.150566Z","shell.execute_reply":"2023-12-27T03:31:48.403943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\nfrom tqdm import tqdm\n\nsave_path = '/kaggle/working'\nmodel = YOLO('/kaggle/input/yolov8n-50-epochs-with-default-params/best_50_23_11_2023.pt')\nimage_path = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images'\n\norgan_dict = {'right kidney' : {'img_path':[], 'conf':[], 'label':[]},\n        'left kidney' : {'img_path':[], 'conf':[], 'label':[]},\n        'spleen' : {'img_path':[], 'conf':[], 'label':[]},\n        'liver' : {'img_path':[], 'conf':[], 'label':[]}}\n\nnames =  ['liver','spleen', 'left kidney','right kidney']\ntarget_organ = 'right kidney'\ndf = pd.read_csv(f'/kaggle/input/balance-dataset/final_kidney.csv')\n\n# for i, path in enumerate(tqdm(df['image_path'])):\nfor i,( images, labels, patient_ids, series_ids, paths) in enumerate(tqdm(dataloader)):\n    new_images = [image for image in images.numpy()]\n    images = new_images\n#     dcm_data = pydicom.dcmread(path)\n#     # Extract the pixel data\n#     image = np.moveaxis(np.stack([dcm_data.pixel_array.astype(np.uint8)]*3),0, 2)\n    target = names.index(target_organ)\n    resultses = model(images, classes = target, conf=0.25, verbose=False)[0]\n    \n    for results in resultses:\n        prev_conf = 0\n        if len(results) != 0:\n            for result in results:\n                klass = result.boxes.cls.item()\n                conf = result.boxes.conf.item()\n                if conf > prev_conf: \n                    prev_conf = conf\n                    x1, y1, x2, y2 = result.boxes.xyxy.cpu().numpy()[0]\n                    x1, y1, x2, y2 = int(x1), int(y1), int(x2), int(y2)\n\n            instance = paths[i].split('/')[-1][:-4]\n            os.makedirs(f'{save_path}/{target_organ}/{patient_ids[i]}_{series_ids[i]}', exist_ok=True)\n            crop=image[y1:y2, x1:x2]\n            img_path = f'{save_path}/{target_organ}/{patient_ids[i]}_{series_ids[i]}/{instance}.jpg'\n            organ_dict[target_organ]['img_path'].append(f'/{target_organ}/{patient_ids[i]}_{series_ids[i]}/{instance}.jpg')\n            organ_dict[target_organ]['conf'].append(round(conf, 5))\n            organ_dict[target_organ]['label'].append(labels[i])\n            if os.path.isfile(img_path):\n                continue\n            cv2.imwrite(img_path, crop)\n        \ncsv = pd.DataFrame(organ_dict[target_organ]).to_csv(save_path+'/right kidney/right_kidney.csv')             \n    ","metadata":{"execution":{"iopub.status.busy":"2023-12-27T03:31:50.363802Z","iopub.execute_input":"2023-12-27T03:31:50.364489Z","iopub.status.idle":"2023-12-27T04:40:44.299717Z","shell.execute_reply.started":"2023-12-27T03:31:50.364457Z","shell.execute_reply":"2023-12-27T04:40:44.298767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(os.listdir('/kaggle/working'))","metadata":{"execution":{"iopub.status.busy":"2023-12-23T09:42:35.415415Z","iopub.execute_input":"2023-12-23T09:42:35.415824Z","iopub.status.idle":"2023-12-23T09:42:35.424219Z","shell.execute_reply.started":"2023-12-23T09:42:35.415795Z","shell.execute_reply":"2023-12-23T09:42:35.423405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"right_csv = pd.DataFrame(organ_dict['right kidney']).to_csv('/kaggle/working/right_k.csv')\nleft_csv = pd.DataFrame(organ_dict['left kidney']).to_csv('/kaggle/working/left_k.csv')\nspleen_csv = pd.DataFrame(organ_dict['spleen']).to_csv('/kaggle/working/spleen.csv')\nliver_csv = pd.DataFrame(organ_dict['liver']).to_csv('/kaggle/working/liver_k.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-23T09:48:53.374412Z","iopub.execute_input":"2023-12-23T09:48:53.375473Z","iopub.status.idle":"2023-12-23T09:48:53.764264Z","shell.execute_reply.started":"2023-12-23T09:48:53.375429Z","shell.execute_reply":"2023-12-23T09:48:53.763462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(organ_dict['spleen'])","metadata":{"execution":{"iopub.status.busy":"2023-12-23T09:53:13.531087Z","iopub.execute_input":"2023-12-23T09:53:13.531756Z","iopub.status.idle":"2023-12-23T09:53:13.549754Z","shell.execute_reply.started":"2023-12-23T09:53:13.531721Z","shell.execute_reply":"2023-12-23T09:53:13.548862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(right_csv['img_path'])","metadata":{"execution":{"iopub.status.busy":"2023-12-23T09:44:34.401784Z","iopub.execute_input":"2023-12-23T09:44:34.402547Z","iopub.status.idle":"2023-12-23T09:44:34.409795Z","shell.execute_reply.started":"2023-12-23T09:44:34.402514Z","shell.execute_reply":"2023-12-23T09:44:34.408867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(right_csv['img_path'].unique())","metadata":{"execution":{"iopub.status.busy":"2023-12-23T09:44:44.110681Z","iopub.execute_input":"2023-12-23T09:44:44.111606Z","iopub.status.idle":"2023-12-23T09:44:44.132308Z","shell.execute_reply.started":"2023-12-23T09:44:44.111573Z","shell.execute_reply":"2023-12-23T09:44:44.131359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Trainning classification","metadata":{}},{"cell_type":"code","source":"class RandomCutout(object):\n    def __init__(self, height_factor=0.2, width_factor=0.2):\n        self.height_factor = height_factor\n        self.width_factor = width_factor\n\n    def __call__(self, image):\n        img_width =  image.size()[1]\n        img_height = image.size()[2]\n\n        cutout_height = int(img_height * self.height_factor)\n        cutout_width = int(img_width * self.width_factor)\n\n        top = torch.randint(low = 0, high = img_height - cutout_height, size = (1,))\n        left = torch.randint(low = 0, high = img_width - cutout_width, size = (1,))\n        bottom = top + cutout_height\n        right = left + cutout_width\n\n        if not isinstance(image, torch.Tensor):\n            image = transforms.functional.to_tensor(image)\n        image[:, top:bottom, left:right] = 0.0\n\n        return transforms.functional.to_pil_image(image)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomImageDataset(Dataset):\n    def __init__(self, path):\n        self.target = []\n        for file in os.listdir(path):\n            file = file.split('_')\n            patient_id = file[0]\n            organ = path.split('/')[4]\n            \n            level = ''\n            if organ == 'kidney':\n                level = file[1]\n            elif organ == 'liver':\n                level = file[2]\n            else:\n                level = file[3]\n                \n            self.target.append(self.one_hot(level))\n        self.target = np.array(self.target)\n        self.path = path\n        self.images_path = os.listdir(self.path)\n        \n    @staticmethod\n    def one_hot(target):\n        one_hot = np.zeros(3)\n        if target == '0':\n            one_hot[0] = 1\n        elif target == '1':\n            one_hot[1] = 1\n        else:\n            one_hot[2] = 1\n        return one_hot\n        \n\n    def __len__(self):\n        return len(self.target)\n\n    def __getitem__(self, idx):\n        image = Image.open(self.path + '/' + str(self.images_path[idx])).convert('RGB')\n        \n        # Need to change the image to match Resnext input\n        T = transforms.Compose([\n            transforms.Resize([256], torchvision.transforms.InterpolationMode.BILINEAR),\n            transforms.CenterCrop([224]),\n            transforms.ToTensor(),\n            transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n        ]) \n        image = T(image)\n        \n        label = torch.tensor(self.target[idx])\n        return image, label","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Kidney data","metadata":{}},{"cell_type":"code","source":"test_kidney_data = CustomImageDataset('/kaggle/input/okdata1/kidney/test')\ntrain_kidney_data = CustomImageDataset('/kaggle/input/okdata1/kidney/train')\n\ngenerator1 = torch.Generator().manual_seed(42)\ntrain_kidney_data, val_kidney_data = torch.utils.data.random_split(train_kidney_data, [0.8, 0.2], generator=generator1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Liver data","metadata":{}},{"cell_type":"code","source":"test_liver_data = CustomImageDataset('/kaggle/input/okdata1/liver/test')\ntrain_liver_data = CustomImageDataset('/kaggle/input/okdata1/liver/train')\n\ngenerator1 = torch.Generator().manual_seed(42)\ntrain_liver_data, val_liver_data = torch.utils.data.random_split(train_liver_data, [0.8, 0.2], generator=generator1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Spleen data","metadata":{}},{"cell_type":"code","source":"test_spleen_data = CustomImageDataset('/kaggle/input/okdata1/spleen/test')\ntrain_spleen_data = CustomImageDataset('/kaggle/input/okdata1/spleen/train')\n\ngenerator1 = torch.Generator().manual_seed(42)\ntrain_spleen_data, val_spleen_data = torch.utils.data.random_split(train_spleen_data, [0.8, 0.2], generator=generator1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Build model","metadata":{}},{"cell_type":"code","source":"BATCH_SIZE = 128","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Kidney\ntrain_kidney_dataloader = DataLoader(train_kidney_data, batch_size=BATCH_SIZE, num_workers=2)\nval_kidney_dataloader = DataLoader(val_kidney_data, batch_size=BATCH_SIZE, num_workers=2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Liver\ntrain_liver_dataloader = DataLoader(train_liver_data, batch_size=BATCH_SIZE, num_workers=2)\nval_liver_dataloader = DataLoader(val_liver_data, batch_size=BATCH_SIZE, num_workers=2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Spleen\ntrain_spleen_dataloader = DataLoader(train_spleen_data, batch_size=BATCH_SIZE, num_workers=2)\nval_spleen_dataloader = DataLoader(val_spleen_data, batch_size=BATCH_SIZE, num_workers=2)\n\n\n# total_train_steps = len(train_dataloader) * config.BATCH_SIZE * config.EPOCHS\n# warmup_steps = int(total_train_steps * 0.10)\n# decay_steps = total_train_steps - warmup_steps\n\n# print(f\"{total_train_steps=}\")\n# print(f\"{warmup_steps=}\")\n# print(f\"{decay_steps=}\")\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = (\n    \"cuda\"\n    if torch.cuda.is_available()\n    else \"mps\"\n    if torch.backends.mps.is_available()\n    else \"cpu\"\n)\nprint(f\"Using {device} device\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define model in pytorch\nclass NNet(nn.Module):\n    def __init__(self):\n        super().__init__()\n        # Backbone\n        #self.backbone = torchvision.models.resnext50_32x4d(weights = 'DEFAULT')\n        self.backbone = torchvision.models.resnet50(weights = 'DEFAULT')\n        # Define 'necks' for each head\n        self.layers = nn.Sequential(\n            nn.Linear(1000, 512),\n            nn.SiLU(),\n            nn.Linear(512, 256),\n            nn.SiLU(),\n            nn.Linear(256, 64),\n            nn.ReLU(),\n            nn.Linear(64, 32),\n            nn.ReLU(),\n            nn.Linear(32, 3)\n        )\n\n    def forward(self, x):\n        x = self.backbone(x)\n        out = self.layers(x)\n        return out","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kidney_model = NNet().to(device)\nliver_model = NNet().to(device)\nspleen_model = NNet().to(device)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define loss\nloss_fn = nn.CrossEntropyLoss()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Optimizer\n#import torch.optim.lr_scheduler as lr_scheduler\n# optimizer = torch.optim.Adam(model.parameters(), lr=1e-3)\n# optimizer_bowel = torch.optim.Adam(model.bowel.parameters(), lr=1e-3)\n# optimizer_extra = torch.optim.Adam(model.extra.parameters(), lr=1e-3)\noptimizer_kidney = torch.optim.SGD(kidney_model.parameters(), lr=0.1)\noptimizer_liver = torch.optim.SGD(liver_model.parameters(), lr=0.1)\noptimizer_spleen = torch.optim.SGD(spleen_model.parameters(), lr=0.1)\n\n# Define the cosine annealing learning rate scheduler\n# scheduler = lr_scheduler.CosineAnnealingLR(optimizer, T_max=10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_checkpoint(state, is_best, organ, epoch):\n    torch.save(state[f'{organ}_state_dict'], f'/kaggle/working/{organ}_resnet50model{epoch}_last.pth')\n    if is_best:\n        print('Found best')\n        torch.save(state[f'{organ}_state_dict'], f'/kaggle/working/{organ}_resnet50model{epoch}_best.pth')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.notebook import tqdm\nimport pandas as pd\n\ndef train(model, epoch, train_dataloader, val_dataloader, optimizer, loss_fn, organ, fine_tune = False):\n    if fine_tune == False:\n        best_acc = 0\n        result = {'epoch':[] ,'train_loss_total':[], \"val_loss_total\":[], 'acc_total':[]}\n    else:\n        r_prev = pd.read_csv(f'{organ}_resnet50result.csv')\n        best_acc = max(r_prev['acc_total'])\n        model.load_state_dict(torch.load(f\"/kaggle/working/{organ}_resnet50model{epoch}_best.pth\"))\n        result = r_prev.to_dict('list')\n    for e in tqdm(range(epoch)):\n        # Training loop\n        print(f\"Epoch {e+1}\")\n        result['epoch'].append(e+1)\n        print('Train-----')\n        loss_epoch = 0\n        model.train()\n\n        for batch_data, batch_labels in tqdm(train_dataloader):\n\n            batch_data = batch_data.to(device)\n            #batch_labels = batch_labels.movedim(0, 1)\n\n            # Forward pass\n            label = batch_labels.to(device)\n\n            output = model(batch_data)\n\n            # Optim\n            optimizer.zero_grad()\n            loss = loss_fn(output, label)\n            loss_epoch += loss.item()\n            loss.backward()\n            optimizer.step()\n            del loss\n\n\n        # Combined loss\n        loss_epoch /= len(batch_labels)\n        result['train_loss_total'].append(loss_epoch)\n\n        print('Loss: ', loss_epoch)\n\n        # Vaidating\n        size = len(val_dataloader.dataset)\n        num_batches = len(val_dataloader)\n        model.eval()\n\n\n        test_loss = 0\n        correct = 0\n    \n        with torch.no_grad():\n            print('Val------')\n            for X, y in tqdm(val_dataloader):\n\n                X = X.to(device)\n                pred = model(X)\n\n                label = y.to(device)    \n\n                test_loss += loss_fn(pred, label).item()\n\n                correct += (pred.argmax(1) == label.argmax(1)).type(torch.float).sum().item()\n\n        test_loss /= num_batches\n        correct /= size\n        \n        result['val_loss_total'].append(test_loss)\n        result['acc_total'].append(correct)\n        \n        # remember best acc@ and save checkpoint\n        is_best = correct > best_acc\n        if is_best:\n            print('Good')\n        best_acc = max(correct, best_acc)\n        save_checkpoint({\n            f'{organ}_state_dict': model.state_dict(),\n        }, is_best, organ, epoch)\n        \n        \n\n        print(f\"\"\"Val Error:\n        Accuracy: {(100*correct)}%\n        Avg loss: {test_loss}\n        \"\"\")\n    r = pd.DataFrame(result)\n    r.to_csv(f'/kaggle/working/{organ}_resnet50result.csv')\n    return result","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_k = train(kidney_model, 20, train_kidney_dataloader, val_kidney_dataloader, optimizer_kidney, loss_fn, 'kidney')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_l = train(liver_model, 20, train_liver_dataloader, val_liver_dataloader, optimizer_liver, loss_fn, 'liver')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_s = train(spleen_model, 20, train_spleen_dataloader, val_spleen_dataloader, optimizer_spleen, loss_fn, 'spleen')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nrk = pd.read_csv('/kaggle/working/kidney_resnet50result.csv')\nfig = plt.figure()\nax1 = fig.add_subplot(3,1,1)\nax1.plot(rk['epoch'], rk['train_loss_total'])\nax1.set_title('Train')\nplt.xticks(rk['epoch'])\nplt.xlabel('Epoch')\nplt.ylabel('Train loss')\nax2 = fig.add_subplot(3, 1, 2)\nax2.plot(rk['epoch'], rk['val_loss_total'])\nax2.set_title('Val')\nplt.xticks(rk['epoch'])\nplt.xlabel('Epoch')\nplt.ylabel('Val loss')\nax3 = fig.add_subplot(3, 1, 3)\nax3.plot(rk['epoch'], rk['acc_total'])\nax3.set_title('Acc')\nplt.xticks(rk['epoch'])\nplt.xlabel('Epoch')\nplt.ylabel('Acc')\nplt.tight_layout()\nplt.show()\nprint(f'Best epoch:{rk[\"epoch\"][rk[\"acc_total\"] == max(rk[\"acc_total\"])].iloc[0]} with acc = {max(rk[\"acc_total\"])}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rl = pd.read_csv('/kaggle/working/liver_resnet50result.csv')\nfig = plt.figure()\nax1 = fig.add_subplot(3,1,1)\nax1.plot(rl['epoch'], rl['train_loss_total'])\nax1.set_title('Train')\nplt.xticks(rk['epoch'])\nplt.xlabel('Epoch')\nplt.ylabel('Train loss')\nax2 = fig.add_subplot(3, 1, 2)\nax2.plot(rl['epoch'], rl['val_loss_total'])\nax2.set_title('Val')\nplt.xticks(rk['epoch'])\nplt.xlabel('Epoch')\nplt.ylabel('Val loss')\nax3 = fig.add_subplot(3, 1, 3)\nax3.plot(rl['epoch'], rl['acc_total'])\nax3.set_title('Acc')\nplt.xticks(rk['epoch'])\nplt.xlabel('Epoch')\nplt.ylabel('Acc')\nplt.tight_layout()\nplt.show()\nprint(f'Best epoch:{rl[\"epoch\"][rl[\"acc_total\"] == max(rl[\"acc_total\"])].iloc[0]} with acc = {max(rl[\"acc_total\"])}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rp = pd.read_csv('/kaggle/working/spleen_resnet50result.csv')\nfig = plt.figure()\nax1 = fig.add_subplot(3,1,1)\nax1.plot(rp['epoch'], rp['train_loss_total'])\nax1.set_title('Train')\nplt.xticks(rk['epoch'])\nplt.xlabel('Epoch')\nplt.ylabel('Train loss')\nax2 = fig.add_subplot(3, 1, 2)\nax2.plot(rp['epoch'], rp['val_loss_total'])\nax2.set_title('Val')\nplt.xticks(rk['epoch'])\nplt.xlabel('Epoch')\nplt.ylabel('Val loss')\nax3 = fig.add_subplot(3, 1, 3)\nax3.plot(rp['epoch'], rp['acc_total'])\nax3.set_title('Acc')\nplt.xticks(rk['epoch'])\nplt.xlabel('Epoch')\nplt.ylabel('Acc')\nplt.tight_layout()\nplt.show()\nprint(f'Best epoch:{rp[\"epoch\"][rp[\"acc_total\"] == max(rp[\"acc_total\"])].iloc[0]} with acc = {max(rp[\"acc_total\"])}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rp[\"epoch\"][rp[\"acc_total\"] == max(rp[\"acc_total\"])]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"code","source":"#Kidney\nkidney_model = NNet().to(device)\nkidney_model.load_state_dict(torch.load(\"/kaggle/working/kidney_resnet50model20_best.pth\"))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Liver\nliver_model = NNet().to(device)\nliver_model.load_state_dict(torch.load(\"/kaggle/working/liver_resnet50model20_best.pth\"))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Spleen\nspleen_model = NNet().to(device)\nspleen_model.load_state_dict(torch.load(\"/kaggle/working/spleen_resnet50model20_best.pth\"))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def infer(model, test_data):\n    model.eval()\n    \n    test_loss = 0\n    correct = 0\n    test_dataloader = DataLoader(test_data, batch_size = 128)\n    size = len(test_dataloader.dataset)\n    num_batches = len(test_dataloader) \n    with torch.no_grad():\n        for X, y in tqdm(test_dataloader):\n\n            X = X.to(device)\n            pred = model(X)\n\n            label = y.to(device)    \n\n            test_loss += loss_fn(pred, label).item()\n\n            correct += (pred.argmax(1) == label.argmax(1)).type(torch.float).sum().item()\n\n        test_loss /= num_batches\n        correct /= size\n\n        print(f\"\"\"Test Error:\n        Accuracy: {(100*correct)}%\n        Avg loss: {test_loss}\n        \"\"\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"infer(kidney_model, test_kidney_data)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"infer(liver_model, test_liver_data)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"infer(spleen_model, test_spleen_data)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = Image.open(f'/kaggle/input/okdata1/kidney/train/10005_0_0_0_18667_87.png')\nplt.imshow(image)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = Image.open(f'/kaggle/input/testrtr/368513372_1154455172181647_946596586229075631_n.png').convert('RGB')\nT = transforms.Compose([\n            transforms.Resize([256], torchvision.transforms.InterpolationMode.BILINEAR),\n            transforms.CenterCrop([224]),\n            transforms.ToTensor(),\n            transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n        ])\nimage = T(image).unsqueeze(dim=0)\nprint(image.shape)\nkidney_model.to('cpu')\nprint(kidney_model(image))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_list = [[i][1] for i in np.argwhere(x>0)]\ny_list = [[i][0] for i in np.argwhere(x>0)]\nxmin = min(x_list)\nxmax = max(x_list)\nymin = min(y_list)\nymax = max(y_list)\nxcenter = ((xmin+xmax)/2)/x.shape[1]\nycenter = ((ymin+ymax)/2)/x.shape[0]\nwidth = (xmax - xmin)\nheight = (ymax - ymin)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"khe, kl, kh = 0, 0, 0\nlhe, ll, lh = 0, 0, 0 \nshe, sl, sh = 0, 0, 0\nfor i in os.listdir('/kaggle/input/blahblah/train_images'):\n    if i.split('_')[1] == 0:\n        khe += 1\n    elif i.split('_')[1] == 1:\n        kl += 1\n    elif i.split('_')[1] == 2:\n        kh += 1\n    ","metadata":{},"execution_count":null,"outputs":[]}]}