{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\nimport shutil\nimport sklearn\nfrom unicodedata import normalize\nimport random\nimport pickle\n%matplotlib inline\nimport math\n\nimport tensorflow as tf\nimport tensorflow.keras as keras\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, BatchNormalization, Dropout, LSTM, Bidirectional\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam, SGD, Nadam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom sklearn.metrics import roc_curve, auc\nfrom sklearn.metrics import confusion_matrix, accuracy_score, precision_score, f1_score, recall_score\nfrom sklearn.model_selection import train_test_split\n\nimport shutil\nimport random\nfrom unicodedata import normalize\nimport collections, numpy\nfrom tensorflow.keras import losses\nfrom tensorflow.keras.utils import Sequence\n\n\nimport os\nimport sys\nfrom PIL import Image\n\nimport cv2\nfrom collections import Counter\nfrom imblearn.over_sampling import SMOTE\nfrom sklearn.datasets import make_classification\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import minmax_scale\nimport pydicom\nimport glob\nfrom sklearn.model_selection import StratifiedShuffleSplit\nfrom tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-10-02T09:06:37.165679Z","iopub.execute_input":"2023-10-02T09:06:37.166018Z","iopub.status.idle":"2023-10-02T09:06:47.427513Z","shell.execute_reply.started":"2023-10-02T09:06:37.165996Z","shell.execute_reply":"2023-10-02T09:06:47.426632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = '/kaggle/input/rsna-2023-abdominal-trauma-detection'\nSTRIDE = 10 # take one patient after each (STRIDE - 1) patients","metadata":{"execution":{"iopub.status.busy":"2023-10-02T09:06:47.429213Z","iopub.execute_input":"2023-10-02T09:06:47.429875Z","iopub.status.idle":"2023-10-02T09:06:47.434366Z","shell.execute_reply.started":"2023-10-02T09:06:47.429844Z","shell.execute_reply":"2023-10-02T09:06:47.433472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"series = glob.glob(f\"{BASE_PATH}/test_images/*/*/*\")\n\nif '/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images/3124/5842/514.dcm' in series:\n    print('remove')\n    series.remove('/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images/3124/5842/514.dcm')","metadata":{"execution":{"iopub.status.busy":"2023-10-02T09:06:47.435661Z","iopub.execute_input":"2023-10-02T09:06:47.436261Z","iopub.status.idle":"2023-10-02T09:06:47.493247Z","shell.execute_reply.started":"2023-10-02T09:06:47.436225Z","shell.execute_reply":"2023-10-02T09:06:47.49245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# meta_df = pd.read_csv(f'{BASE_PATH}/test_series_meta.csv')\n# meta_df['dicom_folder'] = BASE_PATH + '/' + 'test_images/'\\\n#                                     + '/' + meta_df.patient_id.astype(str)\\\n#                                     + '/' + meta_df.series_id.astype(str)\n# test_folders = meta_df.dicom_folder.tolist()\n# test_paths = []\n# for folders in tqdm(test_folders):\n#     test_paths = test_paths+ sorted(glob.glob(os.path.join(folders, '*dcm')))[::STRIDE]","metadata":{"execution":{"iopub.status.busy":"2023-10-02T09:06:47.49536Z","iopub.execute_input":"2023-10-02T09:06:47.496779Z","iopub.status.idle":"2023-10-02T09:06:47.500795Z","shell.execute_reply.started":"2023-10-02T09:06:47.496751Z","shell.execute_reply":"2023-10-02T09:06:47.499788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.DataFrame(series, columns=[\"dicom_path\"])\ntest_df['patient_id'] = test_df.dicom_path.map(lambda x: x.split('/')[-3]).astype(int)\ntest_df['series_id'] = test_df.dicom_path.map(lambda x: x.split('/')[-2]).astype(int)\ntest_df['instance_number'] = test_df.dicom_path.map(lambda x: x.split('/')[-1].replace('.dcm','')).astype(int)\n\n# print('Test:')\n# print(f'# Size: {len(test_df)}')\n# display(test_df.head())","metadata":{"execution":{"iopub.status.busy":"2023-10-02T09:06:47.503017Z","iopub.execute_input":"2023-10-02T09:06:47.503243Z","iopub.status.idle":"2023-10-02T09:06:47.52347Z","shell.execute_reply.started":"2023-10-02T09:06:47.503224Z","shell.execute_reply":"2023-10-02T09:06:47.522605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_ids = test_df['patient_id'].unique()\n# patient_ids\npatient_ids = sorted(patient_ids,key=int)","metadata":{"execution":{"iopub.status.busy":"2023-10-02T09:06:47.525935Z","iopub.execute_input":"2023-10-02T09:06:47.526241Z","iopub.status.idle":"2023-10-02T09:06:47.536301Z","shell.execute_reply.started":"2023-10-02T09:06:47.526213Z","shell.execute_reply":"2023-10-02T09:06:47.535332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_col = [\"bowel_healthy\", \"bowel_injury\", \"extravasation_healthy\",\n                   \"extravasation_injury\", \"kidney_healthy\", \"kidney_low\",\n                   \"kidney_high\", \"liver_healthy\", \"liver_low\", \"liver_high\",\n                   \"spleen_healthy\", \"spleen_low\", \"spleen_high\"]","metadata":{"execution":{"iopub.status.busy":"2023-10-02T09:06:47.537387Z","iopub.execute_input":"2023-10-02T09:06:47.538327Z","iopub.status.idle":"2023-10-02T09:06:47.54737Z","shell.execute_reply.started":"2023-10-02T09:06:47.53828Z","shell.execute_reply":"2023-10-02T09:06:47.546605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def trans_dcm(file_path):\n    \n    window_center = 50\n    window_width = 400\n\n    row = pydicom.read_file(file_path)\n    md = row.pixel_array\n\n    intercept = row.RescaleIntercept\n    slope = row.RescaleSlope\n    hf_image = md * slope + intercept\n\n    # --------------------걍 해야되는거\n\n    img_min = window_center - window_width//2\n    img_max = window_center + window_width//2\n\n    window_image = hf_image.copy()\n    window_image[window_image < img_min] = img_min\n    window_image[window_image > img_max] = img_max\n\n    norm_img = np.array(window_image.copy(), dtype=np.float64)\n    norm_img -= np.min(norm_img)\n    norm_img /= (np.max(norm_img) - np.min(norm_img))\n\n    norm_img = cv2.resize(norm_img,dsize=(512,512))\n\n    # ---------------------window_image가 최종 windowing이 된 것\n\n    \n    \n    return norm_img","metadata":{"execution":{"iopub.status.busy":"2023-10-02T09:06:47.548503Z","iopub.execute_input":"2023-10-02T09:06:47.549057Z","iopub.status.idle":"2023-10-02T09:06:47.557755Z","shell.execute_reply.started":"2023-10-02T09:06:47.549026Z","shell.execute_reply":"2023-10-02T09:06:47.55686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = tf.keras.models.load_model('/kaggle/input/grayscale/fold-0.h5')","metadata":{"execution":{"iopub.status.busy":"2023-10-02T09:06:47.558687Z","iopub.execute_input":"2023-10-02T09:06:47.560786Z","iopub.status.idle":"2023-10-02T09:06:58.007021Z","shell.execute_reply.started":"2023-10-02T09:06:47.560765Z","shell.execute_reply":"2023-10-02T09:06:58.006051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def post_proc_v2(pred):\n    proc_pred = np.empty((pred.shape[0], 2*2 + 3*3), dtype='float32')\n\n    # bowel, extravasation\n    proc_pred[:, 0] = 1 - pred[:, 0]\n    proc_pred[:, 1] = pred[:, 0]\n    proc_pred[:, 2] = 1 - pred[:, 1]\n    proc_pred[:, 3] = pred[:, 1]\n    \n    # liver, kidney, sneel\n    proc_pred[:, 4:7] = pred[:, 2:5]\n    proc_pred[:, 7:10] = pred[:, 5:8]\n    proc_pred[:, 10:13] = pred[:, 8:11]\n\n    return proc_pred","metadata":{"execution":{"iopub.status.busy":"2023-10-02T09:06:58.009895Z","iopub.execute_input":"2023-10-02T09:06:58.01015Z","iopub.status.idle":"2023-10-02T09:06:58.017721Z","shell.execute_reply.started":"2023-10-02T09:06:58.010123Z","shell.execute_reply":"2023-10-02T09:06:58.014829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = np.zeros([1,512,512,3])\npatient_pred = np.zeros([len(patient_ids),13],dtype='float32')\n\nfor i in range(len(patient_ids)): #환자 별로\n    patient_path=[]\n    \n    for j in range(len(series)): #전체 파일 조사\n        if test_df['patient_id'][j] == patient_ids[i]:\n            patient_path.append(test_df['dicom_path'][j])\n#         c = os.path.dirname(test_paths[j]).split(os.path.sep)\n#         if c[5] == str(patient_ids[i]):#해당 환자의 path저장하기\n#             patient_path.append(test_paths[j])\n            \n    prediction = np.zeros([1,len(patient_path),11])\n    \n    for k in range(len(patient_path)):#한 환자에 대해서 prediction 구하기\n        a[0,:,:,0] = trans_dcm(patient_path[k])\n        a[0,:,:,1] = a[0,:,:,0]\n        a[0,:,:,2] = a[0,:,:,0]\n    \n        pred = model.predict(a)\n        pred = np.concatenate(pred, axis=-1).astype('float32') # reducing memory footprint\n        for n in range(0,11):\n            pred[0][n] = \"{:.8f}\".format(float(pred[0][n]))\n            \n        prediction[0,k,:] = pred\n        \n    prediction = np.max(prediction, axis=0) # taking max prediction of all ct scans for a patient\n    patient_pred[i,:] =  post_proc_v2(prediction)\n    \n    del prediction, pred, patient_path","metadata":{"execution":{"iopub.status.busy":"2023-10-02T09:06:58.019025Z","iopub.execute_input":"2023-10-02T09:06:58.019958Z","iopub.status.idle":"2023-10-02T09:07:04.650717Z","shell.execute_reply.started":"2023-10-02T09:06:58.019926Z","shell.execute_reply":"2023-10-02T09:07:04.649878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# patient_preds = np.zeros(shape=(len(patient_ids), 2*2 + 3*3), dtype='float32')\n# a = np.zeros([1,512,512,3])\n# for i in range(len(test_paths)):\n#     a[0,:,:,0] = trans_dcm(test_paths[i])\n#     a[0,:,:,1] = a[0,:,:,0]\n#     a[0,:,:,2] = a[0,:,:,0]\n    \n#     pred = model.predict(a, steps = 1 * len(test_paths) / 3, verbose=1)\n#     pred = np.concatenate(pred, axis=-1).astype('float32') # reducing memory footprint\n#     pred = pred[:1 * len(test_paths), :]\n# #     pred = np.mean(pred.reshape(1, len(test_paths), 11), axis=0)\n\n#     # # Store model's prediction\n#     # model_preds += pred / (len(test_paths)*1)\n\n#     patient_preds[i, :] = post_proc_v2(pred)","metadata":{"execution":{"iopub.status.busy":"2023-10-02T09:07:04.653234Z","iopub.execute_input":"2023-10-02T09:07:04.654115Z","iopub.status.idle":"2023-10-02T09:07:04.658277Z","shell.execute_reply.started":"2023-10-02T09:07:04.654084Z","shell.execute_reply":"2023-10-02T09:07:04.657225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fix_pred = \"{:.16f}\".format(float(2.3283151e-25))","metadata":{"execution":{"iopub.status.busy":"2023-10-02T09:07:04.659491Z","iopub.execute_input":"2023-10-02T09:07:04.659941Z","iopub.status.idle":"2023-10-02T09:07:04.671738Z","shell.execute_reply.started":"2023-10-02T09:07:04.659912Z","shell.execute_reply":"2023-10-02T09:07:04.670894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create Submission\npred_df = pd.DataFrame({'patient_id':patient_ids,})\npred_df[target_col] = patient_pred.astype('float32')\n\n# Align with sample submission\nsub_df = pd.read_csv(f'{BASE_PATH}/sample_submission.csv')\nsub_df = sub_df[['patient_id']]\nsub_df = sub_df.merge(pred_df, on='patient_id', how='left')\n\n# Store submission\nsub_df.to_csv('submission.csv',index=False)\nsub_df","metadata":{"execution":{"iopub.status.busy":"2023-10-02T09:07:04.672909Z","iopub.execute_input":"2023-10-02T09:07:04.673666Z","iopub.status.idle":"2023-10-02T09:07:04.722025Z","shell.execute_reply.started":"2023-10-02T09:07:04.673637Z","shell.execute_reply":"2023-10-02T09:07:04.721149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}