{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport glob   \nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport keras\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\n# for dirname, _, filenames in os.walk('/kaggle\\/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\nimport random, shutil\nimport pydicom as dicom\nimport tensorflow as tf\nimport sklearn\nimport matplotlib.pyplot as plt\nfrom tensorflow.keras import layers, models\nfrom sklearn.model_selection import KFold, StratifiedKFold, GroupKFold, StratifiedGroupKFold\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.utils.class_weight import compute_class_weight\nimport cv2\nfrom tensorflow.keras.utils import Sequence\nfrom sklearn.preprocessing import minmax_scale\nfrom sklearn.model_selection import train_test_split\nfrom csv import writer\nimport pydicom\nfrom tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:52:47.846893Z","iopub.execute_input":"2023-08-28T06:52:47.847321Z","iopub.status.idle":"2023-08-28T06:52:47.858117Z","shell.execute_reply.started":"2023-08-28T06:52:47.847286Z","shell.execute_reply":"2023-08-28T06:52:47.856545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#레이블\ndf_train = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:52:49.174237Z","iopub.execute_input":"2023-08-28T06:52:49.174975Z","iopub.status.idle":"2023-08-28T06:52:49.191288Z","shell.execute_reply.started":"2023-08-28T06:52:49.174927Z","shell.execute_reply":"2023-08-28T06:52:49.19011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#환자 id\nid = df_train.patient_id.to_numpy().astype(str)\n#X는 image data, y는 label\nX, y = [], []","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:52:49.812406Z","iopub.execute_input":"2023-08-28T06:52:49.813945Z","iopub.status.idle":"2023-08-28T06:52:49.823729Z","shell.execute_reply.started":"2023-08-28T06:52:49.8139Z","shell.execute_reply":"2023-08-28T06:52:49.822207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 일단 3번 환자까지만!\n# for x, p_id in enumerate(id[1:3]):\n#     # 이미지 경로(환자 아이디)\n#     dir = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/' + p_id + '/'\n    \n#     # 환자 label을 뽑아냄\n#     features = df_train.iloc[x].to_numpy()[1:]\n    \n#     # Loop through each file in the patient's directory\n#     for file in glob.glob(dir + '*'): # *는 임의 길이의 모든 문자열, 환자 아이디 안의 모든 scan number에 대해서\n#         # Loop through each image file in the current directory\n#         for image_path in glob.glob(file + '/*'): # 하나의 scan number안에 있는 모든 다이콤 파일\n#             # Read the DICOM image and extract the pixel array\n#             a = dicom.dcmread(image_path).pixel_array\n#             image = np.dstack([a,a,a])\n#             X.append(image)\n            \n#             # Append the features (labels) for this image to the y list\n#             y.append(features)\n#             image=[]","metadata":{"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for x, p_id in enumerate(id[:1]):\n    # 이미지 경로(환자 아이디)\n    dir = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/' + p_id + '/'\n    \n    # 환자 label을 뽑아냄\n    features = df_train.iloc[x].to_numpy()[1:]\n    \n    # Loop through each file in the patient's directory\n    for file in glob.glob(dir + '*'): # *는 임의 길이의 모든 문자열, 환자 아이디 안의 모든 scan number에 대해서\n        # Loop through each image file in the current directory\n        for image_path in glob.glob(file + '/*'): # 하나의 scan number안에 있는 모든 다이콤 파일\n            X.append(image_path)\n            # Append the features (labels) for this image to the y list\n            y.append(features)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:52:51.412438Z","iopub.execute_input":"2023-08-28T06:52:51.413295Z","iopub.status.idle":"2023-08-28T06:52:51.434679Z","shell.execute_reply.started":"2023-08-28T06:52:51.413251Z","shell.execute_reply":"2023-08-28T06:52:51.433235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dir = '/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images/'\ntest_id = os.listdir(dir)\ntest_id.sort()","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:52:52.931432Z","iopub.execute_input":"2023-08-28T06:52:52.931923Z","iopub.status.idle":"2023-08-28T06:52:52.938979Z","shell.execute_reply.started":"2023-08-28T06:52:52.931886Z","shell.execute_reply":"2023-08-28T06:52:52.937554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_list = []\nfor x, p_id in enumerate(test_id):\n    dir = '/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images/' + p_id + '/'\n    \n    for file in glob.glob(dir+'*'):\n        for image_path in glob.glob(file + '/*'):\n            test_list.append(image_path)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:52:54.006354Z","iopub.execute_input":"2023-08-28T06:52:54.006777Z","iopub.status.idle":"2023-08-28T06:52:54.019017Z","shell.execute_reply.started":"2023-08-28T06:52:54.006745Z","shell.execute_reply":"2023-08-28T06:52:54.017664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = np.array(X)\ny = np.array(y)\n\nX.shape,y.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:52:54.739784Z","iopub.execute_input":"2023-08-28T06:52:54.740279Z","iopub.status.idle":"2023-08-28T06:52:54.753033Z","shell.execute_reply.started":"2023-08-28T06:52:54.740241Z","shell.execute_reply":"2023-08-28T06:52:54.751491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:52:55.6519Z","iopub.execute_input":"2023-08-28T06:52:55.65237Z","iopub.status.idle":"2023-08-28T06:52:55.660348Z","shell.execute_reply.started":"2023-08-28T06:52:55.652334Z","shell.execute_reply":"2023-08-28T06:52:55.658965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataGenerator(Sequence):\n    \n    def __init__(self, list_IDs,labels, batch_size=8, dim=(512,512), n_channels=3, n_classes=14, shuffle=False):\n        self.dim = dim\n        self.batch_size = batch_size\n        self.labels = labels\n        self.list_IDs = list_IDs\n        self.n_channels = n_channels\n        self.n_classes = n_classes\n        self.shuffle = shuffle\n        self.on_epoch_end()\n        \n    def __len__(self):\n        return int(np.ceil(len(self.list_IDs) / self.batch_size))\n    \n    def __getitem__(self, index):\n        \n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        \n        list_IDs_temp = [self.list_IDs[k] for k in indexes]\n        list_IDs_label = [self.labels[k] for k in indexes]\n        \n        X,y = self.__data_generation(list_IDs_temp,list_IDs_label)\n        \n        return X,y\n    \n    def on_epoch_end(self):\n        self.indexes = np.arange(len(self.list_IDs))\n        if self.shuffle == True:\n            np.random.shuffle(self.indexes)\n            \n    def __data_generation(self, list_IDs_temp, list_IDs_label):\n        \n        X = np.empty((self.batch_size, *self.dim, self.n_channels))\n        y = np.empty((self.batch_size,self.n_classes),dtype=int)\n\n        for i, (fileName,fileLabel) in enumerate(zip(list_IDs_temp,list_IDs_label)):\n            # Store sample\n                def standardize_pixel_array(dcm: pydicom.dataset.FileDataset) -> np.ndarray:\n                    # Correct DICOM pixel_array if PixelRepresentation == 1.\n                    pixel_array = dcm.pixel_array\n                    if dcm.PixelRepresentation == 1:\n                        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n                        dtype = pixel_array.dtype \n                        new_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n                        pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dcm)\n                    return pixel_array\n                \n                dicom = pydicom.dcmread(fileName)\n                data = standardize_pixel_array(dicom)\n                data = data - np.min(data)\n                data = data / (np.max(data) + 1e-5)\n                fix_monochrome = 1\n                if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n                    data = 1.0 - data\n                \n                h, w = data.shape[:2]  # orig hw\n                img = cv2.resize(data, (512,512), cv2.INTER_LINEAR)\n                img = (img * 255).astype(np.uint8)\n                \n                X[i,:,:,0] = img/255\n                X[i,:,:,1] = X[i,:,:,0]\n                X[i,:,:,2] = X[i,:,:,0]\n                \n            \n            # Store class\n                y[i] = fileLabel\n        \n        return X, y","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:52:56.427785Z","iopub.execute_input":"2023-08-28T06:52:56.428251Z","iopub.status.idle":"2023-08-28T06:52:56.450014Z","shell.execute_reply.started":"2023-08-28T06:52:56.428215Z","shell.execute_reply":"2023-08-28T06:52:56.448605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = DataGenerator(list_IDs=X_train,labels=y_train, batch_size=16, shuffle=True)\nvalidation_generator = DataGenerator(list_IDs=X_test,labels=y_test, batch_size=16, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:52:57.140109Z","iopub.execute_input":"2023-08-28T06:52:57.141015Z","iopub.status.idle":"2023-08-28T06:52:57.146315Z","shell.execute_reply.started":"2023-08-28T06:52:57.14098Z","shell.execute_reply":"2023-08-28T06:52:57.145387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from tensorflow.keras.applications import ResNet50V2","metadata":{"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inputs = tf.keras.Input(shape=(512,512, 1))\n\n# base_model = ResNet50V2(input_tensor=inputs, pooling='avg')\n\n# inputs = base_model.input\n# x = base_model.output\n\n\n# x = keras.layers.Dense(512, activation='relu')(x)\n# x = keras.layers.BatchNormalization()(x)\n# x = keras.layers.Dense(256, activation='relu')(x)\n# x = keras.layers.BatchNormalization()(x)\n# x = keras.layers.Dense(128, activation='relu')(x)\n# x = keras.layers.BatchNormalization()(x)\n# x = keras.layers.Dense(64, activation='relu')(x)\n# x = keras.layers.BatchNormalization()(x)\n# x = keras.layers.Dense(32, activation='relu')(x)\n# x = keras.layers.BatchNormalization()(x)\n# x = keras.layers.Dropout(0.25)(x)\n# outputs = tf.keras.layers.Dense(14, activation='sigmoid')(x)\n\n# # Create the model with the defined input and output layers\n# model = tf.keras.Model(inputs=inputs, outputs=outputs)\n# model.summary()","metadata":{"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs = tf.keras.Input(shape=(512,512,3))\n\nx = tf.keras.layers.Conv2D(4, (3, 3), activation='relu')(inputs)\nx = tf.keras.layers.Conv2D(4, (3, 3), activation='relu')(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.MaxPool2D(2)(x)\n\n# Second set of convolutional layers\nx = tf.keras.layers.Conv2D(4, (3, 3), activation='relu')(x)\nx = tf.keras.layers.Conv2D(4, (3, 3), activation='relu')(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.MaxPool2D(2)(x)\n\n# Third set of convolutional layers\nx = tf.keras.layers.Conv2D(4, (3, 3), activation='relu')(x)\nx = tf.keras.layers.Conv2D(4, (3, 3), activation='relu')(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.MaxPool2D(2)(x)\n\n# Fourth set of convolutional layers\nx = tf.keras.layers.Conv2D(8, (3, 3), activation='relu')(x)\nx = tf.keras.layers.Conv2D(8, (3, 3), activation='relu')(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.MaxPool2D(2)(x)\n\n# Fifth set of convolutional layers\nx = tf.keras.layers.Conv2D(8, (3, 3), activation='relu')(x)\nx = tf.keras.layers.Conv2D(8, (3, 3), activation='relu')(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.MaxPool2D(2)(x)\n\n# Flatten the output and connect to a Dense (fully connected) layer\nx = tf.keras.layers.Flatten()(x)\nx = tf.keras.layers.Dense(512, activation='relu')(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.Dense(256, activation='relu')(x)\nx = tf.keras.layers.BatchNormalization()(x)\n\nx = tf.keras.layers.Dense(64, activation='relu')(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.Dense(32, activation='relu')(x)\nx = tf.keras.layers.BatchNormalization()(x)\n\noutputs = tf.keras.layers.Dense(14, activation='softmax')(x)\n\n# Create the model with the defined input and output layers\nmodel = tf.keras.Model(inputs=inputs, outputs=outputs)\n# model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:52:58.183369Z","iopub.execute_input":"2023-08-28T06:52:58.184234Z","iopub.status.idle":"2023-08-28T06:52:58.582071Z","shell.execute_reply.started":"2023-08-28T06:52:58.184188Z","shell.execute_reply":"2023-08-28T06:52:58.580633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss=tf.keras.losses.MeanSquaredError(),\n              optimizer=tf.keras.optimizers.Adam(learning_rate=0.0008, beta_1=0.9, beta_2=0.999, epsilon=1e-07, amsgrad=False),\n              metrics=['accuracy'])\n# modelCheckpoint = keras.callbacks.ModelCheckpoint(filepath = model_path, monitor = 'val_loss', save_best_only=True)\nearlyStopping = keras.callbacks.EarlyStopping(monitor='val_loss', patience=15, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:52:58.897014Z","iopub.execute_input":"2023-08-28T06:52:58.897497Z","iopub.status.idle":"2023-08-28T06:52:58.916801Z","shell.execute_reply.started":"2023-08-28T06:52:58.89746Z","shell.execute_reply":"2023-08-28T06:52:58.915429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model with the specified data (X, y)\nhistory = model.fit(train_generator, validation_data = validation_generator, epochs=1, batch_size=16, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:52:59.421853Z","iopub.execute_input":"2023-08-28T06:52:59.422781Z","iopub.status.idle":"2023-08-28T06:58:48.117549Z","shell.execute_reply.started":"2023-08-28T06:52:59.422744Z","shell.execute_reply":"2023-08-28T06:58:48.11634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_generator = DataGenerator(list_IDs=test_list,labels=y_test[:len(test_list)], batch_size=3, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:58:48.119964Z","iopub.execute_input":"2023-08-28T06:58:48.120388Z","iopub.status.idle":"2023-08-28T06:58:48.126288Z","shell.execute_reply.started":"2023-08-28T06:58:48.120332Z","shell.execute_reply":"2023-08-28T06:58:48.125182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = model.predict(test_generator)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:58:48.127923Z","iopub.execute_input":"2023-08-28T06:58:48.128301Z","iopub.status.idle":"2023-08-28T06:58:48.797186Z","shell.execute_reply.started":"2023-08-28T06:58:48.128269Z","shell.execute_reply":"2023-08-28T06:58:48.796252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:58:48.799022Z","iopub.execute_input":"2023-08-28T06:58:48.799361Z","iopub.status.idle":"2023-08-28T06:58:48.808697Z","shell.execute_reply.started":"2023-08-28T06:58:48.79933Z","shell.execute_reply":"2023-08-28T06:58:48.807525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = [[0.9796631712742294,0.08134731490308231,0.936447410231967,1.7794725135049254,0.94216714331109,0.14617095646647602,0.1277407054337464,0.897998093422307,0.3292024149984112,0.11820781696854148,0.8875119161105816,0.25293930727677155,0.2955195424213537,7.607244995233556],[0.9796631712742294,0.08134731490308231,0.936447410231967,1.7794725135049254,0.94216714331109,0.14617095646647602,0.1277407054337464,0.897998093422307,0.3292024149984112,0.11820781696854148,0.8875119161105816,0.25293930727677155,0.2955195424213537,7.607244995233556],[0.9796631712742294,0.08134731490308231,0.936447410231967,1.7794725135049254,0.94216714331109,0.14617095646647602,0.1277407054337464,0.897998093422307,0.3292024149984112,0.11820781696854148,0.8875119161105816,0.25293930727677155,0.2955195424213537,7.607244995233556]]","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:58:58.272419Z","iopub.execute_input":"2023-08-28T06:58:58.272863Z","iopub.status.idle":"2023-08-28T06:58:58.282654Z","shell.execute_reply.started":"2023-08-28T06:58:58.272828Z","shell.execute_reply":"2023-08-28T06:58:58.281177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(test_list)):\n    df_test.at[i,'bowel_healthy'] = prediction[i][0]\n    df_test.at[i,'bowel_injury'] = prediction[i][1]\n    df_test.at[i,'extravasation_healthy'] = prediction[i][2]\n    df_test.at[i,'extravasation_injury'] = prediction[i][3]\n    df_test.at[i,'kidney_healthy'] = prediction[i][4]\n    df_test.at[i,'kidney_low'] = prediction[i][5]\n    df_test.at[i,'kidney_high'] = prediction[i][6]\n    df_test.at[i,'liver_healthy'] = prediction[i][7]\n    df_test.at[i,'liver_low'] = prediction[i][8]\n    df_test.at[i,'liver_high'] = prediction[i][9]\n    df_test.at[i,'spleen_healthy'] = prediction[i][10]\n    df_test.at[i,'spleen_low'] = prediction[i][11]\n    df_test.at[i,'spleen_high'] = prediction[i][12]","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:58:59.201063Z","iopub.execute_input":"2023-08-28T06:58:59.201501Z","iopub.status.idle":"2023-08-28T06:58:59.212972Z","shell.execute_reply.started":"2023-08-28T06:58:59.20146Z","shell.execute_reply":"2023-08-28T06:58:59.211452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:59:00.17661Z","iopub.execute_input":"2023-08-28T06:59:00.177432Z","iopub.status.idle":"2023-08-28T06:59:00.1964Z","shell.execute_reply.started":"2023-08-28T06:59:00.177395Z","shell.execute_reply":"2023-08-28T06:59:00.19503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:59:03.152517Z","iopub.execute_input":"2023-08-28T06:59:03.152923Z","iopub.status.idle":"2023-08-28T06:59:03.160596Z","shell.execute_reply.started":"2023-08-28T06:59:03.152889Z","shell.execute_reply":"2023-08-28T06:59:03.159242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = pd.read_csv('/kaggle/working/submission.csv')\na.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:59:04.098006Z","iopub.execute_input":"2023-08-28T06:59:04.098973Z","iopub.status.idle":"2023-08-28T06:59:04.126271Z","shell.execute_reply.started":"2023-08-28T06:59:04.098915Z","shell.execute_reply":"2023-08-28T06:59:04.125426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}