{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Peek\n\nFirst, let's see the structure of our files.","metadata":{}},{"cell_type":"code","source":"pip install python-gdcm","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:26:22.384807Z","iopub.execute_input":"2023-06-21T00:26:22.385213Z","iopub.status.idle":"2023-06-21T00:26:32.892815Z","shell.execute_reply.started":"2023-06-21T00:26:22.385179Z","shell.execute_reply":"2023-06-21T00:26:32.891087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport gc\nimport time\nfrom IPython.display import clear_output\nfrom tensorflow.keras.models import load_model\nfrom tensorflow.keras.callbacks import ModelCheckpoint as MC\nfrom tensorflow.keras import backend as K\n\nfrom pydicom.pixel_data_handlers.util import apply_modality_lut\nimport os\nfrom skimage.segmentation import clear_border\nfrom skimage.measure import label, regionprops\nfrom skimage.morphology import disk, dilation, binary_erosion, binary_closing\nfrom skimage.filters import roberts\nfrom scipy import ndimage as ndi\nimport cv2\nimport gdcm\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense\n\nimport pydicom as dcm\n\nfrom tensorflow import keras\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Dense, Dropout, Conv2D\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\nfrom sklearn.metrics import roc_auc_score\n\nimport vtk\nfrom vtk.util import numpy_support\n\n\nroot = '/kaggle/input/rsna-str-pulmonary-embolism-detection'\nfor item in os.listdir(root):\n    path = os.path.join(root, item)\n    if os.path.isfile(path):\n        print(path)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-21T00:26:32.896976Z","iopub.execute_input":"2023-06-21T00:26:32.897302Z","iopub.status.idle":"2023-06-21T00:26:40.56925Z","shell.execute_reply.started":"2023-06-21T00:26:32.897266Z","shell.execute_reply":"2023-06-21T00:26:40.568277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data\n\nNext, we load all '.csv' files into memory and peek into their makeup.","metadata":{}},{"cell_type":"code","source":"print('Reading train data...')\ntrain = pd.read_csv(\"/kaggle/input/bbdd-estratificada/BBDD-estratificada/train-Estratificada-Balanceado.csv\")\ntrain = train.drop(['Unnamed: 0'], axis=1)\nprint(train.shape)\ntrain.head()","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2023-06-20T18:21:20.254055Z","iopub.execute_input":"2023-06-20T18:21:20.254417Z","iopub.status.idle":"2023-06-20T18:21:20.789692Z","shell.execute_reply.started":"2023-06-20T18:21:20.254381Z","shell.execute_reply":"2023-06-20T18:21:20.78876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Reading test data...')\ntest = pd.read_csv(\"/kaggle/input/bbdd-estratificada/BBDD-estratificada/test-Estratificada.csv\")\ntest = test.drop(['Unnamed: 0'], axis=1)\ntest = test.drop(['pe_present_on_image','negative_exam_for_pe','qa_motion','qa_contrast','flow_artifact','rv_lv_ratio_gte_1','rv_lv_ratio_lt_1','leftsided_pe','chronic_pe','true_filling_defect_not_pe','rightsided_pe','acute_and_chronic_pe','central_pe','indeterminate'], axis=1)\nprint(test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T18:21:23.255747Z","iopub.execute_input":"2023-06-20T18:21:23.256149Z","iopub.status.idle":"2023-06-20T18:21:23.319179Z","shell.execute_reply.started":"2023-06-20T18:21:23.256115Z","shell.execute_reply":"2023-06-20T18:21:23.318239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Reading sample data...')\nss = pd.read_csv(\"../input/rsna-str-pulmonary-embolism-detection/sample_submission.csv\")\nprint(ss.shape)\nss.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T18:21:04.403857Z","iopub.execute_input":"2023-06-20T18:21:04.40423Z","iopub.status.idle":"2023-06-20T18:21:04.578266Z","shell.execute_reply.started":"2023-06-20T18:21:04.404193Z","shell.execute_reply":"2023-06-20T18:21:04.576477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = ss.id\ncounter = [1 for _ in range(10)]\nmapper = []\nfor i in ids:\n    n = '_'.join(i.split('_')[1:])\n    if n not in mapper:\n        mapper.append(n)\n    else:\n        counter[mapper.index(n)] += 1\nprint(\"List of keys:\")\nprint(mapper, sep='\\n')\nprint()\nprint(\"Count of items per key:\")\nprint(counter)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T09:00:06.933733Z","iopub.execute_input":"2023-06-20T09:00:06.934084Z","iopub.status.idle":"2023-06-20T09:00:07.125231Z","shell.execute_reply.started":"2023-06-20T09:00:06.934051Z","shell.execute_reply":"2023-06-20T09:00:07.12426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After checking the keys, we will now work on the fuction to get the image array from a DICOM image. We will be using a code snippet by [eladwar](https://www.kaggle.com/eladwar) from his notebook [here](https://www.kaggle.com/eladwar/20-seconds-or-less).","metadata":{}},{"cell_type":"markdown","source":"Lee las imágenes en formato DICOM y luego convierte las imágenes DICOM en un formato de imagen compatible con OpenCV (Open Source Computer Vision Library)","metadata":{}},{"cell_type":"code","source":"import vtk\nfrom vtk.util import numpy_support\nimport cv2\n\nreader = vtk.vtkDICOMImageReader()\ndef get_img(path):\n    reader.SetFileName(path)\n    reader.Update()\n    _extent = reader.GetDataExtent()\n    ConstPixelDims = [_extent[1]-_extent[0]+1, _extent[3]-_extent[2]+1, _extent[5]-_extent[4]+1]\n\n    ConstPixelSpacing = reader.GetPixelSpacing()\n    imageData = reader.GetOutput()\n    pointData = imageData.GetPointData()\n    arrayData = pointData.GetArray(0)\n    ArrayDicom = numpy_support.vtk_to_numpy(arrayData)\n    ArrayDicom = ArrayDicom.reshape(ConstPixelDims, order='F')\n    ArrayDicom = cv2.resize(ArrayDicom,(512,512))\n    return ArrayDicom","metadata":{"execution":{"iopub.status.busy":"2023-06-20T09:00:09.053287Z","iopub.execute_input":"2023-06-20T09:00:09.05421Z","iopub.status.idle":"2023-06-20T09:00:09.070887Z","shell.execute_reply.started":"2023-06-20T09:00:09.054126Z","shell.execute_reply":"2023-06-20T09:00:09.069474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After defining our image reader, we will test it with a sample DICOM image to load.","metadata":{}},{"cell_type":"code","source":"#test read a dcom file and view it\nfpath = \"../input/rsna-str-pulmonary-embolism-detection/train/0003b3d648eb/d2b2960c2bbf/00ac73cfc372.dcm\"\nds = get_img(fpath)\n\nimport matplotlib.pyplot as plt\n\n#Convert dcom file to 8bit color\nfunc = lambda x: int((2**15 + x)*(255/2**16))\nint16_to_uint8 = np.vectorize(func)\n\ndef show_dicom_images(dcom):\n    f, ax = plt.subplots(1,2, figsize=(16,20))\n    data_row_img = int16_to_uint8(ds)\n    ax[0].imshow(data_row_img, cmap=plt.cm.bone)\n    ax[1].imshow(ds, cmap=plt.cm.bone)\n    #print(data_row_img)\n    ax[0].axis('off')\n    ax[0].set_title('8-bit DICOM Image')\n    ax[1].axis('off')\n    ax[1].set_title('16-bit DICOM Image')\n    plt.show()\n    \nshow_dicom_images(ds)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T09:00:11.558837Z","iopub.execute_input":"2023-06-20T09:00:11.559222Z","iopub.status.idle":"2023-06-20T09:00:11.932741Z","shell.execute_reply.started":"2023-06-20T09:00:11.559182Z","shell.execute_reply":"2023-06-20T09:00:11.931872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Creation","metadata":{}},{"cell_type":"markdown","source":"I decided to try and use a pre-trained Xception model as a feature-extractor. To simplify my coding, I did a multi-output model with each output having one node activated by a sigmoid. Also, since we are dealing with numbers between 1 and 0, I decided to use binary_crossentropy as the loss function.","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Dense, Dropout, Conv2D\n\ninputs = Input((512, 512, 3))\n#x = Conv2D(3, (1, 1), activation='relu')(inputs)\nbase_model = keras.applications.Xception(\n    include_top=False,\n    weights=\"imagenet\"\n)\n\nbase_model.trainable = False\n\noutputs = base_model(inputs, training=False)\noutputs = keras.layers.GlobalAveragePooling2D()(outputs)\noutputs = Dropout(0.25)(outputs)\noutputs = Dense(1024, activation='relu')(outputs)\noutputs = Dense(256, activation='relu')(outputs)\noutputs = Dense(64, activation='relu')(outputs)\nppoi = Dense(1, activation='sigmoid', name='pe_present_on_image')(outputs)\n\nmodel = Model(inputs=inputs, outputs={'pe_present_on_image':ppoi})\n\nopt = keras.optimizers.Adam(lr=0.001)\n\nmodel.compile(optimizer=opt,\n              loss='binary_crossentropy',\n              metrics=['accuracy'])\n\nmodel.summary()\nmodel.save('pe_detection_model.h5')\ndel model\nK.clear_session()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T17:07:14.089755Z","iopub.execute_input":"2023-06-20T17:07:14.090153Z","iopub.status.idle":"2023-06-20T17:07:19.287836Z","shell.execute_reply.started":"2023-06-20T17:07:14.090119Z","shell.execute_reply":"2023-06-20T17:07:19.287047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Before we can train our model, we would be needing an image generator. This is so that our training code would look much cleaner and nicer.","metadata":{}},{"cell_type":"code","source":"def convert_to_rgb(array):\n    array = array.reshape((512, 512, 1))\n    return np.stack([array, array, array], axis=2).reshape((512, 512, 3))\n    \ndef custom_dcom_image_generator(batch_size, dataset, test=False, debug=False):\n    \n    fnames = dataset[['StudyInstanceUID', 'SeriesInstanceUID', 'SOPInstanceUID']]\n    \n    if not test:\n        Y = dataset[['pe_present_on_image']]\n        prefix = 'input/rsna-str-pulmonary-embolism-detection/train'\n        \n    else:\n        prefix = 'input/rsna-str-pulmonary-embolism-detection/test'\n    \n    X = []\n    batch = 0\n    for st, sr, so in fnames.values:\n        if debug:\n            print(f\"Current file: ../{prefix}/{st}/{sr}/{so}.dcm\")\n\n        dicom = get_img(f\"../{prefix}/{st}/{sr}/{so}.dcm\")\n        image = convert_to_rgb(dicom)\n        X.append(image)\n        \n        del st, sr, so\n        \n        if len(X) == batch_size:\n            if test:\n                yield np.array(X)\n                del X\n            else:\n                yield np.array(X), Y[batch*batch_size:(batch+1)*batch_size].values\n                del X\n                \n            gc.collect()\n            X = []\n            batch += 1\n        \n    if test:\n        yield np.array(X)\n    else:\n        yield np.array(X), Y[batch*batch_size:(batch+1)*batch_size].values\n        del Y\n    del X\n    gc.collect()\n    return","metadata":{"execution":{"iopub.status.busy":"2023-06-20T09:00:36.614208Z","iopub.execute_input":"2023-06-20T09:00:36.614572Z","iopub.status.idle":"2023-06-20T09:00:36.629409Z","shell.execute_reply.started":"2023-06-20T09:00:36.61454Z","shell.execute_reply":"2023-06-20T09:00:36.627768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Model","metadata":{}},{"cell_type":"markdown","source":"Next, we will be training our model with the train data. We will be using a small batch, fitted with 3 epochs (arbitrary choice) of training to minimize our RAM usage. This is due to our large model, which will eat up quite a large portion of our RAM.\n\nWe will also use sampling for our train data. This will shuffle the data.","metadata":{}},{"cell_type":"code","source":"# Con pesos con las DICOM\n\n# import time\n# from tensorflow.keras.callbacks import ModelCheckpoint as MC\n# from tensorflow.keras.models import load_model\n# from IPython.display import clear_output\n# from sklearn.metrics import roc_auc_score, precision_score\n# import numpy as np\n\n# history = {}\n# start = time.time()\n# debug = 0\n# batch_size = 1000\n# train_size = int(batch_size*0.9)\n\n# max_train_time = 3600 * 4  # horas a segundos de entrenamiento\n\n# checkpoint = MC(filepath='../working/pe_detection_model.h5', monitor='val_loss', save_best_only=True, verbose=1)\n\n# # Bucle de entrenamiento\n# for n, (x, y) in enumerate(custom_dcom_image_generator(batch_size, train.sample(frac=1), False, debug)):\n\n#     if len(x) < 10:  # Intenta filtrar datos vacíos o cortos\n#         break\n\n#     clear_output(wait=True)\n#     print(\"Entrenamiento del lote: %i - %i\" % (batch_size*n, batch_size*(n+1)))\n\n#     model = load_model('../working/pe_detection_model.h5')\n\n#     class_weights = np.ones_like(y[:, 0])  # Inicialmente, asigna un peso de 1 a todas las muestras\n#     class_weights[y[:, 0] == 1] = 10  # Asigna un peso más alto a las muestras de la clase con embolia\n\n#     hist = model.fit(\n#         x,\n#         {'pe_present_on_image': y[:, 0]},\n#         callbacks=checkpoint,\n#         validation_split=0.2,\n#         epochs=3,\n#         batch_size=8,\n#         verbose=debug,\n#         sample_weight=class_weights\n#     )\n\n#     print(\"Métricas de validación del lote:\")\n#     model.evaluate(x[train_size:], {'pe_present_on_image': y[train_size:, 0]})\n\n#     try:\n#         for key in hist.history.keys():\n#             history[key] = np.concatenate([history[key], hist.history[key]], axis=0)\n#     except:\n#         for key in hist.history.keys():\n#             history[key] = hist.history[key]\n\n#     # Para asegurarse de que el modelo no entrene durante demasiado tiempo\n#     if time.time() - start >= max_train_time:\n#         print(\"¡Se acabó el tiempo!\")\n#         break\n\n#     model.save('pe_detection_model.h5')\n#     del model, x, y, hist\n#     K.clear_session()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sin pesos ejecutando las DICOM\n\nhistory = {}\nstart = time.time()\ndebug = 0\nbatch_size = 1000\ntrain_size = int(batch_size*0.9)\n\nmax_train_time = 3600 * 4 #hours to seconds of training\n\ncheckpoint = MC(filepath='../working/pe_detection_model.h5', monitor='val_loss', save_best_only=True, verbose=1)\n#Train loop\nfor n, (x, y) in enumerate(custom_dcom_image_generator(batch_size, train.sample(frac=1), False, debug)):\n    \n    if len(x) < 10: #Tries to filter out empty or short data\n        break\n        \n    clear_output(wait=True)\n    print(\"Training batch: %i - %i\" %(batch_size*n, batch_size*(n+1)))\n    \n    model = load_model('../working/pe_detection_model.h5')\n    hist = model.fit(\n        x[:train_size], #Y values are in a dict as there's more than one target for training output\n        {'pe_present_on_image':y[:train_size, 0]},\n\n        callbacks = checkpoint,\n\n        validation_split=0.2,\n        epochs=3,\n        batch_size=8,\n        verbose=debug\n    )\n    \n    print(\"Metrics for batch validation:\")\n    model.evaluate(x[train_size:],\n                   {'pe_present_on_image':y[train_size:, 0]})\n    \n    try:\n        for key in hist.history.keys():\n            history[key] = np.concatenate([history[key], hist.history[key]], axis=0)\n    except:\n        for key in hist.history.keys():\n            history[key] = hist.history[key]\n            \n    #To make sure that our model don't train overtime\n    if time.time() - start >= max_train_time:\n        print(\"Time's up!\")\n        break\n        \n    model.save('pe_detection_model.h5')\n    del model, x, y, hist\n    K.clear_session()\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T09:00:41.927593Z","iopub.execute_input":"2023-06-20T09:00:41.927947Z","iopub.status.idle":"2023-06-20T11:37:41.718062Z","shell.execute_reply.started":"2023-06-20T09:00:41.927913Z","shell.execute_reply":"2023-06-20T11:37:41.714583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will now look at the history of the training for our data. Since the data is trained across several different batches, there should be some spikes among the values reflected here.","metadata":{}},{"cell_type":"code","source":"for key in history.keys():\n    if key.startswith('val'):\n        continue\n    else:\n        epoch = range(len(history[key]))\n        plt.plot(epoch, history[key]) #X=epoch, Y=value\n        plt.plot(epoch, history['val_'+key])\n        plt.title(key)\n        if 'accuracy' in key:\n            plt.axis([0, len(history[key]), -0.1, 1.1]) #Xmin, Xmax, Ymin, Ymax\n        plt.legend(['train', 'validation'], loc='upper right')\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:37:45.833091Z","iopub.execute_input":"2023-06-20T11:37:45.833495Z","iopub.status.idle":"2023-06-20T11:37:46.182242Z","shell.execute_reply.started":"2023-06-20T11:37:45.833458Z","shell.execute_reply":"2023-06-20T11:37:46.18117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Reading test data...')\ntest = pd.read_csv(\"/kaggle/input/bbdd-estratificada/BBDD-estratificada/test-Estratificada.csv\")\ntest = test.drop(['Unnamed: 0'], axis=1)\ntest = test.drop(['negative_exam_for_pe','qa_motion','qa_contrast','flow_artifact','rv_lv_ratio_gte_1','rv_lv_ratio_lt_1','leftsided_pe','chronic_pe','true_filling_defect_not_pe','rightsided_pe','acute_and_chronic_pe','central_pe','indeterminate'], axis=1)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:38:15.958893Z","iopub.execute_input":"2023-06-20T11:38:15.959266Z","iopub.status.idle":"2023-06-20T11:38:16.024663Z","shell.execute_reply.started":"2023-06-20T11:38:15.959229Z","shell.execute_reply":"2023-06-20T11:38:16.023705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test the model","metadata":{}},{"cell_type":"code","source":"import time\nfrom tensorflow.keras.models import load_model\nfrom IPython.display import clear_output\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\nfrom sklearn.metrics import roc_auc_score\nimport numpy as np\n\npredictions = {}\nstopper = 3600 * 4  # Límite de 4 horas para la predicción\npred_start_time = time.time()\n\np, c = time.time(), time.time()\nbatch_size = 500\n\nl = 0\nn = test.shape[0]\n\n\nfor x in custom_dcom_image_generator(batch_size, test, True, False):\n    clear_output(wait=True)\n    model = load_model(\"../working/pe_detection_model.h5\")\n            \n    preds = model.predict(x, batch_size=8, verbose=1)\n\n    try:\n        for key in preds.keys():\n            if key not in predictions:\n                predictions[key] = []\n            predictions[key].extend(preds[key].flatten().tolist())\n    except Exception as e:\n        print(e)\n        break\n\n    l = (l + batch_size) % n\n    print('Total predicted:', len(predictions['pe_present_on_image']), '/', n)\n    p, c = c, time.time()\n    print(\"One batch time: %.2f seconds\" % (c - p))\n    print(\"ETA: %.2f\" % ((n - l) * (c - p) / batch_size))\n\n    if c - pred_start_time >= stopper:\n        print(\"Time's up!\")\n        break\n        \n    if len(predictions['pe_present_on_image']) == n:\n        # Convertir las predicciones en arrays numpy\n        y_true =  test['pe_present_on_image'].astype(int)\n        y_pred = {key: np.array(value) for key, value in predictions.items()}\n        \n        # Aplicar umbral a las predicciones\n        threshold = 0.5  # Umbral para la conversión\n        y_pred = np.where(y_pred['pe_present_on_image'] > threshold, 1, 0)\n        \n        y_pred = pd.Series(y_pred, name='pe_present_on_image')\n\n        # Calcular las métricas\n        \n        auc = roc_auc_score(y_true, y_pred)\n        print(\"AUC:\", auc)\n        accuracy = accuracy_score(y_true, y_pred)\n        precision = precision_score(y_true, y_pred)\n        recall = recall_score(y_true, y_pred)\n        f1 = f1_score(y_true, y_pred)\n\n        # Imprimir las métricas\n        print(\"Accuracy:\", accuracy)\n        print(\"Precision:\", precision)\n        print(\"Recall:\", recall)\n        print(\"F1 Score:\", f1)\n\n        break\n\n    del model\n    K.clear_session()\n\n    del x, preds\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:38:20.948372Z","iopub.execute_input":"2023-06-20T11:38:20.948717Z","iopub.status.idle":"2023-06-20T11:43:34.813888Z","shell.execute_reply.started":"2023-06-20T11:38:20.948685Z","shell.execute_reply":"2023-06-20T11:43:34.812855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n\ncm = confusion_matrix(y_true, y_pred)\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Muestra la matriz de confusión como un heatmap\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nplt.xlabel(\"Predicción\")\nplt.ylabel(\"Etiqueta verdadera\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:43:49.751266Z","iopub.execute_input":"2023-06-20T11:43:49.751623Z","iopub.status.idle":"2023-06-20T11:43:50.046112Z","shell.execute_reply.started":"2023-06-20T11:43:49.751591Z","shell.execute_reply":"2023-06-20T11:43:50.045239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# End of Part 1","metadata":{}},{"cell_type":"markdown","source":"# Start of Part 2","metadata":{}},{"cell_type":"markdown","source":"# 8. Con las imágenes en HU","metadata":{}},{"cell_type":"code","source":"data_path = \"../input/rsna-str-pulmonary-embolism-detection/\"","metadata":{"execution":{"iopub.status.busy":"2023-06-20T18:21:11.761121Z","iopub.execute_input":"2023-06-20T18:21:11.761606Z","iopub.status.idle":"2023-06-20T18:21:11.76619Z","shell.execute_reply.started":"2023-06-20T18:21:11.761555Z","shell.execute_reply":"2023-06-20T18:21:11.765043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T18:21:31.358616Z","iopub.execute_input":"2023-06-20T18:21:31.359006Z","iopub.status.idle":"2023-06-20T18:21:31.378776Z","shell.execute_reply.started":"2023-06-20T18:21:31.35897Z","shell.execute_reply":"2023-06-20T18:21:31.377712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T18:21:33.346436Z","iopub.execute_input":"2023-06-20T18:21:33.346863Z","iopub.status.idle":"2023-06-20T18:21:33.359815Z","shell.execute_reply.started":"2023-06-20T18:21:33.346825Z","shell.execute_reply":"2023-06-20T18:21:33.357922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_img(path):\n    dcm_img = dcm.dcmread(path)\n    img_data = dcm_img.pixel_array\n    hu_img = apply_modality_lut(img_data, dcm_img)\n    resized_img = cv2.resize(hu_img, (512, 512))\n    return resized_img","metadata":{"execution":{"iopub.status.busy":"2023-06-20T18:21:35.506985Z","iopub.execute_input":"2023-06-20T18:21:35.507384Z","iopub.status.idle":"2023-06-20T18:21:35.513329Z","shell.execute_reply.started":"2023-06-20T18:21:35.507348Z","shell.execute_reply":"2023-06-20T18:21:35.512273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_segmented_lungs(im2, plot=False):\n    im = im2.copy()\n    # Step 1: Convert into a binary image.\n    binary = im < -400\n    \n    if plot:\n        plt.imshow(binary)\n        plt.show()\n        \n    # Step 2: Remove the blobs connected to the border of the image.\n    cleared = clear_border(binary)\n    \n    if plot:\n        plt.imshow(cleared)\n        plt.show()    \n        \n    # Step 3: Label the image.\n    label_image = label(cleared)\n    \n    if plot:\n        plt.imshow(label_image)\n        plt.show()    \n        \n    # Step 4: Keep the labels with 2 largest areas.\n    areas = [r.area for r in regionprops(label_image)]\n    areas.sort()\n    if len(areas) > 0:\n        for region in regionprops(label_image):\n            if region.area < areas[0]:\n                for coordinates in region.coords:\n                       label_image[coordinates[0], coordinates[1]] = 0\n    binary = label_image > 0\n    \n    if plot:\n        plt.imshow(binary)\n        plt.show()  \n        \n    # Step 5: Erosion operation with a disk of radius 2. This operation is to separate the lung nodules attached to the blood vessels.\n    selem = disk(2)\n    binary = binary_erosion(binary, selem)\n    \n    if plot:\n        plt.imshow(binary)\n        plt.show()  \n        \n    # Step 6: Closure operation with a disk of radius 10. This operation is to keep nodules attached to the lung wall.\n    selem = disk(10)\n    binary = binary_closing(binary, selem)\n    \n    if plot:\n        plt.imshow(binary)\n        plt.show() \n        \n    # Step 7: Fill in the small holes inside the binary mask of lungs.\n    edges = roberts(binary)\n    \n    if plot:\n        plt.imshow(edges)\n        plt.show() \n        \n    binary = ndi.binary_fill_holes(edges)\n    \n    if plot:\n        plt.imshow(binary)\n        plt.show() \n        \n    # Step 8: Superimpose the binary mask on the input image.\n    selem = disk(4)\n    binary = dilation(binary, selem)\n    get_high_vals = binary == 0\n    im[get_high_vals] = -2000\n    \n    if plot:\n        plt.imshow(im)\n        plt.show()\n        \n    return im, binary","metadata":{"execution":{"iopub.status.busy":"2023-06-20T18:21:37.506847Z","iopub.execute_input":"2023-06-20T18:21:37.507233Z","iopub.status.idle":"2023-06-20T18:21:37.525359Z","shell.execute_reply.started":"2023-06-20T18:21:37.507196Z","shell.execute_reply":"2023-06-20T18:21:37.524225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Dense, Dropout, Conv2D\n\ninputs = Input((512, 512, 3))\n#x = Conv2D(3, (1, 1), activation='relu')(inputs)\nbase_model = keras.applications.Xception(\n    include_top=False,\n    weights=\"imagenet\"\n)\n\nbase_model.trainable = False\n\noutputs = base_model(inputs, training=False)\noutputs = keras.layers.GlobalAveragePooling2D()(outputs)\noutputs = Dropout(0.25)(outputs)\noutputs = Dense(1024, activation='relu')(outputs)\noutputs = Dense(256, activation='relu')(outputs)\noutputs = Dense(64, activation='relu')(outputs)\nppoi = Dense(1, activation='sigmoid', name='pe_present_on_image')(outputs)\n\nmodel = Model(inputs=inputs, outputs={'pe_present_on_image':ppoi})\n\nopt = keras.optimizers.Adam(lr=0.001)\n\nmodel.compile(optimizer=opt,\n              loss='binary_crossentropy',\n              metrics=['accuracy'])\n\nmodel.summary()\nmodel.save('pe_detection_model.h5')\ndel model\nK.clear_session()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T18:21:41.237749Z","iopub.execute_input":"2023-06-20T18:21:41.238158Z","iopub.status.idle":"2023-06-20T18:21:46.36961Z","shell.execute_reply.started":"2023-06-20T18:21:41.238124Z","shell.execute_reply":"2023-06-20T18:21:46.368621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Definir el generador de imágenes DICOM-HU\ndef custom_dcom_image_generator(batch_size, dataset, test=False, debug=False):\n    fnames = dataset[['StudyInstanceUID', 'SeriesInstanceUID', 'SOPInstanceUID']]\n    \n    if not test:\n        Y = dataset[['pe_present_on_image']]\n        prefix = 'input/rsna-str-pulmonary-embolism-detection/train'\n    else:\n        prefix = 'input/rsna-str-pulmonary-embolism-detection/test'\n    \n    X = []\n    batch = 0\n    for st, sr, so in fnames.values:\n        if debug:\n            print(f\"Current file: ../{prefix}/{st}/{sr}/{so}.dcm\")\n        \n        dicom_path = f\"../{prefix}/{st}/{sr}/{so}.dcm\"\n        hu_image = get_img(dicom_path)\n        segmented_image, _ = get_segmented_lungs(hu_image, plot=False)\n        rgb_image = convert_to_rgb(segmented_image)\n        X.append(rgb_image)\n        \n        del st, sr, so\n        \n        if len(X) == batch_size:\n            if test:\n                yield np.array(X)\n                del X\n            else:\n                yield np.array(X), Y[batch*batch_size:(batch+1)*batch_size].values\n                del X\n                \n            gc.collect()\n            X = []\n            batch += 1\n        \n    if test:\n        yield np.array(X)\n    else:\n        yield np.array(X), Y[batch*batch_size:(batch+1)*batch_size].values\n        del Y\n    del X\n    gc.collect()\n\n# Función para convertir imagen HU a RGB\ndef convert_to_rgb(array):\n    array = array.reshape((512, 512, 1))\n    return np.stack([array, array, array], axis=2).reshape((512, 512, 3))\n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T18:21:59.528131Z","iopub.execute_input":"2023-06-20T18:21:59.528571Z","iopub.status.idle":"2023-06-20T18:21:59.546461Z","shell.execute_reply.started":"2023-06-20T18:21:59.528527Z","shell.execute_reply":"2023-06-20T18:21:59.543258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Entrenamos el modelo","metadata":{}},{"cell_type":"code","source":"# import time\n# from sklearn.model_selection import StratifiedShuffleSplit\n\n# history = {}\n# start = time.time()\n# debug = 0\n# batch_size = 1000\n# train_size = int(batch_size * 0.9)\n# max_train_time = 3600 * 4  # hours to seconds of training\n\n# checkpoint = MC(filepath='../working/pe_detection_model.h5', monitor='val_loss', save_best_only=True, verbose=1)\n\n# # Generate labels for stratification\n# labels = train['pe_present_on_image']\n\n# # Perform stratified shuffle split on indices\n# sss = StratifiedShuffleSplit(n_splits=1, test_size=0.1, random_state=42)\n# train_indices, val_indices = next(sss.split(np.zeros(len(labels)), labels))\n\n# # Create DataFrame for filtered rows in training set\n# train_nuevo = train.loc[train_indices]\n\n# # Create DataFrame for filtered rows in validation set\n# val_nuevo = train.loc[val_indices]\n\n# # Train loop\n# for n, (x, y) in enumerate(custom_dcom_image_generator(batch_size, train_nuevo, False, debug)):\n#     if len(x) < 10:  # Tries to filter out empty or short data\n#         break\n\n#     clear_output(wait=True)\n#     print(\"Training batch: %i - %i\" % (batch_size * n, batch_size * (n + 1)))\n\n#     model = load_model('../working/pe_detection_model.h5')\n#     hist = model.fit(\n#         x[:train_size],\n#         {'pe_present_on_image': y[:train_size, 0]},\n#         callbacks=checkpoint,\n#         validation_data=(x[train_size:], {'pe_present_on_image': y[train_size:, 0]}),\n#         epochs=3,\n#         batch_size=8,\n#         verbose=debug\n#     )\n\n#     try:\n#         for key in hist.history.keys():\n#             history[key] = np.concatenate([history[key], hist.history[key]], axis=0)\n#     except:\n#         for key in hist.history.keys():\n#             history[key] = hist.history[key]\n\n#     # To make sure that our model doesn't train overtime\n#     if time.time() - start >= max_train_time:\n#         print(\"Time's up!\")\n#         break\n\n#     model.save('pe_detection_model.h5')\n#     del model, x, y, hist\n#     K.clear_session()\n#     gc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T18:36:20.453226Z","iopub.execute_input":"2023-06-20T18:36:20.45365Z","iopub.status.idle":"2023-06-20T18:36:20.459775Z","shell.execute_reply.started":"2023-06-20T18:36:20.453611Z","shell.execute_reply":"2023-06-20T18:36:20.458737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# history = {}\n# start = time.time()\n# debug = 0\n# batch_size = 1000\n# train_size = int(batch_size * 0.9)\n\n# max_train_time = 3600 * 4  # hours to seconds of training\n\n# checkpoint = MC(filepath='../working/pe_detection_model.h5', monitor='val_loss', save_best_only=True, verbose=1)\n# print(\"1\")\n# # Train loop\n# for n, (x, y) in enumerate(custom_dcom_image_generator(batch_size, train.sample(frac=1), False, debug)):\n#     if len(x) < 10:  # Tries to filter out empty or short data\n#         break\n#     print(\"2\")\n#     clear_output(wait=True)\n#     print(\"Training batch: %i - %i\" % (batch_size * n, batch_size * (n + 1)))\n#     print(\"3\")\n#     model = load_model('../working/pe_detection_model.h5')\n#     hist = model.fit(\n#         x[:train_size],  # Y values are in a dict as there's more than one target for training output\n#         {'pe_present_on_image': y[:train_size, 0]},\n#         callbacks=checkpoint,\n#         validation_split=0.2,\n#         epochs=3,\n#         batch_size=8,\n#         verbose=debug\n#     )\n\n#     print(\"Metrics for batch validation:\")\n#     model.evaluate(x[train_size:], {'pe_present_on_image': y[train_size:, 0]})\n\n#     try:\n#         for key in hist.history.keys():\n#             history[key] = np.concatenate([history[key], hist.history[key]], axis=0)\n#     except:\n#         for key in hist.history.keys():\n#             history[key] = hist.history[key]\n\n#     # To make sure that our model doesn't train overtime\n#     if time.time() - start >= max_train_time:\n#         print(\"Time's up!\")\n#         break\n\n#     model.save('pe_detection_model.h5')\n#     del model, x, y, hist\n#     K.clear_session()\n#     gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T18:36:16.179569Z","iopub.execute_input":"2023-06-20T18:36:16.179961Z","iopub.status.idle":"2023-06-20T18:36:16.185724Z","shell.execute_reply.started":"2023-06-20T18:36:16.179924Z","shell.execute_reply":"2023-06-20T18:36:16.184686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for key in history.keys():\n    if key.startswith('val'):\n        continue\n    else:\n        epoch = range(len(history[key]))\n        plt.plot(epoch, history[key]) #X=epoch, Y=value\n        plt.plot(epoch, history['val_'+key])\n        plt.title(key)\n        if 'accuracy' in key:\n            plt.axis([0, len(history[key]), -0.1, 1.1]) #Xmin, Xmax, Ymin, Ymax\n        plt.legend(['train', 'validation'], loc='upper right')\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T17:16:11.291346Z","iopub.execute_input":"2023-06-20T17:16:11.291791Z","iopub.status.idle":"2023-06-20T17:16:11.749224Z","shell.execute_reply.started":"2023-06-20T17:16:11.291741Z","shell.execute_reply":"2023-06-20T17:16:11.747788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Testeamos el modelo","metadata":{}},{"cell_type":"code","source":"import time\nfrom tensorflow.keras.models import load_model\nfrom IPython.display import clear_output\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\nfrom sklearn.metrics import roc_auc_score\nimport numpy as np\n\npredictions = {}\nstopper = 3600 * 4  # Límite de 4 horas para la predicción\npred_start_time = time.time()\n\np, c = time.time(), time.time()\nbatch_size = 500\n\nl = 0\nn = test.shape[0]\n\n\nfor x in custom_dcom_image_generator(batch_size, test, True, False):\n    clear_output(wait=True)\n    model = load_model(\"../working/pe_detection_model.h5\")\n            \n    preds = model.predict(x, batch_size=8, verbose=1)\n\n    try:\n        for key in preds.keys():\n            if key not in predictions:\n                predictions[key] = []\n            predictions[key].extend(preds[key].flatten().tolist())\n    except Exception as e:\n        print(e)\n        break\n\n    l = (l + batch_size) % n\n    print('Total predicted:', len(predictions['pe_present_on_image']), '/', n)\n    p, c = c, time.time()\n    print(\"One batch time: %.2f seconds\" % (c - p))\n    print(\"ETA: %.2f\" % ((n - l) * (c - p) / batch_size))\n\n    if c - pred_start_time >= stopper:\n        print(\"Time's up!\")\n        break\n        \n    if len(predictions['pe_present_on_image']) == n:\n        # Convertir las predicciones en arrays numpy\n        y_true =  test['pe_present_on_image'].astype(int)\n        y_pred = {key: np.array(value) for key, value in predictions.items()}\n        \n        # Aplicar umbral a las predicciones\n        threshold = 0.5  # Umbral para la conversión\n        y_pred = np.where(y_pred['pe_present_on_image'] > threshold, 1, 0)\n        \n        y_pred = pd.Series(y_pred, name='pe_present_on_image')\n\n        # Calcular las métricas\n        \n        auc = roc_auc_score(y_true, y_pred)\n        print(\"AUC:\", auc)\n        accuracy = accuracy_score(y_true, y_pred)\n        precision = precision_score(y_true, y_pred)\n        recall = recall_score(y_true, y_pred)\n        f1 = f1_score(y_true, y_pred)\n\n        # Imprimir las métricas\n        print(\"Accuracy:\", accuracy)\n        print(\"Precision:\", precision)\n        print(\"Recall:\", recall)\n        print(\"F1 Score:\", f1)\n\n        break\n\n    del model\n    K.clear_session()\n\n    del x, preds\n    gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n\ncm = confusion_matrix(y_true, y_pred)\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Muestra la matriz de confusión como un heatmap\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nplt.xlabel(\"Predicción\")\nplt.ylabel(\"Etiqueta verdadera\")\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# End Part 2","metadata":{}},{"cell_type":"markdown","source":"# Start Part 3","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport glob\nimport cv2\nimport os\nfrom matplotlib import pyplot as plt\nimport os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom torch.utils.data import TensorDataset, DataLoader,Dataset\nimport albumentations as albu\nfrom skimage.color import gray2rgb\nimport functools\nimport torch\nfrom tqdm.auto import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:26:50.209355Z","iopub.execute_input":"2023-06-21T00:26:50.209744Z","iopub.status.idle":"2023-06-21T00:26:51.54024Z","shell.execute_reply.started":"2023-06-21T00:26:50.209711Z","shell.execute_reply":"2023-06-21T00:26:51.539273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv_path = '/kaggle/input/train-80-20-estratificada/Train-80-20-estratificada.csv'\njpeg_dir = '/kaggle/input/train-jpgs/train-jpegs_partion'","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:54:23.714622Z","iopub.execute_input":"2023-06-21T00:54:23.715025Z","iopub.status.idle":"2023-06-21T00:54:23.720002Z","shell.execute_reply.started":"2023-06-21T00:54:23.714989Z","shell.execute_reply":"2023-06-21T00:54:23.71902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport skimage.io as io\nfrom skimage import exposure, filters, morphology, color\n\n# Paso 1: Carga de la imagen\nimage = io.imread('/kaggle/input/data-binary/data/train/not_normal/8.jpg')\n\n# Paso 2: Conversión a escala de grises\nimage_gray = color.rgb2gray(image)\n\n# Paso 3: Ajuste de contraste\nimage_contrast = exposure.equalize_adapthist(image_gray)\n\n# Paso 4: Normalización\nimage_normalized = image_contrast / np.max(image_contrast)\n\n# Paso 5: Eliminación de ruido\nimage_denoised = filters.median(image_normalized, selem=morphology.disk(3))\n\n# Paso 6: Mejora de bordes\nedges = filters.sobel(image_denoised)\nimage_edges = exposure.rescale_intensity(edges)\n\n# Paso 7: Segmentación\nthreshold = filters.threshold_otsu(image_edges)\nsegmentation = image_edges > threshold\n\n# Paso 8: Postprocesamiento\nsegmentation = morphology.binary_closing(segmentation, selem=morphology.disk(3))\nsegmentation = morphology.remove_small_objects(segmentation, min_size=100)\n\n# Visualización de los resultados\nio.imshow(image)\nio.imshow(segmentation, cmap='gray')\nio.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-21T01:18:15.501546Z","iopub.execute_input":"2023-06-21T01:18:15.50193Z","iopub.status.idle":"2023-06-21T01:18:15.784763Z","shell.execute_reply.started":"2023-06-21T01:18:15.501893Z","shell.execute_reply":"2023-06-21T01:18:15.783896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport skimage.io as io\nfrom skimage import exposure, filters, morphology, color\n\n# Paso 1: Carga de la imagen\nimage = cv2.imread('/kaggle/input/data-binary/data/train/not_normal/106.jpg')\n\n# Paso 2: Redimensionamiento\nresized_image = cv2.resize(image, None, fx=0.5, fy=0.5)\n\n# Paso 3: Filtrado bilateral\nfiltered_image = cv2.bilateralFilter(resized_image, 15, 75, 75)\n\n# Paso 4: Conversión a escala de grises\ngray_image = cv2.cvtColor(filtered_image, cv2.COLOR_BGR2GRAY)\n\n# Paso 5: Umbral adaptativo\nthresh = cv2.adaptiveThreshold(gray_image, 255, cv2.ADAPTIVE_THRESH_MEAN_C, cv2.THRESH_BINARY_INV, 11, 2)\n\n# Paso 6: Detección de bordes con Canny\nedges = cv2.Canny(thresh, 30, 200)\n\n# Paso 7: Dilatación y erosión para cerrar los bordes\nkernel = np.ones((3, 3), np.uint8)\nclosed_edges = cv2.morphologyEx(edges, cv2.MORPH_CLOSE, kernel, iterations=2)\n\n# Paso 8: Eliminación de regiones pequeñas\ncleaned_edges = morphology.remove_small_objects(closed_edges.astype(bool), min_size=500)\n\n# Paso 9: Superposición con la imagen original\noverlay = cv2.bitwise_and(resized_image, resized_image, mask=cleaned_edges.astype(np.uint8))\nresult = cv2.addWeighted(resized_image, 0.7, overlay, 0.3, 0)\n\n# Visualización de los resultados\nio.imshow(resized_image)\nio.imshow(cleaned_edges.astype(np.uint8) * 255, cmap='gray')\nio.imshow(result)\nio.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T01:21:49.782927Z","iopub.execute_input":"2023-06-21T01:21:49.78334Z","iopub.status.idle":"2023-06-21T01:21:50.062008Z","shell.execute_reply.started":"2023-06-21T01:21:49.783304Z","shell.execute_reply":"2023-06-21T01:21:50.060887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport skimage.io as io\nfrom skimage import exposure, filters, morphology, color\n\n# Paso 1: Carga de la imagen\nimage = cv2.imread('/kaggle/input/data-binary/data/train/not_normal/1007.jpg')\n\n# Paso 2: Redimensionamiento\nresized_image = cv2.resize(image, None, fx=0.5, fy=0.5)\n\n# Paso 3: Conversión a escala de grises\ngray_image = cv2.cvtColor(resized_image, cv2.COLOR_BGR2GRAY)\n\n# Paso 4: Mejora del contraste con ecualización de histograma adaptativa\nclahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\nenhanced_image = clahe.apply(gray_image)\n\n# Paso 5: Filtrado bilateral\nfiltered_image = cv2.bilateralFilter(enhanced_image, 15, 75, 75)\n\n# Paso 6: Umbralización adaptativa\nthresh = cv2.adaptiveThreshold(filtered_image, 255, cv2.ADAPTIVE_THRESH_MEAN_C, cv2.THRESH_BINARY_INV, 11, 2)\n\n# Paso 7: Detección de bordes con Canny\nedges = cv2.Canny(thresh, 30, 200)\n\n# Paso 8: Dilatación y erosión para cerrar los bordes\nkernel = np.ones((3, 3), np.uint8)\nclosed_edges = cv2.morphologyEx(edges, cv2.MORPH_CLOSE, kernel, iterations=2)\n\n# Paso 9: Eliminación de regiones pequeñas\ncleaned_edges = morphology.remove_small_objects(closed_edges.astype(bool), min_size=500)\n\n# Paso 10: Relleno de huecos en la máscara\nfilled_mask = morphology.remove_small_holes(cleaned_edges.astype(bool), area_threshold=500)\n\n# Paso 11: Superposición con la imagen original\noverlay = cv2.bitwise_and(resized_image, resized_image, mask=filled_mask.astype(np.uint8))\nresult = cv2.addWeighted(resized_image, 0.7, overlay, 0.3, 0)\n\n# Paso 12: Ajuste de contraste y brillo de la imagen resultante\nresult = exposure.adjust_gamma(result, gamma=1.5)\nresult = exposure.rescale_intensity(result, out_range=(0, 255)).astype(np.uint8)\n\n# Paso 13: Binarización de la imagen resultante para resaltar la embolia\nthreshold = filters.threshold_otsu(result)\nbinary_result = result > threshold\n\n# Paso 14: Eliminación de pequeñas regiones en la máscara binaria\ncleaned_binary = morphology.remove_small_objects(binary_result, min_size=1000)\n\n# Visualización de los resultados\nio.imshow(resized_image)\nio.imshow(filled_mask.astype(np.uint8) * 255, cmap='gray')\nio.imshow(result, cmap='gray')\nio.imshow(binary_result.astype(np.uint8) * 255, cmap='gray')\nio.imshow(cleaned_binary.astype(np.uint8) * 255, cmap='gray')\nio.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T01:28:32.994517Z","iopub.execute_input":"2023-06-21T01:28:32.994895Z","iopub.status.idle":"2023-06-21T01:28:33.32392Z","shell.execute_reply.started":"2023-06-21T01:28:32.994857Z","shell.execute_reply":"2023-06-21T01:28:33.32295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport skimage.io as io\nfrom skimage import exposure, filters, morphology, color\n\n# Paso 1: Carga de la imagen\nimage = cv2.imread('/kaggle/input/data-binary/data/train/not_normal/20.jpg')\n\n# Paso 2: Redimensionamiento\nresized_image = cv2.resize(image, None, fx=0.5, fy=0.5)\n\n# Paso 3: Conversión a escala de grises\ngray_image = cv2.cvtColor(resized_image, cv2.COLOR_BGR2GRAY)\n\n# Paso 4: Mejora del contraste con ecualización de histograma adaptativa\nclahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\nenhanced_image = clahe.apply(gray_image)\n\n# Paso 5: Filtrado bilateral\nfiltered_image = cv2.bilateralFilter(enhanced_image, 15, 75, 75)\n\n# Paso 6: Umbralización adaptativa\nthresh = cv2.adaptiveThreshold(filtered_image, 255, cv2.ADAPTIVE_THRESH_MEAN_C, cv2.THRESH_BINARY_INV, 11, 2)\n\n# Paso 7: Detección de bordes con Canny\nedges = cv2.Canny(thresh, 30, 200)\n\n# Paso 8: Dilatación y erosión para cerrar los bordes\nkernel = np.ones((3, 3), np.uint8)\nclosed_edges = cv2.morphologyEx(edges, cv2.MORPH_CLOSE, kernel, iterations=2)\n\n# Paso 9: Eliminación de regiones pequeñas\ncleaned_edges = morphology.remove_small_objects(closed_edges.astype(bool), min_size=500)\n\n# Paso 10: Relleno de huecos en la máscara\nfilled_mask = morphology.remove_small_holes(cleaned_edges.astype(bool), area_threshold=500)\n\n# Paso 11: Superposición con la imagen original\noverlay = cv2.bitwise_and(resized_image, resized_image, mask=filled_mask.astype(np.uint8))\nresult = cv2.addWeighted(resized_image, 0.7, overlay, 0.3, 0)\n\n# Paso 12: Ajuste de contraste y brillo de la imagen resultante\nresult = exposure.adjust_gamma(result, gamma=1.5)\nresult = exposure.rescale_intensity(result, out_range=(0, 255)).astype(np.uint8)\n\n# Paso 13: Binarización de la imagen resultante para resaltar la embolia\ngray_result = cv2.cvtColor(result, cv2.COLOR_BGR2GRAY)\nthreshold = filters.threshold_otsu(gray_result)\nbinary_result = gray_result > threshold\n\n# Paso 14: Eliminación de pequeñas regiones en la máscara binaria\ncleaned_binary = morphology.remove_small_objects(binary_result, min_size=1000)\n\n# Paso 15: Detección de contornos de la embolia\ncontours, hierarchy = cv2.findContours(cleaned_binary.astype(np.uint8), cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n\n# Paso 16: Dibujar contornos en la imagen original\ncontour_image = np.copy(resized_image)\ncv2.drawContours(contour_image, contours, -1, (0, 255, 0), 2)\n\n# Visualización de los resultados\nio.imshow(resized_image)\nio.imshow(filled_mask.astype(np.uint8) * 255, cmap='gray')\nio.imshow(result)\nio.imshow(cleaned_binary.astype(np.uint8) * 255, cmap='gray')\nio.imshow(contour_image)\nio.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T01:32:28.524216Z","iopub.execute_input":"2023-06-21T01:32:28.524631Z","iopub.status.idle":"2023-06-21T01:32:28.861932Z","shell.execute_reply.started":"2023-06-21T01:32:28.524596Z","shell.execute_reply":"2023-06-21T01:32:28.860854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport skimage.io as io\nfrom skimage import exposure, filters, morphology\n\n# Paso 1: Carga de la imagen\nimage = cv2.imread('/kaggle/input/data-binary/data/train/not_normal/10000.jpg')\n\n# Paso 2: Redimensionamiento\nresized_image = cv2.resize(image, None, fx=0.5, fy=0.5)\n\n# Paso 3: Conversión a escala de grises\ngray_image = cv2.cvtColor(resized_image, cv2.COLOR_BGR2GRAY)\n\n# Paso 4: Mejora del contraste con ecualización de histograma adaptativa\nclahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\nenhanced_image = clahe.apply(gray_image)\n\n# Paso 5: Filtrado bilateral\nfiltered_image = cv2.bilateralFilter(enhanced_image, 15, 75, 75)\n\n# Paso 6: Umbralización adaptativa\nthresh = cv2.adaptiveThreshold(filtered_image, 255, cv2.ADAPTIVE_THRESH_MEAN_C, cv2.THRESH_BINARY_INV, 11, 2)\n\n# Paso 7: Detección de bordes con Canny\nedges = cv2.Canny(thresh, 30, 200)\n\n# Paso 8: Dilatación y erosión para cerrar los bordes\nkernel = np.ones((3, 3), np.uint8)\nclosed_edges = cv2.morphologyEx(edges, cv2.MORPH_CLOSE, kernel, iterations=2)\n\n# Paso 9: Eliminación de regiones pequeñas\ncleaned_edges = morphology.remove_small_objects(closed_edges.astype(bool), min_size=500)\n\n# Paso 10: Relleno de huecos en la máscara\nfilled_mask = morphology.remove_small_holes(cleaned_edges.astype(bool), area_threshold=500)\n\n# Paso 11: Superposición con la imagen original\noverlay = cv2.bitwise_and(resized_image, resized_image, mask=filled_mask.astype(np.uint8))\nresult = cv2.addWeighted(resized_image, 0.7, overlay, 0.3, 0)\n\n# Paso 12: Ajuste de contraste y brillo de la imagen resultante\nresult = exposure.adjust_gamma(result, gamma=1.5)\nresult = exposure.rescale_intensity(result, out_range=(0, 255)).astype(np.uint8)\n\n# Paso 13: Binarización de la imagen resultante para resaltar la embolia\ngray_result = cv2.cvtColor(result, cv2.COLOR_BGR2GRAY)\nthreshold = filters.threshold_otsu(gray_result)\nbinary_result = gray_result > threshold\n\n# Paso 14: Eliminación de pequeñas regiones en la máscara binaria\ncleaned_binary = morphology.remove_small_objects(binary_result, min_size=1000)\n\n# Paso 15: Detección de contornos de la embolia\ncontours, hierarchy = cv2.findContours(cleaned_binary.astype(np.uint8), cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n\n# Paso 16: Crear una máscara de contornos\ncontour_mask = np.zeros_like(cleaned_binary, dtype=np.uint8)\ncv2.drawContours(contour_mask, contours, -1, (255), thickness=cv2.FILLED)\n\n# Paso 17: Dilatación de la máscara de contornos\nkernel = np.ones((5, 5), np.uint8)\ncontour_mask = cv2.dilate(contour_mask, kernel, iterations=2)\n\n# Paso 18: Superposición de la máscara de contornos en la imagen original\ncontour_image = cv2.bitwise_and(resized_image, resized_image, mask=contour_mask)\n\n# Paso 19: Señalización de la embolia con un rectángulo\nx, y, w, h = cv2.boundingRect(contours[0])\ncv2.rectangle(result, (x, y), (x + w, y + h), (0, 255, 0), 2)\n\n# Visualización de los resultados\nio.imshow(resized_image)\nio.imshow(filled_mask.astype(np.uint8) * 255, cmap='gray')\nio.imshow(result)\nio.imshow(cleaned_binary.astype(np.uint8) * 255, cmap='gray')\nio.imshow(contour_image)\nio.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T02:17:41.368797Z","iopub.execute_input":"2023-06-21T02:17:41.369198Z","iopub.status.idle":"2023-06-21T02:17:41.806464Z","shell.execute_reply.started":"2023-06-21T02:17:41.369163Z","shell.execute_reply":"2023-06-21T02:17:41.805479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(train_csv_path)\ntrain_df = train_df.drop(['Unnamed: 0'], axis=1)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:54:30.866667Z","iopub.execute_input":"2023-06-21T00:54:30.867149Z","iopub.status.idle":"2023-06-21T00:54:31.144978Z","shell.execute_reply.started":"2023-06-21T00:54:30.867107Z","shell.execute_reply":"2023-06-21T00:54:31.143934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row = train_df.iloc[101]\nimg = cv2.imread(glob.glob(f\"{jpeg_dir}/{row[0]}/{row[1]}/*{row[2]}.jpg\")[0])\nplt.figure(figsize=[12,6])\nplt.subplot(131)\nplt.imshow(img[:,:,0],cmap='gray')\nplt.subplot(132)\nplt.imshow(img[:,:,1],cmap='gray')\nplt.subplot(133)\nplt.imshow(img[:,:,2],cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2023-06-21T01:08:48.190386Z","iopub.execute_input":"2023-06-21T01:08:48.190755Z","iopub.status.idle":"2023-06-21T01:08:48.653966Z","shell.execute_reply.started":"2023-06-21T01:08:48.190721Z","shell.execute_reply":"2023-06-21T01:08:48.652894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_training_augmentation(y=256,x=256):\n    train_transform = [albu.RandomBrightnessContrast(p=0.3),\n                           albu.VerticalFlip(p=0.5),\n                           albu.HorizontalFlip(p=0.5),\n                           albu.Downscale(p=1.0,scale_min=0.35,scale_max=0.75,),\n                           albu.Resize(y, x)]\n    return albu.Compose(train_transform)\n\n\nformatted_settings = {\n            'input_size': [3, 224, 224],\n            'input_range': [0, 1],\n            'mean': [0.485, 0.456, 0.406],\n            'std': [0.229, 0.224, 0.225],}\ndef preprocess_input(\n    x, mean=None, std=None, input_space=\"RGB\", input_range=None, **kwargs\n):\n\n    if input_space == \"BGR\":\n        x = x[..., ::-1].copy()\n\n    if input_range is not None:\n        if x.max() > 1 and input_range[1] == 1:\n            x = x / 255.0\n\n    if mean is not None:\n        mean = np.array(mean)\n        x = x - mean\n\n    if std is not None:\n        std = np.array(std)\n        x = x / std\n\n    return x\n\ndef get_preprocessing(preprocessing_fn):\n    _transform = [\n        albu.Lambda(image=preprocessing_fn),\n        albu.Lambda(image=to_tensor, mask=to_tensor),\n    ]\n    return albu.Compose(_transform)\n\ndef get_validation_augmentation(y=256,x=256):\n    \"\"\"Add paddings to make image shape divisible by 32\"\"\"\n    test_transform = [albu.Resize(y, x)]\n    return albu.Compose(test_transform)\n\ndef to_tensor(x, **kwargs):\n    \"\"\"\n    Convert image or mask.\n    \"\"\"\n    return x.transpose(2, 0, 1).astype('float32')\n\nclass CTDataset2D(Dataset):\n    def __init__(self,df,transforms = albu.Compose([albu.HorizontalFlip()]),preprocessing=None,size=256,mode='val'):\n        self.df_main = df.values\n        if mode=='val':\n            self.df = self.df_main\n        else:\n            self.update_train_df()\n            \n        self.transforms = transforms\n        self.preprocessing = preprocessing\n        self.size=size\n\n\n    def __getitem__(self, idx):\n        row = self.df[idx]\n        img = cv2.imread(glob.glob(f\"{jpeg_dir}/{row[0]}/{row[1]}/*{row[2]}.jpg\")[0])\n        label = row[3:].astype(int)\n        label[2:] = label[2:] if label[0]==1 else 0\n        if self.transforms:\n            img = self.transforms(image=img)['image']\n        if self.preprocessing:\n            img = self.preprocessing(image=img)['image']\n        return img,torch.from_numpy(label.reshape(-1))\n\n    def __len__(self):\n        return len(self.df)\n    \n    def update_train_df(self):\n        df0 = self.df_main[self.df_main[:,3]==0]\n        df1 = self.df_main[self.df_main[:,3]==1]\n        np.random.shuffle(df0)\n        self.df = np.concatenate([df0[:len(df1)],df1],axis=0)\n        \n\ndef norm(img):\n    img-=img.min()\n    return img/img.max()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T01:08:50.710034Z","iopub.execute_input":"2023-06-21T01:08:50.710401Z","iopub.status.idle":"2023-06-21T01:08:50.74073Z","shell.execute_reply.started":"2023-06-21T01:08:50.710369Z","shell.execute_reply":"2023-06-21T01:08:50.739731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"StudyInstanceUID = list(set(train_df['StudyInstanceUID']))\nprint(len(StudyInstanceUID))\nt_df = train_df[train_df['StudyInstanceUID'].isin(StudyInstanceUID[0:6500])]\nv_df = train_df[train_df['StudyInstanceUID'].isin(StudyInstanceUID[6500:])]","metadata":{"execution":{"iopub.status.busy":"2023-06-21T01:08:53.342156Z","iopub.execute_input":"2023-06-21T01:08:53.342551Z","iopub.status.idle":"2023-06-21T01:08:53.412472Z","shell.execute_reply.started":"2023-06-21T01:08:53.342515Z","shell.execute_reply":"2023-06-21T01:08:53.411264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class config:\n    model_name=\"resnet18\"\n    batch_size = 128\n    WORKERS = 4\n    classes =14\n    resume = False\n    epochs = 5\n    MODEL_PATH = 'log/cpt'\n    if not os.path.exists(MODEL_PATH):\n        os.makedirs(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T01:08:55.06914Z","iopub.execute_input":"2023-06-21T01:08:55.069498Z","iopub.status.idle":"2023-06-21T01:08:55.075131Z","shell.execute_reply.started":"2023-06-21T01:08:55.069466Z","shell.execute_reply":"2023-06-21T01:08:55.074165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessing_fn = functools.partial(preprocess_input, **formatted_settings)\ntrain_dataset = CTDataset2D(t_df,\n                            transforms=get_training_augmentation(),\n                            preprocessing=get_preprocessing(preprocessing_fn),mode='train')\nval_dataset = CTDataset2D(v_df,\n                            transforms=get_validation_augmentation(),\n                            preprocessing=get_preprocessing(preprocessing_fn))","metadata":{"execution":{"iopub.status.busy":"2023-06-21T01:08:56.791549Z","iopub.execute_input":"2023-06-21T01:08:56.791931Z","iopub.status.idle":"2023-06-21T01:08:57.048021Z","shell.execute_reply.started":"2023-06-21T01:08:56.791897Z","shell.execute_reply":"2023-06-21T01:08:57.047052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = DataLoader(train_dataset, batch_size=config.batch_size, shuffle=True, num_workers=config.WORKERS, pin_memory=True)\nval = DataLoader(val_dataset, batch_size=config.batch_size*2, shuffle=False, num_workers=config.WORKERS, pin_memory=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T01:08:58.294368Z","iopub.execute_input":"2023-06-21T01:08:58.294729Z","iopub.status.idle":"2023-06-21T01:08:58.301742Z","shell.execute_reply.started":"2023-06-21T01:08:58.294697Z","shell.execute_reply":"2023-06-21T01:08:58.300736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x,y = train_dataset[-400]\nx.shape,len(y),y,len(train_dataset)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T01:09:00.094767Z","iopub.execute_input":"2023-06-21T01:09:00.095177Z","iopub.status.idle":"2023-06-21T01:09:00.113701Z","shell.execute_reply.started":"2023-06-21T01:09:00.095141Z","shell.execute_reply":"2023-06-21T01:09:00.112732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchvision.models as models\nmodel = models.resnet18(pretrained=True)\nmodel.fc = torch.nn.Linear(in_features=512, out_features=config.classes, bias=True)\nmodel = model.cuda()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:28:08.616789Z","iopub.execute_input":"2023-06-21T00:28:08.617181Z","iopub.status.idle":"2023-06-21T00:28:14.784821Z","shell.execute_reply.started":"2023-06-21T00:28:08.617148Z","shell.execute_reply":"2023-06-21T00:28:14.7839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optimizer = torch.optim.Adam(model.parameters(),lr=5e-4,weight_decay= 0.00001)\nscheduler = torch.optim.lr_scheduler.CosineAnnealingLR(optimizer,T_max= 300,eta_min= 0.000001)\nloss_fn = torch.nn.BCEWithLogitsLoss()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:28:14.786446Z","iopub.execute_input":"2023-06-21T00:28:14.786819Z","iopub.status.idle":"2023-06-21T00:28:14.794064Z","shell.execute_reply.started":"2023-06-21T00:28:14.786778Z","shell.execute_reply":"2023-06-21T00:28:14.792935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.optim.lr_scheduler import ReduceLROnPlateau\nclass trainer:\n    def __init__(self,loss_fn,model,optimizer,scheduler):\n        self.loss_fn = loss_fn\n        self.model = model\n        self.optimizer = optimizer\n        self.scheduler = scheduler\n\n        \n    def batch_train(self, batch_imgs, batch_labels, batch_idx):\n        batch_imgs, batch_labels = batch_imgs.cuda().float(), batch_labels.cuda().float()\n        predicted = self.model(batch_imgs)\n        loss = self.loss_fn(predicted.float(), batch_labels)\n        loss.backward()\n        self.optimizer.step()\n        self.optimizer.zero_grad()\n        return loss.item(), predicted\n    \n    def batch_valid(self, batch_imgs,get_fet):\n        self.model.eval()\n        batch_imgs = batch_imgs.cuda()\n        with torch.no_grad():\n            predicted = self.model(batch_imgs)\n        predicted = torch.sigmoid(predicted)\n        return predicted\n    \n    def train_epoch(self, loader):\n        self.model.train()\n        tqdm_loader = tqdm(loader)\n        current_loss_mean = 0\n        for batch_idx, (imgs,labels) in enumerate(tqdm_loader):\n            loss, predicted = self.batch_train(imgs, labels, batch_idx)\n            current_loss_mean = (current_loss_mean * batch_idx + loss) / (batch_idx + 1)\n            tqdm_loader.set_description('loss: {:.4} lr:{:.6}'.format(\n                    current_loss_mean, self.optimizer.param_groups[0]['lr']))\n            self.scheduler.step(batch_idx)\n        return current_loss_mean\n    \n    def valid_epoch(self, loader,name=\"valid\"):\n        self.model.eval()\n        tqdm_loader = tqdm(loader)\n        current_loss_mean = 0\n        for batch_idx, (imgs,labels) in enumerate(tqdm_loader):\n            with torch.no_grad():\n                batch_imgs = imgs.cuda().float()\n                batch_labels = labels.cuda()\n                predicted = self.model(batch_imgs)\n                loss = self.loss_fn(predicted.float(),batch_labels.float()).item()\n                current_loss_mean = (current_loss_mean * batch_idx + loss) / (batch_idx + 1)\n        score = 1-current_loss_mean\n        print('metric {}'.format(score))\n        return score\n    \n    def run(self,train_loder,val_loder):\n        best_score = -100000\n        for e in range(config.epochs):\n            print(\"----------Epoch {}-----------\".format(e))\n            current_loss_mean = self.train_epoch(train_loder)\n            train_loder.dataset.update_train_df()\n            score = self.valid_epoch(val_loder)\n            if best_score < score:\n                best_score = score\n                torch.save(self.model.state_dict(),config.MODEL_PATH+\"/{}_best.pth\".format(config.model_name))\n\n    def batch_valid_tta(self, batch_imgs):\n        batch_imgs = batch_imgs.cuda()\n        predicted = model(batch_imgs)\n        tta_flip = [[-1],[-2]]\n        for axis in tta_flip:\n            predicted += torch.flip(model(torch.flip(batch_imgs, axis)), axis)\n        predicted = predicted/(1+len(tta_flip))\n        predicted = torch.sigmoid(predicted)\n        return predicted.cpu().numpy()\n            \n    def load_best_model(self):\n        if os.path.exists(config.MODEL_PATH+\"/{}_best.pth\".format(config.model_name)):\n            self.model.load_state_dict(torch.load(config.MODEL_PATH+\"/{}_best.pth\".format(config.model_name)))\n        \n    def predict(self,imgs_tensor,get_fet = False):\n        self.model.train()\n        with torch.no_grad():\n            return self.batch_valid(imgs_tensor,get_fet=get_fet)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:28:15.251743Z","iopub.execute_input":"2023-06-21T00:28:15.252134Z","iopub.status.idle":"2023-06-21T00:28:15.284366Z","shell.execute_reply.started":"2023-06-21T00:28:15.252098Z","shell.execute_reply":"2023-06-21T00:28:15.282947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Trainer = trainer(loss_fn,model,optimizer,scheduler)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:28:19.057186Z","iopub.execute_input":"2023-06-21T00:28:19.057869Z","iopub.status.idle":"2023-06-21T00:28:19.065141Z","shell.execute_reply.started":"2023-06-21T00:28:19.057806Z","shell.execute_reply":"2023-06-21T00:28:19.06208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Trainer.run(train,val)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:28:20.254034Z","iopub.execute_input":"2023-06-21T00:28:20.254439Z","iopub.status.idle":"2023-06-21T00:46:28.232961Z","shell.execute_reply.started":"2023-06-21T00:28:20.254407Z","shell.execute_reply":"2023-06-21T00:46:28.231806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_csv_path = '/kaggle/input/bbdd-estratificada/BBDD-estratificada/test-Estratificada.csv'\ntest_df = pd.read_csv(test_csv_path)\ntest_df = test_df.drop(['Unnamed: 0'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:46:28.235783Z","iopub.execute_input":"2023-06-21T00:46:28.23644Z","iopub.status.idle":"2023-06-21T00:46:28.303152Z","shell.execute_reply.started":"2023-06-21T00:46:28.236391Z","shell.execute_reply":"2023-06-21T00:46:28.302244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Crear el dataset de prueba\ntest_dataset = CTDataset2D(test_df,\n                            transforms=get_validation_augmentation(),\n                            preprocessing=get_preprocessing(preprocessing_fn))\n\ntest_loader = DataLoader(test_dataset, batch_size=config.batch_size*2, shuffle=False, num_workers=config.WORKERS, pin_memory=True)\n\nTrainer.load_best_model()\n\n# Evaluación en el conjunto de prueba\npredictions = []\ntrue_labels = []\n\nfor batch_imgs, batch_labels in test_loader:\n    preds = Trainer.predict(batch_imgs)\n    predictions.append(preds)\n\npredictions = torch.cat(predictions).cpu().numpy()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:46:28.304537Z","iopub.execute_input":"2023-06-21T00:46:28.304982Z","iopub.status.idle":"2023-06-21T00:48:04.612651Z","shell.execute_reply.started":"2023-06-21T00:46:28.304928Z","shell.execute_reply":"2023-06-21T00:48:04.608907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"true_labels = test_df.values[:, 3:]\n\ntrue_labels = true_labels.astype(int)\n\ntrue_labels","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:48:17.565922Z","iopub.execute_input":"2023-06-21T00:48:17.566323Z","iopub.status.idle":"2023-06-21T00:48:17.593449Z","shell.execute_reply.started":"2023-06-21T00:48:17.566288Z","shell.execute_reply":"2023-06-21T00:48:17.592354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Definir el umbral\nthreshold = 0.5\n\n# Convertir las probabilidades en valores binarios\nbinary_predictions = np.where(predictions > threshold, 1, 0)\n\nbinary_predictions","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:48:19.837355Z","iopub.execute_input":"2023-06-21T00:48:19.837774Z","iopub.status.idle":"2023-06-21T00:48:19.847017Z","shell.execute_reply.started":"2023-06-21T00:48:19.837736Z","shell.execute_reply":"2023-06-21T00:48:19.845731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Para todas las variables","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\n\n# Calcular las métricas de evaluación\naccuracy = accuracy_score(true_labels, binary_predictions)\nprecision = precision_score(true_labels, binary_predictions, average='macro', zero_division=1)\nrecall = recall_score(true_labels, binary_predictions, average='macro', zero_division=1)\nf1 = f1_score(true_labels, binary_predictions, average='macro', zero_division=1)\n\n# Imprimir las métricas\nprint(\"Accuracy: {:.4f}\".format(accuracy))\nprint(\"Precision: {:.4f}\".format(precision))\nprint(\"Recall: {:.4f}\".format(recall))\nprint(\"F1-Score: {:.4f}\".format(f1))","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:48:40.52617Z","iopub.execute_input":"2023-06-21T00:48:40.526576Z","iopub.status.idle":"2023-06-21T00:48:40.623075Z","shell.execute_reply.started":"2023-06-21T00:48:40.526524Z","shell.execute_reply":"2023-06-21T00:48:40.621981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\n\n# Calcular el AUC-ROC para cada variable\nauc_scores = {}\nfor col in range(binary_predictions.shape[1]):\n    auc_scores[col] = roc_auc_score(true_labels[:, col], binary_predictions[:, col])\n\n# Imprimir los resultados\nfor col, auc in auc_scores.items():\n    print(\"Variable: {}, AUC-ROC: {:.4f}\".format(col, auc))","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:48:53.087883Z","iopub.execute_input":"2023-06-21T00:48:53.088268Z","iopub.status.idle":"2023-06-21T00:48:53.14361Z","shell.execute_reply.started":"2023-06-21T00:48:53.088234Z","shell.execute_reply":"2023-06-21T00:48:53.142692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Para Pe_present_on_image","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\n\n# Calcular las métricas de evaluación\naccuracy = accuracy_score(true_labels[:,0], binary_predictions[:,0])\nprecision = precision_score(true_labels[:,0], binary_predictions[:,0], average='macro', zero_division=1)\nrecall = recall_score(true_labels[:,0], binary_predictions[:,0], average='macro', zero_division=1)\nf1 = f1_score(true_labels[:,0], binary_predictions[:,0], average='macro', zero_division=1)\n\n# Imprimir las métricas\nprint(\"Accuracy: {:.4f}\".format(accuracy))\nprint(\"Precision: {:.4f}\".format(precision))\nprint(\"Recall: {:.4f}\".format(recall))\nprint(\"F1-Score: {:.4f}\".format(f1))","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:48:59.072139Z","iopub.execute_input":"2023-06-21T00:48:59.072507Z","iopub.status.idle":"2023-06-21T00:48:59.104253Z","shell.execute_reply.started":"2023-06-21T00:48:59.072476Z","shell.execute_reply":"2023-06-21T00:48:59.10347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Calcular la matriz de confusión\ncm = confusion_matrix(true_labels[:, 0], binary_predictions[:,0])\n\n# Mostrar la matriz de confusión como un heatmap\nheatmap = sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nheatmap.set_title(\"Pe_present_on_image\")\nplt.xlabel(\"Predicción\")\nplt.ylabel(\"Etiqueta verdadera\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:49:15.903731Z","iopub.execute_input":"2023-06-21T00:49:15.904137Z","iopub.status.idle":"2023-06-21T00:49:16.121067Z","shell.execute_reply.started":"2023-06-21T00:49:15.904099Z","shell.execute_reply":"2023-06-21T00:49:16.120026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Calcular la matriz de confusión\ncm = confusion_matrix(true_labels[:, 1], binary_predictions[:,1])\n\n# Mostrar la matriz de confusión como un heatmap\nheatmap = sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nheatmap.set_title(\"negative_exam_for_pe\")\nplt.xlabel(\"Predicción\")\nplt.ylabel(\"Etiqueta verdadera\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:49:30.788773Z","iopub.execute_input":"2023-06-21T00:49:30.789163Z","iopub.status.idle":"2023-06-21T00:49:30.983729Z","shell.execute_reply.started":"2023-06-21T00:49:30.789128Z","shell.execute_reply":"2023-06-21T00:49:30.982791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Calcular la matriz de confusión\ncm = confusion_matrix(true_labels[:, 2], binary_predictions[:,2])\n\n# Mostrar la matriz de confusión como un heatmap\nheatmap = sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nheatmap.set_title(\"qa_motion\")\nplt.xlabel(\"Predicción\")\nplt.ylabel(\"Etiqueta verdadera\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:49:33.053591Z","iopub.execute_input":"2023-06-21T00:49:33.056572Z","iopub.status.idle":"2023-06-21T00:49:33.314064Z","shell.execute_reply.started":"2023-06-21T00:49:33.056494Z","shell.execute_reply":"2023-06-21T00:49:33.312947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Calcular la matriz de confusión\ncm = confusion_matrix(true_labels[:, 3], binary_predictions[:,3])\n\n# Mostrar la matriz de confusión como un heatmap\nheatmap = sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nheatmap.set_title(\"qa_contrast\")\nplt.xlabel(\"Predicción\")\nplt.ylabel(\"Etiqueta verdadera\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:49:35.435933Z","iopub.execute_input":"2023-06-21T00:49:35.436347Z","iopub.status.idle":"2023-06-21T00:49:35.646154Z","shell.execute_reply.started":"2023-06-21T00:49:35.436312Z","shell.execute_reply":"2023-06-21T00:49:35.645315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Calcular la matriz de confusión\ncm = confusion_matrix(true_labels[:, 4], binary_predictions[:,4])\n\n# Mostrar la matriz de confusión como un heatmap\nheatmap = sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nheatmap.set_title(\"flow_artifact\")\nplt.xlabel(\"Predicción\")\nplt.ylabel(\"Etiqueta verdadera\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:49:38.653211Z","iopub.execute_input":"2023-06-21T00:49:38.65359Z","iopub.status.idle":"2023-06-21T00:49:38.860255Z","shell.execute_reply.started":"2023-06-21T00:49:38.653555Z","shell.execute_reply":"2023-06-21T00:49:38.859271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Calcular la matriz de confusión\ncm = confusion_matrix(true_labels[:, 5], binary_predictions[:,5])\n\n# Mostrar la matriz de confusión como un heatmap\nheatmap = sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nheatmap.set_title(\"rv_lv_ratio_gte_1\")\nplt.xlabel(\"Predicción\")\nplt.ylabel(\"Etiqueta verdadera\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:49:41.31745Z","iopub.execute_input":"2023-06-21T00:49:41.317888Z","iopub.status.idle":"2023-06-21T00:49:41.510487Z","shell.execute_reply.started":"2023-06-21T00:49:41.31782Z","shell.execute_reply":"2023-06-21T00:49:41.509526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Calcular la matriz de confusión\ncm = confusion_matrix(true_labels[:, 6], binary_predictions[:,6])\n\n# Mostrar la matriz de confusión como un heatmap\nheatmap = sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nheatmap.set_title(\"rv_lv_ratio_lt_1\")\nplt.xlabel(\"Predicción\")\nplt.ylabel(\"Etiqueta verdadera\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:49:44.154872Z","iopub.execute_input":"2023-06-21T00:49:44.155244Z","iopub.status.idle":"2023-06-21T00:49:44.581358Z","shell.execute_reply.started":"2023-06-21T00:49:44.155212Z","shell.execute_reply":"2023-06-21T00:49:44.580398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Calcular la matriz de confusión\ncm = confusion_matrix(true_labels[:, 7], binary_predictions[:,7])\n\n# Mostrar la matriz de confusión como un heatmap\nheatmap = sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nheatmap.set_title(\"leftsided_pe\")\nplt.xlabel(\"Predicción\")\nplt.ylabel(\"Etiqueta verdadera\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:49:46.266857Z","iopub.execute_input":"2023-06-21T00:49:46.267243Z","iopub.status.idle":"2023-06-21T00:49:46.463089Z","shell.execute_reply.started":"2023-06-21T00:49:46.26721Z","shell.execute_reply":"2023-06-21T00:49:46.462052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Calcular la matriz de confusión\ncm = confusion_matrix(true_labels[:, 8], binary_predictions[:,8])\n\n# Mostrar la matriz de confusión como un heatmap\nheatmap = sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nheatmap.set_title(\"chronic_pe\")\nplt.xlabel(\"Predicción\")\nplt.ylabel(\"Etiqueta verdadera\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:49:48.216444Z","iopub.execute_input":"2023-06-21T00:49:48.216801Z","iopub.status.idle":"2023-06-21T00:49:48.422483Z","shell.execute_reply.started":"2023-06-21T00:49:48.216768Z","shell.execute_reply":"2023-06-21T00:49:48.421298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Calcular la matriz de confusión\ncm = confusion_matrix(true_labels[:, 9], binary_predictions[:,9])\n\n# Mostrar la matriz de confusión como un heatmap\nheatmap = sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nheatmap.set_title(\"true_filling_defect_not_pe\")\nplt.xlabel(\"Predicción\")\nplt.ylabel(\"Etiqueta verdadera\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:49:50.395439Z","iopub.execute_input":"2023-06-21T00:49:50.395804Z","iopub.status.idle":"2023-06-21T00:49:50.60525Z","shell.execute_reply.started":"2023-06-21T00:49:50.395771Z","shell.execute_reply":"2023-06-21T00:49:50.603922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Calcular la matriz de confusión\ncm = confusion_matrix(true_labels[:, 10], binary_predictions[:,10])\n\n# Mostrar la matriz de confusión como un heatmap\nheatmap = sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nheatmap.set_title(\"rightsided_pe\")\nplt.xlabel(\"Predicción\")\nplt.ylabel(\"Etiqueta verdadera\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:49:52.181617Z","iopub.execute_input":"2023-06-21T00:49:52.181986Z","iopub.status.idle":"2023-06-21T00:49:52.372216Z","shell.execute_reply.started":"2023-06-21T00:49:52.181952Z","shell.execute_reply":"2023-06-21T00:49:52.371197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Calcular la matriz de confusión\ncm = confusion_matrix(true_labels[:, 11], binary_predictions[:,11])\n\n# Mostrar la matriz de confusión como un heatmap\nheatmap = sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nheatmap.set_title(\"acute_and_chronic_pe\")\nplt.xlabel(\"Predicción\")\nplt.ylabel(\"Etiqueta verdadera\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:49:54.303959Z","iopub.execute_input":"2023-06-21T00:49:54.304324Z","iopub.status.idle":"2023-06-21T00:49:54.514259Z","shell.execute_reply.started":"2023-06-21T00:49:54.304293Z","shell.execute_reply":"2023-06-21T00:49:54.512886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Calcular la matriz de confusión\ncm = confusion_matrix(true_labels[:, 12], binary_predictions[:,12])\n\n# Mostrar la matriz de confusión como un heatmap\nheatmap = sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nheatmap.set_title(\"central_pe\")\nplt.xlabel(\"Predicción\")\nplt.ylabel(\"Etiqueta verdadera\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:49:56.630343Z","iopub.execute_input":"2023-06-21T00:49:56.630714Z","iopub.status.idle":"2023-06-21T00:49:56.835984Z","shell.execute_reply.started":"2023-06-21T00:49:56.630677Z","shell.execute_reply":"2023-06-21T00:49:56.834922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Calcular la matriz de confusión\ncm = confusion_matrix(true_labels[:, 13], binary_predictions[:,13])\n\n# Mostrar la matriz de confusión como un heatmap\nheatmap = sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nheatmap.set_title(\"indeterminate\")\nplt.xlabel(\"Predicción\")\nplt.ylabel(\"Etiqueta verdadera\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:49:58.883693Z","iopub.execute_input":"2023-06-21T00:49:58.884095Z","iopub.status.idle":"2023-06-21T00:49:59.084979Z","shell.execute_reply.started":"2023-06-21T00:49:58.884061Z","shell.execute_reply":"2023-06-21T00:49:59.084037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# End Part 3","metadata":{}}]}