{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":825822,"sourceType":"datasetVersion","datasetId":434854}],"dockerImageVersionId":30747,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\n\n# Función para cargar las imágenes\ndef load_images(patient_folder):\n    images = []\n    labels = []\n    for patient_id in os.listdir(patient_folder):\n        patient_path = os.path.join(patient_folder, patient_id)\n        if os.path.isdir(patient_path):\n            for subfolder in ['bone', 'brain']:\n                subfolder_path = os.path.join(patient_path, subfolder)\n                if os.path.isdir(subfolder_path):\n                    print(f\"Procesando directorio: {subfolder_path}\")\n                    for img_name in os.listdir(subfolder_path):\n                        if img_name.endswith('.jpg'):  # Aseguramos que el formato es .jpg\n                            img_path = os.path.join(subfolder_path, img_name)\n                            print(f\"Cargando imagen: {img_path}\")\n                            try:\n                                img = plt.imread(img_path)\n                                if img.ndim == 2:\n                                    img = np.expand_dims(img, axis=-1)  # Convertir imagen a 3D si es 2D\n                                if img.ndim == 3:\n                                    img_resized = tf.image.resize(img, (224, 224))\n                                    patient_number = int(patient_id)\n                                    slice_number = int(img_name.split('.')[0])  # Asumimos que el nombre del archivo es el número de corte\n                                    label = df[(df['PatientNumber'] == patient_number) & (df['SliceNumber'] == slice_number)]\n                                    if not label.empty:\n                                        images.append(img_resized.numpy())\n                                        labels.append(label.iloc[0]['Subdural'])  # Ajustar según la columna de interés\n                                    else:\n                                        print(f\"No se encontró etiqueta para la imagen {img_path}\")\n                                else:\n                                    print(f\"La imagen {img_path} no tiene las dimensiones adecuadas.\")\n                            except Exception as e:\n                                print(f\"Error al cargar imagen {img_path}: {e}\")\n    if len(images) == 0:\n        print(\"No se encontraron imágenes en los directorios.\")\n    return np.array(images), np.array(labels)\n\n# Directorio de las imágenes\npatients_dir = \"/kaggle/input/computed-tomography-ct-images/computed-tomography-images-for-intracranial-hemorrhage-detection-and-segmentation-1.0.0/Patients_CT\"\n\n# Cargar el CSV con las etiquetas\ndf = pd.read_csv(\"/kaggle/input/computed-tomography-ct-images/computed-tomography-images-for-intracranial-hemorrhage-detection-and-segmentation-1.0.0/hemorrhage_diagnosis.csv\")\n\n# Mostrar las primeras filas del DataFrame para verificar su contenido\nprint(df.head())\n\n# Cargar imágenes y etiquetas\nimages, labels = load_images(patients_dir)\nprint(f\"Forma de images: {images.shape}\")\nprint(f\"Forma de labels: {labels.shape}\")\n\n# Dividir en conjuntos de entrenamiento y prueba\nif len(images) > 0 and len(labels) > 0 and images.shape[0] == labels.shape[0]:\n    X_train, X_test, y_train, y_test = train_test_split(images, labels, test_size=0.2, random_state=42)\n    print(f\"X_train shape: {X_train.shape}\")\n    print(f\"X_test shape: {X_test.shape}\")\nelse:\n    print(\"No hay suficientes datos para dividir en entrenamiento y prueba o las dimensiones no coinciden.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-07-18T07:53:31.574909Z","iopub.execute_input":"2024-07-18T07:53:31.575813Z","iopub.status.idle":"2024-07-18T07:53:57.236944Z","shell.execute_reply.started":"2024-07-18T07:53:31.575777Z","shell.execute_reply":"2024-07-18T07:53:57.236008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.applications import ResNet50\nfrom keras.layers import Dense, Flatten, GlobalAveragePooling2D, Conv2D, Input\nfrom keras.models import Model\n\n# Definir el tamaño de las imágenes y la cantidad de clases\ninput_shape = (224, 224, 1)\nnum_classes = 6  # Esto debe ajustarse según la cantidad de clases en tu problema\n\n# Crear la entrada del modelo\ninput_tensor = Input(shape=input_shape)\n\n# Añadir una capa Conv2D para convertir 1 canal a 3 canales\nx = Conv2D(3, (3, 3), padding='same')(input_tensor)\n\n# Cargar el modelo base de ResNet50 sin pesos preentrenados\nbase_model = ResNet50(include_top=False, weights=None, input_tensor=x)\n\n# Añadir capas adicionales al modelo base\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\nx = Dense(1024, activation='relu')(x)\noutput_tensor = Dense(num_classes, activation='softmax')(x)\n\n# Crear el modelo completo\nmodel = Model(inputs=input_tensor, outputs=output_tensor)\n\n# Compilar el modelo\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Resumen del modelo\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-07-18T07:58:02.159281Z","iopub.execute_input":"2024-07-18T07:58:02.160001Z","iopub.status.idle":"2024-07-18T07:58:03.055357Z","shell.execute_reply.started":"2024-07-18T07:58:02.159968Z","shell.execute_reply":"2024-07-18T07:58:03.05449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compilar el modelo\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Entrenar el modelo\nhistory = model.fit(X_train, y_train, epochs=10, validation_data=(X_test, y_test))\n\n# Guardar el modelo entrenado\nmodel.save('modelo_resnet50.h5')\n\n# Mostrar los resultados del entrenamiento\nimport matplotlib.pyplot as plt\n\n# Función para graficar las métricas\ndef plot_history(history):\n    plt.figure(figsize=(12, 4))\n    \n    plt.subplot(1, 2, 1)\n    plt.plot(history.history['accuracy'], label='train accuracy')\n    plt.plot(history.history['val_accuracy'], label='val accuracy')\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.title('Training and Validation Accuracy')\n    plt.legend()\n    \n    plt.subplot(1, 2, 2)\n    plt.plot(history.history['loss'], label='train loss')\n    plt.plot(history.history['val_loss'], label='val loss')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.title('Training and Validation Loss')\n    plt.legend()\n    \n    plt.show()\n\n# Graficar los resultados del entrenamiento\nplot_history(history)\n","metadata":{"execution":{"iopub.status.busy":"2024-07-18T08:01:28.398534Z","iopub.execute_input":"2024-07-18T08:01:28.399267Z","iopub.status.idle":"2024-07-18T08:01:32.692523Z","shell.execute_reply.started":"2024-07-18T08:01:28.399234Z","shell.execute_reply":"2024-07-18T08:01:32.691394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils import to_categorical\n\n# Convertir las etiquetas a una codificación \"one-hot\"\ny_train = to_categorical(y_train, num_classes=6)\ny_test = to_categorical(y_test, num_classes=6)\n\n# Compilar el modelo\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Entrenar el modelo\nhistory = model.fit(X_train, y_train, epochs=10, validation_data=(X_test, y_test))\n\n# Guardar el modelo entrenado\nmodel.save('modelo_resnet50.h5')\n\n# Mostrar los resultados del entrenamiento\nimport matplotlib.pyplot as plt\n\n# Función para graficar las métricas\ndef plot_history(history):\n    plt.figure(figsize=(12, 4))\n    \n    plt.subplot(1, 2, 1)\n    plt.plot(history.history['accuracy'], label='train accuracy')\n    plt.plot(history.history['val_accuracy'], label='val accuracy')\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.title('Training and Validation Accuracy')\n    plt.legend()\n    \n    plt.subplot(1, 2, 2)\n    plt.plot(history.history['loss'], label='train loss')\n    plt.plot(history.history['val_loss'], label='val loss')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.title('Training and Validation Loss')\n    plt.legend()\n    \n    plt.show()\n\n# Graficar los resultados del entrenamiento\nplot_history(history)\n","metadata":{"execution":{"iopub.status.busy":"2024-07-18T08:02:31.39433Z","iopub.execute_input":"2024-07-18T08:02:31.395204Z","iopub.status.idle":"2024-07-18T08:10:13.415248Z","shell.execute_reply.started":"2024-07-18T08:02:31.39517Z","shell.execute_reply":"2024-07-18T08:10:13.414341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import img_to_array, load_img\nfrom tensorflow.keras.applications.resnet50 import ResNet50\nfrom tensorflow.keras.layers import Input, Dense, GlobalAveragePooling2D, Conv2D\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\n\n# Función para cargar las imágenes y etiquetas\ndef load_images_and_labels(patients_dir, diagnosis_csv):\n    images = []\n    labels = []\n    df = pd.read_csv(diagnosis_csv)\n    for patient_id in os.listdir(patients_dir):\n        patient_path = os.path.join(patients_dir, patient_id, 'brain')\n        if not os.path.isdir(patient_path):\n            continue\n        for img_name in os.listdir(patient_path):\n            if 'HGE_Seg' in img_name:\n                continue\n            img_path = os.path.join(patient_path, img_name)\n            try:\n                img = load_img(img_path, color_mode='rgb', target_size=(224, 224))\n                img_array = img_to_array(img)\n                images.append(img_array)\n                \n                label = df[(df['PatientNumber'] == int(patient_id)) & \n                           (df['SliceNumber'] == int(img_name.split('.')[0]))]\n                if not label.empty:\n                    labels.append(label.iloc[0, 2:].values)  # Asumiendo que las columnas de etiquetas están después de 'SliceNumber'\n            except Exception as e:\n                print(f\"Error al cargar imagen {img_path}: {e}\")\n    \n    return np.array(images), np.array(labels)\n\n# Paths\npatients_dir = '/kaggle/input/computed-tomography-ct-images/computed-tomography-images-for-intracranial-hemorrhage-detection-and-segmentation-1.0.0/Patients_CT'\ndiagnosis_csv = '/kaggle/input/computed-tomography-ct-images/computed-tomography-images-for-intracranial-hemorrhage-detection-and-segmentation-1.0.0/hemorrhage_diagnosis.csv'\n\n# Cargar imágenes y etiquetas\nimages, labels = load_images_and_labels(patients_dir, diagnosis_csv)\n\n# Normalizar imágenes\nimages = images / 255.0\n\n# Convertir etiquetas a categóricas\nlabels = np.array([label for label in labels])\nlabels = np.argmax(labels, axis=1)\nlabels = to_categorical(labels, num_classes=7)  # 7 clases incluyendo 'No Hemorrhage'\n\n# Dividir en conjuntos de entrenamiento y prueba\nX_train, X_test, y_train, y_test = train_test_split(images, labels, test_size=0.2, random_state=42)\nprint(f\"X_train shape: {X_train.shape}\")\nprint(f\"X_test shape: {X_test.shape}\")\n\n# Definir el modelo\ninput_tensor = Input(shape=(224, 224, 3))  # 3 canales desde el inicio\nbase_model = ResNet50(weights='imagenet', include_top=False, input_tensor=input_tensor)\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\noutput_tensor = Dense(7, activation='softmax')(x)  # Asumiendo que hay 7 clases incluyendo 'No Hemorrhage'\nmodel = Model(inputs=input_tensor, outputs=output_tensor)\n\n# Compilar el modelo\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Crear datasets de tf.data\nbatch_size = 32\ntrain_dataset = tf.data.Dataset.from_tensor_slices((X_train, y_train))\ntrain_dataset = train_dataset.shuffle(buffer_size=1024).batch(batch_size).prefetch(buffer_size=tf.data.AUTOTUNE)\n\nval_dataset = tf.data.Dataset.from_tensor_slices((X_test, y_test))\nval_dataset = val_dataset.batch(batch_size).prefetch(buffer_size=tf.data.AUTOTUNE)\n\n# Entrenar el modelo\nhistory = model.fit(train_dataset, epochs=10, validation_data=val_dataset)\n\n# Guardar el modelo entrenado\nmodel.save('modelo_resnet50.h5')\n\n# Graficar la precisión y la pérdida\nplt.figure(figsize=(12, 4))\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='train accuracy')\nplt.plot(history.history['val_accuracy'], label='val accuracy')\nplt.title('Training and Validation Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='train loss')\nplt.plot(history.history['val_loss'], label='val loss')\nplt.title('Training and Validation Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()\n\n# Predicciones en el conjunto de prueba\ny_pred = model.predict(X_test)\ny_pred_classes = np.argmax(y_pred, axis=1)\ny_true_classes = np.argmax(y_test, axis=1)\n\n# Clases de hemorragia\nhemorrhage_types = ['Intraventricular', 'Intraparenchymal', 'Subarachnoid', 'Epidural', 'Subdural', 'No Hemorrhage']\n\n# Crear y graficar la matriz de confusión para cada clase\nfor i, hemorrhage_type in enumerate(hemorrhage_types):\n    # Crear etiquetas binarias para la clase actual\n    y_true_bin = (y_true_classes == i).astype(int)\n    y_pred_bin = (y_pred_classes == i).astype(int)\n    \n    # Calcular la matriz de confusión\n    conf_matrix = confusion_matrix(y_true_bin, y_pred_bin)\n    disp = ConfusionMatrixDisplay(confusion_matrix=conf_matrix, display_labels=[f'Not {hemorrhage_type}', hemorrhage_type])\n    \n    # Graficar la matriz de confusión\n    plt.figure(figsize=(5, 5))\n    disp.plot(cmap=plt.cm.Blues)\n    plt.title(f'Confusion Matrix for {hemorrhage_type}')\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-07-18T09:18:33.129769Z","iopub.execute_input":"2024-07-18T09:18:33.130441Z","iopub.status.idle":"2024-07-18T09:23:59.082095Z","shell.execute_reply.started":"2024-07-18T09:18:33.130409Z","shell.execute_reply":"2024-07-18T09:23:59.080815Z"},"trusted":true},"execution_count":null,"outputs":[]}]}