{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#!pip install pylibjpeg pylibjpeg-libjpeg pylibjpeg-openjpeg\n#!conda install -c conda-forge gdcm -y","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:56:50.537444Z","iopub.execute_input":"2024-09-09T01:56:50.538005Z","iopub.status.idle":"2024-09-09T01:56:50.543219Z","shell.execute_reply.started":"2024-09-09T01:56:50.537952Z","shell.execute_reply":"2024-09-09T01:56:50.541977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport pydicom\nfrom pydicom.data import get_testdata_file\nfrom gzip import GzipFile\nimport matplotlib.pyplot as plt\nimport pandas as pd\nfrom PIL import Image\nimport seaborn as sns\nimport os\nimport pydicom\nimport numpy as np","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-09T01:56:50.545443Z","iopub.execute_input":"2024-09-09T01:56:50.545872Z","iopub.status.idle":"2024-09-09T01:56:50.559077Z","shell.execute_reply.started":"2024-09-09T01:56:50.545813Z","shell.execute_reply":"2024-09-09T01:56:50.557575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path = '/kaggle/input/rsna-breast-cancer-detection/train.csv'\ndf = pd.read_csv(file_path)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:56:50.560516Z","iopub.execute_input":"2024-09-09T01:56:50.560957Z","iopub.status.idle":"2024-09-09T01:56:50.708538Z","shell.execute_reply.started":"2024-09-09T01:56:50.560862Z","shell.execute_reply":"2024-09-09T01:56:50.707213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:56:50.710351Z","iopub.execute_input":"2024-09-09T01:56:50.710774Z","iopub.status.idle":"2024-09-09T01:56:50.719318Z","shell.execute_reply.started":"2024-09-09T01:56:50.71073Z","shell.execute_reply":"2024-09-09T01:56:50.718002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Configurar el estilo del gráfico\nsns.set(style=\"whitegrid\")\n\n# Crear la figura\nplt.figure(figsize=(12, 8))\n\n# Histograma de la distribución de la edad con un gráfico de densidad superpuesto\nsns.histplot(df['age'], kde=True, bins=30, color='skyblue', edgecolor='black')\n\n# Calcular estadísticos\nmean_age = df['age'].mean()\nmedian_age = df['age'].median()\nmin_age = df['age'].min()\nmax_age = df['age'].max()\n\n# Dibujar líneas para la media, mediana, mínimo y máximo\nplt.axvline(mean_age, color='red', linestyle='--', label=f'Media: {mean_age:.2f}')\nplt.axvline(median_age, color='green', linestyle='--', label=f'Mediana: {median_age:.2f}')\nplt.axvline(min_age, color='blue', linestyle='--', label=f'Mínimo: {min_age:.2f}')\nplt.axvline(max_age, color='purple', linestyle='--', label=f'Máximo: {max_age:.2f}')\n\n# Añadir anotaciones en el gráfico\nplt.text(mean_age + 1, plt.gca().get_ylim()[1] * 0.8, f'{mean_age:.2f}', color='red')\nplt.text(median_age + 1, plt.gca().get_ylim()[1] * 0.7, f'{median_age:.2f}', color='green')\nplt.text(min_age + 1, plt.gca().get_ylim()[1] * 0.6, f'{min_age:.2f}', color='blue')\nplt.text(max_age + 1, plt.gca().get_ylim()[1] * 0.5, f'{max_age:.2f}', color='purple')\n\n# Etiquetas y título\nplt.title('Distribución de la Edad de las Pacientes con Media, Mediana, Mínimo y Máximo', fontsize=16)\nplt.xlabel('Edad', fontsize=12)\nplt.ylabel('Frecuencia', fontsize=12)\n\n# Mostrar la leyenda\nplt.legend()\n\n# Mostrar el gráfico\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:56:50.723557Z","iopub.execute_input":"2024-09-09T01:56:50.724416Z","iopub.status.idle":"2024-09-09T01:56:51.765183Z","shell.execute_reply.started":"2024-09-09T01:56:50.724371Z","shell.execute_reply":"2024-09-09T01:56:51.7637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Separar los datos entre pacientes con cáncer y sin cáncer\ncancer_df = df[df['cancer'] == 1]\nno_cancer_df = df[df['cancer'] == 0]\n\n# Crear una figura con dos gráficos (uno para cáncer, otro para no cáncer)\nfig, axes = plt.subplots(1, 2, figsize=(16, 8))\n\n# --- Gráfico para pacientes con cáncer ---\nsns.histplot(cancer_df['age'], kde=True, bins=20, color='lightcoral', edgecolor='black', ax=axes[0])\n\n# Calcular estadísticos para pacientes con cáncer\nmean_age_cancer = cancer_df['age'].mean()\nmedian_age_cancer = cancer_df['age'].median()\nmin_age_cancer = cancer_df['age'].min()\nmax_age_cancer = cancer_df['age'].max()\n\n# Añadir líneas de estadísticos\naxes[0].axvline(mean_age_cancer, color='red', linestyle='--', label=f'Media: {mean_age_cancer:.2f}')\naxes[0].axvline(median_age_cancer, color='green', linestyle='--', label=f'Mediana: {median_age_cancer:.2f}')\naxes[0].axvline(min_age_cancer, color='blue', linestyle='--', label=f'Mínimo: {min_age_cancer:.2f}')\naxes[0].axvline(max_age_cancer, color='purple', linestyle='--', label=f'Máximo: {max_age_cancer:.2f}')\n\n# Añadir anotaciones\naxes[0].text(mean_age_cancer + 1, axes[0].get_ylim()[1] * 0.8, f'{mean_age_cancer:.2f}', color='red')\naxes[0].text(median_age_cancer + 1, axes[0].get_ylim()[1] * 0.7, f'{median_age_cancer:.2f}', color='green')\naxes[0].text(min_age_cancer + 1, axes[0].get_ylim()[1] * 0.6, f'{min_age_cancer:.2f}', color='blue')\naxes[0].text(max_age_cancer + 1, axes[0].get_ylim()[1] * 0.5, f'{max_age_cancer:.2f}', color='purple')\n\n# Título y etiquetas para pacientes con cáncer\naxes[0].set_title('Distribución de la Edad en Pacientes con Cáncer', fontsize=16)\naxes[0].set_xlabel('Edad', fontsize=12)\naxes[0].set_ylabel('Frecuencia', fontsize=12)\naxes[0].legend()\n\n# --- Gráfico para pacientes sin cáncer ---\nsns.histplot(no_cancer_df['age'], kde=True, bins=20, color='skyblue', edgecolor='black', ax=axes[1])\n\n# Calcular estadísticos para pacientes sin cáncer\nmean_age_no_cancer = no_cancer_df['age'].mean()\nmedian_age_no_cancer = no_cancer_df['age'].median()\nmin_age_no_cancer = no_cancer_df['age'].min()\nmax_age_no_cancer = no_cancer_df['age'].max()\n\n# Añadir líneas de estadísticos\naxes[1].axvline(mean_age_no_cancer, color='red', linestyle='--', label=f'Media: {mean_age_no_cancer:.2f}')\naxes[1].axvline(median_age_no_cancer, color='green', linestyle='--', label=f'Mediana: {median_age_no_cancer:.2f}')\naxes[1].axvline(min_age_no_cancer, color='blue', linestyle='--', label=f'Mínimo: {min_age_no_cancer:.2f}')\naxes[1].axvline(max_age_no_cancer, color='purple', linestyle='--', label=f'Máximo: {max_age_no_cancer:.2f}')\n\n# Añadir anotaciones\naxes[1].text(mean_age_no_cancer + 1, axes[1].get_ylim()[1] * 0.8, f'{mean_age_no_cancer:.2f}', color='red')\naxes[1].text(median_age_no_cancer + 1, axes[1].get_ylim()[1] * 0.7, f'{median_age_no_cancer:.2f}', color='green')\naxes[1].text(min_age_no_cancer + 1, axes[1].get_ylim()[1] * 0.6, f'{min_age_no_cancer:.2f}', color='blue')\naxes[1].text(max_age_no_cancer + 1, axes[1].get_ylim()[1] * 0.5, f'{max_age_no_cancer:.2f}', color='purple')\n\n# Título y etiquetas para pacientes sin cáncer\naxes[1].set_title('Distribución de la Edad en Pacientes sin Cáncer', fontsize=16)\naxes[1].set_xlabel('Edad', fontsize=12)\naxes[1].set_ylabel('Frecuencia', fontsize=12)\naxes[1].legend()\n\n# Mostrar el gráfico\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:56:51.766596Z","iopub.execute_input":"2024-09-09T01:56:51.767019Z","iopub.status.idle":"2024-09-09T01:56:53.525172Z","shell.execute_reply.started":"2024-09-09T01:56:51.76697Z","shell.execute_reply":"2024-09-09T01:56:53.523967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Seleccionar solo las columnas numéricas del DataFrame\nnumeric_df = df.select_dtypes(include=['float64', 'int64'])\n\n# Calcular la matriz de correlación\ncorr_matrix = numeric_df.corr()\n\n# Crear el gráfico de correlación\nplt.figure(figsize=(10, 8))\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', fmt='.2f', linewidths=0.5)\nplt.title('Matriz de Correlación de Variables Numéricas')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:56:53.526669Z","iopub.execute_input":"2024-09-09T01:56:53.527094Z","iopub.status.idle":"2024-09-09T01:56:54.377764Z","shell.execute_reply.started":"2024-09-09T01:56:53.527036Z","shell.execute_reply":"2024-09-09T01:56:54.376602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Función para cargar imágenes DICOM con manejo de excepciones\ndef load_images_from_paths(image_paths):\n    images = []\n    for path in image_paths:\n        try:\n            dicom_img = pydicom.dcmread(path)\n            if 'PixelData' in dicom_img:\n                img_array = dicom_img.pixel_array\n                # Normalizar la imagen a 8 bits para visualización\n                img_array_normalized = ((img_array - np.min(img_array)) / (np.max(img_array) - np.min(img_array)) * 255).astype(np.uint8)\n                images.append(img_array_normalized)  # Almacenar la imagen normalizada\n            else:\n                print(f\"No PixelData found in: {path}\")\n        except Exception as e:\n            print(f\"Error loading {path}: {e}\")\n    return images\n\n# Función para visualizar un grid de imágenes y otras características\ndef plot_image_grid(images, patient_data, grid_size=(5, 1), figsize=(25, 25)):  # 5 filas, 1 columna\n    fig, axes = plt.subplots(grid_size[0], grid_size[1], figsize=figsize)\n    axes = axes.flatten()\n    \n    for img, data, ax in zip(images, patient_data, axes):\n        ax.imshow(img, cmap='gray')\n        ax.set_title(f\"Paciente: {data['patient_id']}\\nEdad: {data['age']}, Vista: {data['view']}, Lado: {data['laterality']}, Cáncer: {data['cancer']}\\nBiopsia: {data['biopsy']}, BIRARDS: {data['BIRADS']}\")\n        ax.axis('off')\n\n    plt.tight_layout()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:56:54.379433Z","iopub.execute_input":"2024-09-09T01:56:54.379874Z","iopub.status.idle":"2024-09-09T01:56:54.392227Z","shell.execute_reply.started":"2024-09-09T01:56:54.379827Z","shell.execute_reply":"2024-09-09T01:56:54.390902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filtrar 10 pacientes sin cáncer\nimage_paths = []\npatient_data = []\nnocancer_patients = df[df['cancer'] == 0].drop_duplicates('patient_id').head(10)  \n\nfor index, row in nocancer_patients.iterrows():\n    temp = '/kaggle/input/rsna-breast-cancer-detection/train_images/'\n    img_path = temp + str(row['patient_id']) + '/' + os.listdir(temp + str(row['patient_id']))[0]  # Tomar solo la primera imagen\n    image_paths.append(img_path)\n    patient_data.append({\n        'patient_id': row['patient_id'],\n        'age': row['age'],\n        'view': row['view'],\n        'laterality': row['laterality'],\n        'cancer': row['cancer'],\n        'biopsy': row['biopsy'],\n        'BIRADS': row['BIRADS']\n    })  # Guardar datos del paciente\n\n# Cargar y visualizar las imágenes de pacientes sin cáncer\nimages = load_images_from_paths(image_paths)\nplot_image_grid(images, patient_data, grid_size=(2, 5))  # Visualizar 2 fila, 5 columnas","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:56:54.393551Z","iopub.execute_input":"2024-09-09T01:56:54.393889Z","iopub.status.idle":"2024-09-09T01:57:21.739811Z","shell.execute_reply.started":"2024-09-09T01:56:54.393852Z","shell.execute_reply":"2024-09-09T01:57:21.738334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filtrar 10 pacientes con cáncer\nimage_paths = []\npatient_ids = []\ncancer_patients = df[df['cancer'] == 1].drop_duplicates('patient_id').head(10)  # Limitar a 5 pacientes con cáncer\n\nfor index, row in cancer_patients.iterrows():\n    temp = '/kaggle/input/rsna-breast-cancer-detection/train_images/'\n    img_path = temp + str(row['patient_id']) + '/' + os.listdir(temp + str(row['patient_id']))[0]  # Tomar solo la primera imagen\n    image_paths.append(img_path)\n    patient_data.append({\n        'patient_id': row['patient_id'],\n        'age': row['age'],\n        'view': row['view'],\n        'laterality': row['laterality'],\n        'cancer': row['cancer'],\n        'biopsy': row['biopsy'],\n        'BIRADS': row['BIRADS']\n    })  # Guardar datos del paciente\n\n# Cargar y visualizar las imágenes de pacientes sin cáncer\nimages = load_images_from_paths(image_paths)\nplot_image_grid(images, patient_data, grid_size=(2, 5))  # Visualizar 1 fila, 5 columnas","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:57:21.741308Z","iopub.execute_input":"2024-09-09T01:57:21.741653Z","iopub.status.idle":"2024-09-09T01:57:44.266501Z","shell.execute_reply.started":"2024-09-09T01:57:21.741615Z","shell.execute_reply":"2024-09-09T01:57:44.26505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filtrar los pacientes con cáncer y sin cáncer\ncancer_patients = df[df['cancer'] == 1]\nno_cancer_patients = df[df['cancer'] == 0]\n\n# Configurar estilo visual\nsns.set(style=\"whitegrid\")\n\n# Boxplot para comparar las distribuciones de edad\nplt.figure(figsize=(10, 6))\nsns.boxplot(x='cancer', y='age', data=df, palette=['blue', 'red'])\nplt.title('Comparación de Edad: Pacientes con y sin Cáncer')\nplt.xlabel('Cáncer (0=No, 1=Sí)')\nplt.ylabel('Edad')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:57:44.268373Z","iopub.execute_input":"2024-09-09T01:57:44.268836Z","iopub.status.idle":"2024-09-09T01:57:44.543966Z","shell.execute_reply.started":"2024-09-09T01:57:44.268786Z","shell.execute_reply":"2024-09-09T01:57:44.542881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Histograma para visualizar la distribución de la densidad\nplt.figure(figsize=(14, 7))\n\n# Densidad en pacientes con cáncer\nplt.subplot(1, 2, 1)\nsns.histplot(cancer_patients['density'], kde=True, bins=20, color='red', label='Cáncer')\nplt.title('Distribución de Densidad Mamaria - Pacientes con Cáncer')\nplt.xlabel('Densidad')\nplt.ylabel('Frecuencia')\nplt.legend()\n\n# Densidad en pacientes sin cáncer\nplt.subplot(1, 2, 2)\nsns.histplot(no_cancer_patients['density'], kde=True, bins=20, color='blue', label='Sin Cáncer')\nplt.title('Distribución de Densidad Mamaria - Pacientes sin Cáncer')\nplt.xlabel('Densidad')\nplt.ylabel('Frecuencia')\nplt.legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:57:44.54531Z","iopub.execute_input":"2024-09-09T01:57:44.545664Z","iopub.status.idle":"2024-09-09T01:57:45.674057Z","shell.execute_reply.started":"2024-09-09T01:57:44.545625Z","shell.execute_reply":"2024-09-09T01:57:45.672912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Configurar estilo visual\nsns.set(style=\"whitegrid\")\n\n# Gráfico de barras para tipo de vista en pacientes con cáncer\nplt.figure(figsize=(14, 7))\n\nplt.subplot(1, 2, 1)\nsns.countplot(data=cancer_patients, x='view', palette='Reds')\nplt.title('Distribución de Tipos de Vista - Pacientes con Cáncer')\nplt.xlabel('Vista (CC o MLO)')\nplt.ylabel('Frecuencia')\n\n# Gráfico de barras para tipo de vista en pacientes sin cáncer\nplt.subplot(1, 2, 2)\nsns.countplot(data=no_cancer_patients, x='view', palette='Blues')\nplt.title('Distribución de Tipos de Vista - Pacientes sin Cáncer')\nplt.xlabel('Vista (CC o MLO)')\nplt.ylabel('Frecuencia')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:57:45.675398Z","iopub.execute_input":"2024-09-09T01:57:45.675768Z","iopub.status.idle":"2024-09-09T01:57:46.320081Z","shell.execute_reply.started":"2024-09-09T01:57:45.675728Z","shell.execute_reply":"2024-09-09T01:57:46.31876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filtrar casos difíciles y no difíciles\ndifficult_cases = df[df['difficult_negative_case'] == True]\nnon_difficult_cases = df[df['difficult_negative_case'] == False]\n\n# Configurar estilo visual\nsns.set(style=\"whitegrid\")\n\n# Comparar la edad en casos difíciles y no difíciles\nplt.figure(figsize=(14, 7))\n\nplt.subplot(1, 2, 1)\nsns.histplot(difficult_cases['age'], kde=True, bins=20, color='red', label='Difícil')\nplt.title('Distribución de Edad - Casos Difíciles')\nplt.xlabel('Edad')\nplt.ylabel('Frecuencia')\n\nplt.subplot(1, 2, 2)\nsns.histplot(non_difficult_cases['age'], kde=True, bins=20, color='blue', label='No Difícil')\nplt.title('Distribución de Edad - Casos No Difíciles')\nplt.xlabel('Edad')\nplt.ylabel('Frecuencia')\n\nplt.tight_layout()\nplt.show()\n\n# Comparar el tipo de vista en casos difíciles y no difíciles\nplt.figure(figsize=(14, 7))\n\nplt.subplot(1, 2, 1)\nsns.countplot(data=difficult_cases, x='view', palette='Reds')\nplt.title('Distribución de Tipos de Vista - Casos Difíciles')\nplt.xlabel('Vista (CC o MLO)')\nplt.ylabel('Frecuencia')\n\nplt.subplot(1, 2, 2)\nsns.countplot(data=non_difficult_cases, x='view', palette='Blues')\nplt.title('Distribución de Tipos de Vista - Casos No Difíciles')\nplt.xlabel('Vista (CC o MLO)')\nplt.ylabel('Frecuencia')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:57:46.324183Z","iopub.execute_input":"2024-09-09T01:57:46.324598Z","iopub.status.idle":"2024-09-09T01:57:48.435577Z","shell.execute_reply.started":"2024-09-09T01:57:46.324557Z","shell.execute_reply":"2024-09-09T01:57:48.434374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import shutil\n#os.makedirs('/kaggle/working/cancer', exist_ok=True)\n#os.makedirs('/kaggle/working/no_cancer', exist_ok=True)\n\n# Filtrar y mover imágenes a las carpetas correspondientes\n#for index, row in df.iterrows():\n#    image_path = f\"/kaggle/input/rsna-breast-cancer-detection/train_images/{row['patient_id']}/{row['image_id']}.dcm\"\n    \n#    if row['cancer'] == 1:\n#        shutil.copy(image_path, '/kaggle/working/cancer')\n#    else:\n#        shutil.copy(image_path, '/kaggle/working/no_cancer')","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:57:48.437235Z","iopub.execute_input":"2024-09-09T01:57:48.43771Z","iopub.status.idle":"2024-09-09T01:57:48.444427Z","shell.execute_reply.started":"2024-09-09T01:57:48.437653Z","shell.execute_reply":"2024-09-09T01:57:48.443088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convertir DICOM a JPG\ndef convert_dicom_to_jpg(input_folder, output_folder):\n    os.makedirs(output_folder, exist_ok=True)\n    for root, dirs, files in os.walk(input_folder):\n        for file in files:\n            if file.endswith('.dcm'):\n                dicom_img = pydicom.dcmread(os.path.join(root, file))\n                img_array = dicom_img.pixel_array\n                img = Image.fromarray(img_array)\n                img.save(os.path.join(output_folder, file.replace('.dcm', '.jpg')))\n\n# Convertir las imágenes de ambas carpetas\nconvert_dicom_to_jpg('/kaggle/working/cancer', '/kaggle/working/cancer_jpg')\nconvert_dicom_to_jpg('/kaggle/working/no_cancer', '/kaggle/working/no_cancer_jpg')","metadata":{"execution":{"iopub.status.busy":"2024-09-09T02:02:19.613354Z","iopub.execute_input":"2024-09-09T02:02:19.614202Z","iopub.status.idle":"2024-09-09T02:02:19.852549Z","shell.execute_reply.started":"2024-09-09T02:02:19.614124Z","shell.execute_reply":"2024-09-09T02:02:19.850202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Crear generadores de imágenes\ndatagen = ImageDataGenerator(rescale=1./255, validation_split=0.2)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:57:48.446061Z","iopub.execute_input":"2024-09-09T01:57:48.446545Z","iopub.status.idle":"2024-09-09T01:57:48.455677Z","shell.execute_reply.started":"2024-09-09T01:57:48.446492Z","shell.execute_reply":"2024-09-09T01:57:48.454328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = datagen.flow_from_directory(\n    '/kaggle/working/',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='binary',\n    subset='training'\n)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:59:42.065088Z","iopub.execute_input":"2024-09-09T01:59:42.065654Z","iopub.status.idle":"2024-09-09T01:59:42.086964Z","shell.execute_reply.started":"2024-09-09T01:59:42.065604Z","shell.execute_reply":"2024-09-09T01:59:42.085984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_generator = datagen.flow_from_directory(\n    '/kaggle/working/',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='binary',\n    subset='validation'\n)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:57:48.48886Z","iopub.execute_input":"2024-09-09T01:57:48.489397Z","iopub.status.idle":"2024-09-09T01:57:48.510512Z","shell.execute_reply.started":"2024-09-09T01:57:48.489343Z","shell.execute_reply":"2024-09-09T01:57:48.509412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Crear el modelo CNN\nmodel = models.Sequential([\n    layers.Conv2D(32, (3, 3), activation='relu', input_shape=(224, 224, 3)),\n    layers.MaxPooling2D((2, 2)),\n    layers.Conv2D(64, (3, 3), activation='relu'),\n    layers.MaxPooling2D((2, 2)),\n    layers.Conv2D(128, (3, 3), activation='relu'),\n    layers.MaxPooling2D((2, 2)),\n    layers.Flatten(),\n    layers.Dense(512, activation='relu'),\n    layers.Dense(1, activation='sigmoid')\n])\n\n# Compilar el modelo\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\n# Entrenar el modelo\nhistory = model.fit(train_generator, validation_data=validation_generator, epochs=10)\n\n# Guardar el modelo\nmodel.save('mammo_cancer_detection.h5')\n","metadata":{"execution":{"iopub.status.busy":"2024-09-09T01:57:48.511909Z","iopub.execute_input":"2024-09-09T01:57:48.512311Z","iopub.status.idle":"2024-09-09T01:57:48.966757Z","shell.execute_reply.started":"2024-09-09T01:57:48.512271Z","shell.execute_reply":"2024-09-09T01:57:48.965331Z"},"trusted":true},"execution_count":null,"outputs":[]}]}