{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":2021025,"sourceType":"datasetVersion","datasetId":1209633},{"sourceId":2052874,"sourceType":"datasetVersion","datasetId":981292},{"sourceId":4810680,"sourceType":"datasetVersion","datasetId":2785611},{"sourceId":5205373,"sourceType":"datasetVersion","datasetId":3027365},{"sourceId":11335828,"sourceType":"datasetVersion","datasetId":7091095}],"dockerImageVersionId":30302,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install ultralytics\n!pip install xlrd \n!pip install /kaggle/input/rsnawhl/{pydicom-2.3.0-py3-none-any.whl,pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}\n!pip install -q scikit-learn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T13:44:40.039287Z","iopub.execute_input":"2025-05-06T13:44:40.04023Z","iopub.status.idle":"2025-05-06T13:45:07.042443Z","shell.execute_reply.started":"2025-05-06T13:44:40.040192Z","shell.execute_reply":"2025-05-06T13:45:07.041394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import libraries\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nfrom glob import glob\nimport cv2\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut, apply_windowing\nfrom tqdm.notebook import tqdm\nimport time\nfrom datetime import datetime\nfrom IPython import display\nimport os\nimport torch\nfrom ultralytics import YOLO\nimport multiprocessing as mp\nimport warnings\nwarnings.filterwarnings('ignore')\n\n\n%matplotlib inline\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\ncores = mp.cpu_count()\n\nplt.rcParams.update({'font.size': 10})\nplt.rcParams['figure.figsize'] = (8, 6)\n\nprint('Cores:', cores)\nprint('Device:', device)\nprint('Day: ', datetime.now())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T13:45:07.045041Z","iopub.execute_input":"2025-05-06T13:45:07.045321Z","iopub.status.idle":"2025-05-06T13:45:07.058435Z","shell.execute_reply.started":"2025-05-06T13:45:07.045293Z","shell.execute_reply":"2025-05-06T13:45:07.057547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Pre-processing Libraries\nimport os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport cv2\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut, apply_windowing\nfrom tqdm.notebook import tqdm\nfrom glob import glob\nfrom datetime import datetime\nfrom IPython import display\nimport warnings\n\n# TensorFlow and Keras Libraries\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.keras.applications import DenseNet169\nfrom tensorflow.keras.layers import (Input, Dense, Dropout, Flatten, Conv2D, BatchNormalization, DepthwiseConv2D, Activation, Multiply, Concatenate, GlobalAveragePooling2D)\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.utils import class_weight\nfrom sklearn.metrics import classification_report\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.regularizers import l2\n\n# Additional Libraries for DICOM Handling, YOLO, and multiprocessing\nimport torch\nfrom ultralytics import YOLO\nimport multiprocessing as mp","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T13:45:07.060047Z","iopub.execute_input":"2025-05-06T13:45:07.060575Z","iopub.status.idle":"2025-05-06T13:45:07.075509Z","shell.execute_reply.started":"2025-05-06T13:45:07.060549Z","shell.execute_reply":"2025-05-06T13:45:07.074542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the base path to check the directories\nbase_path = \"/kaggle/input/miniddsm2/MINI-DDSM-Complete-PNG-16\"\n\n# List the subdirectories (Benign, Malignant, Normal)\nclass_names = ['Benign', 'Cancer', 'Normal']\n\n# Function to show and save images\ndef show_and_save_images(class_name, image_paths, save_dir):\n    # Select the first 4 images for display\n    fig, axes = plt.subplots(1, 4, figsize=(20, 5))\n    \n    for i, img_path in enumerate(image_paths[:4]):\n        # Read the image in grayscale\n        img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)\n        \n        # Resize the image for better display\n        img_resized = cv2.resize(img, (300, 300))\n        \n        # Display the image\n        axes[i].imshow(img_resized, cmap='gray')\n        axes[i].axis('off')\n        axes[i].set_title(f\"{class_name} Image {i+1}\")\n    \n    # Save the figure to a file\n    save_path = os.path.join(save_dir, f\"{class_name}_images_grid.png\")\n    fig.savefig(save_path)\n    plt.close(fig)  # Close the figure to free up memory\n    print(f\"Image grid for {class_name} saved at {save_path}\")\n    \n# Create a directory to save the images if it doesn't exist\nsave_dir = \"output_image_grids\"\nos.makedirs(save_dir, exist_ok=True)\n\n# Check if the base directory exists\nif os.path.exists(base_path):\n    print(\"Dataset directory found.\")\n    \n    # Loop through each class and list image paths\n    for class_name in class_names:\n        class_path = os.path.join(base_path, class_name)\n        \n        if os.path.exists(class_path):\n            # Find all PNG images in the class folder and its subdirectories\n            # Filtering for filenames that contain 'LEFT_CC', 'RIGHT_CC', 'LEFT_MLO', or 'RIGHT_MLO'\n            image_paths = glob(f\"{class_path}/**/*LEFT_CC.png\", recursive=True) + \\\n                          glob(f\"{class_path}/**/*RIGHT_CC.png\", recursive=True) + \\\n                          glob(f\"{class_path}/**/*LEFT_MLO.png\", recursive=True) + \\\n                          glob(f\"{class_path}/**/*RIGHT_MLO.png\", recursive=True)\n            \n            print(f\"Number of images found in {class_name} class: {len(image_paths)}\")\n            \n            # Show and save the first 4 images if available\n            if len(image_paths) >= 4:\n                print(f\"Displaying and saving images for {class_name}...\")\n                show_and_save_images(class_name, image_paths, save_dir)\n            else:\n                print(f\"Not enough images available in {class_name} class.\")\n        else:\n            print(f\"Class folder {class_name} not found in the base directory.\")\nelse:\n    print(\"Dataset directory not found.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T13:45:07.077371Z","iopub.execute_input":"2025-05-06T13:45:07.077644Z","iopub.status.idle":"2025-05-06T13:45:21.834666Z","shell.execute_reply.started":"2025-05-06T13:45:07.077605Z","shell.execute_reply":"2025-05-06T13:45:21.833592Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA rsna","metadata":{}},{"cell_type":"code","source":"%cd '/kaggle'\n%pwd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T13:45:21.835999Z","iopub.execute_input":"2025-05-06T13:45:21.836288Z","iopub.status.idle":"2025-05-06T13:45:21.844575Z","shell.execute_reply.started":"2025-05-06T13:45:21.83626Z","shell.execute_reply":"2025-05-06T13:45:21.843276Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# all dataset","metadata":{}},{"cell_type":"code","source":"def get_ddsm_df():\n    is_cancer = {'Benign': 1, 'Cancer': 2, 'Normal': 0} \n    data = {'Filename': [], 'Age':[], 'Density': [], 'Cancer':[], 'View': [], 'Laterality': [], 'Path': []}\n    \n    # Extract the path and view\n    paths = glob(\"input/miniddsm2/MINI-DDSM-Complete-PNG-16/*/*\")\n    for head in tqdm(paths, desc=\"Loading MINI-DDSM dataset\"):\n        path = glob(f\"{head}/*\")\n        path_ics = [x for x in path if \"ics\" in x][0]\n        path_img = [x for x in path if (\"png\" in x and 'Mask' not in x)]\n        \n        if len(path_img) >= 1:\n            # Get information from file *.png\n            for txt in path_img:\n                view = txt.split('.')[-2].split('_')[1]\n                laterality = txt.split('.')[-2].split('_')[0]\n                data['View'].append(view)\n                data['Laterality'].append('L' if laterality == 'LEFT' else 'R')\n                data['Path'].append(txt)\n                data['Cancer'].append(is_cancer[head.split('/')[-2]])\n\n                # Get information from file *.ics\n                with open(path_ics, \"r\") as f:\n                    ics_text = f.read().strip().split(\"\\n\")\n                    for txt in ics_text:\n                        if txt.split()[0].upper() == 'FILENAME':\n                            data['Filename'].append(txt.split()[1] if len(txt.split()) > 1 else 'NaN')\n                        if txt.split()[0].upper() == 'PATIENT_AGE':\n                            data['Age'].append(txt.split()[1] if len(txt.split()) > 1 else 'NaN')\n                        if txt.split()[0].upper() == 'DENSITY':\n                            data['Density'].append(txt.split()[1] if len(txt.split()) > 1 else 'NaN')\n\n    df = pd.DataFrame(data)\n    return df\n\n# Load the DDSM dataset\nddsm = get_ddsm_df()[['Path', 'Cancer', 'View', 'Laterality']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T13:45:21.846074Z","iopub.execute_input":"2025-05-06T13:45:21.846423Z","iopub.status.idle":"2025-05-06T13:45:37.552281Z","shell.execute_reply.started":"2025-05-06T13:45:21.846388Z","shell.execute_reply":"2025-05-06T13:45:37.551389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(ddsm)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T13:45:37.553542Z","iopub.execute_input":"2025-05-06T13:45:37.553835Z","iopub.status.idle":"2025-05-06T13:45:37.563033Z","shell.execute_reply.started":"2025-05-06T13:45:37.553808Z","shell.execute_reply":"2025-05-06T13:45:37.562155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check the number of entries\nprint(ddsm['Cancer'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T13:45:37.564599Z","iopub.execute_input":"2025-05-06T13:45:37.564989Z","iopub.status.idle":"2025-05-06T13:45:37.580713Z","shell.execute_reply.started":"2025-05-06T13:45:37.564951Z","shell.execute_reply":"2025-05-06T13:45:37.579661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Map the numeric 'Cancer' values to their corresponding labels\ncancer_labels = {0: 'Normal', 1: 'Benign', 2: 'Cancer'}\nddsm['Cancer'] = ddsm['Cancer'].map(cancer_labels)\n\n# Prepare total cancer counts per dataset for Benign, Cancer, and Normal\ntotal_cancer = {\n    'MINI-DDSM': ddsm['Cancer'].value_counts()\n}\n\nframe_cancer = pd.DataFrame(total_cancer).T\n\n# Total data for DDSM dataset\ndataset_sum = frame_cancer.sum(axis=1)\nfor name_dataset, total in dataset_sum.to_dict().items():\n    print(f'- {name_dataset}: {total}')\n\n# Total counts for Benign, Cancer, and Normal\ncancer_sum = frame_cancer.sum(axis=0)\n\n# Pie chart showing the distribution of cancer categories (Benign, Cancer, Normal)\nplt.pie(cancer_sum, labels=cancer_sum.index, autopct='%1.1f%%', startangle=90)\nplt.title(\"Percent of Benign, Cancer, and Normal\")\nplt.legend(cancer_sum.index)\nplt.show()\n\n# Resetting index and renaming columns for better readability\nframe_cancer.reset_index(inplace=True)\nframe_cancer.rename(columns={'index': 'dataset', 'Benign': 'Benign', 'Cancer': 'Cancer', 'Normal': 'Normal'}, inplace=True)\n\n# Print the DataFrame to inspect the columns\nprint(frame_cancer)\n\n# Plotting double bar chart for Benign, Cancer, and Normal counts\nx = np.arange(len(frame_cancer.dataset))\nw = 0.25\n\n# Check column names in frame_cancer\nplt.bar(x - w, frame_cancer['Benign'], label='Benign', width=w)\nplt.bar(x, frame_cancer['Cancer'], label='Cancer', width=w)\nplt.bar(x + w, frame_cancer['Normal'], label='Normal', width=w)\nplt.xticks(x, frame_cancer['dataset'])\n\nplt.title(\"Number of Benign, Cancer, and Normal cases\")\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T13:45:37.582089Z","iopub.execute_input":"2025-05-06T13:45:37.582733Z","iopub.status.idle":"2025-05-06T13:45:37.952657Z","shell.execute_reply.started":"2025-05-06T13:45:37.582676Z","shell.execute_reply":"2025-05-06T13:45:37.951718Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Pre Precossing","metadata":{}},{"cell_type":"markdown","source":"## ROI Cropping","metadata":{}},{"cell_type":"markdown","source":"model yolov8: https://colab.research.google.com/drive/1WFSiMeTixqm4anSMNUY_FdzK8TdHB_W5?usp=sharing","metadata":{}},{"cell_type":"code","source":"yolo_model = YOLO(\"/kaggle/input/checkpoint-yolov8l/best.pt\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T13:45:37.955677Z","iopub.execute_input":"2025-05-06T13:45:37.95598Z","iopub.status.idle":"2025-05-06T13:45:39.152756Z","shell.execute_reply.started":"2025-05-06T13:45:37.955955Z","shell.execute_reply":"2025-05-06T13:45:39.151796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class extract_ROI():\n    def __init__(self, df):\n        self.df = df\n        self.detect_model = yolo_model\n        self.folder = None\n        self.df_loc = None\n        self.count_access = 0\n        self.count_error = 0\n\n    def __len__(self):\n        return len(self.df)\n\n    def load_image(self, idx=0):\n        '''\n        Method to load the image\n        Parameter:\n            - path (int or str): index or path of the image\n        Return (numpy.ndarray): image 8-bit 3-channel\n        '''\n        path = idx\n        if isinstance(idx, int):\n            self.df_loc = self.df.iloc[idx]\n            path = self.df_loc.Path\n            \n        else:\n            self.df_loc = self.df[self.df['Path']==idx]\n        mode = path.split('.')[-1]\n        img = None\n        if mode == 'dcm':\n            ds = pydicom.dcmread(path)\n            img2d = ds.pixel_array\n            # apply voi_lut\n            voi_lut = apply_voi_lut(img2d, ds)\n            if np.sum(voi_lut) > 0:\n                img2d = voi_lut\n            # min-max scale\n            img2d = (img2d - img2d.min()) / (img2d.max() - img2d.min())\n            # convert to uint8\n            img2d = (img2d * 255).astype(np.uint8)\n            # convert to float to avoid overflow or underflow losses.\n            if ds.PhotometricInterpretation == 'MONOCHROME1':\n                img2d = np.invert(img2d)\n            # convert to 3-channel\n            img = cv2.cvtColor(img2d, cv2.COLOR_GRAY2BGR)\n        else:\n            img = cv2.imread(path)\n        return img\n    \n    \n    def crop(self, img):\n        results = self.detect_model(img)\n        boxes = results[0].boxes\n        box = boxes[0]\n        xy = box.xyxy\n        x1 = int(xy[0][0].item())\n        y1 = int(xy[0][1].item())\n        x2 = int(xy[0][2].item())\n        y2 = int(xy[0][3].item())\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        img = img[y1: y2, x1:x2]\n        return img\n\n    def plot_image(self, idx=0):\n        img = self.load_image(idx)\n        # convert to grayscale\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        # plot\n        fig, ax = plt.subplots(1, 1)\n        fig.suptitle(f\"Path: {self.df_loc.Path}\")\n        ax.imshow(img, cmap=plt.cm.gray)\n        ax.set_title(f'Image, shape: {img.shape}')\n        ax.axis('off')\n        fig.tight_layout()\n        plt.show()\n\n    def plot(self, idx=0):\n        img = self.load_image(idx)\n        cropped = self.crop(img)\n        # convert to grayscale\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        # plot\n        fig, ax = plt.subplots(1, 2)\n        fig.suptitle(f\"Path: {self.df_loc.Path}\")\n        ax[0].imshow(img, cmap=plt.cm.gray)\n        ax[0].set_title(f'Image, shape: {img.shape}')\n        ax[0].axis('off')\n        ax[1].imshow(cropped, cmap=plt.cm.gray)\n        ax[1].set_title(f'Cropped, shape: {cropped.shape}')\n        ax[1].axis('off')\n        fig.tight_layout()\n        plt.show()\n        \n    def plot_sample(self, resize=256):\n        '''\n        Method to plot a sample of the images\n        Parameter:\n            - cropped (bool): True is origin image, False is cropped image\n            - resize (int): resize the image\n        '''\n        imgs = []\n        df_sample = self.df.sample(100, random_state=42)\n        for path in tqdm(df_sample.Path.values):\n            print(path)\n            img = self.load_image(path)\n            cropped = self.crop(img)\n            display.clear_output(wait=True)\n            imgs.append(cropped)\n        if resize:\n            imgs = [cv2.resize(img, (resize, resize)) for img in imgs]\n        # plot\n        n_cols = 10\n        n_rows = 10\n        fig, ax = plt.subplots(n_rows, n_cols, figsize=(n_cols*5,n_rows*5))\n        i = 0\n        for r in tqdm(range(0,n_rows)):\n            for c in range(0,n_cols):\n                idx = r*n_cols + c\n                ax_idx = ax[r,c]\n                ax_idx.imshow(imgs[i],cmap=plt.cm.gray)\n                i+=1\n                ax_idx.axis('off')\n        plt.tight_layout()\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T13:45:39.154042Z","iopub.execute_input":"2025-05-06T13:45:39.154317Z","iopub.status.idle":"2025-05-06T13:45:39.17402Z","shell.execute_reply.started":"2025-05-06T13:45:39.154293Z","shell.execute_reply":"2025-05-06T13:45:39.173208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"roi_ddsm = extract_ROI(ddsm)\nprint(roi_ddsm.__len__())\nroi_ddsm.df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T13:45:39.175036Z","iopub.execute_input":"2025-05-06T13:45:39.175298Z","iopub.status.idle":"2025-05-06T13:45:39.194Z","shell.execute_reply.started":"2025-05-06T13:45:39.175275Z","shell.execute_reply":"2025-05-06T13:45:39.192975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in range (10):\n    roi_ddsm.plot(i)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T13:45:39.195198Z","iopub.execute_input":"2025-05-06T13:45:39.195447Z","iopub.status.idle":"2025-05-06T13:45:48.095954Z","shell.execute_reply.started":"2025-05-06T13:45:39.195425Z","shell.execute_reply":"2025-05-06T13:45:48.095037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming your DataFrame is named roi_ddsm\ncancer_class_counts = roi_ddsm.df['Cancer'].value_counts()\n\n# Display the counts for each cancer class\nprint(cancer_class_counts)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T13:45:48.097461Z","iopub.execute_input":"2025-05-06T13:45:48.097753Z","iopub.status.idle":"2025-05-06T13:45:48.104139Z","shell.execute_reply.started":"2025-05-06T13:45:48.097726Z","shell.execute_reply":"2025-05-06T13:45:48.103131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"roi_ddsm.plot_image(idx=0)  # Change idx to view other images","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T13:45:48.105293Z","iopub.execute_input":"2025-05-06T13:45:48.105606Z","iopub.status.idle":"2025-05-06T13:45:48.623191Z","shell.execute_reply.started":"2025-05-06T13:45:48.10558Z","shell.execute_reply":"2025-05-06T13:45:48.622188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\n\n# Base directory for saving cropped images\noutput_base_dir = \"output/cropped_images\"\n\n# Mapping for cancer classes\nclass_mapping = {0: \"Normal\", 1: \"Benign\", 2: \"Cancer\"}\n\n# Create output directories for each class\nfor class_name in class_mapping.values():\n    os.makedirs(os.path.join(output_base_dir, class_name), exist_ok=True)\n\ndef apply_roi_to_all_images_by_class(extract_roi_obj, save_dir, class_map):\n    \"\"\"\n    Apply the extract_ROI crop to all images and save them into folders based on the cancer class.\n    Parameters:\n        - extract_roi_obj: Instance of the extract_ROI class.\n        - save_dir: Base directory to save the cropped images.\n        - class_map: Dictionary mapping cancer class numbers to class names.\n    \"\"\"\n    for idx in range(len(extract_roi_obj)):\n        try:\n            # Load the image\n            img = extract_roi_obj.load_image(idx)\n            \n            # Crop the image using extract_ROI\n            cropped_img = extract_roi_obj.crop(img)\n            \n            # Get the cancer class as an integer\n            cancer_class = extract_roi_obj.df.iloc[idx][\"Cancer\"]\n            \n            # Check if cancer_class is a string (e.g., 'Cancer') or an integer (e.g., 2)\n            if isinstance(cancer_class, str):\n                # If it's a string, map it to the integer value\n                cancer_class = {'Normal': 0, 'Benign': 1, 'Cancer': 2}.get(cancer_class, None)\n            \n            # Ensure cancer_class is a valid key in class_map\n            if cancer_class is None or cancer_class not in class_map:\n                print(f\"Invalid cancer class for image at index {idx}\")\n                continue\n            \n            class_name = class_map[cancer_class]\n            \n            # Create the save path\n            file_name = os.path.basename(extract_roi_obj.df.iloc[idx][\"Path\"])\n            save_path = os.path.join(save_dir, class_name, file_name)\n            \n            # Save the cropped image\n            cv2.imwrite(save_path, cropped_img)\n            print(f\"Processed and saved: {save_path}\")\n        except Exception as e:\n            print(f\"Error processing image at index {idx}: {e}\")\n\n# Apply the ROI extraction and save cropped images\napply_roi_to_all_images_by_class(roi_ddsm, output_base_dir, class_mapping)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T13:45:48.624465Z","iopub.execute_input":"2025-05-06T13:45:48.624756Z","iopub.status.idle":"2025-05-06T14:15:16.788042Z","shell.execute_reply.started":"2025-05-06T13:45:48.624726Z","shell.execute_reply":"2025-05-06T14:15:16.787054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ndef count_files_in_class_folders(base_dir, class_map):\n    \"\"\"\n    Counts the number of files in each class folder.\n    Parameters:\n        - base_dir: Base directory containing class folders.\n        - class_map: Dictionary mapping cancer class numbers to class names.\n    Returns:\n        - counts: Dictionary with class names as keys and file counts as values.\n    \"\"\"\n    counts = {}\n    for class_num, class_name in class_map.items():\n        class_dir = os.path.join(base_dir, class_name)\n        if os.path.exists(class_dir):\n            counts[class_name] = len([f for f in os.listdir(class_dir) if os.path.isfile(os.path.join(class_dir, f))])\n        else:\n            counts[class_name] = 0  # If the folder does not exist, set count to 0\n    return counts\n\n# Count files in each class folder\nfile_counts = count_files_in_class_folders(output_base_dir, class_mapping)\n\n# Print the counts\nfor class_name, count in file_counts.items():\n    print(f\"Class '{class_name}' has {count} files.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T14:15:16.789395Z","iopub.execute_input":"2025-05-06T14:15:16.790252Z","iopub.status.idle":"2025-05-06T14:15:16.875275Z","shell.execute_reply.started":"2025-05-06T14:15:16.790223Z","shell.execute_reply":"2025-05-06T14:15:16.874415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nprint(os.listdir(\"output/cropped_images\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T14:15:16.876574Z","iopub.execute_input":"2025-05-06T14:15:16.876955Z","iopub.status.idle":"2025-05-06T14:15:16.882263Z","shell.execute_reply.started":"2025-05-06T14:15:16.876919Z","shell.execute_reply":"2025-05-06T14:15:16.881386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nfrom tensorflow.keras.preprocessing.image import load_img\n\n# Define the base directory and class names\nbase_dir = 'output/cropped_images'\nclass_names = ['Benign', 'Cancer', 'Normal']\n\n# Number of images to display per class\nnum_images_per_class = 5\n\n# Create a plot for images\nplt.figure(figsize=(15, 10))\n\n# Iterate through each class\nfor class_idx, class_name in enumerate(class_names):\n    class_dir = os.path.join(base_dir, class_name)  # Path to class directory\n    image_files = os.listdir(class_dir)[:num_images_per_class]  # Get first N images\n    \n    for i, image_file in enumerate(image_files):\n        # Load the image\n        img_path = os.path.join(class_dir, image_file)\n        img = load_img(img_path)  # Load the image as is\n        img_size = img.size  # Get image size (width, height)\n        \n        # Plot the image\n        plt.subplot(len(class_names), num_images_per_class, class_idx * num_images_per_class + i + 1)\n        plt.imshow(img)\n        plt.axis('off')\n        plt.title(f\"{class_name}\\n{img_size[0]}x{img_size[1]}\")  # Display size as title\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T14:15:16.883343Z","iopub.execute_input":"2025-05-06T14:15:16.883619Z","iopub.status.idle":"2025-05-06T14:15:21.131471Z","shell.execute_reply.started":"2025-05-06T14:15:16.883597Z","shell.execute_reply":"2025-05-06T14:15:21.13055Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Hasil ROI CROPPING","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tensorflow.keras.preprocessing.image import load_img\nimport os\n\n# Define paths to the original and cropped images\noriginal_img_path = 'input/miniddsm2/MINI-DDSM-Complete-PNG-16/Cancer/1520/A_1520_1.LEFT_CC.png'\nroi_img_path = 'output/cropped_images/Cancer/A_1520_1.LEFT_CC.png'\n\n# Load images in RGB mode\noriginal_img = load_img(original_img_path)\nroi_img = load_img(roi_img_path)\n\n# Create the plot\nplt.figure(figsize=(10, 5))\n\n# Plot Original Image\nplt.subplot(1, 2, 1)\nplt.imshow(original_img)\nplt.title(\"Original Image\")\nplt.axis('off')\n\n# Plot ROI Image\nplt.subplot(1, 2, 2)\nplt.imshow(roi_img)\nplt.title(\"ROI Cropped Image\")\nplt.axis('off')\n\nplt.tight_layout()\n\n# Save the figure\noutput_path = \"/kaggle/working/comparison_output(1024).png\"\nplt.savefig(output_path, bbox_inches='tight')\n\n# Display the plot\nplt.show()\n\nprint(f\"✅ Image displayed and saved successfully to: {output_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T14:15:21.132889Z","iopub.execute_input":"2025-05-06T14:15:21.133756Z","iopub.status.idle":"2025-05-06T14:15:22.726081Z","shell.execute_reply.started":"2025-05-06T14:15:21.133711Z","shell.execute_reply":"2025-05-06T14:15:22.725029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ndef get_folder_size_and_count(folder_path):\n    total_size = 0\n    total_files = 0\n    \n    # Walk through the directory\n    for dirpath, dirnames, filenames in os.walk(folder_path):\n        total_files += len(filenames)\n        total_size += sum(os.path.getsize(os.path.join(dirpath, filename)) for filename in filenames)\n    \n    # Convert size to GB\n    total_size_gb = total_size / (1024 * 1024 * 1024)\n    \n    return total_files, total_size_gb\n\n# Folder path to check\nfolder_path = 'output/original_images'  # Adjust path if needed\n\n# Get number of files and folder size\nfile_count, folder_size_gb = get_folder_size_and_count(folder_path)\n\nprint(f\"Total files: {file_count}\")\nprint(f\"Total size: {folder_size_gb:.2f} GB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T14:15:22.727308Z","iopub.execute_input":"2025-05-06T14:15:22.727578Z","iopub.status.idle":"2025-05-06T14:15:22.734528Z","shell.execute_reply.started":"2025-05-06T14:15:22.727553Z","shell.execute_reply":"2025-05-06T14:15:22.733633Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Resizing","metadata":{}},{"cell_type":"markdown","source":"### NO PADDING","metadata":{}},{"cell_type":"code","source":"import os\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array, save_img\nimport tensorflow as tf\nfrom tqdm import tqdm  # Progress bar for large datasets\n\n# Define the base directory and output directory\nbase_dir = 'output/cropped_images'\noutput_dir = 'output/resized_images'\nclass_names = ['Benign', 'Cancer', 'Normal']\n\n# Target size\ntarget_size = (512, 512)\n\n# Create output directory structure\nos.makedirs(output_dir, exist_ok=True)\nfor class_name in class_names:\n    os.makedirs(os.path.join(output_dir, class_name), exist_ok=True)\n\n# Function to resize image without padding\ndef resize_image(image_path, target_size):\n    img = load_img(image_path)  # Load image\n    img_array = img_to_array(img)  # Convert to array\n    resized_img = tf.image.resize(img_array, target_size, method=tf.image.ResizeMethod.BILINEAR)  # Resize without padding\n    return resized_img\n\n# Resize images for each class\nfor class_name in class_names:\n    class_dir = os.path.join(base_dir, class_name)\n    output_class_dir = os.path.join(output_dir, class_name)\n    \n    print(f\"Processing class: {class_name}\")\n    for image_file in tqdm(os.listdir(class_dir)):\n        img_path = os.path.join(class_dir, image_file)\n        output_img_path = os.path.join(output_class_dir, image_file)\n        \n        # Resize and save the image\n        resized_img = resize_image(img_path, target_size)\n        resized_img = tf.cast(resized_img, tf.uint8)  # Convert back to uint8 for saving\n        save_img(output_img_path, resized_img.numpy())  # Save the resized image\n\nprint(\"All images have been resized and saved!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T14:17:10.446724Z","iopub.execute_input":"2025-05-06T14:17:10.447096Z","iopub.status.idle":"2025-05-06T14:34:37.10601Z","shell.execute_reply.started":"2025-05-06T14:17:10.447067Z","shell.execute_reply":"2025-05-06T14:34:37.105072Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### PADDING","metadata":{}},{"cell_type":"code","source":"#import os\n#from tensorflow.keras.preprocessing.image import load_img, img_to_array, save_img\n#import tensorflow as tf\n#from tqdm import tqdm  # Progress bar for large datasets\n\n# Define the base directory and output directory\n#base_dir = 'output/cropped_images'\n#output_dir = 'output/resized_images'\n#class_names = ['Benign', 'Cancer', 'Normal']\n\n# Target size\n#target_size = (1024, 1024)\n\n# Create output directory structure\n#os.makedirs(output_dir, exist_ok=True)\n#for class_name in class_names:\n#    os.makedirs(os.path.join(output_dir, class_name), exist_ok=True)\n\n# Function to resize and pad image\n#def resize_with_padding(image_path, target_size):\n#    img = load_img(image_path)  # Load image\n#    img_array = img_to_array(img)  # Convert to array\n#    padded_img = tf.image.resize_with_pad(img_array, target_size[0], target_size[1])  # Resize with padding\n#    return padded_img\n\n# Resize images for each class\n#for class_name in class_names:\n#    class_dir = os.path.join(base_dir, class_name)\n#    output_class_dir = os.path.join(output_dir, class_name)\n    \n#    print(f\"Processing class: {class_name}\")\n#    for image_file in tqdm(os.listdir(class_dir)):\n#        img_path = os.path.join(class_dir, image_file)\n#        output_img_path = os.path.join(output_class_dir, image_file)\n        \n        # Resize and save the image\n#        resized_img = resize_with_padding(img_path, target_size)\n#        resized_img = tf.cast(resized_img, tf.uint8)  # Convert back to uint8 for saving\n#        save_img(output_img_path, resized_img.numpy())  # Save the resized image\n\n#print(\"All images have been resized and saved!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T14:34:37.107845Z","iopub.execute_input":"2025-05-06T14:34:37.108525Z","iopub.status.idle":"2025-05-06T14:34:37.11458Z","shell.execute_reply.started":"2025-05-06T14:34:37.108487Z","shell.execute_reply":"2025-05-06T14:34:37.113547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nfrom tensorflow.keras.preprocessing.image import load_img\n\n# Define the base directory of resized images and class names\nresized_dir = 'output/resized_images'\nclass_names = ['Benign', 'Cancer', 'Normal']\n\n# Number of images to display per class\nnum_images_per_class = 3\n\n# Create a plot for resized images\nplt.figure(figsize=(15, 10))\n\n# Iterate through each class\nfor class_idx, class_name in enumerate(class_names):\n    class_dir = os.path.join(resized_dir, class_name)  # Path to class directory\n    image_files = os.listdir(class_dir)[:num_images_per_class]  # Get first N images\n    \n    for i, image_file in enumerate(image_files):\n        # Load the resized image\n        img_path = os.path.join(class_dir, image_file)\n        img = load_img(img_path)  # Load the image as is\n        img_size = img.size  # Get image size (width, height)\n        \n        # Plot the resized image\n        plt.subplot(len(class_names), num_images_per_class, class_idx * num_images_per_class + i + 1)\n        plt.imshow(img)\n        plt.axis('off')\n        plt.title(f\"{class_name}\\n{img_size[0]}x{img_size[1]}\")  # Show new size in title\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T14:34:37.115786Z","iopub.execute_input":"2025-05-06T14:34:37.116128Z","iopub.status.idle":"2025-05-06T14:34:37.944489Z","shell.execute_reply.started":"2025-05-06T14:34:37.116094Z","shell.execute_reply":"2025-05-06T14:34:37.943475Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Hasil Resizing","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tensorflow.keras.preprocessing.image import load_img\nimport os\n\n# Define paths to the original and cropped images\noriginal_img_path = 'output/cropped_images/Cancer/A_1520_1.LEFT_CC.png'\nresized_img_path = 'output/resized_images/Cancer/A_1520_1.LEFT_CC.png'\n\n# Load images in RGB mode\noriginal_img = load_img(original_img_path)\nroi_img = load_img(resized_img_path)\n\n# Create the plot\nplt.figure(figsize=(10, 5))\n\n# Plot Original Image\nplt.subplot(1, 2, 1)\nplt.imshow(original_img)\nplt.title(\"Original Image\")\nplt.axis('off')\n\n# Plot ROI Image\nplt.subplot(1, 2, 2)\nplt.imshow(roi_img)\nplt.title(\"Rezised Image\")\nplt.axis('off')\n\nplt.tight_layout()\n\n# Save the figure\noutput_path = \"/kaggle/working/comparison_output(1024)-2.png\"\nplt.savefig(output_path, bbox_inches='tight')\n\n# Display the plot\nplt.show()\n\nprint(f\"✅ Image displayed and saved successfully to: {output_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T14:34:37.947288Z","iopub.execute_input":"2025-05-06T14:34:37.947717Z","iopub.status.idle":"2025-05-06T14:34:38.804328Z","shell.execute_reply.started":"2025-05-06T14:34:37.947656Z","shell.execute_reply":"2025-05-06T14:34:38.8034Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Applying Filter","metadata":{}},{"cell_type":"code","source":"#import cv2\n#import numpy as np\n#from scipy.signal import wiener\n#from skimage import io, util\n#import os\n#from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Define CLAHE parameters\n#clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n\n# Preprocessing function for a single image\n#def preprocess_image(image_path, save_path):\n    # Read the image in grayscale\n    #img = cv2.imread(image_path, cv2.IMREAD_GRAYSCALE)\n    \n    # Apply CLAHE\n    #img_clahe = clahe.apply(img)\n    \n    # Apply Wiener filter for noise reduction\n    #img_wiener = wiener(img_clahe, (5, 5))\n    #img_wiener = np.clip(img_wiener, 0, 255).astype(np.uint8)\n    \n    # Apply Median filter for further noise reduction\n    #img_median = cv2.medianBlur(img_wiener, 5)\n    \n    # Save preprocessed image\n    #cv2.imwrite(save_path, img_median)\n\n# Process all images in a directory\n#def preprocess_images_in_directory(input_dir, output_dir):\n    #os.makedirs(output_dir, exist_ok=True)\n    #for root, _, files in os.walk(input_dir):\n        #for file in files:\n            #if file.endswith(('png', 'jpg', 'jpeg', 'dcm')):  # Process valid image files\n                #input_path = os.path.join(root, file)\n                #output_path = os.path.join(output_dir, file)\n                #preprocess_image(input_path, output_path)\n\n# Input and output directories\n#input_dir = 'output/cropped_images'\n#output_dir = 'output/preprocessed_images'\n\n# Preprocess all images\n#preprocess_images_in_directory(input_dir, output_dir)\n\n\nimport os\nfrom tqdm import tqdm  # Progress bar for large datasets\nimport cv2\nimport numpy as np\nfrom scipy.signal import wiener\n\n# Define CLAHE parameters\nclahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n\n# Define the base directory and output directory\nbase_dir = 'output/resized_images'\noutput_dir = 'output/preprocessed_images'\nclass_names = ['Benign', 'Cancer', 'Normal']\n\n# Create output directory structure\nos.makedirs(output_dir, exist_ok=True)\nfor class_name in class_names:\n    os.makedirs(os.path.join(output_dir, class_name), exist_ok=True)\n\n# Preprocessing function for a single image\ndef preprocess_image(image_path):\n    # Read the image in grayscale\n    img = cv2.imread(image_path, cv2.IMREAD_GRAYSCALE)\n    \n    # Apply CLAHE\n    img_clahe = clahe.apply(img)\n    \n    # Apply Wiener filter for noise reduction\n    #img_wiener = wiener(img_clahe, (5, 5))\n    #img_wiener = np.clip(img_wiener, 0, 255).astype(np.uint8)\n    \n    img_wiener = img_clahe.astype(np.float32) / 255.0  # Normalize to [0,1]\n    img_wiener = wiener(img_wiener, (3, 3))  # Apply Wiener filter\n    img_wiener = np.clip(img_wiener * 255, 0, 255).astype(np.uint8)  # Convert back to uint8\n    \n    #img_wiener = img_clahe.astype(np.float32)  # Convert to float32\n    #img_wiener /= 255.0  # Normalize to range [0, 1]\n    #img_wiener = wiener(img, (3, 3))  # Apply Wiener filter\n    #img_wiener = np.clip(img_wiener * 255, 0, 255).astype(np.uint8)  # Convert back to uint8\n    \n    # Apply Median filter for further noise reduction\n    img_median = cv2.medianBlur(img_wiener, 5)\n    \n    return img_median\n\n# Process and save images for each class\nfor class_name in class_names:\n    class_dir = os.path.join(base_dir, class_name)\n    output_class_dir = os.path.join(output_dir, class_name)\n    \n    print(f\"Processing class: {class_name}\")\n    for image_file in tqdm(os.listdir(class_dir)):\n        img_path = os.path.join(class_dir, image_file)\n        output_img_path = os.path.join(output_class_dir, image_file)\n        \n        # Preprocess and save the image\n        preprocessed_img = preprocess_image(img_path)\n        cv2.imwrite(output_img_path, preprocessed_img)\n\nprint(\"All images have been preprocessed and saved!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T14:34:38.805655Z","iopub.execute_input":"2025-05-06T14:34:38.806034Z","iopub.status.idle":"2025-05-06T14:38:48.699438Z","shell.execute_reply.started":"2025-05-06T14:34:38.805985Z","shell.execute_reply":"2025-05-06T14:38:48.698502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nfrom tensorflow.keras.preprocessing.image import load_img\n\n# Define the base directory of preprocessed images and class names\npreprocessed_dir = 'output/preprocessed_images'\nclass_names = ['Benign', 'Cancer', 'Normal']\n\n# Number of images to display per class\nnum_images_per_class = 5\n\n# Create a plot for preprocessed images\nplt.figure(figsize=(15, 10))\n\n# Iterate through each class\nfor class_idx, class_name in enumerate(class_names):\n    class_dir = os.path.join(preprocessed_dir, class_name)  # Path to class directory\n    image_files = os.listdir(class_dir)[:num_images_per_class]  # Get first N images\n    \n    for i, image_file in enumerate(image_files):\n        # Load the preprocessed image\n        img_path = os.path.join(class_dir, image_file)\n        img = load_img(img_path, color_mode='grayscale')  # Load in grayscale if applicable\n        img_size = img.size  # Get image size (width, height)\n        \n        # Plot the preprocessed image\n        plt.subplot(len(class_names), num_images_per_class, class_idx * num_images_per_class + i + 1)\n        plt.imshow(img, cmap='gray')  # Display in grayscale\n        plt.axis('off')\n        plt.title(f\"{class_name}\\n{img_size[0]}x{img_size[1]}\")  # Show new size in title\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T14:38:48.700602Z","iopub.execute_input":"2025-05-06T14:38:48.700907Z","iopub.status.idle":"2025-05-06T14:38:49.921272Z","shell.execute_reply.started":"2025-05-06T14:38:48.70088Z","shell.execute_reply":"2025-05-06T14:38:49.920364Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Hasil Filtering","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tensorflow.keras.preprocessing.image import load_img\nimport os\n\n# Define paths to the original and cropped images\noriginal_img_path = 'output/resized_images/Cancer/A_1520_1.LEFT_CC.png'\nresized_img_path = 'output/preprocessed_images/Cancer/A_1520_1.LEFT_CC.png'\n\n# Load images in RGB mode\noriginal_img = load_img(original_img_path)\nroi_img = load_img(resized_img_path)\n\n# Create the plot\nplt.figure(figsize=(10, 5))\n\n# Plot Original Image\nplt.subplot(1, 2, 1)\nplt.imshow(original_img)\nplt.title(\"Original Image\")\nplt.axis('off')\n\n# Plot ROI Image\nplt.subplot(1, 2, 2)\nplt.imshow(roi_img)\nplt.title(\"Filtered Image\")\nplt.axis('off')\n\nplt.tight_layout()\n\n# Save the figure\noutput_path = \"/kaggle/working/comparison_output(1024)-3.png\"\nplt.savefig(output_path, bbox_inches='tight')\n\n# Display the plot\nplt.show()\n\nprint(f\"✅ Image displayed and saved successfully to: {output_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T14:38:49.922506Z","iopub.execute_input":"2025-05-06T14:38:49.922796Z","iopub.status.idle":"2025-05-06T14:38:50.335939Z","shell.execute_reply.started":"2025-05-06T14:38:49.922769Z","shell.execute_reply":"2025-05-06T14:38:50.334931Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Hasil 3 Jenis Filter","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nfrom scipy.signal import wiener\n\n# Define CLAHE parameters\nclahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n\n# Define the specific image path\nimage_path = 'output/resized_images/Cancer/A_1520_1.LEFT_CC.png'\n\n# Define the new output directory for each filter type\noutput_dir = 'output/filter'\nos.makedirs(output_dir, exist_ok=True)\nos.makedirs(os.path.join(output_dir, 'CLAHE'), exist_ok=True)\nos.makedirs(os.path.join(output_dir, 'Wiener'), exist_ok=True)\nos.makedirs(os.path.join(output_dir, 'Median'), exist_ok=True)\n\n# Preprocessing function for CLAHE filter\ndef apply_clahe(image_path):\n    img = cv2.imread(image_path, cv2.IMREAD_GRAYSCALE)\n    img_clahe = clahe.apply(img)\n    return img_clahe\n\n# Preprocessing function for Wiener filter\ndef apply_wiener(image_path):\n    img = cv2.imread(image_path, cv2.IMREAD_GRAYSCALE)\n    img_float = img.astype(np.float32) / 255.0  # Normalize\n    img_wiener = wiener(img_float, (19, 19))  # Apply Wiener filter\n    img_wiener = np.clip(img_wiener * 255, 0, 255).astype(np.uint8)  # Convert back to uint8\n    return img_wiener\n\n# Preprocessing function for Median filter\ndef apply_median(image_path):\n    img = cv2.imread(image_path, cv2.IMREAD_GRAYSCALE)\n    img_median = cv2.medianBlur(img, 19)\n    return img_median\n\n# Apply and save CLAHE filter\nclahe_img = apply_clahe(image_path)\nclahe_output_path = os.path.join(output_dir, 'CLAHE', 'A_1520_1.LEFT_CC.png')\ncv2.imwrite(clahe_output_path, clahe_img)\n\n# Apply and save Wiener filter\nwiener_img = apply_wiener(image_path)\nwiener_output_path = os.path.join(output_dir, 'Wiener', 'A_1520_1.LEFT_CC.png')\ncv2.imwrite(wiener_output_path, wiener_img)\n\n# Apply and save Median filter\nmedian_img = apply_median(image_path)\nmedian_output_path = os.path.join(output_dir, 'Median', 'A_1520_1.LEFT_CC.png')\ncv2.imwrite(median_output_path, median_img)\n\nprint(\"The image has been preprocessed and saved with each filter separately in the 'output/filter' directory!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T14:38:50.337223Z","iopub.execute_input":"2025-05-06T14:38:50.338041Z","iopub.status.idle":"2025-05-06T14:38:50.410545Z","shell.execute_reply.started":"2025-05-06T14:38:50.338005Z","shell.execute_reply":"2025-05-06T14:38:50.409613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tensorflow.keras.preprocessing.image import load_img\nimport os\n\n# Define paths to the original image and the filtered images for each filter\noriginal_img_path = 'output/resized_images/Cancer/A_1520_1.LEFT_CC.png'\nclahe_img_path = 'output/filter/CLAHE/A_1520_1.LEFT_CC.png'\nwiener_img_path = 'output/filter/Wiener/A_1520_1.LEFT_CC.png'\nmedian_img_path = 'output/filter/Median/A_1520_1.LEFT_CC.png'\n\n# Load images in RGB mode\noriginal_img = load_img(original_img_path)\nclahe_img = load_img(clahe_img_path)\nwiener_img = load_img(wiener_img_path)\nmedian_img = load_img(median_img_path)\n\n# Function to create and save comparison plots for each filter\ndef plot_and_save_comparison(original, filtered, filter_name):\n    plt.figure(figsize=(10, 5))\n\n    # Plot Original Image\n    plt.subplot(1, 2, 1)\n    plt.imshow(original)\n    plt.title(\"Original Image\")\n    plt.axis('off')\n\n    # Plot Filtered Image\n    plt.subplot(1, 2, 2)\n    plt.imshow(filtered)\n    plt.title(f\"{filter_name} Filtered Image\")\n    plt.axis('off')\n\n    plt.tight_layout()\n\n    # Save the figure\n    output_path = f\"/kaggle/working/comparison_{filter_name}_output.png\"\n    plt.savefig(output_path, bbox_inches='tight')\n\n    # Display the plot\n    plt.show()\n\n    print(f\"✅ {filter_name} comparison displayed and saved successfully to: {output_path}\")\n\n# Create and save comparisons for each filter\nplot_and_save_comparison(original_img, clahe_img, \"CLAHE\")\nplot_and_save_comparison(original_img, wiener_img, \"Wiener\")\nplot_and_save_comparison(original_img, median_img, \"Median\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T14:38:50.41181Z","iopub.execute_input":"2025-05-06T14:38:50.412186Z","iopub.status.idle":"2025-05-06T14:38:51.826538Z","shell.execute_reply.started":"2025-05-06T14:38:50.412156Z","shell.execute_reply":"2025-05-06T14:38:51.825393Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Zero Mean","metadata":{}},{"cell_type":"code","source":"import os\nfrom tqdm import tqdm  # Progress bar for large datasets\nimport cv2\nimport numpy as np\n\n# Define the base directory and output directory\nbase_dir = 'output/preprocessed_images'\noutput_dir = 'output/normalized_images'\nclass_names = ['Benign', 'Cancer', 'Normal']\n\n# Create output directory structure\nos.makedirs(output_dir, exist_ok=True)\nfor class_name in class_names:\n    os.makedirs(os.path.join(output_dir, class_name), exist_ok=True)\n\n# Zero-mean normalization function\ndef zero_mean_normalization(img):\n    \"\"\"\n    Normalize an image to have zero mean and unit variance.\n    \"\"\"\n    img_normalized = (img - np.mean(img)) / np.std(img)  # Zero mean and unit variance\n    return img_normalized\n\n# Preprocessing function for a single image\ndef preprocess_image(image_path):\n    # Read the image in grayscale\n    img = cv2.imread(image_path, cv2.IMREAD_GRAYSCALE)\n    \n    # Apply zero-mean normalization\n    img_normalized = zero_mean_normalization(img)\n    \n    # Convert back to uint8 for saving (optional, depending on use case)\n    img_normalized = np.clip(img_normalized * 255, 0, 255).astype(np.uint8)\n    \n    return img_normalized\n\n# Process and save images for each class\nfor class_name in class_names:\n    class_dir = os.path.join(base_dir, class_name)\n    output_class_dir = os.path.join(output_dir, class_name)\n    \n    print(f\"Processing class: {class_name}\")\n    for image_file in tqdm(os.listdir(class_dir)):\n        img_path = os.path.join(class_dir, image_file)\n        output_img_path = os.path.join(output_class_dir, image_file)\n        \n        # Preprocess and save the image\n        preprocessed_img = preprocess_image(img_path)\n        cv2.imwrite(output_img_path, preprocessed_img)\n\nprint(\"All images have been zero-mean normalized and saved!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T14:38:51.828987Z","iopub.execute_input":"2025-05-06T14:38:51.829272Z","iopub.status.idle":"2025-05-06T14:39:50.959976Z","shell.execute_reply.started":"2025-05-06T14:38:51.829244Z","shell.execute_reply":"2025-05-06T14:39:50.959029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tensorflow.keras.preprocessing.image import load_img\nimport os\n\n# Define paths to the original and preprocessed images\noriginal_img_path = '/kaggle/input/preprocessed-miniddsm/Cancer/A_1520_1.LEFT_CC.png'\nfiltered_img_path = 'output/normalized_images/Cancer/A_1520_1.LEFT_CC.png'  # Assuming zero-mean normalized images\n\n# Load images in RGB mode (or grayscale, depending on how they are processed)\noriginal_img = load_img(original_img_path)\nfiltered_img = load_img(filtered_img_path)\n\n# Create the plot\nplt.figure(figsize=(10, 5))\n\n# Plot Original Image\nplt.subplot(1, 2, 1)\nplt.imshow(original_img)\nplt.title(\"Original Image\")\nplt.axis('off')\n\n# Plot Filtered Image\nplt.subplot(1, 2, 2)\nplt.imshow(filtered_img)\nplt.title(\"Normalized Image\")\nplt.axis('off')\n\nplt.tight_layout()\n\n# Save the figure\noutput_path = \"/kaggle/working/comparison_output_filtered_image.png\"\nplt.savefig(output_path, bbox_inches='tight')\n\n# Display the plot\nplt.show()\n\nprint(f\"✅ Image displayed and saved successfully to: {output_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T14:39:50.961367Z","iopub.execute_input":"2025-05-06T14:39:50.962066Z","iopub.status.idle":"2025-05-06T14:39:51.705929Z","shell.execute_reply.started":"2025-05-06T14:39:50.962027Z","shell.execute_reply":"2025-05-06T14:39:51.704882Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Modelling","metadata":{}},{"cell_type":"markdown","source":"## Testing Model","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Parameters\nIMG_SIZE = 256 #1024 \nBATCH_SIZE = 32  # Increased from 2\nEPOCHS = 100\nNUM_CLASSES = 3\nDROPOUT_RATE = 0.5  # Reduced from 0.7\nFOLDS = 2\nLEARNING_RATE = 1e-4\nDATASET_PATH = \"/kaggle/input/preprocessed-miniddsm\"\n\n# Preprocess all data manually for k-fold\nclasses = ['Benign', 'Cancer', 'Normal']\nimage_paths, labels = [], []\nfor idx, class_name in enumerate(classes):\n    class_dir = os.path.join(DATASET_PATH, class_name)\n    for fname in os.listdir(class_dir):\n        image_paths.append(os.path.join(class_dir, fname))\n        labels.append(idx)\n\nimage_paths = np.array(image_paths)\nlabels = np.array(labels)\n\n# Load and preprocess images\nX = np.zeros((len(image_paths), IMG_SIZE, IMG_SIZE, 1), dtype=np.float32)\nfor i, path in enumerate(image_paths):\n    img = load_img(path, color_mode='grayscale', target_size=(IMG_SIZE, IMG_SIZE))\n    arr = img_to_array(img).astype(np.float32)\n    arr = (arr - np.mean(arr)) / np.std(arr)  # zero-mean normalization\n    X[i] = arr\n\ny = to_categorical(labels, num_classes=NUM_CLASSES)\n\n# Data augmentation setup\ndef create_data_generators(X_train, y_train, X_val, y_val):\n    train_datagen = ImageDataGenerator(\n        rotation_range=20,\n        width_shift_range=0.2,\n        height_shift_range=0.2,\n        zoom_range=0.2,\n        horizontal_flip=True,\n        fill_mode='nearest'\n    )\n    \n    val_datagen = ImageDataGenerator()\n    \n    # Convert grayscale to RGB for transfer learning\n    X_train_rgb = np.repeat(X_train, 3, axis=-1)\n    X_val_rgb = np.repeat(X_val, 3, axis=-1)\n    \n    train_generator = train_datagen.flow(X_train_rgb, y_train, batch_size=BATCH_SIZE)\n    val_generator = val_datagen.flow(X_val_rgb, y_val, batch_size=BATCH_SIZE, shuffle=False)\n    \n    return train_generator, val_generator, X_val_rgb\n\n# Feature extractor with reduced trainable layers\ndef build_feature_extractor():\n    base_model = DenseNet169(include_top=False, input_shape=(IMG_SIZE, IMG_SIZE, 3), weights='imagenet')\n    \n    # Freeze most layers\n    for layer in base_model.layers[:-20]:  # Only train last 20 layers\n        layer.trainable = False\n    \n    return base_model\n\n# Improved bi-stream CNN with regularization\ndef bi_stream_cnn(input_tensor):\n    # Add BatchNormalization at the beginning\n    x = BatchNormalization()(input_tensor)\n    \n    # First stream with regularization\n    x1 = DepthwiseConv2D(kernel_size=3, strides=1, padding='same')(x)\n    x1 = BatchNormalization()(x1)\n    x1 = Activation('relu')(x1)\n    x1 = Conv2D(32, kernel_size=3, padding='same', kernel_regularizer=l2(0.001))(x1)\n    x1 = BatchNormalization()(x1)\n    x1 = Activation('relu')(x1)\n    \n    # Second stream with regularization\n    x2 = DepthwiseConv2D(kernel_size=5, strides=1, padding='same')(x)\n    x2 = BatchNormalization()(x2)\n    x2 = Activation('selu')(x2)\n    x2 = Conv2D(32, kernel_size=3, padding='same', kernel_regularizer=l2(0.001))(x2)\n    x2 = BatchNormalization()(x2)\n    x2 = Activation('selu')(x2)\n    \n    # Combine streams\n    x = Concatenate()([x1, x2])\n    \n    # Global pooling instead of flattening to reduce parameters\n    x = GlobalAveragePooling2D()(x)\n    \n    # Fully connected layers with dropout and regularization\n    x = Dense(64, activation='relu', kernel_regularizer=l2(0.001))(x)\n    x = BatchNormalization()(x)\n    x = Dropout(DROPOUT_RATE)(x)\n    x = Dense(32, activation='relu', kernel_regularizer=l2(0.001))(x)\n    x = BatchNormalization()(x)\n    x = Dropout(DROPOUT_RATE)(x)\n    \n    # Output layer\n    output = Dense(NUM_CLASSES, activation='softmax')(x)\n    \n    return output\n\n# Setup callbacks\ndef create_callbacks(fold):\n    # Learning rate scheduler\n    reduce_lr = tf.keras.callbacks.ReduceLROnPlateau(\n        monitor='val_loss', factor=0.2, patience=5, min_lr=1e-7, verbose=1)\n    \n    # Model checkpoint\n    checkpoint = tf.keras.callbacks.ModelCheckpoint(\n        f'best_model_fold_{fold}.h5', monitor='val_loss', save_best_only=True, verbose=1)\n    \n    # Early stopping\n    early_stopping = tf.keras.callbacks.EarlyStopping(\n        monitor='val_loss', patience=10, restore_best_weights=True, verbose=1)\n    \n    return [early_stopping, reduce_lr, checkpoint]\n\n# 5-fold cross-validation\nresults = []\nskf = StratifiedKFold(n_splits=FOLDS, shuffle=True, random_state=42)\nfold = 1\n\nfor train_idx, val_idx in skf.split(X, labels):\n    print(f\"\\nFold {fold}\")\n    X_train, X_val = X[train_idx], X[val_idx]\n    y_train, y_val = y[train_idx], y[val_idx]\n    \n    # Create data generators with augmentation\n    train_generator, val_generator, X_val_rgb = create_data_generators(X_train, y_train, X_val, y_val)\n    \n    # Compute class weights to handle imbalance\n    class_weights = class_weight.compute_class_weight(\n        class_weight='balanced', \n        classes=np.unique(labels), \n        y=labels[train_idx]\n    )\n    class_weights_dict = dict(enumerate(class_weights))\n\n    # Build model\n    base_model = build_feature_extractor()\n    x = base_model.output\n    output = bi_stream_cnn(x)\n    model = Model(inputs=base_model.input, outputs=output)\n    \n    # Compile with Adam optimizer\n    model.compile(\n        optimizer=Adam(learning_rate=LEARNING_RATE),\n        loss='categorical_crossentropy',\n        metrics=['accuracy']\n    )\n    \n    # Get callbacks\n    callbacks = create_callbacks(fold)\n    \n    # Train model\n    history = model.fit(\n        train_generator,\n        epochs=EPOCHS,\n        validation_data=val_generator,\n        callbacks=callbacks,\n        class_weight=class_weights_dict,\n        verbose=1\n    )\n    \n    # Evaluate on validation set\n    print(\"\\nEvaluating on validation set:\")\n    val_preds = model.predict(X_val_rgb, verbose=0)\n    val_report = classification_report(\n        np.argmax(y_val, axis=1), \n        np.argmax(val_preds, axis=1), \n        target_names=classes\n    )\n    print(val_report)\n    \n    # Save results\n    fold_results = {\n        'fold': fold,\n        'val_accuracy': np.max(history.history['val_accuracy']),\n        'val_loss': np.min(history.history['val_loss']),\n        'classification_report': val_report\n    }\n    results.append(fold_results)\n    \n    fold += 1\n\n# Print summary of results\nprint(\"\\n===== CROSS-VALIDATION RESULTS =====\")\navg_accuracy = np.mean([r['val_accuracy'] for r in results])\navg_loss = np.mean([r['val_loss'] for r in results])\nprint(f\"Average validation accuracy: {avg_accuracy:.4f}\")\nprint(f\"Average validation loss: {avg_loss:.4f}\")\n\n# Save final model from last fold\nmodel.save('improved_bi_stream_cnn_final.h5')\nprint(\"Final model saved.\")\n\n# Optional: Plot learning curves\ntry:\n    import matplotlib.pyplot as plt\n    \n    plt.figure(figsize=(12, 4))\n    plt.subplot(1, 2, 1)\n    plt.plot(history.history['accuracy'], label='Training Accuracy')\n    plt.plot(history.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('Accuracy')\n    plt.xlabel('Epoch')\n    plt.ylabel('Accuracy')\n    plt.legend()\n    \n    plt.subplot(1, 2, 2)\n    plt.plot(history.history['loss'], label='Training Loss')\n    plt.plot(history.history['val_loss'], label='Validation Loss')\n    plt.title('Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend()\n    \n    plt.tight_layout()\n    plt.savefig('learning_curves_final_fold.png')\n    plt.close()\n    print(\"Learning curves plot saved.\")\nexcept:\n    print(\"Could not create learning curve plots.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T13:02:10.345967Z","iopub.execute_input":"2025-04-25T13:02:10.346599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import (Input, Conv2D, BatchNormalization, Dense, DepthwiseConv2D, ReLU, \n                                     Activation, GlobalAveragePooling2D, Flatten, Dropout, Multiply)\nfrom tensorflow.keras.applications import DenseNet121\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, EarlyStopping\nfrom sklearn.utils.class_weight import compute_class_weight\n\n# Directory containing the dataset\ndata_dir = '/kaggle/input/preprocessed-miniddsm'  # Adjust path\n\n# Data augmentation and preprocessing\ndata_gen = ImageDataGenerator(\n    rescale=1.0/255,\n    validation_split=0.15,\n    rotation_range=40,\n    width_shift_range=0.3,\n    height_shift_range=0.3,\n    zoom_range=0.3,\n    horizontal_flip=True,\n    vertical_flip=True,\n    brightness_range=[0.7, 1.3],\n    shear_range=0.2,\n    fill_mode='reflect'\n)\n\n# Training and validation generators\ntrain_gen = data_gen.flow_from_directory(\n    directory=data_dir,\n    target_size=(256, 256),  # Reduce size to speed up training 512, 512 from 1024, 1024 \n    batch_size=32, #16 / 8\n    class_mode='categorical',\n    subset='training',\n    shuffle=True\n)\n\nval_gen = data_gen.flow_from_directory(\n    directory=data_dir,\n    target_size=(256, 256), # Reduce size to speed up training 512, 512 from 1024, 1024 \n    batch_size=32, #16 / 8\n    class_mode='categorical',\n    subset='validation',\n    shuffle=False\n)\n\n# Compute class weights for imbalanced datasets\nclass_weights = compute_class_weight(\n    class_weight='balanced',\n    classes=np.unique(train_gen.classes),\n    y=train_gen.classes\n)\nclass_weights = dict(enumerate(class_weights))\n\n# Stage 1: Feature Extraction using DenseNet\n\ndef build_stage1_feature_extractor(input_shape):\n    base_model = DenseNet121(include_top=False, weights='imagenet', input_shape=input_shape)\n    for layer in base_model.layers[:-40]:  # Freeze all but the last 30 layers\n        layer.trainable = False\n    inputs = Input(shape=input_shape)\n    x = base_model(inputs)\n    model = Model(inputs, x, name=\"Stage1_FeatureExtractor\")\n    return model\n\n# Stage 2: Bi-Stream CNN for Classification\n\ndef build_stage2_classifier(feature_shape):\n    inputs = Input(shape=feature_shape)\n\n    # Upper Stream\n    upper_stream = DepthwiseConv2D(kernel_size=3, padding='same')(inputs)\n    upper_stream = ReLU()(upper_stream)\n    upper_stream = Dropout(0.3)(upper_stream)\n    upper_stream = Conv2D(48, kernel_size=3, padding='same', \n                         kernel_regularizer=tf.keras.regularizers.l2(0.02))(upper_stream)\n    upper_stream = BatchNormalization()(upper_stream)\n    upper_stream = Dropout(0.4)(upper_stream)\n\n    # Lower Stream\n    lower_stream = DepthwiseConv2D(kernel_size=3, padding='same')(inputs)\n    lower_stream = Activation(\"selu\")(lower_stream)\n    lower_stream = Dropout(0.3)(lower_stream)\n    lower_stream = Conv2D(48, kernel_size=3, padding='same', \n                         kernel_regularizer=tf.keras.regularizers.l2(0.02))(lower_stream)\n    lower_stream = BatchNormalization()(lower_stream)\n    lower_stream = Dropout(0.4)(lower_stream)\n\n    # Element-wise Multiplication\n    multiplied = Multiply()([upper_stream, lower_stream])\n\n    # Fully Connected Layers\n    x = Flatten()(multiplied)\n    x = Dense(96, activation='relu', \n             kernel_regularizer=tf.keras.regularizers.l2(0.02))(x)\n    x = Dropout(0.6)(x)\n    x = Dense(48, activation='relu',\n             kernel_regularizer=tf.keras.regularizers.l2(0.01))(x)\n    x = Dropout(0.5)(x)\n    outputs = Dense(3, activation='softmax')(x)\n\n    return Model(inputs, outputs, name=\"Stage2_Classifier\")\n\n# Combine Stage 1 and Stage 2 into the Two-Stage CNN\ndef build_two_stage_cnn(input_shape):\n    stage1_model = build_stage1_feature_extractor(input_shape)\n    feature_shape = stage1_model.output_shape[1:]  # Retain spatial dimensions\n    stage2_model = build_stage2_classifier(feature_shape)\n\n    inputs = Input(shape=input_shape)\n    features = stage1_model(inputs)\n    outputs = stage2_model(features)\n\n    model = Model(inputs, outputs, name=\"TwoStageCNN\")\n    return model\n\n# Define input shape\ninput_shape = (256, 256, 3) #1024, 1024\nmodel = build_two_stage_cnn(input_shape)\n\n# Compile the model\nmodel.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=5e-5),\n              loss='categorical_crossentropy',\n              metrics=['accuracy'])\n\n# Model Summary\nmodel.summary()\n\n# Learning rate scheduler and early stopping\n# Add more aggressive learning rate reduction\nlr_scheduler = ReduceLROnPlateau(\n    monitor='val_loss',\n    factor=0.3,  # More aggressive reduction\n    patience=3,\n    min_lr=1e-6,\n    verbose=1\n)\nearly_stopping = EarlyStopping(\n    monitor='val_loss',\n    patience=10,\n    restore_best_weights=True,\n    min_delta=0.001\n)\n\n# Train the model\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=100,\n    class_weight=class_weights,\n    callbacks=[ #lr_scheduler, #early_stopping\n              ]\n)\n\n# Save the trained model\nmodel.save('improved_two_stage_cnn_model.h5')\n\n# Evaluate the model\nloss, accuracy = model.evaluate(val_gen)\nprint(f\"Validation Loss: {loss}\")\nprint(f\"Validation Accuracy: {accuracy}\")\n\n# Classification report\nfrom sklearn.metrics import classification_report\nval_gen.reset()\npredictions = model.predict(val_gen)\npredicted_classes = np.argmax(predictions, axis=1)\nprint(classification_report(val_gen.classes, predicted_classes, target_names=list(val_gen.class_indices.keys())))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## OTHER CNN TEST","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.applications import VGG16\nfrom tensorflow.keras.layers import (Dense, Dropout, GlobalAveragePooling2D,\n                                     BatchNormalization)\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import class_weight\nfrom sklearn.metrics import classification_report, accuracy_score, precision_score, recall_score, f1_score\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.keras.regularizers import l2\n\n# Clear any previous session\ntf.keras.backend.clear_session()\n\n# Parameters\nIMG_SIZE = 128\nBATCH_SIZE = 32\nEPOCHS = 100\nNUM_CLASSES = 3\nDROPOUT_RATE = 0.5\nLEARNING_RATE = 1e-4\nDATASET_PATH = \"/kaggle/input/preprocessed-miniddsm\"\n\n# Preprocess all data\nclasses = ['Benign', 'Cancer', 'Normal']\nimage_paths, labels = [], []\nfor idx, class_name in enumerate(classes):\n    class_dir = os.path.join(DATASET_PATH, class_name)\n    for fname in os.listdir(class_dir):\n        image_paths.append(os.path.join(class_dir, fname))\n        labels.append(idx)\n\nimage_paths = np.array(image_paths)\nlabels = np.array(labels)\n\n# Load and preprocess images\nX = np.zeros((len(image_paths), IMG_SIZE, IMG_SIZE, 1), dtype=np.float32)\nfor i, path in enumerate(image_paths):\n    img = load_img(path, color_mode='grayscale', target_size=(IMG_SIZE, IMG_SIZE))\n    arr = img_to_array(img).astype(np.float32)\n    #arr = (arr - np.mean(arr)) / np.std(arr) \n    X[i] = arr\n\n# Split data into training and validation sets BEFORE one-hot encoding\nX_train, X_val, y_train_labels, y_val_labels = train_test_split(X, labels, test_size=0.2, random_state=42, stratify=labels)\n\n# Convert to one-hot encoding AFTER splitting\ny_train = to_categorical(y_train_labels, num_classes=NUM_CLASSES)\ny_val = to_categorical(y_val_labels, num_classes=NUM_CLASSES)\n\n# Data augmentation setup\ntrain_datagen = ImageDataGenerator(\n    #rotation_range=90,\n    #width_shift_range=0.35,\n    #height_shift_range=0.35,\n    #zoom_range=0.7,\n    #shear_range=0.7,\n    #horizontal_flip=True,\n    #fill_mode='nearest',  # Changed from 'nearest' to 'reflect'\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest',\n    preprocessing_function=lambda x: x + np.random.normal(0, 15, x.shape)  # Add mild noise\n)\n\nval_datagen = ImageDataGenerator()\n\n# Convert grayscale to RGB for transfer learning (required for VGG16)\nX_train_rgb = np.repeat(X_train, 3, axis=-1)\nX_val_rgb = np.repeat(X_val, 3, axis=-1)\n\ntrain_generator = train_datagen.flow(X_train_rgb, y_train, batch_size=BATCH_SIZE)\nval_generator = val_datagen.flow(X_val_rgb, y_val, batch_size=BATCH_SIZE, shuffle=False)\n\n# Compute class weights to handle imbalance\nclass_weights = class_weight.compute_class_weight(\n    class_weight='balanced', \n    classes=np.unique(labels), \n    y=y_train_labels\n)\nclass_weights_dict = dict(enumerate(class_weights))\n\n# Setup callbacks\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_loss', factor=0.2, patience=5, min_lr=1e-7, verbose=1)\n\nearly_stopping = tf.keras.callbacks.EarlyStopping(\n    monitor='val_loss', patience=10, restore_best_weights=True, verbose=1)\n\n# Define evaluation function\ndef evaluate_model(model, model_name):\n    print(f\"\\nEvaluating {model_name} on validation set:\")\n    val_preds = model.predict(X_val_rgb, verbose=0)\n    val_pred_classes = np.argmax(val_preds, axis=1)\n    val_true_classes = np.argmax(y_val, axis=1)\n\n    # Calculate the metrics\n    accuracy = accuracy_score(val_true_classes, val_pred_classes)\n    precision_per_class = precision_score(val_true_classes, val_pred_classes, average=None)\n    precision_macro = precision_score(val_true_classes, val_pred_classes, average='macro')\n    precision_weighted = precision_score(val_true_classes, val_pred_classes, average='weighted')\n    recall_per_class = recall_score(val_true_classes, val_pred_classes, average=None)\n    recall_macro = recall_score(val_true_classes, val_pred_classes, average='macro')\n    recall_weighted = recall_score(val_true_classes, val_pred_classes, average='weighted')\n    f1_per_class = f1_score(val_true_classes, val_pred_classes, average=None)\n    f1_macro = f1_score(val_true_classes, val_pred_classes, average='macro')\n    f1_weighted = f1_score(val_true_classes, val_pred_classes, average='weighted')\n\n    # Display results\n    print(f\"\\n===== {model_name} EVALUATION METRICS =====\")\n    print(f\"Overall Accuracy: {accuracy:.4f}\")\n\n    print(\"\\nPer-class Metrics:\")\n    print(\"-\" * 70)\n    print(f\"{'Class':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\")\n    print(\"-\" * 70)\n    for i, class_name in enumerate(classes):\n        print(f\"{class_name:20} {precision_per_class[i]:.4f}{' ':10} {recall_per_class[i]:.4f}{' ':10} {f1_per_class[i]:.4f}\")\n    print(\"-\" * 70)\n\n    print(\"\\nAveraged Metrics:\")\n    print(\"-\" * 70)\n    print(f\"{'Metric Type':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\")\n    print(\"-\" * 70)\n    print(f\"{'Macro Average':20} {precision_macro:.4f}{' ':10} {recall_macro:.4f}{' ':10} {f1_macro:.4f}\")\n    print(f\"{'Weighted Average':20} {precision_weighted:.4f}{' ':10} {recall_weighted:.4f}{' ':10} {f1_weighted:.4f}\")\n    print(\"-\" * 70)\n\n    # Save the detailed metrics to a text file\n    with open(f'{model_name}_evaluation_metrics.txt', 'w') as f:\n        f.write(f\"===== {model_name} EVALUATION METRICS =====\\n\")\n        f.write(f\"Overall Accuracy: {accuracy:.4f}\\n\\n\")\n        \n        f.write(\"Per-class Metrics:\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        f.write(f\"{'Class':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        for i, class_name in enumerate(classes):\n            f.write(f\"{class_name:20} {precision_per_class[i]:.4f}{' ':10} {recall_per_class[i]:.4f}{' ':10} {f1_per_class[i]:.4f}\\n\")\n        f.write(\"-\" * 70 + \"\\n\\n\")\n        \n        f.write(\"Averaged Metrics:\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        f.write(f\"{'Metric Type':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        f.write(f\"{'Macro Average':20} {precision_macro:.4f}{' ':10} {recall_macro:.4f}{' ':10} {f1_macro:.4f}\\n\")\n        f.write(f\"{'Weighted Average':20} {precision_weighted:.4f}{' ':10} {recall_weighted:.4f}{' ':10} {f1_weighted:.4f}\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n\n    # Also save the standard classification report for reference\n    with open(f'{model_name}_classification_report.txt', 'w') as f:\n        f.write(\"\\nStandard Classification Report:\\n\")\n        f.write(classification_report(val_true_classes, val_pred_classes, target_names=classes))\n\n    print(f\"Detailed metrics saved to '{model_name}_evaluation_metrics.txt'\")\n    print(f\"Classification report saved to '{model_name}_classification_report.txt'\")\n    \n    return model\n\n# ---- VGG16 Model with Transfer Learning ----\ndef build_vgg16():\n    base_model = VGG16(include_top=False, input_shape=(IMG_SIZE, IMG_SIZE, 3), weights='imagenet')\n    \n    # Freeze most layers\n    for layer in base_model.layers[:-4]:  # Train only last few layers\n        layer.trainable = False\n    \n    x = base_model.output\n    x = GlobalAveragePooling2D()(x)\n    x = Dense(512, activation='relu', kernel_regularizer=l2(0.001))(x)\n    x = BatchNormalization()(x)\n    x = Dropout(DROPOUT_RATE)(x)\n    x = Dense(256, activation='relu', kernel_regularizer=l2(0.001))(x)\n    x = BatchNormalization()(x)\n    x = Dropout(DROPOUT_RATE)(x)\n    \n    # Output layer\n    output = Dense(NUM_CLASSES, activation='softmax')(x)\n    \n    model = Model(inputs=base_model.input, outputs=output)\n    \n    model.compile(\n        optimizer=Adam(learning_rate=LEARNING_RATE),\n        loss='categorical_crossentropy',\n        metrics=['accuracy']\n    )\n    \n    return model\n\n# ---- Train VGG16 Model ----\nprint(\"\\n===== TRAINING VGG16 MODEL =====\")\nvgg16_model = build_vgg16()\nvgg16_checkpoint = tf.keras.callbacks.ModelCheckpoint(\n    'vgg16_best_model.h5', monitor='val_loss', save_best_only=True, verbose=1)\nvgg16_callbacks = [early_stopping, reduce_lr, vgg16_checkpoint]\n\nvgg16_history = vgg16_model.fit(\n    train_generator,\n    epochs=EPOCHS,\n    validation_data=val_generator,\n    callbacks=vgg16_callbacks,\n    class_weight=class_weights_dict,\n    verbose=1\n)\n\nvgg16_model = evaluate_model(vgg16_model, \"VGG16\")\n# Commented out to avoid memory error\n# vgg16_model.save('vgg16_final.h5')\nprint(\"VGG16 model evaluated.\")\n\n# Plot learning curves for VGG16 only\ntry:\n    import matplotlib.pyplot as plt\n    \n    # Plot accuracy curves\n    plt.figure(figsize=(12, 5))\n    \n    plt.subplot(1, 2, 1)\n    plt.plot(vgg16_history.history['accuracy'], label='Training Accuracy')\n    plt.plot(vgg16_history.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('VGG16 Accuracy')\n    plt.xlabel('Epoch')\n    plt.ylabel('Accuracy')\n    plt.legend()\n    \n    # Plot loss curves\n    plt.subplot(1, 2, 2)\n    plt.plot(vgg16_history.history['loss'], label='Training Loss')\n    plt.plot(vgg16_history.history['val_loss'], label='Validation Loss')\n    plt.title('VGG16 Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend()\n    \n    plt.tight_layout()\n    plt.savefig('vgg16_learning_curves.png')\n    plt.close()\n    print(\"VGG16 learning curves plot saved.\")\nexcept Exception as e:\n    print(f\"Could not create learning curve plots. Error: {str(e)}\")\n\n# Create a VGG16 model summary\nwith open('vgg16_model_summary.txt', 'w') as f:\n    f.write(\"===== VGG16 MODEL SUMMARY =====\\n\\n\")\n    \n    f.write(\"Parameters used:\\n\")\n    f.write(f\"- Image Size: {IMG_SIZE}x{IMG_SIZE}\\n\")\n    f.write(f\"- Batch Size: {BATCH_SIZE}\\n\")\n    f.write(f\"- Maximum Epochs: {EPOCHS}\\n\")\n    f.write(f\"- Dropout Rate: {DROPOUT_RATE}\\n\")\n    f.write(f\"- Learning Rate: {LEARNING_RATE}\\n\")\n    f.write(f\"- Classes: {', '.join(classes)}\\n\\n\")\n    \n    # VGG16 summary\n    vgg16_val_accuracy = max(vgg16_history.history['val_accuracy'])\n    vgg16_val_loss = min(vgg16_history.history['val_loss'])\n    f.write(f\"VGG16 Results:\\n\")\n    f.write(f\"- Best Validation Accuracy: {vgg16_val_accuracy:.4f}\\n\")\n    f.write(f\"- Best Validation Loss: {vgg16_val_loss:.4f}\\n\")\n    f.write(f\"- Training Epochs Completed: {len(vgg16_history.history['accuracy'])}\\n\")\n\nprint(\"VGG16 model summary saved to 'vgg16_model_summary.txt'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T07:42:58.649215Z","iopub.execute_input":"2025-04-28T07:42:58.649532Z","iopub.status.idle":"2025-04-28T08:12:17.224223Z","shell.execute_reply.started":"2025-04-28T07:42:58.649506Z","shell.execute_reply":"2025-04-28T08:12:17.223333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.applications import VGG16\nfrom tensorflow.keras.layers import (Dense, Dropout, GlobalAveragePooling2D,\n                                     BatchNormalization)\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import class_weight\nfrom sklearn.metrics import classification_report, accuracy_score, precision_score, recall_score, f1_score\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.keras.regularizers import l2\n\n# Clear any previous session\ntf.keras.backend.clear_session()\n\n# Parameters\nIMG_SIZE = 128\nBATCH_SIZE = 32\nEPOCHS = 100\nNUM_CLASSES = 3\nDROPOUT_RATE = 0.5\nLEARNING_RATE = 1e-4\nDATASET_PATH = \"/kaggle/input/preprocessed-miniddsm\"\n\n# Preprocess all data\nclasses = ['Benign', 'Cancer', 'Normal']\nimage_paths, labels = [], []\nfor idx, class_name in enumerate(classes):\n    class_dir = os.path.join(DATASET_PATH, class_name)\n    for fname in os.listdir(class_dir):\n        image_paths.append(os.path.join(class_dir, fname))\n        labels.append(idx)\n\nimage_paths = np.array(image_paths)\nlabels = np.array(labels)\n\n# Load and preprocess images\nX = np.zeros((len(image_paths), IMG_SIZE, IMG_SIZE, 1), dtype=np.float32)\nfor i, path in enumerate(image_paths):\n    img = load_img(path, color_mode='grayscale', target_size=(IMG_SIZE, IMG_SIZE))\n    arr = img_to_array(img).astype(np.float32)\n    #arr = (arr - np.mean(arr)) / np.std(arr) \n    X[i] = arr\n\n# Split data into training and validation sets BEFORE one-hot encoding\nX_train, X_val, y_train_labels, y_val_labels = train_test_split(X, labels, test_size=0.2, random_state=42, stratify=labels)\n\n# Convert to one-hot encoding AFTER splitting\ny_train = to_categorical(y_train_labels, num_classes=NUM_CLASSES)\ny_val = to_categorical(y_val_labels, num_classes=NUM_CLASSES)\n\n# Data augmentation setup\ntrain_datagen = ImageDataGenerator(\n    rotation_range=90,\n    width_shift_range=0.35,\n    height_shift_range=0.35,\n    zoom_range=0.7,\n    shear_range=0.7,\n    horizontal_flip=True,\n    fill_mode='nearest',  # Changed from 'nearest' to 'reflect'\n    #rotation_range=20,\n    #width_shift_range=0.2,\n    #height_shift_range=0.2,\n    #shear_range=0.2,\n    #zoom_range=0.2,\n    #horizontal_flip=True,\n    #fill_mode='nearest',\n    preprocessing_function=lambda x: x + np.random.normal(0, 15, x.shape)  # Add mild noise\n)\n\nval_datagen = ImageDataGenerator()\n\n# Convert grayscale to RGB for transfer learning (required for VGG16)\nX_train_rgb = np.repeat(X_train, 3, axis=-1)\nX_val_rgb = np.repeat(X_val, 3, axis=-1)\n\ntrain_generator = train_datagen.flow(X_train_rgb, y_train, batch_size=BATCH_SIZE)\nval_generator = val_datagen.flow(X_val_rgb, y_val, batch_size=BATCH_SIZE, shuffle=False)\n\n# Compute class weights to handle imbalance\nclass_weights = class_weight.compute_class_weight(\n    class_weight='balanced', \n    classes=np.unique(labels), \n    y=y_train_labels\n)\nclass_weights_dict = dict(enumerate(class_weights))\n\n# Setup callbacks\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_loss', factor=0.2, patience=5, min_lr=1e-7, verbose=1)\n\nearly_stopping = tf.keras.callbacks.EarlyStopping(\n    monitor='val_loss', patience=10, restore_best_weights=True, verbose=1)\n\n# Define evaluation function\ndef evaluate_model(model, model_name):\n    print(f\"\\nEvaluating {model_name} on validation set:\")\n    val_preds = model.predict(X_val_rgb, verbose=0)\n    val_pred_classes = np.argmax(val_preds, axis=1)\n    val_true_classes = np.argmax(y_val, axis=1)\n\n    # Calculate the metrics\n    accuracy = accuracy_score(val_true_classes, val_pred_classes)\n    precision_per_class = precision_score(val_true_classes, val_pred_classes, average=None)\n    precision_macro = precision_score(val_true_classes, val_pred_classes, average='macro')\n    precision_weighted = precision_score(val_true_classes, val_pred_classes, average='weighted')\n    recall_per_class = recall_score(val_true_classes, val_pred_classes, average=None)\n    recall_macro = recall_score(val_true_classes, val_pred_classes, average='macro')\n    recall_weighted = recall_score(val_true_classes, val_pred_classes, average='weighted')\n    f1_per_class = f1_score(val_true_classes, val_pred_classes, average=None)\n    f1_macro = f1_score(val_true_classes, val_pred_classes, average='macro')\n    f1_weighted = f1_score(val_true_classes, val_pred_classes, average='weighted')\n\n    # Display results\n    print(f\"\\n===== {model_name} EVALUATION METRICS =====\")\n    print(f\"Overall Accuracy: {accuracy:.4f}\")\n\n    print(\"\\nPer-class Metrics:\")\n    print(\"-\" * 70)\n    print(f\"{'Class':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\")\n    print(\"-\" * 70)\n    for i, class_name in enumerate(classes):\n        print(f\"{class_name:20} {precision_per_class[i]:.4f}{' ':10} {recall_per_class[i]:.4f}{' ':10} {f1_per_class[i]:.4f}\")\n    print(\"-\" * 70)\n\n    print(\"\\nAveraged Metrics:\")\n    print(\"-\" * 70)\n    print(f\"{'Metric Type':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\")\n    print(\"-\" * 70)\n    print(f\"{'Macro Average':20} {precision_macro:.4f}{' ':10} {recall_macro:.4f}{' ':10} {f1_macro:.4f}\")\n    print(f\"{'Weighted Average':20} {precision_weighted:.4f}{' ':10} {recall_weighted:.4f}{' ':10} {f1_weighted:.4f}\")\n    print(\"-\" * 70)\n\n    # Save the detailed metrics to a text file\n    with open(f'{model_name}_evaluation_metrics.txt', 'w') as f:\n        f.write(f\"===== {model_name} EVALUATION METRICS =====\\n\")\n        f.write(f\"Overall Accuracy: {accuracy:.4f}\\n\\n\")\n        \n        f.write(\"Per-class Metrics:\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        f.write(f\"{'Class':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        for i, class_name in enumerate(classes):\n            f.write(f\"{class_name:20} {precision_per_class[i]:.4f}{' ':10} {recall_per_class[i]:.4f}{' ':10} {f1_per_class[i]:.4f}\\n\")\n        f.write(\"-\" * 70 + \"\\n\\n\")\n        \n        f.write(\"Averaged Metrics:\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        f.write(f\"{'Metric Type':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        f.write(f\"{'Macro Average':20} {precision_macro:.4f}{' ':10} {recall_macro:.4f}{' ':10} {f1_macro:.4f}\\n\")\n        f.write(f\"{'Weighted Average':20} {precision_weighted:.4f}{' ':10} {recall_weighted:.4f}{' ':10} {f1_weighted:.4f}\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n\n    # Also save the standard classification report for reference\n    with open(f'{model_name}_classification_report.txt', 'w') as f:\n        f.write(\"\\nStandard Classification Report:\\n\")\n        f.write(classification_report(val_true_classes, val_pred_classes, target_names=classes))\n\n    print(f\"Detailed metrics saved to '{model_name}_evaluation_metrics.txt'\")\n    print(f\"Classification report saved to '{model_name}_classification_report.txt'\")\n    \n    return model\n\n# ---- VGG16 Model with Transfer Learning ----\ndef build_vgg16():\n    base_model = VGG16(include_top=False, input_shape=(IMG_SIZE, IMG_SIZE, 3), weights='imagenet')\n    \n    # Freeze most layers\n    for layer in base_model.layers[:-4]:  # Train only last few layers\n        layer.trainable = False\n    \n    x = base_model.output\n    x = GlobalAveragePooling2D()(x)\n    x = Dense(512, activation='relu', kernel_regularizer=l2(0.001))(x)\n    x = BatchNormalization()(x)\n    x = Dropout(DROPOUT_RATE)(x)\n    x = Dense(256, activation='relu', kernel_regularizer=l2(0.001))(x)\n    x = BatchNormalization()(x)\n    x = Dropout(DROPOUT_RATE)(x)\n    \n    # Output layer\n    output = Dense(NUM_CLASSES, activation='softmax')(x)\n    \n    model = Model(inputs=base_model.input, outputs=output)\n    \n    model.compile(\n        optimizer=Adam(learning_rate=LEARNING_RATE),\n        loss='categorical_crossentropy',\n        metrics=['accuracy']\n    )\n    \n    return model\n\n# ---- Train VGG16 Model ----\nprint(\"\\n===== TRAINING VGG16 MODEL =====\")\nvgg16_model = build_vgg16()\nvgg16_checkpoint = tf.keras.callbacks.ModelCheckpoint(\n    'vgg16_best_model.h5', monitor='val_loss', save_best_only=True, verbose=1)\nvgg16_callbacks = [early_stopping, reduce_lr, vgg16_checkpoint]\n\nvgg16_history = vgg16_model.fit(\n    train_generator,\n    epochs=EPOCHS,\n    validation_data=val_generator,\n    callbacks=vgg16_callbacks,\n    class_weight=class_weights_dict,\n    verbose=1\n)\n\nvgg16_model = evaluate_model(vgg16_model, \"VGG16\")\n# Commented out to avoid memory error\n# vgg16_model.save('vgg16_final.h5')\nprint(\"VGG16 model evaluated.\")\n\n# Plot learning curves for VGG16 only\ntry:\n    import matplotlib.pyplot as plt\n    \n    # Plot accuracy curves\n    plt.figure(figsize=(12, 5))\n    \n    plt.subplot(1, 2, 1)\n    plt.plot(vgg16_history.history['accuracy'], label='Training Accuracy')\n    plt.plot(vgg16_history.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('VGG16 Accuracy')\n    plt.xlabel('Epoch')\n    plt.ylabel('Accuracy')\n    plt.legend()\n    \n    # Plot loss curves\n    plt.subplot(1, 2, 2)\n    plt.plot(vgg16_history.history['loss'], label='Training Loss')\n    plt.plot(vgg16_history.history['val_loss'], label='Validation Loss')\n    plt.title('VGG16 Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend()\n    \n    plt.tight_layout()\n    plt.savefig('vgg16_learning_curves.png')\n    plt.close()\n    print(\"VGG16 learning curves plot saved.\")\nexcept Exception as e:\n    print(f\"Could not create learning curve plots. Error: {str(e)}\")\n\n# Create a VGG16 model summary\nwith open('vgg16_model_summary.txt', 'w') as f:\n    f.write(\"===== VGG16 MODEL SUMMARY =====\\n\\n\")\n    \n    f.write(\"Parameters used:\\n\")\n    f.write(f\"- Image Size: {IMG_SIZE}x{IMG_SIZE}\\n\")\n    f.write(f\"- Batch Size: {BATCH_SIZE}\\n\")\n    f.write(f\"- Maximum Epochs: {EPOCHS}\\n\")\n    f.write(f\"- Dropout Rate: {DROPOUT_RATE}\\n\")\n    f.write(f\"- Learning Rate: {LEARNING_RATE}\\n\")\n    f.write(f\"- Classes: {', '.join(classes)}\\n\\n\")\n    \n    # VGG16 summary\n    vgg16_val_accuracy = max(vgg16_history.history['val_accuracy'])\n    vgg16_val_loss = min(vgg16_history.history['val_loss'])\n    f.write(f\"VGG16 Results:\\n\")\n    f.write(f\"- Best Validation Accuracy: {vgg16_val_accuracy:.4f}\\n\")\n    f.write(f\"- Best Validation Loss: {vgg16_val_loss:.4f}\\n\")\n    f.write(f\"- Training Epochs Completed: {len(vgg16_history.history['accuracy'])}\\n\")\n\nprint(\"VGG16 model summary saved to 'vgg16_model_summary.txt'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T08:28:01.660445Z","iopub.execute_input":"2025-04-28T08:28:01.660817Z","iopub.status.idle":"2025-04-28T08:59:53.961382Z","shell.execute_reply.started":"2025-04-28T08:28:01.660791Z","shell.execute_reply":"2025-04-28T08:59:53.960423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.applications import VGG16, ResNet50\nfrom tensorflow.keras.layers import (Input, Dense, Dropout, Flatten, Conv2D, BatchNormalization,\n                                     MaxPooling2D, Activation, GlobalAveragePooling2D)\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import class_weight\nfrom sklearn.metrics import classification_report, accuracy_score, precision_score, recall_score, f1_score\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.keras.regularizers import l2\n\n# Clear any previous session\ntf.keras.backend.clear_session()\n\n# Parameters\nIMG_SIZE = 128\nBATCH_SIZE = 32\nEPOCHS = 100\nNUM_CLASSES = 3\nDROPOUT_RATE = 0.5\nLEARNING_RATE = 1e-4\nDATASET_PATH = \"/kaggle/input/preprocessed-miniddsm\"\n\n# Preprocess all data\nclasses = ['Benign', 'Cancer', 'Normal']\nimage_paths, labels = [], []\nfor idx, class_name in enumerate(classes):\n    class_dir = os.path.join(DATASET_PATH, class_name)\n    for fname in os.listdir(class_dir):\n        image_paths.append(os.path.join(class_dir, fname))\n        labels.append(idx)\n\nimage_paths = np.array(image_paths)\nlabels = np.array(labels)\n\n# Load and preprocess images\nX = np.zeros((len(image_paths), IMG_SIZE, IMG_SIZE, 1), dtype=np.float32)\nfor i, path in enumerate(image_paths):\n    img = load_img(path, color_mode='grayscale', target_size=(IMG_SIZE, IMG_SIZE))\n    arr = img_to_array(img).astype(np.float32)\n    #arr = (arr - np.mean(arr)) / np.std(arr) \n    X[i] = arr\n\n# Split data into training and validation sets BEFORE one-hot encoding\nX_train, X_val, y_train_labels, y_val_labels = train_test_split(X, labels, test_size=0.2, random_state=42, stratify=labels)\n\n# Convert to one-hot encoding AFTER splitting\ny_train = to_categorical(y_train_labels, num_classes=NUM_CLASSES)\ny_val = to_categorical(y_val_labels, num_classes=NUM_CLASSES)\n\n# Data augmentation setup\ntrain_datagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\nval_datagen = ImageDataGenerator()\n\n# Convert grayscale to RGB for transfer learning\nX_train_rgb = np.repeat(X_train, 3, axis=-1)\nX_val_rgb = np.repeat(X_val, 3, axis=-1)\n\ntrain_generator = train_datagen.flow(X_train_rgb, y_train, batch_size=BATCH_SIZE)\nval_generator = val_datagen.flow(X_val_rgb, y_val, batch_size=BATCH_SIZE, shuffle=False)\n\n# Compute class weights to handle imbalance\nclass_weights = class_weight.compute_class_weight(\n    class_weight='balanced', \n    classes=np.unique(labels), \n    y=y_train_labels\n)\nclass_weights_dict = dict(enumerate(class_weights))\n\n# Setup callbacks\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_loss', factor=0.2, patience=5, min_lr=1e-7, verbose=1)\n\nearly_stopping = tf.keras.callbacks.EarlyStopping(\n    monitor='val_loss', patience=10, restore_best_weights=True, verbose=1)\n\n# Define evaluation function\ndef evaluate_model(model, model_name):\n    print(f\"\\nEvaluating {model_name} on validation set:\")\n    val_preds = model.predict(X_val_rgb, verbose=0)\n    val_pred_classes = np.argmax(val_preds, axis=1)\n    val_true_classes = np.argmax(y_val, axis=1)\n\n    # Calculate the metrics\n    accuracy = accuracy_score(val_true_classes, val_pred_classes)\n    precision_per_class = precision_score(val_true_classes, val_pred_classes, average=None)\n    precision_macro = precision_score(val_true_classes, val_pred_classes, average='macro')\n    precision_weighted = precision_score(val_true_classes, val_pred_classes, average='weighted')\n    recall_per_class = recall_score(val_true_classes, val_pred_classes, average=None)\n    recall_macro = recall_score(val_true_classes, val_pred_classes, average='macro')\n    recall_weighted = recall_score(val_true_classes, val_pred_classes, average='weighted')\n    f1_per_class = f1_score(val_true_classes, val_pred_classes, average=None)\n    f1_macro = f1_score(val_true_classes, val_pred_classes, average='macro')\n    f1_weighted = f1_score(val_true_classes, val_pred_classes, average='weighted')\n\n    # Display results\n    print(f\"\\n===== {model_name} EVALUATION METRICS =====\")\n    print(f\"Overall Accuracy: {accuracy:.4f}\")\n\n    print(\"\\nPer-class Metrics:\")\n    print(\"-\" * 70)\n    print(f\"{'Class':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\")\n    print(\"-\" * 70)\n    for i, class_name in enumerate(classes):\n        print(f\"{class_name:20} {precision_per_class[i]:.4f}{' ':10} {recall_per_class[i]:.4f}{' ':10} {f1_per_class[i]:.4f}\")\n    print(\"-\" * 70)\n\n    print(\"\\nAveraged Metrics:\")\n    print(\"-\" * 70)\n    print(f\"{'Metric Type':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\")\n    print(\"-\" * 70)\n    print(f\"{'Macro Average':20} {precision_macro:.4f}{' ':10} {recall_macro:.4f}{' ':10} {f1_macro:.4f}\")\n    print(f\"{'Weighted Average':20} {precision_weighted:.4f}{' ':10} {recall_weighted:.4f}{' ':10} {f1_weighted:.4f}\")\n    print(\"-\" * 70)\n\n    # Save the detailed metrics to a text file\n    with open(f'{model_name}_evaluation_metrics.txt', 'w') as f:\n        f.write(f\"===== {model_name} EVALUATION METRICS =====\\n\")\n        f.write(f\"Overall Accuracy: {accuracy:.4f}\\n\\n\")\n        \n        f.write(\"Per-class Metrics:\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        f.write(f\"{'Class':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        for i, class_name in enumerate(classes):\n            f.write(f\"{class_name:20} {precision_per_class[i]:.4f}{' ':10} {recall_per_class[i]:.4f}{' ':10} {f1_per_class[i]:.4f}\\n\")\n        f.write(\"-\" * 70 + \"\\n\\n\")\n        \n        f.write(\"Averaged Metrics:\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        f.write(f\"{'Metric Type':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        f.write(f\"{'Macro Average':20} {precision_macro:.4f}{' ':10} {recall_macro:.4f}{' ':10} {f1_macro:.4f}\\n\")\n        f.write(f\"{'Weighted Average':20} {precision_weighted:.4f}{' ':10} {recall_weighted:.4f}{' ':10} {f1_weighted:.4f}\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n\n    # Also save the standard classification report for reference\n    with open(f'{model_name}_classification_report.txt', 'w') as f:\n        f.write(\"\\nStandard Classification Report:\\n\")\n        f.write(classification_report(val_true_classes, val_pred_classes, target_names=classes))\n\n    print(f\"Detailed metrics saved to '{model_name}_evaluation_metrics.txt'\")\n    print(f\"Classification report saved to '{model_name}_classification_report.txt'\")\n    \n    return model\n\n# ---- AlexNet Model ----\ndef build_alexnet():\n    model = Sequential([\n        # First convolutional block\n        Conv2D(96, kernel_size=11, strides=4, padding='same', input_shape=(IMG_SIZE, IMG_SIZE, 3), \n               kernel_regularizer=l2(0.001)),\n        BatchNormalization(),\n        Activation('relu'),\n        MaxPooling2D(pool_size=3, strides=2),\n        \n        # Second convolutional block\n        Conv2D(256, kernel_size=5, padding='same', kernel_regularizer=l2(0.001)),\n        BatchNormalization(),\n        Activation('relu'),\n        MaxPooling2D(pool_size=3, strides=2),\n        \n        # Third convolutional block\n        Conv2D(384, kernel_size=3, padding='same', kernel_regularizer=l2(0.001)),\n        BatchNormalization(),\n        Activation('relu'),\n        \n        # Fourth convolutional block\n        Conv2D(384, kernel_size=3, padding='same', kernel_regularizer=l2(0.001)),\n        BatchNormalization(),\n        Activation('relu'),\n        \n        # Fifth convolutional block\n        Conv2D(256, kernel_size=3, padding='same', kernel_regularizer=l2(0.001)),\n        BatchNormalization(),\n        Activation('relu'),\n        MaxPooling2D(pool_size=3, strides=2),\n        \n        # Flatten layer\n        Flatten(),\n        \n        # Dense layers\n        Dense(4096, activation='relu', kernel_regularizer=l2(0.001)),\n        BatchNormalization(),\n        Dropout(DROPOUT_RATE),\n        \n        Dense(4096, activation='relu', kernel_regularizer=l2(0.001)),\n        BatchNormalization(),\n        Dropout(DROPOUT_RATE),\n        \n        # Output layer\n        Dense(NUM_CLASSES, activation='softmax')\n    ])\n    \n    model.compile(\n        optimizer=Adam(learning_rate=LEARNING_RATE),\n        loss='categorical_crossentropy',\n        metrics=['accuracy']\n    )\n    \n    return model\n\n# ---- Train ALEXNET Model ----\nprint(\"\\n===== TRAINING ALEXNET MODEL =====\")\nalexnet_model = build_alexnet()\nalexnet_checkpoint = tf.keras.callbacks.ModelCheckpoint(\n    'alexnet_best_model.h5', monitor='val_loss', save_best_only=True, verbose=1)\nalexnet_callbacks = [early_stopping, reduce_lr, alexnet_checkpoint]\n\nalexnet_history = alexnet_model.fit(\n    train_generator,\n    epochs=EPOCHS,\n    validation_data=val_generator,\n    callbacks=alexnet_callbacks,\n    class_weight=class_weights_dict,\n    verbose=1\n)\n\nalexnet_model = evaluate_model(alexnet_model, \"ALEXNET\")\n# Commented out to avoid memory error\n# alexnet_model.save('alexnet_final.h5')\nprint(\"ALEXNET model evaluated.\")\n\n# Plot learning curves for ALEXNET only\ntry:\n    import matplotlib.pyplot as plt\n    \n    # Plot accuracy curves\n    plt.figure(figsize=(12, 5))\n    \n    plt.subplot(1, 2, 1)\n    plt.plot(alexnet_history.history['accuracy'], label='Training Accuracy')\n    plt.plot(alexnet_history.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('AlexNet Accuracy')\n    plt.xlabel('Epoch')\n    plt.ylabel('Accuracy')\n    plt.legend()\n    \n    # Plot loss curves\n    plt.subplot(1, 2, 2)\n    plt.plot(alexnet_history.history['loss'], label='Training Loss')\n    plt.plot(alexnet_history.history['val_loss'], label='Validation Loss')\n    plt.title('AlexNet Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend()\n    \n    plt.tight_layout()\n    plt.savefig('alexnet_learning_curves.png')\n    plt.close()\n    print(\"AlexNet learning curves plot saved.\")\nexcept Exception as e:\n    print(f\"Could not create learning curve plots. Error: {str(e)}\")\n\n# Create a AlexNet model summary\nwith open('alexnet_model_summary.txt', 'w') as f:\n    f.write(\"===== AlexNet MODEL SUMMARY =====\\n\\n\")\n    \n    f.write(\"Parameters used:\\n\")\n    f.write(f\"- Image Size: {IMG_SIZE}x{IMG_SIZE}\\n\")\n    f.write(f\"- Batch Size: {BATCH_SIZE}\\n\")\n    f.write(f\"- Maximum Epochs: {EPOCHS}\\n\")\n    f.write(f\"- Dropout Rate: {DROPOUT_RATE}\\n\")\n    f.write(f\"- Learning Rate: {LEARNING_RATE}\\n\")\n    f.write(f\"- Classes: {', '.join(classes)}\\n\\n\")\n    \n    # AlexNet summary\n    alexnet_val_accuracy = max(alexnet_history.history['val_accuracy'])\n    alexnet_val_loss = min(alexnet_history.history['val_loss'])\n    f.write(f\"AlexNet Results:\\n\")\n    f.write(f\"- Best Validation Accuracy: {alexnet_val_accuracy:.4f}\\n\")\n    f.write(f\"- Best Validation Loss: {alexnet_val_loss:.4f}\\n\")\n    f.write(f\"- Training Epochs Completed: {len(alexnet_history.history['accuracy'])}\\n\")\n\nprint(\"AlexNet model summary saved to 'alexnet_model_summary.txt'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T11:29:46.151345Z","iopub.execute_input":"2025-04-28T11:29:46.151865Z","iopub.status.idle":"2025-04-28T12:10:51.392723Z","shell.execute_reply.started":"2025-04-28T11:29:46.151764Z","shell.execute_reply":"2025-04-28T12:10:51.391793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.applications import VGG16, ResNet50\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import (Input, Dense, Dropout, Flatten, Conv2D, BatchNormalization,\n                                     MaxPooling2D, Activation, GlobalAveragePooling2D)\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import class_weight\nfrom sklearn.metrics import classification_report, accuracy_score, precision_score, recall_score, f1_score\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.keras.regularizers import l2\n\n# Clear any previous session\ntf.keras.backend.clear_session()\n\n# Parameters\nIMG_SIZE = 128\nBATCH_SIZE = 32\nEPOCHS = 100\nNUM_CLASSES = 3\nDROPOUT_RATE = 0.5\nLEARNING_RATE = 1e-4\nDATASET_PATH = \"/kaggle/input/preprocessed-miniddsm\"\n\n# Preprocess all data\nclasses = ['Benign', 'Cancer', 'Normal']\nimage_paths, labels = [], []\nfor idx, class_name in enumerate(classes):\n    class_dir = os.path.join(DATASET_PATH, class_name)\n    for fname in os.listdir(class_dir):\n        image_paths.append(os.path.join(class_dir, fname))\n        labels.append(idx)\n\nimage_paths = np.array(image_paths)\nlabels = np.array(labels)\n\n# Load and preprocess images\nX = np.zeros((len(image_paths), IMG_SIZE, IMG_SIZE, 1), dtype=np.float32)\nfor i, path in enumerate(image_paths):\n    img = load_img(path, color_mode='grayscale', target_size=(IMG_SIZE, IMG_SIZE))\n    arr = img_to_array(img).astype(np.float32)\n    #arr = (arr - np.mean(arr)) / np.std(arr) \n    X[i] = arr\n\n# Split data into training and validation sets BEFORE one-hot encoding\nX_train, X_val, y_train_labels, y_val_labels = train_test_split(X, labels, test_size=0.2, random_state=42, stratify=labels)\n\n# Convert to one-hot encoding AFTER splitting\ny_train = to_categorical(y_train_labels, num_classes=NUM_CLASSES)\ny_val = to_categorical(y_val_labels, num_classes=NUM_CLASSES)\n\n# Data augmentation setup\ntrain_datagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\nval_datagen = ImageDataGenerator()\n\n# Convert grayscale to RGB for transfer learning\nX_train_rgb = np.repeat(X_train, 3, axis=-1)\nX_val_rgb = np.repeat(X_val, 3, axis=-1)\n\ntrain_generator = train_datagen.flow(X_train_rgb, y_train, batch_size=BATCH_SIZE)\nval_generator = val_datagen.flow(X_val_rgb, y_val, batch_size=BATCH_SIZE, shuffle=False)\n\n# Compute class weights to handle imbalance\nclass_weights = class_weight.compute_class_weight(\n    class_weight='balanced', \n    classes=np.unique(labels), \n    y=y_train_labels\n)\nclass_weights_dict = dict(enumerate(class_weights))\n\n# Setup callbacks\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_loss', factor=0.2, patience=5, min_lr=1e-7, verbose=1)\n\nearly_stopping = tf.keras.callbacks.EarlyStopping(\n    monitor='val_loss', patience=10, restore_best_weights=True, verbose=1)\n\n# Define evaluation function\ndef evaluate_model(model, model_name):\n    print(f\"\\nEvaluating {model_name} on validation set:\")\n    val_preds = model.predict(X_val_rgb, verbose=0)\n    val_pred_classes = np.argmax(val_preds, axis=1)\n    val_true_classes = np.argmax(y_val, axis=1)\n\n    # Calculate the metrics\n    accuracy = accuracy_score(val_true_classes, val_pred_classes)\n    precision_per_class = precision_score(val_true_classes, val_pred_classes, average=None)\n    precision_macro = precision_score(val_true_classes, val_pred_classes, average='macro')\n    precision_weighted = precision_score(val_true_classes, val_pred_classes, average='weighted')\n    recall_per_class = recall_score(val_true_classes, val_pred_classes, average=None)\n    recall_macro = recall_score(val_true_classes, val_pred_classes, average='macro')\n    recall_weighted = recall_score(val_true_classes, val_pred_classes, average='weighted')\n    f1_per_class = f1_score(val_true_classes, val_pred_classes, average=None)\n    f1_macro = f1_score(val_true_classes, val_pred_classes, average='macro')\n    f1_weighted = f1_score(val_true_classes, val_pred_classes, average='weighted')\n\n    # Display results\n    print(f\"\\n===== {model_name} EVALUATION METRICS =====\")\n    print(f\"Overall Accuracy: {accuracy:.4f}\")\n\n    print(\"\\nPer-class Metrics:\")\n    print(\"-\" * 70)\n    print(f\"{'Class':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\")\n    print(\"-\" * 70)\n    for i, class_name in enumerate(classes):\n        print(f\"{class_name:20} {precision_per_class[i]:.4f}{' ':10} {recall_per_class[i]:.4f}{' ':10} {f1_per_class[i]:.4f}\")\n    print(\"-\" * 70)\n\n    print(\"\\nAveraged Metrics:\")\n    print(\"-\" * 70)\n    print(f\"{'Metric Type':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\")\n    print(\"-\" * 70)\n    print(f\"{'Macro Average':20} {precision_macro:.4f}{' ':10} {recall_macro:.4f}{' ':10} {f1_macro:.4f}\")\n    print(f\"{'Weighted Average':20} {precision_weighted:.4f}{' ':10} {recall_weighted:.4f}{' ':10} {f1_weighted:.4f}\")\n    print(\"-\" * 70)\n\n    # Save the detailed metrics to a text file\n    with open(f'{model_name}_evaluation_metrics.txt', 'w') as f:\n        f.write(f\"===== {model_name} EVALUATION METRICS =====\\n\")\n        f.write(f\"Overall Accuracy: {accuracy:.4f}\\n\\n\")\n        \n        f.write(\"Per-class Metrics:\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        f.write(f\"{'Class':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        for i, class_name in enumerate(classes):\n            f.write(f\"{class_name:20} {precision_per_class[i]:.4f}{' ':10} {recall_per_class[i]:.4f}{' ':10} {f1_per_class[i]:.4f}\\n\")\n        f.write(\"-\" * 70 + \"\\n\\n\")\n        \n        f.write(\"Averaged Metrics:\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        f.write(f\"{'Metric Type':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        f.write(f\"{'Macro Average':20} {precision_macro:.4f}{' ':10} {recall_macro:.4f}{' ':10} {f1_macro:.4f}\\n\")\n        f.write(f\"{'Weighted Average':20} {precision_weighted:.4f}{' ':10} {recall_weighted:.4f}{' ':10} {f1_weighted:.4f}\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n\n    # Also save the standard classification report for reference\n    with open(f'{model_name}_classification_report.txt', 'w') as f:\n        f.write(\"\\nStandard Classification Report:\\n\")\n        f.write(classification_report(val_true_classes, val_pred_classes, target_names=classes))\n\n    print(f\"Detailed metrics saved to '{model_name}_evaluation_metrics.txt'\")\n    print(f\"Classification report saved to '{model_name}_classification_report.txt'\")\n    \n    return model\n\n# ---- ResNet50 Model ----\ndef build_resnet50():\n    base_model = ResNet50(include_top=False, input_shape=(IMG_SIZE, IMG_SIZE, 3), weights='imagenet')\n    \n    # Freeze early layers\n    for layer in base_model.layers[:-15]:  # Train only last 15 layers\n        layer.trainable = False\n    \n    x = base_model.output\n    x = GlobalAveragePooling2D()(x)\n    x = Dense(512, activation='relu', kernel_regularizer=l2(0.001))(x)\n    x = BatchNormalization()(x)\n    x = Dropout(DROPOUT_RATE)(x)\n    x = Dense(256, activation='relu', kernel_regularizer=l2(0.001))(x)\n    x = BatchNormalization()(x)\n    x = Dropout(DROPOUT_RATE)(x)\n    \n    # Output layer\n    output = Dense(NUM_CLASSES, activation='softmax')(x)\n    \n    model = Model(inputs=base_model.input, outputs=output)\n    \n    model.compile(\n        optimizer=Adam(learning_rate=LEARNING_RATE),\n        loss='categorical_crossentropy',\n        metrics=['accuracy']\n    )\n    \n    return model\n\n# ---- Train ResNet50 Model ----\nprint(\"\\n===== TRAINING RESNET50 MODEL =====\")\nresnet50_model = build_resnet50()\nresnet50_checkpoint = tf.keras.callbacks.ModelCheckpoint(\n    'resnet50_best_model.h5', monitor='val_loss', save_best_only=True, verbose=1)\nresnet50_callbacks = [early_stopping, reduce_lr, resnet50_checkpoint]\n\nresnet50_history = resnet50_model.fit(\n    train_generator,\n    epochs=EPOCHS,\n    validation_data=val_generator,\n    callbacks=resnet50_callbacks,\n    class_weight=class_weights_dict,\n    verbose=1\n)\n\nresnet50_model = evaluate_model(resnet50_model, \"ResNet50\")\n# Commented out to avoid memory error\n# resnet50_model.save('resnet50_final.h5')\nprint(\"ResNet50 model evaluated.\")\n\n# Plot learning curves for resnet50 only\ntry:\n    import matplotlib.pyplot as plt\n    \n    # Plot accuracy curves\n    plt.figure(figsize=(12, 5))\n    \n    plt.subplot(1, 2, 1)\n    plt.plot(resnet50_history.history['accuracy'], label='Training Accuracy')\n    plt.plot(resnet50_history.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('ResNet50 Accuracy')\n    plt.xlabel('Epoch')\n    plt.ylabel('Accuracy')\n    plt.legend()\n    \n    # Plot loss curves\n    plt.subplot(1, 2, 2)\n    plt.plot(resnet50_history.history['loss'], label='Training Loss')\n    plt.plot(resnet50_history.history['val_loss'], label='Validation Loss')\n    plt.title('ResNet50 Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend()\n    \n    plt.tight_layout()\n    plt.savefig('resnet50_learning_curves.png')\n    plt.close()\n    print(\"ResNet50 learning curves plot saved.\")\nexcept Exception as e:\n    print(f\"Could not create learning curve plots. Error: {str(e)}\")\n\n# Create a ResNet50 model summary\nwith open('resnet50_model_summary.txt', 'w') as f:\n    f.write(\"===== ResNet50 MODEL SUMMARY =====\\n\\n\")\n    \n    f.write(\"Parameters used:\\n\")\n    f.write(f\"- Image Size: {IMG_SIZE}x{IMG_SIZE}\\n\")\n    f.write(f\"- Batch Size: {BATCH_SIZE}\\n\")\n    f.write(f\"- Maximum Epochs: {EPOCHS}\\n\")\n    f.write(f\"- Dropout Rate: {DROPOUT_RATE}\\n\")\n    f.write(f\"- Learning Rate: {LEARNING_RATE}\\n\")\n    f.write(f\"- Classes: {', '.join(classes)}\\n\\n\")\n    \n    # ResNet50 summary\n    resnet50_val_accuracy = max(resnet50_history.history['val_accuracy'])\n    resnet50_val_loss = min(resnet50_history.history['val_loss'])\n    f.write(f\"ResNet50 Results:\\n\")\n    f.write(f\"- Best Validation Accuracy: {resnet50_val_accuracy:.4f}\\n\")\n    f.write(f\"- Best Validation Loss: {resnet50_val_loss:.4f}\\n\")\n    f.write(f\"- Training Epochs Completed: {len(resnet50_history.history['accuracy'])}\\n\")\n\nprint(\"ResNet50 model summary saved to 'ResNet50_model_summary.txt'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T12:26:14.862871Z","iopub.execute_input":"2025-04-28T12:26:14.863244Z","iopub.status.idle":"2025-04-28T12:56:21.579254Z","shell.execute_reply.started":"2025-04-28T12:26:14.863214Z","shell.execute_reply":"2025-04-28T12:56:21.578127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.applications import VGG16, ResNet50\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import (Input, Dense, Dropout, Flatten, Conv2D, BatchNormalization,\n                                     MaxPooling2D, Activation, GlobalAveragePooling2D)\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import class_weight\nfrom sklearn.metrics import classification_report, accuracy_score, precision_score, recall_score, f1_score\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.keras.regularizers import l2\n\n# Clear any previous session\ntf.keras.backend.clear_session()\n\n# Parameters\nIMG_SIZE = 128\nBATCH_SIZE = 32\nEPOCHS = 100\nNUM_CLASSES = 3\nDROPOUT_RATE = 0.5\nLEARNING_RATE = 1e-4\nDATASET_PATH = \"/kaggle/input/preprocessed-miniddsm\"\n\n# Preprocess all data\nclasses = ['Benign', 'Cancer', 'Normal']\nimage_paths, labels = [], []\nfor idx, class_name in enumerate(classes):\n    class_dir = os.path.join(DATASET_PATH, class_name)\n    for fname in os.listdir(class_dir):\n        image_paths.append(os.path.join(class_dir, fname))\n        labels.append(idx)\n\nimage_paths = np.array(image_paths)\nlabels = np.array(labels)\n\n# Load and preprocess images\nX = np.zeros((len(image_paths), IMG_SIZE, IMG_SIZE, 1), dtype=np.float32)\nfor i, path in enumerate(image_paths):\n    img = load_img(path, color_mode='grayscale', target_size=(IMG_SIZE, IMG_SIZE))\n    arr = img_to_array(img).astype(np.float32)\n    #arr = (arr - np.mean(arr)) / np.std(arr) \n    X[i] = arr\n\n# Split data into training and validation sets BEFORE one-hot encoding\nX_train, X_val, y_train_labels, y_val_labels = train_test_split(X, labels, test_size=0.2, random_state=42, stratify=labels)\n\n# Convert to one-hot encoding AFTER splitting\ny_train = to_categorical(y_train_labels, num_classes=NUM_CLASSES)\ny_val = to_categorical(y_val_labels, num_classes=NUM_CLASSES)\n\n# Data augmentation setup\ntrain_datagen = ImageDataGenerator(\n    rotation_range=90,\n    width_shift_range=0.35,\n    height_shift_range=0.35,\n    zoom_range=0.7,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\nval_datagen = ImageDataGenerator()\n\n# Convert grayscale to RGB for transfer learning\nX_train_rgb = np.repeat(X_train, 3, axis=-1)\nX_val_rgb = np.repeat(X_val, 3, axis=-1)\n\ntrain_generator = train_datagen.flow(X_train_rgb, y_train, batch_size=BATCH_SIZE)\nval_generator = val_datagen.flow(X_val_rgb, y_val, batch_size=BATCH_SIZE, shuffle=False)\n\n# Compute class weights to handle imbalance\nclass_weights = class_weight.compute_class_weight(\n    class_weight='balanced', \n    classes=np.unique(labels), \n    y=y_train_labels\n)\nclass_weights_dict = dict(enumerate(class_weights))\n\n# Setup callbacks\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_loss', factor=0.2, patience=5, min_lr=1e-7, verbose=1)\n\nearly_stopping = tf.keras.callbacks.EarlyStopping(\n    monitor='val_loss', patience=10, restore_best_weights=True, verbose=1)\n\n# Define evaluation function\ndef evaluate_model(model, model_name):\n    print(f\"\\nEvaluating {model_name} on validation set:\")\n    val_preds = model.predict(X_val_rgb, verbose=0)\n    val_pred_classes = np.argmax(val_preds, axis=1)\n    val_true_classes = np.argmax(y_val, axis=1)\n\n    # Calculate the metrics\n    accuracy = accuracy_score(val_true_classes, val_pred_classes)\n    precision_per_class = precision_score(val_true_classes, val_pred_classes, average=None)\n    precision_macro = precision_score(val_true_classes, val_pred_classes, average='macro')\n    precision_weighted = precision_score(val_true_classes, val_pred_classes, average='weighted')\n    recall_per_class = recall_score(val_true_classes, val_pred_classes, average=None)\n    recall_macro = recall_score(val_true_classes, val_pred_classes, average='macro')\n    recall_weighted = recall_score(val_true_classes, val_pred_classes, average='weighted')\n    f1_per_class = f1_score(val_true_classes, val_pred_classes, average=None)\n    f1_macro = f1_score(val_true_classes, val_pred_classes, average='macro')\n    f1_weighted = f1_score(val_true_classes, val_pred_classes, average='weighted')\n\n    # Display results\n    print(f\"\\n===== {model_name} EVALUATION METRICS =====\")\n    print(f\"Overall Accuracy: {accuracy:.4f}\")\n\n    print(\"\\nPer-class Metrics:\")\n    print(\"-\" * 70)\n    print(f\"{'Class':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\")\n    print(\"-\" * 70)\n    for i, class_name in enumerate(classes):\n        print(f\"{class_name:20} {precision_per_class[i]:.4f}{' ':10} {recall_per_class[i]:.4f}{' ':10} {f1_per_class[i]:.4f}\")\n    print(\"-\" * 70)\n\n    print(\"\\nAveraged Metrics:\")\n    print(\"-\" * 70)\n    print(f\"{'Metric Type':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\")\n    print(\"-\" * 70)\n    print(f\"{'Macro Average':20} {precision_macro:.4f}{' ':10} {recall_macro:.4f}{' ':10} {f1_macro:.4f}\")\n    print(f\"{'Weighted Average':20} {precision_weighted:.4f}{' ':10} {recall_weighted:.4f}{' ':10} {f1_weighted:.4f}\")\n    print(\"-\" * 70)\n\n    # Save the detailed metrics to a text file\n    with open(f'{model_name}_evaluation_metrics.txt', 'w') as f:\n        f.write(f\"===== {model_name} EVALUATION METRICS =====\\n\")\n        f.write(f\"Overall Accuracy: {accuracy:.4f}\\n\\n\")\n        \n        f.write(\"Per-class Metrics:\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        f.write(f\"{'Class':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        for i, class_name in enumerate(classes):\n            f.write(f\"{class_name:20} {precision_per_class[i]:.4f}{' ':10} {recall_per_class[i]:.4f}{' ':10} {f1_per_class[i]:.4f}\\n\")\n        f.write(\"-\" * 70 + \"\\n\\n\")\n        \n        f.write(\"Averaged Metrics:\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        f.write(f\"{'Metric Type':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n        f.write(f\"{'Macro Average':20} {precision_macro:.4f}{' ':10} {recall_macro:.4f}{' ':10} {f1_macro:.4f}\\n\")\n        f.write(f\"{'Weighted Average':20} {precision_weighted:.4f}{' ':10} {recall_weighted:.4f}{' ':10} {f1_weighted:.4f}\\n\")\n        f.write(\"-\" * 70 + \"\\n\")\n\n    # Also save the standard classification report for reference\n    with open(f'{model_name}_classification_report.txt', 'w') as f:\n        f.write(\"\\nStandard Classification Report:\\n\")\n        f.write(classification_report(val_true_classes, val_pred_classes, target_names=classes))\n\n    print(f\"Detailed metrics saved to '{model_name}_evaluation_metrics.txt'\")\n    print(f\"Classification report saved to '{model_name}_classification_report.txt'\")\n    \n    return model\n\n# ---- ResNet50 Model ----\ndef build_resnet50():\n    base_model = ResNet50(include_top=False, input_shape=(IMG_SIZE, IMG_SIZE, 3), weights='imagenet')\n    \n    # Freeze early layers\n    for layer in base_model.layers[:-15]:  # Train only last 15 layers\n        layer.trainable = False\n    \n    x = base_model.output\n    x = GlobalAveragePooling2D()(x)\n    x = Dense(512, activation='relu', kernel_regularizer=l2(0.001))(x)\n    x = BatchNormalization()(x)\n    x = Dropout(DROPOUT_RATE)(x)\n    x = Dense(256, activation='relu', kernel_regularizer=l2(0.001))(x)\n    x = BatchNormalization()(x)\n    x = Dropout(DROPOUT_RATE)(x)\n    \n    # Output layer\n    output = Dense(NUM_CLASSES, activation='softmax')(x)\n    \n    model = Model(inputs=base_model.input, outputs=output)\n    \n    model.compile(\n        optimizer=Adam(learning_rate=LEARNING_RATE),\n        loss='categorical_crossentropy',\n        metrics=['accuracy']\n    )\n    \n    return model\n\n# ---- Train ResNet50 Model ----\nprint(\"\\n===== TRAINING RESNET50 MODEL =====\")\nresnet50_model = build_resnet50()\nresnet50_checkpoint = tf.keras.callbacks.ModelCheckpoint(\n    'resnet50_best_model.h5', monitor='val_loss', save_best_only=True, verbose=1)\nresnet50_callbacks = [early_stopping, reduce_lr, resnet50_checkpoint]\n\nresnet50_history = resnet50_model.fit(\n    train_generator,\n    epochs=EPOCHS,\n    validation_data=val_generator,\n    callbacks=resnet50_callbacks,\n    class_weight=class_weights_dict,\n    verbose=1\n)\n\nresnet50_model = evaluate_model(resnet50_model, \"ResNet50\")\n# Commented out to avoid memory error\n# resnet50_model.save('resnet50_final.h5')\nprint(\"ResNet50 model evaluated.\")\n\n# Plot learning curves for resnet50 only\ntry:\n    import matplotlib.pyplot as plt\n    \n    # Plot accuracy curves\n    plt.figure(figsize=(12, 5))\n    \n    plt.subplot(1, 2, 1)\n    plt.plot(resnet50_history.history['accuracy'], label='Training Accuracy')\n    plt.plot(resnet50_history.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('ResNet50 Accuracy')\n    plt.xlabel('Epoch')\n    plt.ylabel('Accuracy')\n    plt.legend()\n    \n    # Plot loss curves\n    plt.subplot(1, 2, 2)\n    plt.plot(resnet50_history.history['loss'], label='Training Loss')\n    plt.plot(resnet50_history.history['val_loss'], label='Validation Loss')\n    plt.title('ResNet50 Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend()\n    \n    plt.tight_layout()\n    plt.savefig('resnet50_learning_curves.png')\n    plt.close()\n    print(\"ResNet50 learning curves plot saved.\")\nexcept Exception as e:\n    print(f\"Could not create learning curve plots. Error: {str(e)}\")\n\n# Create a ResNet50 model summary\nwith open('resnet50_model_summary.txt', 'w') as f:\n    f.write(\"===== ResNet50 MODEL SUMMARY =====\\n\\n\")\n    \n    f.write(\"Parameters used:\\n\")\n    f.write(f\"- Image Size: {IMG_SIZE}x{IMG_SIZE}\\n\")\n    f.write(f\"- Batch Size: {BATCH_SIZE}\\n\")\n    f.write(f\"- Maximum Epochs: {EPOCHS}\\n\")\n    f.write(f\"- Dropout Rate: {DROPOUT_RATE}\\n\")\n    f.write(f\"- Learning Rate: {LEARNING_RATE}\\n\")\n    f.write(f\"- Classes: {', '.join(classes)}\\n\\n\")\n    \n    # ResNet50 summary\n    resnet50_val_accuracy = max(resnet50_history.history['val_accuracy'])\n    resnet50_val_loss = min(resnet50_history.history['val_loss'])\n    f.write(f\"ResNet50 Results:\\n\")\n    f.write(f\"- Best Validation Accuracy: {resnet50_val_accuracy:.4f}\\n\")\n    f.write(f\"- Best Validation Loss: {resnet50_val_loss:.4f}\\n\")\n    f.write(f\"- Training Epochs Completed: {len(resnet50_history.history['accuracy'])}\\n\")\n\nprint(\"ResNet50 model summary saved to 'ResNet50_model_summary.txt'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T13:53:11.554926Z","iopub.execute_input":"2025-04-28T13:53:11.555319Z","iopub.status.idle":"2025-04-28T14:37:05.750389Z","shell.execute_reply.started":"2025-04-28T13:53:11.55529Z","shell.execute_reply":"2025-04-28T14:37:05.749445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.applications import DenseNet169\nfrom tensorflow.keras.layers import (Input, Dense, Dropout, Flatten, Conv2D, BatchNormalization,\n                                     DepthwiseConv2D, Activation, Multiply, Concatenate, GlobalAveragePooling2D)\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import class_weight\nfrom sklearn.metrics import classification_report, accuracy_score, precision_score, recall_score, f1_score\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.keras.regularizers import l2\n\n# Parameters\nIMG_SIZE = 128\nBATCH_SIZE = 32\nEPOCHS = 100\nNUM_CLASSES = 3\nDROPOUT_RATE = 0.5\nLEARNING_RATE = 1e-4\nDATASET_PATH = \"/kaggle/input/preprocessed-miniddsm\"\n # Your dataset path\n\n# Preprocess all data\nclasses = ['Benign', 'Cancer', 'Normal']\nimage_paths, labels = [], []\nfor idx, class_name in enumerate(classes):\n    class_dir = os.path.join(DATASET_PATH, class_name)\n    for fname in os.listdir(class_dir):\n        image_paths.append(os.path.join(class_dir, fname))\n        labels.append(idx)\n\nimage_paths = np.array(image_paths)\nlabels = np.array(labels)\n\n# Load and preprocess images\nX = np.zeros((len(image_paths), IMG_SIZE, IMG_SIZE, 1), dtype=np.float32)\nfor i, path in enumerate(image_paths):\n    img = load_img(path, color_mode='grayscale', target_size=(IMG_SIZE, IMG_SIZE))\n    arr = img_to_array(img).astype(np.float32)\n    arr = (arr - np.mean(arr)) / np.std(arr) \n    X[i] = arr\n\n# Split data into training and validation sets BEFORE one-hot encoding\nX_train, X_val, y_train_labels, y_val_labels = train_test_split(X, labels, test_size=0.2, random_state=42, stratify=labels)\n\n# Convert to one-hot encoding AFTER splitting\ny_train = to_categorical(y_train_labels, num_classes=NUM_CLASSES)\ny_val = to_categorical(y_val_labels, num_classes=NUM_CLASSES)\n\n# Data augmentation setup\ntrain_datagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\nval_datagen = ImageDataGenerator()\n\n# Convert grayscale to RGB for transfer learning\nX_train_rgb = np.repeat(X_train, 3, axis=-1)\nX_val_rgb = np.repeat(X_val, 3, axis=-1)\n\ntrain_generator = train_datagen.flow(X_train_rgb, y_train, batch_size=BATCH_SIZE)\nval_generator = val_datagen.flow(X_val_rgb, y_val, batch_size=BATCH_SIZE, shuffle=False)\n\n# Feature extractor with reduced trainable layers\ndef build_feature_extractor():\n    base_model = DenseNet169(include_top=False, input_shape=(IMG_SIZE, IMG_SIZE, 3), weights='imagenet')\n    \n    # Freeze most layers\n    for layer in base_model.layers[:-20]:  # Only train last 20 layers\n        layer.trainable = False\n    \n    return base_model\n\n# Improved bi-stream CNN with regularization\ndef bi_stream_cnn(input_tensor):\n    # Add BatchNormalization at the beginning\n    x = BatchNormalization()(input_tensor)\n    \n    # First stream with regularization\n    x1 = DepthwiseConv2D(kernel_size=3, strides=1, padding='same')(x)\n    x1 = BatchNormalization()(x1)\n    x1 = Activation('relu')(x1)\n    x1 = Conv2D(32, kernel_size=3, padding='same', kernel_regularizer=l2(0.001))(x1)\n    x1 = BatchNormalization()(x1)\n    x1 = Activation('relu')(x1)\n    \n    # Second stream with regularization\n    x2 = DepthwiseConv2D(kernel_size=5, strides=1, padding='same')(x)\n    x2 = BatchNormalization()(x2)\n    x2 = Activation('selu')(x2)\n    x2 = Conv2D(32, kernel_size=3, padding='same', kernel_regularizer=l2(0.001))(x2)\n    x2 = BatchNormalization()(x2)\n    x2 = Activation('selu')(x2)\n    \n    # Combine streams\n    x = Concatenate()([x1, x2])\n    \n    # Global pooling instead of flattening to reduce parameters\n    x = GlobalAveragePooling2D()(x)\n    \n    # Fully connected layers with dropout and regularization\n    x = Dense(64, activation='relu', kernel_regularizer=l2(0.001))(x)\n    x = BatchNormalization()(x)\n    x = Dropout(DROPOUT_RATE)(x)\n    x = Dense(32, activation='relu', kernel_regularizer=l2(0.001))(x)\n    x = BatchNormalization()(x)\n    x = Dropout(DROPOUT_RATE)(x)\n    \n    # Output layer\n    output = Dense(NUM_CLASSES, activation='softmax')(x)\n    \n    return output\n\n# Setup callbacks\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_loss', factor=0.2, patience=5, min_lr=1e-7, verbose=1)\n\ncheckpoint = tf.keras.callbacks.ModelCheckpoint(\n    'best_model.h5', monitor='val_loss', save_best_only=True, verbose=1)\n\nearly_stopping = tf.keras.callbacks.EarlyStopping(\n    monitor='val_loss', patience=10, restore_best_weights=True, verbose=1)\n\ncallbacks = [early_stopping, reduce_lr, checkpoint]\n\n# Compute class weights to handle imbalance\n# Use raw class labels (before one-hot encoding)\nclass_weights = class_weight.compute_class_weight(\n    class_weight='balanced', \n    classes=np.unique(labels), \n    y=y_train_labels\n)\nclass_weights_dict = dict(enumerate(class_weights))\n\n# Build model\nbase_model = build_feature_extractor()\nx = base_model.output\noutput = bi_stream_cnn(x)\nmodel = Model(inputs=base_model.input, outputs=output)\n\n# Compile with Adam optimizer\nmodel.compile(\n    optimizer=Adam(learning_rate=LEARNING_RATE),\n    loss='categorical_crossentropy',\n    metrics=['accuracy']\n)\n\n# Train model\nprint(\"Training model...\")\nhistory = model.fit(\n    train_generator,\n    epochs=EPOCHS,\n    validation_data=val_generator,\n    callbacks=callbacks,\n    class_weight=class_weights_dict,\n    verbose=1\n)\n\n# Enhanced evaluation section with detailed metrics\nprint(\"\\nEvaluating on validation set:\")\nval_preds = model.predict(X_val_rgb, verbose=0)\nval_pred_classes = np.argmax(val_preds, axis=1)\nval_true_classes = np.argmax(y_val, axis=1)\n\n# Calculate the four key metrics\n# 1. Overall accuracy\naccuracy = accuracy_score(val_true_classes, val_pred_classes)\n\n# 2. Per-class and averaged precision\nprecision_per_class = precision_score(val_true_classes, val_pred_classes, average=None)\nprecision_macro = precision_score(val_true_classes, val_pred_classes, average='macro')\nprecision_weighted = precision_score(val_true_classes, val_pred_classes, average='weighted')\n\n# 3. Per-class and averaged recall\nrecall_per_class = recall_score(val_true_classes, val_pred_classes, average=None)\nrecall_macro = recall_score(val_true_classes, val_pred_classes, average='macro')\nrecall_weighted = recall_score(val_true_classes, val_pred_classes, average='weighted')\n\n# 4. Per-class and averaged F1-score\nf1_per_class = f1_score(val_true_classes, val_pred_classes, average=None)\nf1_macro = f1_score(val_true_classes, val_pred_classes, average='macro')\nf1_weighted = f1_score(val_true_classes, val_pred_classes, average='weighted')\n\n# Display results in a more detailed way\nprint(\"\\n===== DETAILED EVALUATION METRICS =====\")\nprint(f\"Overall Accuracy: {accuracy:.4f}\")\n\nprint(\"\\nPer-class Metrics:\")\nprint(\"-\" * 70)\nprint(f\"{'Class':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\")\nprint(\"-\" * 70)\nfor i, class_name in enumerate(classes):\n    print(f\"{class_name:20} {precision_per_class[i]:.4f}{' ':10} {recall_per_class[i]:.4f}{' ':10} {f1_per_class[i]:.4f}\")\nprint(\"-\" * 70)\n\nprint(\"\\nAveraged Metrics:\")\nprint(\"-\" * 70)\nprint(f\"{'Metric Type':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\")\nprint(\"-\" * 70)\nprint(f\"{'Macro Average':20} {precision_macro:.4f}{' ':10} {recall_macro:.4f}{' ':10} {f1_macro:.4f}\")\nprint(f\"{'Weighted Average':20} {precision_weighted:.4f}{' ':10} {recall_weighted:.4f}{' ':10} {f1_weighted:.4f}\")\nprint(\"-\" * 70)\n\n# Save the detailed metrics to a text file\nwith open('evaluation_metrics.txt', 'w') as f:\n    f.write(\"===== DETAILED EVALUATION METRICS =====\\n\")\n    f.write(f\"Overall Accuracy: {accuracy:.4f}\\n\\n\")\n    \n    f.write(\"Per-class Metrics:\\n\")\n    f.write(\"-\" * 70 + \"\\n\")\n    f.write(f\"{'Class':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\\n\")\n    f.write(\"-\" * 70 + \"\\n\")\n    for i, class_name in enumerate(classes):\n        f.write(f\"{class_name:20} {precision_per_class[i]:.4f}{' ':10} {recall_per_class[i]:.4f}{' ':10} {f1_per_class[i]:.4f}\\n\")\n    f.write(\"-\" * 70 + \"\\n\\n\")\n    \n    f.write(\"Averaged Metrics:\\n\")\n    f.write(\"-\" * 70 + \"\\n\")\n    f.write(f\"{'Metric Type':20} {'Precision':15} {'Recall':15} {'F1-Score':15}\\n\")\n    f.write(\"-\" * 70 + \"\\n\")\n    f.write(f\"{'Macro Average':20} {precision_macro:.4f}{' ':10} {recall_macro:.4f}{' ':10} {f1_macro:.4f}\\n\")\n    f.write(f\"{'Weighted Average':20} {precision_weighted:.4f}{' ':10} {recall_weighted:.4f}{' ':10} {f1_weighted:.4f}\\n\")\n    f.write(\"-\" * 70 + \"\\n\")\n\nprint(\"Detailed metrics saved to 'evaluation_metrics.txt'\")\n\n# Also print the standard classification report for reference\nprint(\"\\nStandard Classification Report:\")\nprint(classification_report(val_true_classes, val_pred_classes, target_names=classes))\n\n# Save final model\nmodel.save('improved_bi_stream_cnn_final.h5')\nprint(\"Final model saved.\")\n\n# Optional: Plot learning curves\ntry:\n    import matplotlib.pyplot as plt\n    \n    plt.figure(figsize=(12, 4))\n    plt.subplot(1, 2, 1)\n    plt.plot(history.history['accuracy'], label='Training Accuracy')\n    plt.plot(history.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('Accuracy')\n    plt.xlabel('Epoch')\n    plt.ylabel('Accuracy')\n    plt.legend()\n    \n    plt.subplot(1, 2, 2)\n    plt.plot(history.history['loss'], label='Training Loss')\n    plt.plot(history.history['val_loss'], label='Validation Loss')\n    plt.title('Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend()\n    \n    plt.tight_layout()\n    plt.savefig('learning_curves.png')\n    plt.close()\n    print(\"Learning curves plot saved.\")\nexcept:\n    print(\"Could not create learning curve plots.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T17:05:10.524098Z","iopub.execute_input":"2025-04-28T17:05:10.524481Z","iopub.status.idle":"2025-04-28T17:38:57.180525Z","shell.execute_reply.started":"2025-04-28T17:05:10.524449Z","shell.execute_reply":"2025-04-28T17:38:57.179561Z"}},"outputs":[],"execution_count":null}]}