{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":18647,"databundleVersionId":1126921,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-01T09:58:07.160965Z","iopub.execute_input":"2024-04-01T09:58:07.16138Z","iopub.status.idle":"2024-04-01T09:58:48.183174Z","shell.execute_reply.started":"2024-04-01T09:58:07.161353Z","shell.execute_reply":"2024-04-01T09:58:48.182097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/prostate-cancer-grade-assessment/train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:48.184886Z","iopub.execute_input":"2024-04-01T09:58:48.185285Z","iopub.status.idle":"2024-04-01T09:58:48.226443Z","shell.execute_reply.started":"2024-04-01T09:58:48.18526Z","shell.execute_reply":"2024-04-01T09:58:48.225624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:48.227618Z","iopub.execute_input":"2024-04-01T09:58:48.227957Z","iopub.status.idle":"2024-04-01T09:58:48.251076Z","shell.execute_reply.started":"2024-04-01T09:58:48.22792Z","shell.execute_reply":"2024-04-01T09:58:48.250122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_gleason_score(score):\n    try:\n        parts = score.split('+')\n        return int(parts[0]) + int(parts[1])\n    except (ValueError, AttributeError):\n        return np.nan\n\ntrain_df['gleason_score_numeric'] = train_df['gleason_score'].apply(convert_gleason_score)","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:48.253674Z","iopub.execute_input":"2024-04-01T09:58:48.254364Z","iopub.status.idle":"2024-04-01T09:58:48.274902Z","shell.execute_reply.started":"2024-04-01T09:58:48.254339Z","shell.execute_reply":"2024-04-01T09:58:48.273998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:48.275898Z","iopub.execute_input":"2024-04-01T09:58:48.276181Z","iopub.status.idle":"2024-04-01T09:58:48.299834Z","shell.execute_reply.started":"2024-04-01T09:58:48.276157Z","shell.execute_reply":"2024-04-01T09:58:48.298568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:48.30104Z","iopub.execute_input":"2024-04-01T09:58:48.3019Z","iopub.status.idle":"2024-04-01T09:58:49.752869Z","shell.execute_reply.started":"2024-04-01T09:58:48.301869Z","shell.execute_reply":"2024-04-01T09:58:49.751989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:49.754083Z","iopub.execute_input":"2024-04-01T09:58:49.754626Z","iopub.status.idle":"2024-04-01T09:58:49.759335Z","shell.execute_reply.started":"2024-04-01T09:58:49.754586Z","shell.execute_reply":"2024-04-01T09:58:49.758326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nplt.subplot(1, 2, 1)\nsns.countplot(x='isup_grade', data=train_df)\nplt.title('Distribution of ISUP Grade')\n\nfor p in plt.gca().patches:\n    plt.gca().annotate(f\"{p.get_height()}\", (p.get_x() + p.get_width() / 2., p.get_height()), ha='center', va='center', fontsize=10, color='black', xytext=(0, 5), textcoords='offset points')\n\nplt.subplot(1, 2, 2)\nsns.countplot(x='gleason_score_numeric', data=train_df)\nplt.title('Distribution of Gleason Score (Numeric)')\n\nfor p in plt.gca().patches:\n    plt.gca().annotate(f\"{p.get_height()}\", (p.get_x() + p.get_width() / 2., p.get_height()), ha='center', va='center', fontsize=10, color='black', xytext=(0, 5), textcoords='offset points')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:49.760696Z","iopub.execute_input":"2024-04-01T09:58:49.761049Z","iopub.status.idle":"2024-04-01T09:58:50.492214Z","shell.execute_reply.started":"2024-04-01T09:58:49.761016Z","shell.execute_reply":"2024-04-01T09:58:50.491191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:50.493628Z","iopub.execute_input":"2024-04-01T09:58:50.494064Z","iopub.status.idle":"2024-04-01T09:58:50.756069Z","shell.execute_reply.started":"2024-04-01T09:58:50.494029Z","shell.execute_reply":"2024-04-01T09:58:50.755226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_dir = '/kaggle/input/prostate-cancer-grade-assessment/train_images/'","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:50.759546Z","iopub.execute_input":"2024-04-01T09:58:50.759944Z","iopub.status.idle":"2024-04-01T09:58:50.766175Z","shell.execute_reply.started":"2024-04-01T09:58:50.7599Z","shell.execute_reply":"2024-04-01T09:58:50.765226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_size = 256","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:50.767299Z","iopub.execute_input":"2024-04-01T09:58:50.767563Z","iopub.status.idle":"2024-04-01T09:58:50.774853Z","shell.execute_reply.started":"2024-04-01T09:58:50.767541Z","shell.execute_reply":"2024-04-01T09:58:50.773755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_ids = train_df['image_id'].iloc[:9].tolist()","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:50.775937Z","iopub.execute_input":"2024-04-01T09:58:50.776244Z","iopub.status.idle":"2024-04-01T09:58:50.786549Z","shell.execute_reply.started":"2024-04-01T09:58:50.776221Z","shell.execute_reply":"2024-04-01T09:58:50.785296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import openslide","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:50.788938Z","iopub.execute_input":"2024-04-01T09:58:50.789372Z","iopub.status.idle":"2024-04-01T09:58:50.920625Z","shell.execute_reply.started":"2024-04-01T09:58:50.789333Z","shell.execute_reply":"2024-04-01T09:58:50.919703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id = train_df['image_id'].iloc[0]\nfull_image_path = image_dir + image_id + '.tiff'\n\ntry:\n\n    example = openslide.OpenSlide(full_image_path)\n\n    clipped_example = example.read_region((5000, 5000), 0, (image_size, image_size))\n\n    plt.imshow(clipped_example)\n    plt.title(f\"Image ID: {image_id}\")\n    plt.axis('off')\n\n    example.close()\nexcept Exception as e:\n    print(f\"Error loading image {image_id}: {e}\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:50.921852Z","iopub.execute_input":"2024-04-01T09:58:50.922138Z","iopub.status.idle":"2024-04-01T09:58:51.350371Z","shell.execute_reply.started":"2024-04-01T09:58:50.922114Z","shell.execute_reply":"2024-04-01T09:58:51.349375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"image_path\"] = [image_dir+image_id+\".tiff\" for image_id in train_df[\"image_id\"]]","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:51.351629Z","iopub.execute_input":"2024-04-01T09:58:51.351946Z","iopub.status.idle":"2024-04-01T09:58:51.363042Z","shell.execute_reply.started":"2024-04-01T09:58:51.35192Z","shell.execute_reply":"2024-04-01T09:58:51.362145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:51.364471Z","iopub.execute_input":"2024-04-01T09:58:51.364789Z","iopub.status.idle":"2024-04-01T09:58:51.386261Z","shell.execute_reply.started":"2024-04-01T09:58:51.364764Z","shell.execute_reply":"2024-04-01T09:58:51.385346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='isup_grade')\nplt.title('Distribution of ISUP grades')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:51.387474Z","iopub.execute_input":"2024-04-01T09:58:51.387781Z","iopub.status.idle":"2024-04-01T09:58:51.646419Z","shell.execute_reply.started":"2024-04-01T09:58:51.387758Z","shell.execute_reply":"2024-04-01T09:58:51.645508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nplt.pie(train_df['isup_grade'].value_counts(), labels=train_df['isup_grade'].unique(), autopct='%1.1f%%', startangle=140)\nplt.title('Distribution of ISUP grades')\nplt.axis('equal')  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:51.647502Z","iopub.execute_input":"2024-04-01T09:58:51.64779Z","iopub.status.idle":"2024-04-01T09:58:51.868686Z","shell.execute_reply.started":"2024-04-01T09:58:51.647767Z","shell.execute_reply":"2024-04-01T09:58:51.867247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='data_provider')\nplt.title('Distribution of Data Providers')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:51.870516Z","iopub.execute_input":"2024-04-01T09:58:51.871691Z","iopub.status.idle":"2024-04-01T09:58:52.131039Z","shell.execute_reply.started":"2024-04-01T09:58:51.871621Z","shell.execute_reply":"2024-04-01T09:58:52.130036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nplt.pie(train_df['data_provider'].value_counts(), labels=train_df['data_provider'].unique(), autopct='%1.1f%%', startangle=140)\nplt.title('Distribution of Data Providers')\nplt.axis('equal')  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:52.132416Z","iopub.execute_input":"2024-04-01T09:58:52.132809Z","iopub.status.idle":"2024-04-01T09:58:52.301877Z","shell.execute_reply.started":"2024-04-01T09:58:52.132771Z","shell.execute_reply":"2024-04-01T09:58:52.300227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='gleason_score_numeric')\nplt.title('Distribution of Gleason Scores')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:52.303991Z","iopub.execute_input":"2024-04-01T09:58:52.305217Z","iopub.status.idle":"2024-04-01T09:58:52.587772Z","shell.execute_reply.started":"2024-04-01T09:58:52.305166Z","shell.execute_reply":"2024-04-01T09:58:52.586715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlation_matrix = train_df[['isup_grade', 'gleason_score_numeric']].corr()\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm')\nplt.title('Correlation Matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:52.588809Z","iopub.execute_input":"2024-04-01T09:58:52.589057Z","iopub.status.idle":"2024-04-01T09:58:52.891463Z","shell.execute_reply.started":"2024-04-01T09:58:52.589035Z","shell.execute_reply":"2024-04-01T09:58:52.890317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from random import randint\nfrom sklearn.utils import shuffle\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.layers import Conv2D, BatchNormalization, Activation, MaxPooling2D, GlobalAveragePooling2D, Dense, Dropout\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.models import Model","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:58:52.892885Z","iopub.execute_input":"2024-04-01T09:58:52.89318Z","iopub.status.idle":"2024-04-01T09:59:06.872258Z","shell.execute_reply.started":"2024-04-01T09:58:52.893156Z","shell.execute_reply":"2024-04-01T09:59:06.871155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = shuffle(train_df)\ntraining_item_count = int(len(train_df) * 0.8)\nvalidation_df = train_df[training_item_count:]\ntrain_df = train_df[:training_item_count]","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:59:06.873517Z","iopub.execute_input":"2024-04-01T09:59:06.874131Z","iopub.status.idle":"2024-04-01T09:59:06.884451Z","shell.execute_reply.started":"2024-04-01T09:59:06.874102Z","shell.execute_reply":"2024-04-01T09:59:06.88345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_single_sample(image_path, image_size=256, training=False, display=False):\n    image = openslide.OpenSlide(image_path)\n    mask_path = image_path.replace(\"train_images\", \"train_label_masks\").replace(\".tiff\", \"_mask.tiff\")\n    mask = openslide.OpenSlide(mask_path)\n    \n    stacked_image = []\n    groundtruth_per_image = []\n    \n    maximum_iteration = 0\n    selected_sample = False\n    while not selected_sample:\n        sampling_start_x = randint(image_size, image.dimensions[0] - image_size)\n        sampling_start_y = randint(image_size, image.dimensions[1] - image_size)\n\n        clipped_sample = image.read_region((sampling_start_x, sampling_start_y), 0, (256, 256))\n        clipped_array = np.asarray(clipped_sample)\n        \n        if (not np.all(clipped_array == 255) and np.std(clipped_array) > 20) or maximum_iteration > 200:\n            if display:\n                plt.imshow(clipped_sample)\n                plt.show()\n                \n            sampled_image = clipped_array[:, :, :3]\n            \n            if training:\n                clipped_mask = mask.read_region((sampling_start_x, sampling_start_y), 0, (256, 256))\n                groundtruth_per_image.append(np.mean(np.asarray(clipped_mask)[:, :, 0]))\n            \n            selected_sample = True\n        maximum_iteration += 1\n    \n    if training: \n        return np.array(sampled_image), np.array(groundtruth_per_image)\n    else:\n        return np.array(sampled_image)","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:59:06.885961Z","iopub.execute_input":"2024-04-01T09:59:06.886333Z","iopub.status.idle":"2024-04-01T09:59:06.910954Z","shell.execute_reply.started":"2024-04-01T09:59:06.886303Z","shell.execute_reply":"2024-04-01T09:59:06.909876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_random_samples(image_path, image_size=256, display=False):\n    image = openslide.OpenSlide(image_path)\n    stacked_image = []\n    \n    selected_samples = 0\n    maximum_iteration = 0\n    while selected_samples < 3:\n        sampling_start_x = randint(image_size, image.dimensions[0] - image_size)\n        sampling_start_y = randint(image_size, image.dimensions[1] - image_size)\n\n        clipped_sample = image.read_region((sampling_start_x, sampling_start_y), 0, (256, 256))\n        clipped_array = np.asarray(clipped_sample)\n        \n        if (not np.all(clipped_array == 255) and np.std(clipped_array) > 20) or maximum_iteration > 200:\n            if display:\n                plt.imshow(clipped_sample)\n                plt.show()\n\n            stacked_image.append(clipped_array[:, :, :3])\n            selected_samples += 1\n        maximum_iteration += 1\n    return np.array(stacked_image)","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:59:06.912322Z","iopub.execute_input":"2024-04-01T09:59:06.912607Z","iopub.status.idle":"2024-04-01T09:59:06.924942Z","shell.execute_reply.started":"2024-04-01T09:59:06.912582Z","shell.execute_reply":"2024-04-01T09:59:06.924045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_single_image_generator(image_path_list, batch_size=16):\n    while True:\n        for start in range(0, len(image_path_list), batch_size):\n            X_batch = []\n            Y_batch = []\n            end = min(start + batch_size, training_item_count)\n\n            image_info_list = [get_single_sample(image_path, training=True) for image_path in image_path_list[start:end]]\n            X_batch = np.array([image_info[0]/255. for image_info in image_info_list])\n            Y_batch = np.array([image_info[1] for image_info in image_info_list])\n            \n            yield X_batch, Y_batch ","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:59:06.926034Z","iopub.execute_input":"2024-04-01T09:59:06.926277Z","iopub.status.idle":"2024-04-01T09:59:06.940428Z","shell.execute_reply.started":"2024-04-01T09:59:06.926256Z","shell.execute_reply":"2024-04-01T09:59:06.939546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_samples = get_random_samples(train_df.iloc[0].image_path, display=True)\nprint(\"Random samples shape:\", random_samples.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:59:06.946013Z","iopub.execute_input":"2024-04-01T09:59:06.946261Z","iopub.status.idle":"2024-04-01T09:59:08.663833Z","shell.execute_reply.started":"2024-04-01T09:59:06.94624Z","shell.execute_reply":"2024-04-01T09:59:08.662923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_path = train_df.iloc[0].image_path  \nsample_image, groundtruth = get_single_sample(image_path, training=True, display=True)\n\nprint(\"Sample Image Shape:\", sample_image.shape)\nprint(\"Ground Truth:\", groundtruth)","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:59:08.66509Z","iopub.execute_input":"2024-04-01T09:59:08.665421Z","iopub.status.idle":"2024-04-01T09:59:09.149853Z","shell.execute_reply.started":"2024-04-01T09:59:08.665394Z","shell.execute_reply":"2024-04-01T09:59:09.148853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def branch(input_image):\n    x = Conv2D(64, (3, 3), padding='same')(input_image)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = Conv2D(64, (3, 3), padding='same')(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = MaxPooling2D(pool_size=(2, 2))(x)\n    \n    x = Conv2D(128, (3, 3), padding='same')(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = Conv2D(128, (3, 3), padding='same')(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = MaxPooling2D(pool_size=(2, 2))(x)\n    \n    x = Conv2D(256, (3, 3), padding='same')(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = Conv2D(256, (3, 3), padding='same')(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = GlobalAveragePooling2D()(x)\n    \n    x = Dense(256)(x)\n    x = Activation('relu')(x)\n    \n    return Dropout(0.3)(x)","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:59:09.150981Z","iopub.execute_input":"2024-04-01T09:59:09.151252Z","iopub.status.idle":"2024-04-01T09:59:09.161571Z","shell.execute_reply.started":"2024-04-01T09:59:09.15123Z","shell.execute_reply":"2024-04-01T09:59:09.160687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_image = layers.Input(shape=(256, 256, 3))\noutput = branch(input_image)\nmodel = Model(inputs=input_image, outputs=output)","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:59:09.162744Z","iopub.execute_input":"2024-04-01T09:59:09.163103Z","iopub.status.idle":"2024-04-01T09:59:10.178485Z","shell.execute_reply.started":"2024-04-01T09:59:09.163069Z","shell.execute_reply":"2024-04-01T09:59:10.177647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.layers import Input","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:59:10.179767Z","iopub.execute_input":"2024-04-01T09:59:10.180047Z","iopub.status.idle":"2024-04-01T09:59:10.184956Z","shell.execute_reply.started":"2024-04-01T09:59:10.180023Z","shell.execute_reply":"2024-04-01T09:59:10.183758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_image = Input(shape=(256, 256, 3))\n\ncore_branch = branch(input_image)\n\noutput = Dense(1, activation='linear')(core_branch)\n\nbranch_model = Model(inputs=input_image, outputs=output)","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:59:10.186319Z","iopub.execute_input":"2024-04-01T09:59:10.186639Z","iopub.status.idle":"2024-04-01T09:59:10.329734Z","shell.execute_reply.started":"2024-04-01T09:59:10.186574Z","shell.execute_reply":"2024-04-01T09:59:10.328842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"branch_model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:59:10.331066Z","iopub.execute_input":"2024-04-01T09:59:10.331403Z","iopub.status.idle":"2024-04-01T09:59:10.376503Z","shell.execute_reply.started":"2024-04-01T09:59:10.331376Z","shell.execute_reply":"2024-04-01T09:59:10.375692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.optimizers import SGD","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:59:10.377486Z","iopub.execute_input":"2024-04-01T09:59:10.377734Z","iopub.status.idle":"2024-04-01T09:59:10.381962Z","shell.execute_reply.started":"2024-04-01T09:59:10.377714Z","shell.execute_reply":"2024-04-01T09:59:10.381095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optimizer = SGD(learning_rate=0.01, momentum=0.9)\nbranch_model.compile(optimizer=optimizer, loss='mean_squared_error')","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:59:10.383142Z","iopub.execute_input":"2024-04-01T09:59:10.383419Z","iopub.status.idle":"2024-04-01T09:59:10.399629Z","shell.execute_reply.started":"2024-04-01T09:59:10.383389Z","shell.execute_reply":"2024-04-01T09:59:10.398621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 32","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:59:10.400803Z","iopub.execute_input":"2024-04-01T09:59:10.401071Z","iopub.status.idle":"2024-04-01T09:59:10.404928Z","shell.execute_reply.started":"2024-04-01T09:59:10.401049Z","shell.execute_reply":"2024-04-01T09:59:10.403961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_paths = train_df[\"image_path\"].tolist()[:batch_size]","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:59:10.406014Z","iopub.execute_input":"2024-04-01T09:59:10.406264Z","iopub.status.idle":"2024-04-01T09:59:10.415763Z","shell.execute_reply.started":"2024-04-01T09:59:10.406244Z","shell.execute_reply":"2024-04-01T09:59:10.414815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_image_paths = validation_df[\"image_path\"].tolist()\nvalidation_steps = len(validation_image_paths) // batch_size","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:59:10.41694Z","iopub.execute_input":"2024-04-01T09:59:10.417279Z","iopub.status.idle":"2024-04-01T09:59:10.426541Z","shell.execute_reply.started":"2024-04-01T09:59:10.417237Z","shell.execute_reply":"2024-04-01T09:59:10.425684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"steps_per_epoch = len(train_df) // batch_size\nepochs = 3","metadata":{"execution":{"iopub.status.busy":"2024-04-01T09:59:10.427498Z","iopub.execute_input":"2024-04-01T09:59:10.42779Z","iopub.status.idle":"2024-04-01T09:59:10.436141Z","shell.execute_reply.started":"2024-04-01T09:59:10.427768Z","shell.execute_reply":"2024-04-01T09:59:10.435264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_single_image_generator1(image_path_list, batch_size=16):\n    while True:\n        for start in range(0, len(image_path_list), batch_size):\n            X_batch = []\n            Y_batch = []\n            end = min(start + batch_size, len(image_path_list))\n\n            image_info_list = []\n            for image_path in image_path_list[start:end]:\n                try:\n                    image_info_list.append(get_single_sample(image_path, training=True))\n                except openslide.OpenSlideUnsupportedFormatError as e:\n                    print(f\"Ignoring unsupported image file: {image_path}\")\n                    continue\n                    \n            if not image_info_list:\n                continue\n\n            X_batch = np.array([image_info[0]/255. for image_info in image_info_list])\n            Y_batch = np.array([image_info[1] for image_info in image_info_list])\n            \n            yield X_batch, Y_batch","metadata":{"execution":{"iopub.status.busy":"2024-04-01T10:25:13.803646Z","iopub.execute_input":"2024-04-01T10:25:13.804551Z","iopub.status.idle":"2024-04-01T10:25:13.812419Z","shell.execute_reply.started":"2024-04-01T10:25:13.804521Z","shell.execute_reply":"2024-04-01T10:25:13.811425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = branch_model.fit(\n    custom_single_image_generator1(image_paths, batch_size=batch_size),\n    steps_per_epoch=steps_per_epoch,\n    epochs=epochs,\n    validation_data=custom_single_image_generator1(validation_image_paths, batch_size=batch_size),\n    validation_steps=validation_steps\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-01T10:25:20.771351Z","iopub.execute_input":"2024-04-01T10:25:20.771708Z","iopub.status.idle":"2024-04-01T11:58:18.529341Z","shell.execute_reply.started":"2024-04-01T10:25:20.771681Z","shell.execute_reply":"2024-04-01T11:58:18.528417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate_model(model, data_generator, steps):\n    return model.evaluate(data_generator, steps=steps)\n\ndef plot_history(history):\n   \n    plt.plot(history.history['loss'], label='Training Loss')\n    plt.plot(history.history['val_loss'], label='Validation Loss')\n    plt.title('Model Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-01T12:01:30.731137Z","iopub.execute_input":"2024-04-01T12:01:30.731617Z","iopub.status.idle":"2024-04-01T12:01:30.739197Z","shell.execute_reply.started":"2024-04-01T12:01:30.731583Z","shell.execute_reply":"2024-04-01T12:01:30.73812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_generator = custom_single_image_generator1(validation_image_paths, batch_size=batch_size)\nvalidation_steps = len(validation_image_paths) // batch_size\nloss = evaluate_model(branch_model, validation_generator, validation_steps)\nprint(\"Validation Loss:\", loss)","metadata":{"execution":{"iopub.status.busy":"2024-04-01T12:03:20.982591Z","iopub.execute_input":"2024-04-01T12:03:20.983527Z","iopub.status.idle":"2024-04-01T12:13:19.307312Z","shell.execute_reply.started":"2024-04-01T12:03:20.983491Z","shell.execute_reply":"2024-04-01T12:13:19.306134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_history(history)","metadata":{"execution":{"iopub.status.busy":"2024-04-01T12:13:51.55034Z","iopub.execute_input":"2024-04-01T12:13:51.550736Z","iopub.status.idle":"2024-04-01T12:13:51.880901Z","shell.execute_reply.started":"2024-04-01T12:13:51.550704Z","shell.execute_reply":"2024-04-01T12:13:51.879872Z"},"trusted":true},"execution_count":null,"outputs":[]}]}