{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":37333,"databundleVersionId":3949526,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\nimport pandas as pd\nimport numpy as np\nimport torch\nimport torchvision\nimport torchvision.transforms as transforms\nfrom torch import nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import models\nfrom PIL import Image\nimport matplotlib.pyplot as plt\n# 设置数据集路径\nDATASET_FOLDER = \"/kaggle/input/mayo-clinic-strip-ai/\"\nDATASET_SMALL_FOLDER = \"/kaggle/input/newtrain\"\n\n# 验证路径是否存在\nif not os.path.exists(DATASET_FOLDER):\n    print(f\"警告: {DATASET_FOLDER} 不存在\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport keras.backend as K #to define custom loss function\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\nimport warnings\nwarnings.filterwarnings('ignore')\n\nfrom pprint import pprint\nfrom collections import defaultdict\nimport openslide\nfrom openslide import OpenSlide\n\nfrom glob import glob\n\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Dropout, Flatten, Dense\nfrom tensorflow.keras.layers import GlobalMaxPooling2D\nfrom keras.models import load_model\n\nprint(keras.__version__)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\n\ntrain_df = pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')\ntest_df  = pd.read_csv('../input/mayo-clinic-strip-ai/test.csv')\n\n# Specify patient_ids to remove\npatient_ids_to_remove = ['006388', '008e5c', '00c058', '01adc5']\n\n# Filter out the rows with specified patient_ids\ntrain_df = train_df[~train_df['patient_id'].isin(patient_ids_to_remove)].reset_index(drop=True)\ntrain_df = train_df.drop_duplicates(subset=['patient_id'])\n\ndf1 = train_df[train_df['label'] == 'CE']\ndf2 = train_df[train_df['label'] == 'LAA']\n#adjust n to change number of CE data\nsampled= df1.sample(n=200, random_state=42)\ntrain_df = pd.concat([sampled, df2],ignore_index=True)\n# Print the cleaned DataFrame\nprint(\"Cleaned DataFrame:\")\nprint(train_df.head())\ntrain_df['label'].value_counts()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['patient_id'].nunique","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_images = glob(\"/kaggle/input/mayo-clinic-strip-ai/train/*\")\ntest_images = glob(\"/kaggle/input/mayo-clinic-strip-ai/test/*\")\nother_images = glob(\"/kaggle/input/mayo-clinic-strip-ai/other/*\")\nprint(f\"Number of images in a training set: {len(train_images)}\")\nprint(f\"Number of images in a training set: {len(test_images)}\")\nprint(f\"Number of other: {len(other_images)}\")\n\n# Filtering out images based on the cleaned patient_ids in train_df\nimages_to_remove = [\n    '/kaggle/input/mayo-clinic-strip-ai/train/006388_0.tif',\n    '/kaggle/input/mayo-clinic-strip-ai/train/008e5c_0.tif',\n    '/kaggle/input/mayo-clinic-strip-ai/train/00c058_0.tif',\n    '/kaggle/input/mayo-clinic-strip-ai/train/01adc5_0.tif',\n]\n\n# Remove images associated with the patient_ids\ntrain_images = [img for img in train_images if img not in images_to_remove]\n\n# Check the total number of images after deletion\ntotal_images_after_deletion = len(train_images)\nprint(\"Total number of images after deletion:\", total_images_after_deletion)\n\n# Print the paths of the cleaned list of images\nprint(\"First 5 image paths after deletion:\")\nprint(train_images[:5])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[\"file_path\"] = train_df[\"image_id\"].apply(lambda x: \"../input/mayo-clinic-strip-ai/train/\" + x + \".tif\")\ntest_df[\"file_path\"]  = test_df[\"image_id\"].apply(lambda x: \"../input/mayo-clinic-strip-ai/test/\" + x + \".tif\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# labelling CE class as 1 and LAA as 0\ntrain_df[\"target\"] = train_df[\"label\"].apply(lambda x : 1 if x==\"CE\" else 0)\ntrain_df.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ntrain_df[\"file_path\"] = train_df[\"image_id\"].apply(lambda x: \"../input/mayo-clinic-strip-ai/train/\" + x + \".tif\")\n\ndef preprocess(image_path):\n    slide=OpenSlide(image_path)\n    region= (2500,2500)    \n    size  = (5000, 5000)\n    image = slide.read_region(region, 0, size)\n    image = image.resize((128, 128))\n    image = np.array(image)    \n    return image\n\nX_train=[]\nfor i in tqdm(train_df['file_path']):\n    x1=preprocess(i)\n    X_train.append(x1)\n\nY_train=[]    \nY_train=train_df['target']\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import train_test_split\n\n# defain xy train\nX_train = np.array(X_train)\nX_train = X_train / 255.0  # 归一化\nY_train = np.array(Y_train)\n\n# train split\nX_train_split, X_val, Y_train_split, Y_val = train_test_split(X_train, Y_train, test_size=0.2, random_state=42)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install tensorflow numpy pandas tqdm","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.applications import EfficientNetB6\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, Dropout, GlobalAveragePooling2D\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Y_train shape:\", Y_train.shape)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Unique values in Y_train:\", np.unique(Y_train))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Unique values in Y_val:\", np.unique(Y_val))\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.utils import to_categorical\n\n# check for one-hot code\ndef ensure_one_hot(labels, num_classes):\n    # \n    if len(labels.shape) == 2 and labels.shape[1] == num_classes:\n        return labels\n    # one-hot code\n    return to_categorical(labels, num_classes=num_classes)\n\n# 假num_classes = 2\nnum_classes = 2\n\n# 确保 Y_train 和 Y_val 是正确的 one-hot 编码形式\nY_train = ensure_one_hot(Y_train, num_classes)\nY_val = ensure_one_hot(Y_val, num_classes)\n\n# \nprint(\"Y_train shape after one-hot encoding:\", Y_train.shape)\nprint(\"Y_val shape after one-hot encoding:\", Y_val.shape)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# agumation（train）\ntrain_datagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\n# val generator\nval_datagen = ImageDataGenerator()\n\n# train generator\ntrain_generator = train_datagen.flow(\n    X_train, \n    Y_train, \n    batch_size=32  # 去掉 class_mode，因为已经有标签\n)\n\n# val （no aug）\nval_generator = val_datagen.flow(\n    X_val, \n    Y_val, \n    batch_size=32  #\n)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetB6\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, Dropout, GlobalAveragePooling2D\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.utils import to_categorical\n\n\n\n# Load the pre-trained model for EfficientNet-B6\nbase_model = EfficientNetB6(weights=None, include_top=False, input_shape=(128, 128, 4))\n\n# Freeze the convolution layer of the pre-trained model\nbase_model.trainable = False\n\n# Add a custom category header\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\nx = Dropout(0.5)(x)\nx = Dense(256, activation='relu')(x)\nx = Dropout(0.5)(x)\npredictions = Dense(Y_train.shape[1], activation='softmax')(x)  # 输出层\n\n# define model\nmodel = Model(inputs=base_model.input, outputs=predictions)\n\n# Compilation model\nmodel.compile(optimizer=Adam(learning_rate=0.001), loss='categorical_crossentropy', metrics=['accuracy'])\n\n# print model\nmodel.summary()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\n\n# Define a callback function\nearly_stopping = EarlyStopping(\n    monitor='val_loss',  #  loss\n    patience=5,          # If the validation set loss does not improve for 5 consecutive times, the training is stopped\n    restore_best_weights=True  # \n)\n\nmodel_checkpoint = ModelCheckpoint(\n    'efficientnet_b6_best_model.keras',  # The path to save the model file\n    save_best_only=True,                 # Save only the models whose validation sets perform best\n    monitor='val_loss'                   # loss\n)\n\n\nhistory = model.fit(\n    train_generator,         # Training data generator\n    validation_data=val_generator,  # Validation data generator\n    epochs=100,               #\n    callbacks=[ model_checkpoint]\n)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# split\nX_train_final, X_test, Y_train_final, Y_test = train_test_split(X_train_split, Y_train_split, test_size=0.2, random_state=42)\n\n\nprint(f\"训练集大小: {X_train_final.shape[0]}\")\nprint(f\"验证集大小: {X_val.shape[0]}\")\nprint(f\"测试集大小: {X_test.shape[0]}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n\n\nbatch_size = 32  \n\ntest_datagen = ImageDataGenerator(rescale=1./255)\ntest_generator = test_datagen.flow(\n    X_test, Y_test,\n    batch_size=batch_size,\n    shuffle=False  \n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Y_test 的形状：\", Y_test.shape)\nprint(\"Y_test 的内容：\", Y_test[:10])  \n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.utils import to_categorical\nimport numpy as np\n\n\n\n\nnum_classes = len(np.unique(Y_test))  \n\n# onehot\nY_test_one_hot = to_categorical(Y_test, num_classes=num_classes)\n\n\nprint(\"独热编码后的标签形状：\", Y_test_one_hot.shape)\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_generator = test_datagen.flow(\n    X_test, Y_test_one_hot,\n    batch_size=batch_size,\n    shuffle=False  \n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\n\n\nbest_model = load_model('efficientnet_b6_best_model.keras')\n\n\ntest_loss, test_accuracy = best_model.evaluate(test_generator)\n\nprint(f\"测试集上的损失: {test_loss}\")\nprint(f\"测试集上的准确率: {test_accuracy}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\nfrom sklearn.metrics import log_loss, f1_score\nimport numpy as np\n\n\nbest_model = load_model('efficientnet_b6_best_model.keras')\n\n\npredictions = best_model.predict(test_generator)\n\n\nY_true = []\nfor i in range(len(test_generator)):\n    _, labels = test_generator[i]\n    Y_true.extend(labels)\n\nY_true = np.array(Y_true)\nY_true = np.argmax(Y_true, axis=1)  # trans back from one-hot\n\n#  Log Loss\nlog_loss_value = log_loss(Y_true, predictions)\nprint(f\"Log Loss: {log_loss_value}\")\n\n# trans percent to class\nY_pred_classes = np.argmax(predictions, axis=1)\n\n# F1 score\nf1 = f1_score(Y_true, Y_pred_classes, average='weighted')\nprint(f\" F1 : {f1}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}