{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":37333,"databundleVersionId":3949526,"sourceType":"competition"},{"sourceId":9866064,"sourceType":"datasetVersion","datasetId":6012343}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Import library\nimport openslide\nfrom openslide import OpenSlide  \nimport pandas as pd\nimport tifffile as tiff\nimport matplotlib.pyplot as plt \nimport numpy as np\nimport tensorflow as tf\nimport os\nfrom PIL import Image\n\nfrom sklearn.model_selection import train_test_split\n\nfrom tensorflow import keras\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import layers\nfrom tensorflow.python.keras.layers import Dense, Flatten\nfrom tensorflow.keras.optimizers import Adam\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-11-14T08:32:43.575591Z","iopub.execute_input":"2024-11-14T08:32:43.576404Z","iopub.status.idle":"2024-11-14T08:32:43.587913Z","shell.execute_reply.started":"2024-11-14T08:32:43.576358Z","shell.execute_reply":"2024-11-14T08:32:43.586874Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Prediction","metadata":{}},{"cell_type":"markdown","source":"## 4.1. Prepare test data","metadata":{}},{"cell_type":"markdown","source":"### 4.1.1 Test Image Reset","metadata":{}},{"cell_type":"code","source":"def clean_test_data_output():\n    output = \"/kaggle/working/\"\n    if os.path.exists(output+\"processed_test_image.csv\"):\n        df_processed_test_image = pd.read_csv(output+\"processed_test_image.csv\")\n        df_processed_test_image.drop(df_processed_test_image.index, inplace=True)\n        print(df_processed_test_image)\n        pd.DataFrame(processed_test_image).to_csv(output_path+\"processed_test_image.csv\")\n        df_newtest = pd.read_csv(output+\"new_test.csv\")\n        df_newtest.drop(df_newtest.index, inplace=True)\n        print(df_newtest)\n        pd.DataFrame(df_newtest).to_csv(output_path+\"new_test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T08:32:43.590163Z","iopub.execute_input":"2024-11-14T08:32:43.590535Z","iopub.status.idle":"2024-11-14T08:32:43.60656Z","shell.execute_reply.started":"2024-11-14T08:32:43.590497Z","shell.execute_reply":"2024-11-14T08:32:43.605277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"clean_test_data_output()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T08:32:43.608205Z","iopub.execute_input":"2024-11-14T08:32:43.608861Z","iopub.status.idle":"2024-11-14T08:32:43.643116Z","shell.execute_reply.started":"2024-11-14T08:32:43.608799Z","shell.execute_reply":"2024-11-14T08:32:43.6418Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 4.1.2. Proces Test Image","metadata":{}},{"cell_type":"markdown","source":"Image  processing functions","metadata":{}},{"cell_type":"code","source":"# Image processing functions for test image\nimport math\nimport gc\nimport random\nimport itertools\ndef show_img(img_path, size):\n    slide = OpenSlide(img_path) # 512 x 512 or 5120 x 5120\n    region = (0, 0)\n    image = slide.read_region(region, 0, size)\n    plt.figure(figsize=(8, 8))\n    plt.imshow(image)\n    plt.show() \ndef show_tiffimg(img):\n    plt.imshow(img, cmap='gray')\n    plt.axis('off')\n    plt.show()\ndef show_tiff(path):\n    tiff_image = tiff.imread(path)\n    plt.imshow(tiff_image, cmap='gray')\n    plt.axis('off')\n    plt.show()    \n\ndef show_tiffpath(img_path):\n    # Read the image from the file path using tifffile\n    img = tiff.imread(img_path)\n    \n    # Display the image\n    plt.imshow(img, cmap='gray')  # Use cmap='gray' if it's a grayscale image; remove it for color\n    plt.axis('off')\n    plt.show()\n    \ndef is_quality_img(img,dim, ratio):\n    n_0 = np.count_nonzero(img == 0)\n    n_255 = np.count_nonzero(img == 255)\n\n    dim_ratio = ratio*dim[0]*dim[1]\n    if n_0 + n_255 > ratio * dim[0] * dim[1] * dim[2]:\n        return False\n    if n_0 > 0.7 * dim[0] * dim[1] * dim[2]:\n        return False\n\n    n_color1 =  np.count_nonzero(img[:,:,0]==np.median(img[:,:,0]))\n    n_color2 =  np.count_nonzero(img[:,:,1]==np.median(img[:,:,1]))\n    n_color3 =  np.count_nonzero(img[:,:,0]==np.median(img[:,:,2]))\n    if n_color1 > dim_ratio and n_color2 > dim_ratio and n_color2 > dim_ratio:\n        return False\n    if n_color1 != 0 and n_color1 != 255 and n_color2 != 0 and n_color2 != 255 and n_color3 != 0 and n_color3 != 255:\n        if (n_0/3 + n_255/3 +  n_color2 > dim_ratio) and (n_0/3 + n_255/3 +  n_color3 > dim_ratio):\n            return False\n    return True\ndef preprocess(path, region, size):\n    slide = OpenSlide(path)\n    image = slide.read_region(region, 0, size).convert('RGB')  # Convert to RGB\n    image = np.array(image)\n    return image\n\ndef test_image_processing(working_item, new_image_dir=\"\",method = \"small_tile\", sample = -1):\n    '''\n        method:\n            small_tile -> get all tiles sequentially upto k sample number\n            max_size -> maxium possible square frame as a tile (aka size = min of 2 dim of original image)\n            sampling -> randomly select tiles as k sampling \n    '''\n    dim = (256,256,3)\n    image_partition = []\n    chunk_size = 2500\n    min_ratio = 0.9\n    new_input = []\n    stepx = chunk_size\n    stepy = chunk_size\n    # Get the image dimensions using OpenSlide\n    slide = OpenSlide(working_item[\"image_path\"])\n    width, height = slide.dimensions\n    print(f\"Image dimensions: {height}x{width}\")\n    if method == \"max_size\":\n        if width >= height:\n            chunk_size = height\n            stepy = chunk_size\n            n_step = math.ceil(width/height)\n            if n_step > 1:\n                stepx=chunk_size-int((n_step*chunk_size - width)/(n_step-1))\n            else:\n                stepx=chunk_size\n        else:\n            chunk_size = width\n            stepx = chunk_size\n            n_step = math.ceil(height/width)\n            stepy=chunk_size-int((n_step*chunk_size - height)/(n_step-1))\n        if height == 0 or width == 0:\n            print(\"Error: Image dimensions are invalid.\")\n    \n    # Process chunks of the image\n    y = list(range(0, height-chunk_size+1, stepy))\n    x = list(range(0, width-chunk_size+1, stepx))\n    combination = list(itertools.product(y, x))\n    if method == \"sample\":\n        random.shuffle(combination)\n    for y, x in combination:\n        # Extract a chunk using OpenSlide\n        print(\"Process tile partition\", x, y)\n        img = preprocess(working_item[\"image_path\"], (x, y), (chunk_size, chunk_size))\n        dim_prod = chunk_size * chunk_size * 3\n        # Check if the image has sufficient information\n        if img is not None and img.size > 0 and is_quality_img(img,(chunk_size,chunk_size,3),min_ratio):# and np.count_nonzero(img) > (1-min_ratio) * dim_prod and np.count_nonzero(img == 255) < min_ratio * dim_prod:\n            img = Image.fromarray(img)\n            img= img.resize((dim[0], dim[1]), Image.LANCZOS)\n            img = np.array(img, dtype=np.uint8)\n            show_tiffimg(img) # to be commented\n            image_partition.append(img)\n        else:\n            print(\"Abort the image as not qualified\")\n        if sample != -1 and len(image_partition) >= sample:\n            break\n    return image_partition","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T08:32:43.645608Z","iopub.execute_input":"2024-11-14T08:32:43.645985Z","iopub.status.idle":"2024-11-14T08:32:43.675548Z","shell.execute_reply.started":"2024-11-14T08:32:43.645944Z","shell.execute_reply":"2024-11-14T08:32:43.674246Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Load the test image for processing\ninput_path = \"../input/mayo-clinic-strip-ai/\"\ntest_df = pd.read_csv(input_path+\"test.csv\")\ntest_df[\"image_path\"] = test_df[\"image_id\"].apply(lambda x: input_path + \"test/\"+x + \".tif\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T08:32:43.677142Z","iopub.execute_input":"2024-11-14T08:32:43.677608Z","iopub.status.idle":"2024-11-14T08:32:43.706322Z","shell.execute_reply.started":"2024-11-14T08:32:43.67756Z","shell.execute_reply":"2024-11-14T08:32:43.705072Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4.2. Process Test Input","metadata":{}},{"cell_type":"markdown","source":"## 4.3. Process Trainded Models","metadata":{}},{"cell_type":"markdown","source":"### Load models","metadata":{}},{"cell_type":"code","source":"output_path = \"/kaggle/working/\"\ninput_path = \"/kaggle/input/mayo-clinic-processed-image-110-220/\"\nresnet = \"resnet_model.h5\"\ninception = \"inception_model.h5\"\nvgg = \"vgg16_model.h5\"\nefficientnet = \"efficientnet_model.h5\"\n\nif os.path.exists(input_path+resnet):\n    resnet_model = tf.keras.models.load_model(input_path+resnet)\n    resnet_ready = True\n    print(\"GET MODEL FROM INPUT\")\nelif os.path.exists(output_path+resnet):\n    resnet_model = tf.keras.models.load_model(output_path+resnet)\n    resnet_ready = True\nif os.path.exists(input_path+inception):\n    inception_model = tf.keras.models.load_model(input_path+inception)\n    inception_ready = True\n    print(\"GET MODEL FROM INPUT\")\nelif os.path.exists(output_path+inception):\n    inception_model = tf.keras.models.load_model(output_path+inception)\n    inception_ready = True\nif os.path.exists(input_path+efficientnet):\n    efficientnet_model = tf.keras.models.load_model(input_path+efficientnet)\n    efficientnet_ready = True\n    print(\"GET MODEL FROM INPUT\")\nelif os.path.exists(output_path+efficientnet):\n    efficientnet_model = tf.keras.models.load_model(output_path+efficientnet)\n    efficientnet_ready = True\nif os.path.exists(input_path+vgg):\n    vgg_model = tf.keras.models.load_model(input_path+vgg)\n    vgg_ready = True\n    print(\"GET MODEL FROM INPUT\")\nelif os.path.exists(output_path+vgg):\n    vgg_model = tf.keras.models.load_model(output_path+vgg)\n    vgg_ready = True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T08:32:43.707824Z","iopub.execute_input":"2024-11-14T08:32:43.708217Z","iopub.status.idle":"2024-11-14T08:32:52.018932Z","shell.execute_reply.started":"2024-11-14T08:32:43.708175Z","shell.execute_reply":"2024-11-14T08:32:52.017862Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4.4. Prediction","metadata":{}},{"cell_type":"code","source":"# Function to check a patient with different models\nimport gc\ndef predict_patient(patient_id, models):\n    chunk_images = []\n    predict = []\n    row = test_df[test_df[\"patient_id\"]==patient_id]\n    chunks = test_image_processing(row.iloc[0],\"\",\"sample\",3)\n    chunks = [chunk/255.0 for chunk in chunks]\n    chunk_images_tensor = tf.convert_to_tensor(chunks)\n    for model in models: \n        predict.append(model.predict(chunk_images_tensor))\n    return predict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T08:32:52.020231Z","iopub.execute_input":"2024-11-14T08:32:52.020552Z","iopub.status.idle":"2024-11-14T08:32:52.028301Z","shell.execute_reply.started":"2024-11-14T08:32:52.020516Z","shell.execute_reply":"2024-11-14T08:32:52.027038Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Prediction for each person with each model","metadata":{}},{"cell_type":"code","source":"# Test the test data with selected models\nimport gc\n# Load the saved models and uncomment when model is availble \n\n# Models and their weight in \nmodels = [inception_model,efficientnet_model,vgg_model] #resnet_model,\n\npatient_ids = test_df['patient_id'].unique()\n\npatient_predictions = {}\n\nfor patient_id in patient_ids:\n    # Get predictions from models\n    patient_predictions[patient_id] = predict_patient(patient_id, models)\n    collected = gc.collect()\n    #print(\"Garbage collector: collected\", \"%d objects.\" % collected)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T08:32:52.029853Z","iopub.execute_input":"2024-11-14T08:32:52.03031Z","iopub.status.idle":"2024-11-14T08:34:05.465388Z","shell.execute_reply.started":"2024-11-14T08:32:52.030259Z","shell.execute_reply":"2024-11-14T08:34:05.464003Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Ensemble Prediction result","metadata":{}},{"cell_type":"code","source":"#Ensemble the result of selected models with weights\nimport numpy\nweights = [0.4,0.3,0.4]\npatient_probs = {}\nfor patient_id in patient_ids:\n\n    predict = patient_predictions[patient_id]\n\n    ensemble_preds = 0\n    for a, w in zip(predict,weights):\n        ensemble_preds += a*w\n\n    # Average probabilities for the patient\n    patient_prob = np.mean(ensemble_preds, axis=0)\n    patient_probs[patient_id] = patient_prob","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T08:34:05.469658Z","iopub.execute_input":"2024-11-14T08:34:05.470215Z","iopub.status.idle":"2024-11-14T08:34:05.479052Z","shell.execute_reply.started":"2024-11-14T08:34:05.470157Z","shell.execute_reply":"2024-11-14T08:34:05.477526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"patient_probs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T08:34:05.480933Z","iopub.execute_input":"2024-11-14T08:34:05.481461Z","iopub.status.idle":"2024-11-14T08:34:05.50667Z","shell.execute_reply.started":"2024-11-14T08:34:05.481407Z","shell.execute_reply":"2024-11-14T08:34:05.505364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create submission DataFrame\n\nsubmission_data = []\n\nfor patient_id, probs in patient_probs.items():\n    submission_data.append([patient_id, probs[0], probs[1]])  # Assuming [CE, LAA] order\n\nsubmission_df = pd.DataFrame(submission_data, columns=['patient_id', 'CE', 'LAA'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T08:34:05.508151Z","iopub.execute_input":"2024-11-14T08:34:05.508521Z","iopub.status.idle":"2024-11-14T08:34:05.525404Z","shell.execute_reply.started":"2024-11-14T08:34:05.508481Z","shell.execute_reply":"2024-11-14T08:34:05.523661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T08:34:05.52704Z","iopub.execute_input":"2024-11-14T08:34:05.527468Z","iopub.status.idle":"2024-11-14T08:34:05.555375Z","shell.execute_reply.started":"2024-11-14T08:34:05.527427Z","shell.execute_reply":"2024-11-14T08:34:05.553767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"output_path = \"/kaggle/working/\"\nsubmission_df.to_csv('submission.csv', index = False)\n!head submission.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T08:34:05.557139Z","iopub.execute_input":"2024-11-14T08:34:05.557563Z","iopub.status.idle":"2024-11-14T08:34:05.566272Z","shell.execute_reply.started":"2024-11-14T08:34:05.557518Z","shell.execute_reply":"2024-11-14T08:34:05.564893Z"}},"outputs":[],"execution_count":null}]}