{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":37333,"databundleVersionId":3949526,"sourceType":"competition"},{"sourceId":9848586,"sourceType":"datasetVersion","datasetId":6011086},{"sourceId":160605,"sourceType":"modelInstanceVersion","modelInstanceId":136565,"modelId":159291}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"colab":{"provenance":[],"gpuType":"T4"},"widgets":{"application/vnd.jupyter.widget-state+json":{"12980315754f48728c7775386c2bf8f9":{"model_module":"@jupyter-widgets/controls","model_name":"VBoxModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"VBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"VBoxView","box_style":"","children":["IPY_MODEL_06a439ec4cae49bf9f7c787ac2d59cf5","IPY_MODEL_b2a09f52686448dd86dd699cc575bddc","IPY_MODEL_4bb9bf2b55da49c1ab34a42ade0e4bfa","IPY_MODEL_ea2f9f8da9594f6a99281a9e527cfd37","IPY_MODEL_d7ca4eebff1b4101a7e254f95280f52d"],"layout":"IPY_MODEL_036cc348112946be83e1906e6b0a70bf"}},"06a439ec4cae49bf9f7c787ac2d59cf5":{"model_module":"@jupyter-widgets/controls","model_name":"HTMLModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_acc67eb8ef85496da98f82feff6c857b","placeholder":"​","style":"IPY_MODEL_953dd2739c724633961a11a505c9d10c","value":"<center> <img\nsrc=https://www.kaggle.com/static/images/site-logo.png\nalt='Kaggle'> <br> Create an API token from <a\nhref=\"https://www.kaggle.com/settings/account\" target=\"_blank\">your Kaggle\nsettings page</a> and paste it below along with your Kaggle username. <br> </center>"}},"b2a09f52686448dd86dd699cc575bddc":{"model_module":"@jupyter-widgets/controls","model_name":"TextModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"TextModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"TextView","continuous_update":true,"description":"Username:","description_tooltip":null,"disabled":false,"layout":"IPY_MODEL_0b410bd4a92549d48a87edcd09cb0c68","placeholder":"​","style":"IPY_MODEL_3acdf55e278d40948c7f2fc4f0cc9f35","value":""}},"4bb9bf2b55da49c1ab34a42ade0e4bfa":{"model_module":"@jupyter-widgets/controls","model_name":"PasswordModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"PasswordModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"PasswordView","continuous_update":true,"description":"Token:","description_tooltip":null,"disabled":false,"layout":"IPY_MODEL_9f249f1770fc40aab0e9becdef7d3224","placeholder":"​","style":"IPY_MODEL_ad2c6f7966e741e287e0c797d105b9b4","value":""}},"ea2f9f8da9594f6a99281a9e527cfd37":{"model_module":"@jupyter-widgets/controls","model_name":"ButtonModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ButtonModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ButtonView","button_style":"","description":"Login","disabled":false,"icon":"","layout":"IPY_MODEL_10b0c6ebb6184bc3b32c0dfeffc4d255","style":"IPY_MODEL_6c7a0aa42ede48989574a4ca2af7ffd2","tooltip":""}},"d7ca4eebff1b4101a7e254f95280f52d":{"model_module":"@jupyter-widgets/controls","model_name":"HTMLModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_e17bc92c8c5844d8b2dec5b79cace5be","placeholder":"​","style":"IPY_MODEL_58ba6682e81443a58a3984610a0574de","value":"\n<b>Thank You</b></center>"}},"036cc348112946be83e1906e6b0a70bf":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":"center","align_self":null,"border":null,"bottom":null,"display":"flex","flex":null,"flex_flow":"column","grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"50%"}},"acc67eb8ef85496da98f82feff6c857b":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"953dd2739c724633961a11a505c9d10c":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"0b410bd4a92549d48a87edcd09cb0c68":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"3acdf55e278d40948c7f2fc4f0cc9f35":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"9f249f1770fc40aab0e9becdef7d3224":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"ad2c6f7966e741e287e0c797d105b9b4":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"10b0c6ebb6184bc3b32c0dfeffc4d255":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"6c7a0aa42ede48989574a4ca2af7ffd2":{"model_module":"@jupyter-widgets/controls","model_name":"ButtonStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ButtonStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","button_color":null,"font_weight":""}},"e17bc92c8c5844d8b2dec5b79cace5be":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"58ba6682e81443a58a3984610a0574de":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}}}},"accelerator":"GPU"},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import openslide\n\nfrom openslide import OpenSlide\n\nimport pandas as pd\n\nimport tifffile as tiff\n\nimport matplotlib.pyplot as plt\n\nimport numpy as np\n\nimport tensorflow as tf\n\nimport os\n\nfrom PIL import Image\n\n\n\nfrom sklearn.model_selection import train_test_split\n\n\n\nfrom tensorflow import keras\n\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom tensorflow.keras import layers\n\nfrom tensorflow.python.keras.layers import Dense, Flatten\n\nfrom tensorflow.keras.optimizers import Adam\n\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-11-09T05:48:07.020669Z","iopub.execute_input":"2024-11-09T05:48:07.021188Z","iopub.status.idle":"2024-11-09T05:48:23.043685Z","shell.execute_reply.started":"2024-11-09T05:48:07.021115Z","shell.execute_reply":"2024-11-09T05:48:23.042553Z"},"trusted":true,"id":"gbhhaeoESe6j"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Data Processing","metadata":{"id":"6NlqC3p6Se6j"}},{"cell_type":"markdown","source":"## 1.1. Data Loading","metadata":{"id":"tx8iHE-BSe6k"}},{"cell_type":"code","source":"#Reading data\n\ninput_path = \"../input/mayo-clinic-strip-ai/\"\n\ntrain_df = pd.read_csv(input_path+\"train.csv\")\n\ntest_df = pd.read_csv(input_path+\"test.csv\")\n\nother_df = pd.read_csv(input_path+\"other.csv\")","metadata":{"trusted":true,"id":"5Xs0wS9_Se6k"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1.2. Data Description and visualization","metadata":{"id":"1tXJDiS5Se6k"}},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"id":"Cqv_7Ai8Se6k"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head(5)","metadata":{"trusted":true,"id":"sOBdK5jGSe6k"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1.3. Data Preprocessing","metadata":{"id":"mjymGxk6Se6k"}},{"cell_type":"code","source":"train_df[\"image_path\"] = train_df[\"image_id\"].apply(lambda x: input_path +\"train/\" + x + \".tif\")","metadata":{"trusted":true,"id":"I3wbHOZfSe6k"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head(5)","metadata":{"trusted":true,"id":"a3gzPfy-Se6l"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1.4. Image preprocessing","metadata":{"id":"kI6TwkLaSe6l"}},{"cell_type":"markdown","source":"### 1.4.1. Image viewers","metadata":{"id":"eedRwd7_Se6l"}},{"cell_type":"code","source":"def show_img(img_path, size):\n\n    slide = OpenSlide(img_path) # 512 x 512 or 5120 x 5120\n\n    region = (0, 0)\n\n    image = slide.read_region(region, 0, size)\n\n    plt.figure(figsize=(8, 8))\n\n    plt.imshow(image)\n\n    plt.show()\n\ndef show_tiffimg(img):\n\n    plt.imshow(img, cmap='gray')\n\n    plt.axis('off')\n\n    plt.show()\n\ndef show_tiff(path):\n\n    tiff_image = tiff.imread(path)\n\n    plt.imshow(tiff_image, cmap='gray')\n\n    plt.axis('off')\n\n    plt.show()\n\n\n\ndef show_tiffpath(img_path):\n\n    # Read the image from the file path using tifffile\n\n    img = tiff.imread(img_path)\n\n\n\n    # Display the image\n\n    plt.imshow(img, cmap='gray')  # Use cmap='gray' if it's a grayscale image; remove it for color\n\n    plt.axis('off')\n\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:48:30.308857Z","iopub.execute_input":"2024-11-09T05:48:30.310461Z","iopub.status.idle":"2024-11-09T05:48:30.319192Z","shell.execute_reply.started":"2024-11-09T05:48:30.310381Z","shell.execute_reply":"2024-11-09T05:48:30.317794Z"},"id":"0CbIpwyjSe6l"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1.4.2. Image tiling","metadata":{"id":"hxWDFKtWSe6l"}},{"cell_type":"code","source":"import math\n\ndef is_quality_img(img,dim, ratio):\n\n    return ((np.count_nonzero(img[:,:,0]==np.median(img[:,:,0])) < ratio * dim[0] * dim[1])\n\n            and (np.count_nonzero(img[:,:,1]==np.median(img[:,:,1])) < ratio * dim[0] * dim[1])\n\n            and (np.count_nonzero(img[:,:,2]==np.median(img[:,:,2])) < ratio * dim[0] * dim[1]))\n\n    #return True\n\ndef preprocess(path, region, size):\n\n    slide = OpenSlide(path)\n\n    image = slide.read_region(region, 0, size).convert('RGB')  # Convert to RGB\n\n    image = np.array(image)\n\n    return image\n\n\n\ndef image_processing(working_item, new_image_dir=\"\"):\n\n    dim = (256,256,3)\n\n    image_partition = []\n\n    chunk_size = 5000\n\n    min_ratio = 0.8\n\n    new_input = []\n\n    stepx = chunk_size\n\n    stepy = chunk_size\n\n    # Get the image dimensions using OpenSlide\n\n    slide = OpenSlide(working_item[\"image_path\"])\n\n    width, height = slide.dimensions\n\n    print(f\"Image dimensions: {height}x{width}\")\n\n    if width >= height:\n\n        chunk_size = height\n\n        stepy = chunk_size\n\n        n_step = math.ceil(width/height)\n\n        if n_step > 1:\n\n            stepx=chunk_size-int((n_step*chunk_size - width)/(n_step-1))\n\n        else:\n\n            stepx=chunk_size\n\n    else:\n\n        chunk_size = width\n\n        stepx = chunk_size\n\n        n_step = math.ceil(height/width)\n\n        stepy=chunk_size-int((n_step*chunk_size - height)/(n_step-1))\n\n    if height == 0 or width == 0:\n\n        print(\"Error: Image dimensions are invalid.\")\n\n        return []\n\n\n\n    # Process chunks of the image\n\n    for y in range(0, height-chunk_size+1, stepy):\n\n        for x in range(0, width-chunk_size+1, stepx):\n\n            # Extract a chunk using OpenSlide\n\n            print(\"Process tile partition\", x, y)\n\n            img = preprocess(working_item[\"image_path\"], (x, y), (chunk_size, chunk_size))\n\n            dim_prod = chunk_size * chunk_size * 3\n\n            # Check if the image has sufficient information\n\n            if img is not None and img.size > 0 and np.count_nonzero(img) > (1-min_ratio) * dim_prod and np.count_nonzero(img == 255) < min_ratio * dim_prod:\n\n                img = Image.fromarray(img)\n\n                img= img.resize((dim[0], dim[1]), Image.LANCZOS)\n\n                img = np.array(img, dtype=np.uint8)\n\n                show_tiffimg(img) # to be commented\n\n                if is_quality_img(img,dim,0.99):\n\n                    image_partition.append(img)\n\n                else:\n\n                    print(\"Abort the image as not qualified\")\n\n\n\n    num_image = len(image_partition)\n\n    for i, img in enumerate(image_partition):\n\n        # Confirm the validity of the image before saving\n\n        if img is None or img.size == 0:\n\n            print(f\"Error: Skipping empty or None image at index {i}\")\n\n            continue\n\n\n\n        # Collect metadata for each image\n\n        new_input.append(\n\n            {\n\n                'image_id': working_item[\"image_id\"] + \"_\" + str(i),\n\n                'center_id': working_item[\"center_id\"],\n\n                'patient_id': working_item[\"patient_id\"],\n\n                'image_num': num_image,\n\n                'label': working_item[\"label\"],\n\n                'image_path': working_item[\"image_id\"] + \"_\" + str(i) + \".tif\"\n\n            }\n\n        )\n\n        # Ensure the output directory exists\n\n        if not os.path.exists(new_image_dir):\n\n            os.makedirs(new_image_dir)\n\n\n\n        # Save the image\n\n        tiff.imwrite(\n\n            new_image_dir + working_item[\"image_id\"] + \"_\" + str(i) + \".tif\",\n\n            img.astype(np.uint8)  # Ensure the image is saved in 8-bit format\n\n        )\n\n\n\n    return new_input","metadata":{"id":"p9F6BVuY4znl","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Process the images, only run once","metadata":{"id":"PzuWyXXZSe6l"}},{"cell_type":"code","source":"#imgs = [\"0ba49d_0\",\"05a1ec_0\"]\n\n#df_sample = train_df.loc[[i for i, img_id in enumerate(train_df[\"image_id\"]) if img_id in imgs]]\n\n#df_sample = train_df","metadata":{"trusted":true,"id":"dBMqheo7Se6l"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\n\n%%time\n\nimport gc\n\nnew_input = []\n\nprocessed_image = []\n\noutput_path = \"/kaggle/working/\"\n\nprocessed_img_dir = output_path+\"processed_image/\"\n\nfor index, row in df_sample.iterrows():\n\n    print(\"Image \",str(index),row[\"image_id\"])\n\n    new_input.extend(image_processing(row,new_image_dir=processed_img_dir))\n\n    processed_image.append(row[\"image_id\"])\n\n    collected = gc.collect()\n\n    print(\"Garbage collector: collected\", \"%d objects.\" % collected)\n\npd.DataFrame(new_input).to_csv(output_path+\"new_input.csv\")\n\npd.DataFrame(processed_image).to_csv(output_path+\"processed_image.csv\")\n\n\"\"\"","metadata":{"trusted":true,"id":"5gkrwDkkSe6l"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Training Data Preparation","metadata":{"id":"zw5Y5ZJISe6l"}},{"cell_type":"markdown","source":"## 2.1. Load processed images","metadata":{"id":"qavS61eCSe6l"}},{"cell_type":"code","source":"folder1 = \"/kaggle/input/testing/images-0-110\"\nfolder2 = \"/kaggle/input/testing/images-111-220\"\nfolder3 = \"/kaggle/input/testing/images-221-330\"\nfolder4 = \"/kaggle/input/testing/images-331-440\"\nfolder5 = \"/kaggle/input/testing/images-441-550\"\nfolder6 = \"/kaggle/input/testing/images-551-660\"\nfolder7 = \"/kaggle/input/testing/images-661-754\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:48:46.719158Z","iopub.execute_input":"2024-11-09T05:48:46.719606Z","iopub.status.idle":"2024-11-09T05:48:46.725306Z","shell.execute_reply.started":"2024-11-09T05:48:46.719567Z","shell.execute_reply":"2024-11-09T05:48:46.724078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df1 = pd.read_csv(folder1+\"/new_input.csv\")\ntrain_df2 = pd.read_csv(folder2+\"/new_input.csv\")\ntrain_df3 = pd.read_csv(folder3+\"/new_input.csv\")\ntrain_df4 = pd.read_csv(folder4+\"/new_input.csv\")\ntrain_df5 = pd.read_csv(folder5+\"/new_input.csv\")\ntrain_df6 = pd.read_csv(folder6+\"/new_input.csv\")\ntrain_df7 = pd.read_csv(folder7+\"/new_input.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:48:49.09733Z","iopub.execute_input":"2024-11-09T05:48:49.098033Z","iopub.status.idle":"2024-11-09T05:48:49.161451Z","shell.execute_reply.started":"2024-11-09T05:48:49.097882Z","shell.execute_reply":"2024-11-09T05:48:49.160169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df1[\"image_path\"] = train_df1[\"image_id\"].apply(lambda x: folder1+\"/processed_image/\" + x + \".tif\")\ntrain_df2[\"image_path\"] = train_df2[\"image_id\"].apply(lambda x: folder2+\"/processed_image/\" + x + \".tif\")\ntrain_df3[\"image_path\"] = train_df3[\"image_id\"].apply(lambda x: folder3+\"/processed_image/\" + x + \".tif\")\ntrain_df4[\"image_path\"] = train_df4[\"image_id\"].apply(lambda x: folder4+\"/processed_image/\" + x + \".tif\")\ntrain_df5[\"image_path\"] = train_df5[\"image_id\"].apply(lambda x: folder5+\"/processed_image/\" + x + \".tif\")\ntrain_df6[\"image_path\"] = train_df6[\"image_id\"].apply(lambda x: folder6+\"/processed_image/\" + x + \".tif\")\ntrain_df7[\"image_path\"] = train_df7[\"image_id\"].apply(lambda x: folder7+\"/processed_image/\" + x + \".tif\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:48:50.515774Z","iopub.execute_input":"2024-11-09T05:48:50.516174Z","iopub.status.idle":"2024-11-09T05:48:50.53627Z","shell.execute_reply.started":"2024-11-09T05:48:50.51614Z","shell.execute_reply":"2024-11-09T05:48:50.534876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dfs = [train_df1, train_df2, train_df3, train_df4, train_df5, train_df6, train_df7]\ntrain_df = pd.concat(dfs, ignore_index=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:48:51.961983Z","iopub.execute_input":"2024-11-09T05:48:51.962432Z","iopub.status.idle":"2024-11-09T05:48:51.970701Z","shell.execute_reply.started":"2024-11-09T05:48:51.962374Z","shell.execute_reply":"2024-11-09T05:48:51.969491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:48:55.048561Z","iopub.execute_input":"2024-11-09T05:48:55.048973Z","iopub.status.idle":"2024-11-09T05:48:55.077464Z","shell.execute_reply.started":"2024-11-09T05:48:55.048934Z","shell.execute_reply":"2024-11-09T05:48:55.076108Z"},"id":"211uX33mSe6m","outputId":"0c8ef2a2-a925-431e-91fc-b9f461f478f9"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2.2. Check Data","metadata":{"id":"-aoOtfDiSe6m"}},{"cell_type":"code","source":"print(train_df.shape)\n\ntrain_df['image_id'].nunique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:48:59.595938Z","iopub.execute_input":"2024-11-09T05:48:59.596478Z","iopub.status.idle":"2024-11-09T05:48:59.610635Z","shell.execute_reply.started":"2024-11-09T05:48:59.596434Z","shell.execute_reply":"2024-11-09T05:48:59.609183Z"},"id":"VZMX4m8bSe6m","outputId":"ebf763a8-0fd8-4082-a221-1c690961df8c"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Unique Values for: ')\n\ntrain_df.nunique()\n\n\n\nprint('\\n')\n\n\n\nnu = train_df.nunique().reset_index()\n\nnu.columns = ['feature','nunique']\n\nplt.figure(figsize=(12,4))\n\nax = sns.barplot(x='feature', y='nunique', data=nu)\n\nax.bar_label(ax.containers[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:49:01.153524Z","iopub.execute_input":"2024-11-09T05:49:01.153963Z","iopub.status.idle":"2024-11-09T05:49:01.492715Z","shell.execute_reply.started":"2024-11-09T05:49:01.153921Z","shell.execute_reply":"2024-11-09T05:49:01.491352Z"},"id":"k1pFBZPLSe6m","outputId":"f33455e5-a507-4725-d142-6fbb4039550d"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['label'].value_counts()\n\nprint('\\n')\n\nsns.countplot(data=train_df, x='label')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:49:50.191558Z","iopub.execute_input":"2024-11-09T05:49:50.192021Z","iopub.status.idle":"2024-11-09T05:49:50.36551Z","shell.execute_reply.started":"2024-11-09T05:49:50.191959Z","shell.execute_reply":"2024-11-09T05:49:50.364336Z"},"id":"o4sZ42jPSe6m","outputId":"4d8ee2db-434f-4015-d342-cc284b8ce72c"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"CE class covers most of the data so we have class imbalance.","metadata":{"id":"RQd7lvk1Se6m"}},{"cell_type":"code","source":"show_tiff(train_df.iloc[44]['image_path'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:49:53.61028Z","iopub.execute_input":"2024-11-09T05:49:53.610826Z","iopub.status.idle":"2024-11-09T05:49:53.734785Z","shell.execute_reply.started":"2024-11-09T05:49:53.610774Z","shell.execute_reply":"2024-11-09T05:49:53.733449Z"},"id":"vZW67XPnSe6m","outputId":"1fd2429b-2e0a-49dd-f9ee-8f001b2bd90f"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2.3. Prepare Train Data","metadata":{"id":"J1NASrxFSe6m"}},{"cell_type":"code","source":"# Include both x and y\n\ntrain, val = train_test_split(\n\n    train_df,\n\n    test_size=0.2,       # 20% for validation; adjust as needed\n\n    stratify=train_df['label'],  # maintain class distribution with stratify\n\n    random_state=42\n\n)\n\n\n\n\n\nprint(f\"train type: {type(train)}\")\n\nprint(f\"val type: {type(val)}\")\n\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:49:56.937528Z","iopub.execute_input":"2024-11-09T05:49:56.938588Z","iopub.status.idle":"2024-11-09T05:49:56.961519Z","shell.execute_reply.started":"2024-11-09T05:49:56.938543Z","shell.execute_reply":"2024-11-09T05:49:56.960335Z"},"id":"oe4Ku-QVSe6m","outputId":"088029e3-ac85-4b04-8e59-c553f1fb840e"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE= 256\n\n\n\nimage_shape = (IMG_SIZE, IMG_SIZE)\n\n\n\ntrain_datagen = ImageDataGenerator(\n\n                            rescale=1./255,  # Keep rescaling\n\n                            rotation_range=30,\n\n                            horizontal_flip=True,\n\n                            vertical_flip=True,\n\n                            zoom_range=0.2,\n\n                            width_shift_range=0.2,\n\n                            height_shift_range=0.2,\n\n                            brightness_range=[0.8, 1.2])\n\n\n\ntest_datagen = ImageDataGenerator(rescale=1./255)\n\n\n\ntrain_gen = train_datagen.flow_from_dataframe(\n\n                         train,\n\n\n\n                         x_col = 'image_path',\n\n                         y_col = 'label',\n\n                         target_size=image_shape,\n\n                         class_mode = 'sparse', # Automatically converts labels to integers\n\n                         color_mode = 'rgb',\n\n                         shuffle=True,\n\n                         batch_size=16,\n\n                         seed=19,\n\n                         )\n\nval_gen = test_datagen.flow_from_dataframe(\n\n                         val,\n\n\n\n                         x_col = 'image_path',\n\n                         y_col = 'label',\n\n                         target_size=image_shape,\n\n                         class_mode = 'sparse',\n\n                         color_mode = 'rgb',\n\n                         shuffle=True,\n\n                         batch_size=16,\n\n                         seed=19,\n\n                         )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:49:59.458234Z","iopub.execute_input":"2024-11-09T05:49:59.459051Z","iopub.status.idle":"2024-11-09T05:50:01.873171Z","shell.execute_reply.started":"2024-11-09T05:49:59.459006Z","shell.execute_reply":"2024-11-09T05:50:01.871975Z"},"id":"CB1YpukySe6n","outputId":"4929c5fc-5b80-41bb-c9e3-3fb1dbfaf98d"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_gen.class_indices)","metadata":{"id":"C8LHbASsAtau","outputId":"8ba771c1-d4b4-4ea3-aaa5-a8888745272b","trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:50:03.625374Z","iopub.execute_input":"2024-11-09T05:50:03.625821Z","iopub.status.idle":"2024-11-09T05:50:03.631706Z","shell.execute_reply.started":"2024-11-09T05:50:03.625773Z","shell.execute_reply":"2024-11-09T05:50:03.630447Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Model Training","metadata":{"id":"JJXYpqhaSe6n"}},{"cell_type":"code","source":"import warnings\n\nimport os\n\n\n\n# Suppress Python warnings\n\nwarnings.filterwarnings(\"ignore\")\n\n\n\n# Suppress TensorFlow warnings\n\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'  # '3' hides all TensorFlow messages\n\n\n\n# Re-import TensorFlow to apply changes\n\ntf.get_logger().setLevel('ERROR')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:50:07.308932Z","iopub.execute_input":"2024-11-09T05:50:07.310054Z","iopub.status.idle":"2024-11-09T05:50:07.315474Z","shell.execute_reply.started":"2024-11-09T05:50:07.309915Z","shell.execute_reply":"2024-11-09T05:50:07.314162Z"},"id":"v0lxTZO4Se6n"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras import models\n\nfrom tensorflow.keras import layers, Model\n\nimport tensorflow as tf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:50:09.024389Z","iopub.execute_input":"2024-11-09T05:50:09.024825Z","iopub.status.idle":"2024-11-09T05:50:09.030191Z","shell.execute_reply.started":"2024-11-09T05:50:09.024789Z","shell.execute_reply":"2024-11-09T05:50:09.029079Z"},"id":"hYT5RTFtSe6n"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"reduce_lr = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.2, verbose=1,mode='min',patience=3, min_lr=1E-5)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:50:11.205825Z","iopub.execute_input":"2024-11-09T05:50:11.206265Z","iopub.status.idle":"2024-11-09T05:50:11.212657Z","shell.execute_reply.started":"2024-11-09T05:50:11.206215Z","shell.execute_reply":"2024-11-09T05:50:11.21123Z"},"id":"EMMMwNsTSe6o"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#model = tf.keras.models.load_model('path/to/model.h5')","metadata":{"trusted":true,"id":"496LwhRmSe6n"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3.1. Resnet 50 with Batch Normalization and Dropout\n\n### Unfreezed last 2 layers","metadata":{"id":"LVyk-y4zSe6n"}},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint\n\n\n\ncheckpoint = ModelCheckpoint(\n\n    filepath='res_net.keras',  # Path to save the best model\n\n    monitor='val_accuracy',   # Metric to monitor (e.g., validation accuracy)\n\n    save_best_only=True,      # Save only the best model\n\n    mode='max',                # 'max' for accuracy, 'min' for loss\n\n    verbose=1                 # Print messages when a checkpoint is saved\n\n)","metadata":{"id":"oo48XFGCzn1K","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = 256\n\n\n\n# Load ResNet50 as the base model\n\nresnet_model = tf.keras.applications.ResNet50(\n\n    include_top=False,\n\n    input_shape=(IMG_SIZE, IMG_SIZE, 3),\n\n    pooling='avg',\n\n    weights='imagenet'\n\n)\n\n\n\n# Freeze most of the base model layers, but unfreeze the last few\n\nfor layer in resnet_model.layers[:-2]:  # Unfreeze the last 2 layers\n\n    layer.trainable = False\n\n\n\n# Build the model using the Functional API\n\ninputs = resnet_model.input\n\nx = resnet_model.output\n\n\n\n# Add a dense layer with Batch Normalization and Dropout\n\nx = layers.Dense(256)(x)                    # Dense layer without activation\n\nx = layers.BatchNormalization()(x)           # Batch Normalization before activation\n\nx = layers.Activation('relu')(x)             # Activation after Batch Normalization\n\nx = layers.Dropout(0.5)(x)\n\n\n\n# Final output layer\n\noutputs = layers.Dense(2, activation='softmax')(x)\n\n\n\n# Create the final model\n\nmodel = Model(inputs=inputs, outputs=outputs)\n\n\n\n# Print the summary\n\nmodel.summary()\n","metadata":{"trusted":true,"scrolled":true,"id":"nxiyUIUNSe6n","outputId":"ab2c65a8-24c8-4d91-9335-559039435706"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(optimizer=Adam(learning_rate=0.001),loss='sparse_categorical_crossentropy',metrics=['accuracy']) # Assumes from_logits=False by default\n\n\n\nhistory = model.fit(train_gen, epochs=50, validation_data=val_gen, callbacks=[reduce_lr,checkpoint],verbose = 1)\n\n\n\nprint(\"Model training complete!\")\n","metadata":{"trusted":true,"id":"cU7jiEZ9Se6n","outputId":"884b0c74-4c28-46e8-c352-142bad73ae60"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Inception Model","metadata":{"id":"FGeYrwTF1aKi"}},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint\n\n\n\ncheckpoint = ModelCheckpoint(\n\n    filepath='inception_model.keras',  # Path to save the best model\n\n    monitor='val_accuracy',   # Metric to monitor (e.g., validation accuracy)\n\n    save_best_only=True,      # Save only the best model\n\n    mode='max',                # 'max' for accuracy, 'min' for loss\n\n    verbose=1                 # Print messages when a checkpoint is saved\n\n)","metadata":{"id":"Oo7adgxpTwiz","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = 256  # Image input size\n\n\n\n# Load InceptionV3 as the base model\n\ninception_model = tf.keras.applications.InceptionV3(\n\n    include_top=False,            # Exclude the top layers (fully connected)\n\n    input_shape=(IMG_SIZE, IMG_SIZE, 3),\n\n    pooling='avg',                # Global average pooling for a 1D output\n\n    weights='imagenet'            # Load pretrained weights\n\n)\n\n\n\n# Freeze most of the base model layers, but unfreeze the last few\n\nfor layer in inception_model.layers[:-2]:  # Unfreeze the last 2 layers\n\n    layer.trainable = False\n\n\n\n# Build the model using the Functional API\n\ninputs = inception_model.input\n\nx = inception_model.output\n\n\n\n\n\nx = layers.Dense(256)(x)\n\nx = layers.BatchNormalization()(x)\n\nx = layers.Activation('relu')(x)\n\nx = layers.Dropout(0.5)(x)\n\n\n\n# Final output layer for binary classification\n\noutputs = layers.Dense(2, activation='softmax')(x)\n\n\n\n# Create the final model\n\nmodel = Model(inputs=inputs, outputs=outputs)\n\n\n\n# Print the summary\n\nmodel.summary()","metadata":{"trusted":true,"scrolled":true,"id":"mALLiLFrSe6n","outputId":"2bfb8400-f19e-44ca-dd6c-edcb85f59a4f"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(optimizer=Adam(learning_rate=0.0001),loss='sparse_categorical_crossentropy',metrics=['accuracy']) # Assumes from_logits=False by default\n\n\n\nhistory = model.fit(train_gen, epochs=30, shuffle=True, validation_data=val_gen, callbacks=[reduce_lr,checkpoint],verbose = 1)\n\n\n\nprint(\"Model training complete!\")","metadata":{"trusted":true,"id":"MFqdqge4Se6n","outputId":"1bd73da4-a492-4900-b9b7-96ef5243c819"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EfficientNet Model","metadata":{"id":"liY8XLWD1iDD"}},{"cell_type":"code","source":"checkpoint = ModelCheckpoint(\n\n    filepath='efficientnet_model.keras',  # Path to save the best model\n\n    monitor='val_accuracy',   # Metric to monitor (e.g., validation accuracy)\n\n    save_best_only=True,      # Save only the best model\n\n    mode='max',                # 'max' for accuracy, 'min' for loss\n\n    verbose=1                 # Print messages when a checkpoint is saved\n\n)","metadata":{"id":"zbwTYj0dT3Sf","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = 256  # Define the input image size\n\n\n\n# Load EfficientNetB0 as the base model\n\nefficientnet_model = tf.keras.applications.EfficientNetB0(\n\n    include_top=False,            # Exclude the top layers (fully connected)\n\n    input_shape=(IMG_SIZE, IMG_SIZE, 3),\n\n    pooling='avg',                # Global average pooling for a 1D output\n\n    weights='imagenet'            # Load pretrained weights\n\n)\n\n\n\n# Freeze most of the base model layers, but unfreeze the last few\n\nfor layer in efficientnet_model.layers:\n\n    layer.trainable = False\n\n\n\n# Build the model using the Functional API\n\n# Use efficientnet_model.input and efficientnet_model.output directly\n\nx = efficientnet_model.output\n\n\n\n# Add a dense layer with Batch Normalization and Dropout\n\nx = layers.Dense(256)(x)                     # Dense layer without activation\n\nx = layers.BatchNormalization()(x)            # Batch Normalization before activation\n\nx = layers.Activation('relu')(x)              # Activation after Batch Normalization\n\nx = layers.Dropout(0.5)(x)\n\n\n\n\n\n# Final output layer for binary classification\n\noutputs = layers.Dense(2, activation='softmax')(x)\n\n\n\n# Create the final model\n\nmodel = Model(inputs=efficientnet_model.input, outputs=outputs) # Use efficientnet_model.input here\n\n\n\n# Print the model summary\n\nmodel.summary()","metadata":{"id":"qngkbIp_W6Cl","outputId":"50bfc846-685f-4238-8a1e-a715666f8704","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(optimizer=Adam(learning_rate=0.0001),loss='sparse_categorical_crossentropy',metrics=['accuracy']) # Assumes from_logits=False by default\n\n\n\nhistory = model.fit(train_gen, epochs=30, shuffle=True, validation_data=val_gen, callbacks=[reduce_lr,checkpoint],verbose = 1)\n\n\n\nprint(\"Model training complete!\")\n\n\n\nmodel.save('efficientnet_model.h5')","metadata":{"trusted":true,"id":"FAhFVDbeSe6o","outputId":"b2d55b15-a78a-4588-cc4f-4a6cb7fbaa39"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## VGG16 Model","metadata":{"id":"Uk1XNbxn1mmY"}},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint\n\n\n\ncheckpoint = ModelCheckpoint(\n\n    filepath='vgg16_net.keras',  # Path to save the best model\n\n    monitor='val_accuracy',   # Metric to monitor (e.g., validation accuracy)\n\n    save_best_only=True,      # Save only the best model\n\n    mode='max',                # 'max' for accuracy, 'min' for loss\n\n    verbose=1                 # Print messages when a checkpoint is saved\n\n)","metadata":{"id":"vw1T7lATT-5G","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\nfrom tensorflow.keras import layers, Model\n\nfrom tensorflow.keras.applications import VGG16\n\nfrom tensorflow.keras.optimizers import Adam\n\n\n\n# Set image size\n\nIMG_SIZE = 256\n\n\n\n# Load the VGG16 model, excluding the top (fully connected) layers\n\nvgg16_base = VGG16(\n\n    include_top=False,              # Exclude the top dense layers\n\n    input_shape=(IMG_SIZE, IMG_SIZE, 3),\n\n    weights='imagenet',             # Use pretrained ImageNet weights\n\n    pooling='avg'                   # Apply global average pooling to reduce output dimensions\n\n)\n\n\n\n# Freeze the base model layers (optional, depending on whether you want to fine-tune)\n\nfor layer in vgg16_base.layers:\n\n    layer.trainable = False\n\n\n\n# Add custom layers on top of VGG16\n\ninputs = vgg16_base.input\n\nx = vgg16_base.output\n\n\n\n# Add dense layers for classification\n\nx = layers.Dense(256, activation='relu')(x)\n\nx = layers.BatchNormalization()(x)\n\nx = layers.Dropout(0.5)(x)\n\n\n\nx = layers.Dense(128, activation='relu')(x)\n\nx = layers.BatchNormalization()(x)\n\nx = layers.Dropout(0.3)(x)\n\n\n\n# Final output layer (for binary classification, use 2 units; for multi-class, adjust as needed)\n\noutputs = layers.Dense(2, activation='softmax')(x)  # using softmax as we need weighted multi-class logarithmic loss\n\n\n\n# Create the final model\n\nmodel = Model(inputs=inputs, outputs=outputs)\n\n\n\n# Compile the model\n\nmodel.compile(\n\n    optimizer=Adam(learning_rate=0.001),\n\n    loss='sparse_categorical_crossentropy',  # Use categorical cross-entropy for multi-class or binary cross-entropy if using sigmoid\n\n    metrics=['accuracy']\n\n)\n\n\n\n# Model summary\n\nmodel.summary()\n","metadata":{"trusted":true,"id":"Bk1fXjlpSe6o"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(optimizer=Adam(learning_rate=0.0001),loss='sparse_categorical_crossentropy',metrics=['accuracy']) # Assumes from_logits=False by default\n\n\n\nhistory = model.fit(train_gen, epochs=10, shuffle=True, validation_data=val_gen, callbacks=[reduce_lr,checkpoint],verbose = 1)\n\n\n\nprint(\"Model training complete!\")\n\n\n\nmodel.save('vgg16_model.h5')","metadata":{"id":"qrnBOQBk10Qw","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Test Data Evaluation and Submission","metadata":{"id":"Hy68klMw1tjM"}},{"cell_type":"code","source":"import math\ndef is_quality_img(img,dim, ratio):\n    return ((np.count_nonzero(img[:,:,0]==np.median(img[:,:,0])) < ratio * dim[0] * dim[1]) \n            and (np.count_nonzero(img[:,:,1]==np.median(img[:,:,1])) < ratio * dim[0] * dim[1])\n            and (np.count_nonzero(img[:,:,2]==np.median(img[:,:,2])) < ratio * dim[0] * dim[1]))\n    #return True\ndef preprocess(path, region, size):\n    slide = OpenSlide(path)\n    image = slide.read_region(region, 0, size).convert('RGB')  # Convert to RGB\n    image = np.array(image)\n    return image\n\ndef test_image_processing(working_item, new_image_dir=\"\"):\n    dim = (256,256,3)\n    image_partition = []\n    chunk_size = 5000\n    min_ratio = 0.8\n    new_input = []\n    stepx = chunk_size\n    stepy = chunk_size\n    # Get the image dimensions using OpenSlide\n    slide = OpenSlide(working_item[\"image_path\"])\n    width, height = slide.dimensions\n    print(f\"Image dimensions: {height}x{width}\")\n    if width >= height:\n        chunk_size = height\n        stepy = chunk_size\n        n_step = math.ceil(width/height)\n        if n_step > 1:\n            stepx=chunk_size-int((n_step*chunk_size - width)/(n_step-1))\n        else:\n            stepx=chunk_size\n    else:\n        chunk_size = width\n        stepx = chunk_size\n        n_step = math.ceil(height/width)\n        stepy=chunk_size-int((n_step*chunk_size - height)/(n_step-1))\n    if height == 0 or width == 0:\n        print(\"Error: Image dimensions are invalid.\")\n        return []\n\n    # Process chunks of the image\n    for y in range(0, height-chunk_size+1, stepy):\n        for x in range(0, width-chunk_size+1, stepx):\n            # Extract a chunk using OpenSlide\n            print(\"Process tile partition\", x, y)\n            img = preprocess(working_item[\"image_path\"], (x, y), (chunk_size, chunk_size))\n            dim_prod = chunk_size * chunk_size * 3\n            # Check if the image has sufficient information\n            if img is not None and img.size > 0 and np.count_nonzero(img) > (1-min_ratio) * dim_prod and np.count_nonzero(img == 255) < min_ratio * dim_prod:\n                img = Image.fromarray(img)\n                img= img.resize((dim[0], dim[1]), Image.LANCZOS)\n                img = np.array(img, dtype=np.uint8)\n                show_tiffimg(img) # to be commented\n                if is_quality_img(img,dim,0.99):\n                    image_partition.append(img)\n                else:\n                    print(\"Abort the image as not qualified\")\n            else:\n                print(\"Number of non-zero\",np.count_nonzero(img),\"should larger \",(1-min_ratio) * dim_prod)\n                print(\"Number of 255\",np.count_nonzero(img==255),\"should less than \",min_ratio* dim_prod)\n\n    num_image = len(image_partition)\n    for i, img in enumerate(image_partition):\n        # Confirm the validity of the image before saving\n        if img is None or img.size == 0:\n            print(f\"Error: Skipping empty or None image at index {i}\")\n            continue\n\n        # Collect metadata for each image\n        new_input.append(\n            {\n                'image_id': working_item[\"image_id\"] + \"_\" + str(i),\n                'center_id': working_item[\"center_id\"],\n                'patient_id': working_item[\"patient_id\"],\n                'image_num': num_image,\n                'parentid': working_item[\"image_id\"],\n                'image_path': new_image_dir+working_item[\"image_id\"] + \"_\" + str(i) + \".tif\"\n            }\n        )\n        # Ensure the output directory exists\n        if not os.path.exists(new_image_dir):\n            os.makedirs(new_image_dir)\n\n        # Save the image\n        tiff.imwrite(\n            new_image_dir + working_item[\"image_id\"] + \"_\" + str(i) + \".tif\",\n            img.astype(np.uint8)  # Ensure the image is saved in 8-bit format\n        )\n\n    return new_input, image_partition\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:59:51.793135Z","iopub.execute_input":"2024-11-09T05:59:51.794003Z","iopub.status.idle":"2024-11-09T05:59:51.814354Z","shell.execute_reply.started":"2024-11-09T05:59:51.793957Z","shell.execute_reply":"2024-11-09T05:59:51.813125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nimport gc\n\ntest_data = pd.read_csv(\"/kaggle/input/mayo-clinic-strip-ai/test.csv\")\ntest_path = \"/kaggle/input/mayo-clinic-strip-ai/test/\"\ntest_data[\"image_path\"] = test_data[\"image_id\"].apply(lambda x: test_path + x + \".tif\")\n\nimages = []\nnew_test = []\noutput_path = \"/kaggle/working/\"\nprocessed_img_dir = output_path+\"processed_test_image/\"\nfor index, row in test_data.iterrows():\n    print(\"Image \",str(index),row[\"image_id\"])\n    new_test_img,image_partition = test_image_processing(row,new_image_dir=processed_img_dir)\n    new_test.extend(new_test_img)\n    images.extend(image_partition)\n    collected = gc.collect()\n    print(\"Garbage collector: collected\", \"%d objects.\" % collected)\npd.DataFrame(new_test).to_csv(output_path+\"new_test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T05:59:55.212648Z","iopub.execute_input":"2024-11-09T05:59:55.213066Z","iopub.status.idle":"2024-11-09T06:14:39.863874Z","shell.execute_reply.started":"2024-11-09T05:59:55.213029Z","shell.execute_reply":"2024-11-09T06:14:39.862431Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## enter path of test data\noutput_path = \"/kaggle/working/\"\ntest_data = pd.read_csv(output_path + \"new_test.csv\")\ntest_img_path = output_path + \"processed_test_image/\"\ntest_data[\"image_path\"] = test_data[\"image_id\"].apply(lambda x: test_img_path + x + \".tif\")\ntest_data","metadata":{"id":"wPMd1PXy7dNz","outputId":"3ad2bc50-ae06-4403-b42f-8918f99064b1","trusted":true,"execution":{"iopub.status.busy":"2024-11-09T06:17:40.136501Z","iopub.execute_input":"2024-11-09T06:17:40.13695Z","iopub.status.idle":"2024-11-09T06:17:40.157023Z","shell.execute_reply.started":"2024-11-09T06:17:40.136907Z","shell.execute_reply":"2024-11-09T06:17:40.155834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#vgg_model = tf.keras.models.load_model('')","metadata":{"id":"Oo4q83Ty77q9","outputId":"d3b9d8a0-f491-4e13-c4ff-1bce68d6f56d","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inception_model = tf.keras.models.load_model('/kaggle/input/models/keras/default/1/inception_model_new.h5')","metadata":{"id":"xpdlzTB2QMkf","trusted":true,"execution":{"iopub.status.busy":"2024-11-09T06:17:50.020992Z","iopub.execute_input":"2024-11-09T06:17:50.021442Z","iopub.status.idle":"2024-11-09T06:17:53.556456Z","shell.execute_reply.started":"2024-11-09T06:17:50.021387Z","shell.execute_reply":"2024-11-09T06:17:53.554005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"resnet_model = tf.keras.models.load_model('/kaggle/input/models/keras/default/1/inception_model_new.h5')","metadata":{"id":"1BG5XUJpQZZd","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#efficientnet_model = tf.keras.models.load_model('')","metadata":{"id":"sugCt7RaQdWH","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inception_model.summary()","metadata":{"id":"RE8O1OVACmhJ","trusted":true,"execution":{"iopub.status.busy":"2024-11-09T06:18:11.527289Z","iopub.execute_input":"2024-11-09T06:18:11.527762Z","iopub.status.idle":"2024-11-09T06:18:11.977731Z","shell.execute_reply.started":"2024-11-09T06:18:11.527718Z","shell.execute_reply":"2024-11-09T06:18:11.976234Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Prediction 1","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\nimport numpy as np\n\nimport pandas as pd\n\n\n\n\npatient_ids = test_data['patient_id'].unique()\n\npatient_predictions = {}\n\n\n\nfor patient_id in patient_ids:\n\n    patient_chunks = test_data[test_data['patient_id'] == patient_id]\n\n    chunk_images = []\n\n    for img_path in patient_chunks[\"image_path\"]:\n\n        img = tf.keras.preprocessing.image.load_img(img_path, target_size=(256, 256))\n\n        img_array = tf.keras.preprocessing.image.img_to_array(img)\n\n        img_array = img_array / 255.0\n\n        img_array = tf.expand_dims(img_array, 0)\n\n        chunk_images.append(img_array)\n\n\n\n    chunk_images_tensor = tf.concat(chunk_images, axis=0)\n\n    chunk_results = inception_model.predict(chunk_images_tensor) #predict individual model . Change as needed\n\n    chunk_results = tf.nn.softmax(chunk_results)\n\n\n\n    # Average probabilities for the patient\n\n    patient_probs = np.mean(chunk_results, axis=0)\n\n    patient_predictions[patient_id] = patient_probs\n\n\n\n# Create submission DataFrame\n\nsubmission_data = []\n\nfor patient_id, probs in patient_predictions.items():\n\n    submission_data.append([patient_id, probs[0], probs[1]]) # Assuming [CE, LAA] order\n\n\n\nsubmission_df = pd.DataFrame(submission_data, columns=['patient_id', 'CE', 'LAA'])","metadata":{"id":"th8qpjZQFlvl","outputId":"4319f702-9ce1-4d47-eb1b-d8bcf8f318f9","trusted":true,"execution":{"iopub.status.busy":"2024-11-09T06:19:17.536954Z","iopub.execute_input":"2024-11-09T06:19:17.537516Z","iopub.status.idle":"2024-11-09T06:19:22.945486Z","shell.execute_reply.started":"2024-11-09T06:19:17.537471Z","shell.execute_reply":"2024-11-09T06:19:22.944249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df","metadata":{"id":"O-4kQvjuCI-f","outputId":"be3ff694-b848-48cb-b69a-02094b1f3e9a","trusted":true,"execution":{"iopub.status.busy":"2024-11-09T06:19:27.069196Z","iopub.execute_input":"2024-11-09T06:19:27.069665Z","iopub.status.idle":"2024-11-09T06:19:27.083021Z","shell.execute_reply.started":"2024-11-09T06:19:27.069621Z","shell.execute_reply":"2024-11-09T06:19:27.081689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', index = False)\n\n!head submission.csv","metadata":{"id":"h6rIce2VCCf1","outputId":"a7640639-3c33-47b4-b423-bdcad109e20e","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Ensemble","metadata":{"id":"eZo_QZL1EAxG"}},{"cell_type":"code","source":"'''# Load the saved models and uncomment when model is availble \n\n#efficientnet_model = tf.keras.models.load_model('efficientnet_model.h5')\n\ninception_model = tf.keras.models.load_model('/kaggle/input/models/keras/default/1/inception_model_new.h5')\n\nresnet_model = tf.keras.models.load_model('/kaggle/input/models/keras/default/1/inception_model_new.h5')\n# Define weights for each model (adjust these based on validation performance)\n\nresnet_weight = 0.4\n\ninception_weight = 0.6\n\n#efficientnet_weight = 0.3\n\n\n\n\n\npatient_ids = test_data['patient_id'].unique()\n\npatient_predictions = {}\n\n\n\nfor patient_id in patient_ids:\n\n    patient_chunks = test_data[test_data['patient_id'] == patient_id]\n\n    chunk_images = []\n\n    for img_path in patient_chunks[\"image_path\"]:\n\n        img = tf.keras.preprocessing.image.load_img(img_path, target_size=(256, 256))\n\n        img_array = tf.keras.preprocessing.image.img_to_array(img)\n\n        img_array = img_array / 255.0  # Assuming your models were trained with this scaling\n\n        img_array = tf.expand_dims(img_array, 0)  # Add batch dimension\n\n        chunk_images.append(img_array)\n\n\n\n    chunk_images_tensor = tf.concat(chunk_images, axis=0)\n\n\n\n    # Get predictions from each model\n\n    resnet_preds = resnet_model.predict(chunk_images_tensor)\n\n    inception_preds = inception_model.predict(chunk_images_tensor)\n\n    #efficientnet_preds = efficientnet_model.predict(chunk_images_tensor)\n\n\n\n    # Average the predictions (simple averaging)\n\n\n\n    ensemble_preds = (\n\n    resnet_weight * resnet_preds +\n\n    inception_weight * inception_preds #+\n\n    #efficientnet_weight * efficientnet_preds\n\n)\n\n    # Average probabilities for the patient\n\n    patient_probs = np.mean(ensemble_preds, axis=0)\n\n    patient_predictions[patient_id] = patient_probs\n\n\n\n# Create submission DataFrame\n\nsubmission_data = []\n\nfor patient_id, probs in patient_predictions.items():\n\n    submission_data.append([patient_id, probs[0], probs[1]])  # Assuming [CE, LAA] order\n\n\n\nsubmission_df = pd.DataFrame(submission_data, columns=['patient_id', 'CE', 'LAA'])\n\nsubmission_df\n\n'''\n","metadata":{"id":"SbOc2TZ_EC8N","trusted":true,"execution":{"iopub.status.busy":"2024-11-09T06:20:05.613445Z","iopub.execute_input":"2024-11-09T06:20:05.613918Z","iopub.status.idle":"2024-11-09T06:20:21.587312Z","shell.execute_reply.started":"2024-11-09T06:20:05.613874Z","shell.execute_reply":"2024-11-09T06:20:21.585928Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null}]}