{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":18647,"databundleVersionId":1126921,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport PIL\nfrom IPython.display import Image, display\nimport seaborn as sns\nimport openslide\n\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Input\n\n\nfrom PIL import Image\n\n# Increase the maximum image pixel limit to avoid DecompressionBombError\nImage.MAX_IMAGE_PIXELS = None  # Disable the limit entirely (use with caution)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-22T02:41:24.980193Z","iopub.execute_input":"2024-10-22T02:41:24.981121Z","iopub.status.idle":"2024-10-22T02:41:24.988808Z","shell.execute_reply.started":"2024-10-22T02:41:24.981073Z","shell.execute_reply":"2024-10-22T02:41:24.987613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image and mask directories\ndata_dir = f'{BASE_PATH}/train_images'\nmask_dir = f'{BASE_PATH}/train_label_masks'\n\n\n# Location of training labels\ntrain = pd.read_csv(f'{BASE_PATH}/train.csv').set_index('image_id')\ntest = pd.read_csv(f'{BASE_PATH}/test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-22T01:58:22.05837Z","iopub.execute_input":"2024-10-22T01:58:22.058807Z","iopub.status.idle":"2024-10-22T01:58:22.134893Z","shell.execute_reply.started":"2024-10-22T01:58:22.058763Z","shell.execute_reply":"2024-10-22T01:58:22.133619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#SHOW DATA\ndisplay(train.head())\nprint(\"Shape of training data :\", train.shape)\nprint(\"Count of unique data providers :\", len(train.data_provider.unique()))\nprint(\"Count of unique isup_grade(target) :\", len(train.isup_grade.unique()))\nprint(\"Count of unique gleason_score :\", len(train.gleason_score.unique()))\n\ndisplay(test.head())\nprint(\"Shape of test data :\", test.shape)\nprint(\"Count of unique data providers :\", len(test.data_provider.unique()))\n","metadata":{"execution":{"iopub.status.busy":"2024-10-22T01:58:22.13666Z","iopub.execute_input":"2024-10-22T01:58:22.137322Z","iopub.status.idle":"2024-10-22T01:58:22.186578Z","shell.execute_reply.started":"2024-10-22T01:58:22.13727Z","shell.execute_reply":"2024-10-22T01:58:22.185356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#TODO - format and normalize data","metadata":{"execution":{"iopub.status.busy":"2024-10-22T01:58:22.188348Z","iopub.execute_input":"2024-10-22T01:58:22.188728Z","iopub.status.idle":"2024-10-22T01:58:22.194218Z","shell.execute_reply.started":"2024-10-22T01:58:22.188688Z","shell.execute_reply":"2024-10-22T01:58:22.193018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_count(df, feature, title='', size=2):\n    f, ax = plt.subplots(1,1, figsize=(4*size,3*size))\n    total = float(len(df))\n    sns.countplot(x=feature, data=df, order=df[feature].value_counts().index, palette='Set2', ax=ax)\n    plt.title(title)\n    for p in ax.patches:\n        height = p.get_height()\n        ax.text(p.get_x()+p.get_width()/2.,\n                height + 3,\n                '{:1.2f}%'.format(100*height/total),\n                ha=\"center\") \n    plt.show()\n\nplot_count(df=train, feature='isup_grade', title = 'isup_grade count and %age plot')\n# plot_count(df=train, feature='gleason_score', title = 'gleason_score count and %age plot', size=3)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-22T01:58:22.197261Z","iopub.execute_input":"2024-10-22T01:58:22.197693Z","iopub.status.idle":"2024-10-22T01:58:22.546658Z","shell.execute_reply.started":"2024-10-22T01:58:22.197653Z","shell.execute_reply":"2024-10-22T01:58:22.545447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_images(slides): \n    f, ax = plt.subplots(5,3, figsize=(18,22))\n    for i, slide in enumerate(slides):\n        image = openslide.OpenSlide(os.path.join(data_dir, f'{slide}.tiff'))\n        spacing = 1 / (float(image.properties['tiff.XResolution']) / 10000)\n        patch = image.read_region((1780,1950), 0, (256, 256))\n        ax[i//3, i%3].imshow(patch) \n        image.close()       \n        ax[i//3, i%3].axis('off')\n        \n        image_id = slide\n        data_provider = train.loc[slide, 'data_provider']\n        isup_grade = train.loc[slide, 'isup_grade']\n        gleason_score = train.loc[slide, 'gleason_score']\n        ax[i//3, i%3].set_title(f\"ID: {image_id}\\nSource: {data_provider} ISUP: {isup_grade} Gleason: {gleason_score}\")\n\n    plt.show() \n\n\nimages = [\n    '07a7ef0ba3bb0d6564a73f4f3e1c2293',\n    '037504061b9fba71ef6e24c48c6df44d',\n    '035b1edd3d1aeeffc77ce5d248a01a53',\n    '059cbf902c5e42972587c8d17d49efed',\n    '06a0cbd8fd6320ef1aa6f19342af2e68',\n    '06eda4a6faca84e84a781fee2d5f47e1',\n    '0a4b7a7499ed55c71033cefb0765e93d',\n    '0838c82917cd9af681df249264d2769c',\n    '046b35ae95374bfb48cdca8d7c83233f',\n    '074c3e01525681a275a42282cd21cbde',\n    '05abe25c883d508ecc15b6e857e59f32',\n    '05f4e9415af9fdabc19109c980daf5ad',\n    '060121a06476ef401d8a21d6567dee6d',\n    '068b0e3be4c35ea983f77accf8351cc8',\n    '08f055372c7b8a7e1df97c6586542ac8'\n]\n\ndisplay_images(images)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-22T01:58:22.548445Z","iopub.execute_input":"2024-10-22T01:58:22.548937Z","iopub.status.idle":"2024-10-22T01:58:26.904599Z","shell.execute_reply.started":"2024-10-22T01:58:22.548884Z","shell.execute_reply":"2024-10-22T01:58:26.903304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_masks(slides): \n    f, ax = plt.subplots(5,3, figsize=(18,22))\n    for i, slide in enumerate(slides):\n        \n        mask = openslide.OpenSlide(os.path.join(mask_dir, f'{slide}_mask.tiff'))\n        mask_data = mask.read_region((0,0), mask.level_count - 1, mask.level_dimensions[-1])\n        cmap = matplotlib.colors.ListedColormap(['black', 'gray', 'green', 'yellow', 'orange', 'red'])\n\n        ax[i//3, i%3].imshow(np.asarray(mask_data)[:,:,0], cmap=cmap, interpolation='nearest', vmin=0, vmax=5) \n        mask.close()       \n        ax[i//3, i%3].axis('off')\n        \n        image_id = slide\n        data_provider = train.loc[slide, 'data_provider']\n        isup_grade = train.loc[slide, 'isup_grade']\n        gleason_score = train.loc[slide, 'gleason_score']\n        ax[i//3, i%3].set_title(f\"ID: {image_id}\\nSource: {data_provider} ISUP: {isup_grade} Gleason: {gleason_score}\")\n        f.tight_layout()\n        \n    plt.show()\ndisplay_masks(images)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-22T01:58:26.906212Z","iopub.execute_input":"2024-10-22T01:58:26.906621Z","iopub.status.idle":"2024-10-22T01:58:33.756253Z","shell.execute_reply.started":"2024-10-22T01:58:26.90658Z","shell.execute_reply":"2024-10-22T01:58:33.755024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df = pd.read_csv(TRAIN_CSV_PATH)\n\n# # Preprocess data - Filter only rows with images available\n# available_images = set(os.listdir(TRAIN_IMAGES_PATH))\n\n# # Ensure correct paths for images\n# train_df['image_path'] = (train_df['image_id'] + '.tiff').apply(lambda x: os.path.join(TRAIN_IMAGES_PATH, x))\n# train_df = train_df[train_df['image_path'].apply(lambda x: os.path.exists(x))]\n","metadata":{"execution":{"iopub.status.busy":"2024-10-22T01:58:33.757856Z","iopub.execute_input":"2024-10-22T01:58:33.758286Z","iopub.status.idle":"2024-10-22T01:58:33.763264Z","shell.execute_reply.started":"2024-10-22T01:58:33.758245Z","shell.execute_reply":"2024-10-22T01:58:33.762124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom PIL import Image\n\n# Set paths\nBASE_PATH = '/kaggle/input/prostate-cancer-grade-assessment/'\nTRAIN_CSV_PATH = os.path.join(BASE_PATH, 'train.csv')\nTRAIN_IMAGES_PATH = os.path.join(BASE_PATH, 'train_images/')\n\n# Load CSV\ntrain_df = pd.read_csv(TRAIN_CSV_PATH)\n\n# Ensure correct paths for images\ntrain_df['image_path'] = (train_df['image_id'] + '.tiff').apply(lambda x: os.path.join(TRAIN_IMAGES_PATH, x))\n\n# Filter only rows with available image files\ntrain_df = train_df[train_df['image_path'].apply(lambda x: os.path.exists(x))]\nprint(f\"Filtered DataFrame dimensions: {train_df.shape}\")\n\n# Function to get image dimensions\ndef get_image_size(image_path):\n    with Image.open(image_path) as img:\n        return img.size  # Returns (width, height)\n\n# Apply the function to get dimensions for all images\nimage_sizes = train_df['image_path'].apply(get_image_size)\n\n# Check if all images have the same size\nunique_sizes = image_sizes.unique()\nprint(f\"Unique image sizes: {len(unique_sizes) }\")\n\nif len(unique_sizes) == 1:\n    print(f\"All images have the same size: {unique_sizes[0]}\")\nelse:\n    print(\"Images have varying sizes.\", )\n","metadata":{"execution":{"iopub.status.busy":"2024-10-22T02:40:47.731247Z","iopub.execute_input":"2024-10-22T02:40:47.731922Z","iopub.status.idle":"2024-10-22T02:40:56.238349Z","shell.execute_reply.started":"2024-10-22T02:40:47.731865Z","shell.execute_reply":"2024-10-22T02:40:56.236376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_size = (4096, 4096)\n\nmodel = Sequential([\n    Input(shape=(image_size[0], image_size[1], 3)),\n    Conv2D(32, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Flatten(),\n    Dense(128, activation='relu'),\n    Dense(1, activation='linear')  # Regression output for ISUP score\n])\n","metadata":{"execution":{"iopub.status.busy":"2024-10-22T02:41:30.025096Z","iopub.execute_input":"2024-10-22T02:41:30.02596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}