{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":52254,"databundleVersionId":6863140,"sourceType":"competition"},{"sourceId":8449671,"sourceType":"datasetVersion","datasetId":5035335}],"dockerImageVersionId":30699,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cv2\nimport pydicom\nimport matplotlib.pyplot as plt\nfrom math import ceil\nimport seaborn as sb\nimport nibabel as nib\nfrom glob import glob\nfrom tqdm import tqdm, trange\nimport os\nos.environ[\"KERAS_BACKEND\"] = \"tensorflow\"  # Or \"jax\" or \"torch\"!\nimport tensorflow as tf\nimport keras_cv\nimport keras\nfrom sklearn.model_selection import train_test_split\n\n%matplotlib inline\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os","metadata":{"execution":{"iopub.status.busy":"2024-05-18T13:23:13.66265Z","iopub.execute_input":"2024-05-18T13:23:13.663282Z","iopub.status.idle":"2024-05-18T13:23:13.672283Z","shell.execute_reply.started":"2024-05-18T13:23:13.663246Z","shell.execute_reply":"2024-05-18T13:23:13.671055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# Define the paths for the directories\nselected_images_dir = \"/kaggle/working/selected_images\"\n\n# Create the directories if they don't exist\nos.makedirs(selected_images_dir, exist_ok=True)\nos.makedirs(augmented_images_dir, exist_ok=True)\n\n# Check if directories are created\nif os.path.exists(selected_images_dir):\n    print(\"Selected images directory created successfully.\")\nelse:\n    print(\"Error creating selected images directory.\")\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-18T13:41:38.483149Z","iopub.execute_input":"2024-05-18T13:41:38.483984Z","iopub.status.idle":"2024-05-18T13:41:38.490377Z","shell.execute_reply.started":"2024-05-18T13:41:38.483953Z","shell.execute_reply":"2024-05-18T13:41:38.489482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#BASE_PATH = \"/kaggle/input/llllll\"\nimport pandas as pd\n\n# Path to the CSV file\ncsv_file_path = f\"{BASE_PATH}/Selected images.csv\"\n\n# Load the CSV file into a DataFrame\nmeta_df = pd.read_csv(csv_file_path)\n\n# Extract the image paths from the 'image_path' column and put them into a list\nimage_paths = meta_df['image_path'].tolist()\n\n# Print the first few image paths to verify\nprint(image_paths[:10])\n","metadata":{"execution":{"iopub.status.busy":"2024-05-18T13:25:04.782309Z","iopub.execute_input":"2024-05-18T13:25:04.782944Z","iopub.status.idle":"2024-05-18T13:25:05.114393Z","shell.execute_reply.started":"2024-05-18T13:25:04.782912Z","shell.execute_reply":"2024-05-18T13:25:05.113443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras.utils.set_random_seed(seed=config.SEED)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T13:26:42.668866Z","iopub.execute_input":"2024-05-18T13:26:42.669803Z","iopub.status.idle":"2024-05-18T13:26:42.679082Z","shell.execute_reply.started":"2024-05-18T13:26:42.669753Z","shell.execute_reply":"2024-05-18T13:26:42.677989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_df = pd.read_csv(f\"{BASE_PATH}/Selected images.csv\")\n\n# Checking if patients are repeated by finding the number of unique patient IDs\nnum_rows = meta_df.shape[0]\nunique_patients = meta_df[\"patient_id\"].nunique()\n\nprint(f\"{num_rows=}\")\nprint(f\"{unique_patients=}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:19:22.996005Z","iopub.execute_input":"2024-05-18T12:19:22.996793Z","iopub.status.idle":"2024-05-18T12:19:23.333619Z","shell.execute_reply.started":"2024-05-18T12:19:22.996762Z","shell.execute_reply":"2024-05-18T12:19:23.332719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_df[\"dicom_folder\"] = BASE_PATH + \"/\" + \"/Selected images.csv\"\\\n                                    + \"/\" + meta_df.patient_id.astype(str)\\\n                                    + \"/\" + meta_df.series_id.astype(str)\n\ntest_folders = meta_df.dicom_folder.tolist()\ntest_paths = []\nfor folder in tqdm(test_folders):\n    test_paths += sorted(glob(os.path.join(folder, \"*dcm\")))[::STRIDE]","metadata":{"execution":{"iopub.status.busy":"2024-05-18T13:26:57.772176Z","iopub.execute_input":"2024-05-18T13:26:57.772839Z","iopub.status.idle":"2024-05-18T13:27:01.114489Z","shell.execute_reply.started":"2024-05-18T13:26:57.772808Z","shell.execute_reply":"2024-05-18T13:27:01.113469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:27:34.024581Z","iopub.execute_input":"2024-05-18T12:27:34.025552Z","iopub.status.idle":"2024-05-18T12:27:34.052551Z","shell.execute_reply.started":"2024-05-18T12:27:34.02551Z","shell.execute_reply":"2024-05-18T12:27:34.051424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport tensorflow as tf\nimport pandas as pd\nfrom glob import glob\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nimport keras_cv\n\n\nconfig = Config()\n\n# Setting the random seed for reproducibility\ntf.keras.utils.set_random_seed(config.SEED)\n\n# Loading metadata\nmeta_df = pd.read_csv(f\"{BASE_PATH}/Selected images.csv\")\n\n# Checking if patients are repeated by finding the number of unique patient IDs\nnum_rows = meta_df.shape[0]\nunique_patients = meta_df[\"patient_id\"].nunique()\n\nprint(f\"num_rows={num_rows}\")\nprint(f\"unique_patients={unique_patients}\")\n\n# Constructing the DICOM folder paths\nmeta_df[\"dicom_folder\"] = BASE_PATH + \"/\" + meta_df[\"patient_id\"].astype(str) + \"/\" + meta_df[\"series_id\"].astype(str)\n\n# Gathering test paths with the specified stride\ntest_folders = meta_df[\"dicom_folder\"].tolist()\ntest_paths = []\nfor folder in tqdm(test_folders):\n    dicom_files = sorted(glob(os.path.join(folder, \"*.dcm\")))\n    test_paths.extend(dicom_files[::STRIDE])\n\n# Function to split data for each group\ndef split_group(group, test_size=0.2):\n    if len(group) == 1:\n        return (group, pd.DataFrame()) if np.random.rand() < test_size else (pd.DataFrame(), group)\n    else:\n        return train_test_split(group, test_size=test_size, random_state=config.SEED)\n\n# Initialize the train and validation datasets\ntrain_data = pd.DataFrame()\nval_data = pd.DataFrame()\n\n# Verify the TARGET_COLS exist in the dataframe\nexisting_target_cols = [col for col in config.TARGET_COLS if col in meta_df.columns]\n\n# Iterate through the groups and split them, handling single-sample groups\nfor _, group in meta_df.groupby(existing_target_cols):\n    train_group, val_group = split_group(group)\n    train_data = pd.concat([train_data, train_group], ignore_index=True)\n    val_data = pd.concat([val_data, val_group], ignore_index=True)\n\n# Outputs to verify\nprint(f\"Training Data Shape: {train_data.shape}\")\nprint(f\"Validation Data Shape: {val_data.shape}\")\n\n# Prepare image paths and labels for the dataset\ntrain_image_paths = train_data['image_path'].tolist()\ntrain_labels = train_data[existing_target_cols].values\n\nval_image_paths = val_data['image_path'].tolist()\nval_labels = val_data[existing_target_cols].values\n\n# Define the image and label decoding function\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-18T13:27:27.436063Z","iopub.execute_input":"2024-05-18T13:27:27.436453Z","iopub.status.idle":"2024-05-18T13:27:31.122726Z","shell.execute_reply.started":"2024-05-18T13:27:27.436422Z","shell.execute_reply":"2024-05-18T13:27:31.121828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"def decode_image_and_label(image_path, label):\n    file_bytes = tf.io.read_file(image_path)\n    image = tf.io.decode_png(file_bytes, channels=3, dtype=tf.uint8)\n    image = tf.image.resize(image, config.IMAGE_SIZE, method=\"bilinear\")\n    image = tf.cast(image, tf.float32) / 255.0\n    \n    label = tf.cast(label, tf.float32)\n    #         bowel       fluid       kidney      liver       spleen\n    labels = (label[0:1], label[1:2], label[2:5], label[5:8], label[8:11])\n    \n    return (image, labels)\n\n\ndef apply_augmentation(images, labels):\n    augmenter = keras_cv.layers.Augmenter(\n        [\n            keras_cv.layers.RandomFlip(mode=\"horizontal_and_vertical\"),\n            keras_cv.layers.RandomCutout(height_factor=0.2, width_factor=0.2),\n            \n        ]\n    )\n    \n    return (augmenter(images), labels)\n\n\ndef build_dataset(image_paths, labels):\n    ds = (\n        tf.data.Dataset.from_tensor_slices((image_paths, labels))\n        .map(decode_image_and_label, num_parallel_calls=config.AUTOTUNE)\n        .shuffle(config.BATCH_SIZE * 10)\n        .batch(config.BATCH_SIZE)\n        .map(apply_augmentation, num_parallel_calls=config.AUTOTUNE)\n        .prefetch(config.AUTOTUNE)\n    )\n    \n    return ds","metadata":{"execution":{"iopub.status.busy":"2024-05-18T13:27:39.702822Z","iopub.execute_input":"2024-05-18T13:27:39.70317Z","iopub.status.idle":"2024-05-18T13:27:39.713159Z","shell.execute_reply.started":"2024-05-18T13:27:39.703143Z","shell.execute_reply":"2024-05-18T13:27:39.712145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def standardize_image(dicom_image):\n    pixel_array = dicom_image.pixel_array   \n    \n    if dicom_image.PixelRepresentation == 1:\n        bit_shift = dicom_image.BitsAllocated - dicom_image.BitsStored\n        dtype = pixel_array.dtype \n        pixel_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n    \n    if dicom_image.PhotometricInterpretation == \"MONOCHROME1\":\n        pixel_array = 1 - pixel_array\n    \n    # transform to hounsfield units\n    intercept = dicom_image.RescaleIntercept\n    slope = dicom_image.RescaleSlope\n    pixel_array = pixel_array * slope + intercept\n    return pixel_array\ndef apply_window(image_stack, window, normalize = True):\n    \n    window_width = window[0]\n    window_center = window[1]\n    img_min = window_center - window_width//2 #minimum HU level\n    img_max = window_center + window_width//2 #maximum HU level\n    img = np.clip(image_stack, img_min, img_max)\n    \n    if normalize:\n        img = (img - img_min) / (img_max - img_min)\n    return img\ndef dicom_to_png(dicom_image):\n    \"\"\"\n    Read the dicom file and preprocess appropriately.\n    \"\"\"\n    pixel_array = standardize_image(dicom_image)\n    \n    windows = [(2,1),(2048,1),(300,150)]\n    imgs = []\n    for window in windows:\n        windowed_array = apply_window(pixel_array,window, normalize = True)\n        img = (windowed_array * 255).astype(np.uint8)\n        img = cv2.resize(img, (256,256))\n        imgs.append(img)\n    \n    return np.stack(imgs,axis=-1)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T13:27:43.763255Z","iopub.execute_input":"2024-05-18T13:27:43.764106Z","iopub.status.idle":"2024-05-18T13:27:43.774353Z","shell.execute_reply.started":"2024-05-18T13:27:43.764074Z","shell.execute_reply":"2024-05-18T13:27:43.77343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# Define the paths for the directories\nselected_images_dir = \"/kaggle/working/selected_images\"\naugmented_images_dir = \"/kaggle/working/augmented_images\"\n\n# Create the directories if they don't exist\nos.makedirs(selected_images_dir, exist_ok=True)\nos.makedirs(augmented_images_dir, exist_ok=True)\n\n# Check if directories are created\nif os.path.exists(selected_images_dir):\n    print(\"Selected images directory created successfully.\")\nelse:\n    print(\"Error creating selected images directory.\")\n\nif os.path.exists(augmented_images_dir):\n    print(\"Augmented images directory created successfully.\")\nelse:\n    print(\"Error creating augmented images directory.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-05-18T13:31:52.023282Z","iopub.execute_input":"2024-05-18T13:31:52.024132Z","iopub.status.idle":"2024-05-18T13:31:52.031082Z","shell.execute_reply.started":"2024-05-18T13:31:52.024098Z","shell.execute_reply":"2024-05-18T13:31:52.030091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pydicom\nfrom math import ceil\n\nnum_images = len(image_paths[:])\nnum_rows = ceil(num_images / 3)\nfig, axes = plt.subplots(num_rows, 3, figsize=(15, 5 * num_rows))\n\nfor i, path in enumerate(image_paths[:15]):\n    dicom_data = pydicom.dcmread(path)\n    img = dicom_data.pixel_array\n    row = i // 3\n    col = i % 3\n    axes[row, col].imshow(img, cmap='gray')\n    axes[row, col].axis('off')\n    axes[row, col].set_title(f\"Image {i+1}\")\n\n# Remove empty subplots\nfor i in range(num_images, num_rows*3):\n    row = i // 3\n    col = i % 3\n    fig.delaxes(axes[row, col])\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-18T13:28:59.214143Z","iopub.execute_input":"2024-05-18T13:28:59.214558Z","iopub.status.idle":"2024-05-18T13:29:02.038264Z","shell.execute_reply.started":"2024-05-18T13:28:59.214527Z","shell.execute_reply":"2024-05-18T13:29:02.037302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pydicom\nimport os\nfrom math import ceil\n\n# Define the directory to save the images\nselected_images_dir = \"/kaggle/working/selected_images\"\n\n# Create the directory if it doesn't exist\nos.makedirs(selected_images_dir, exist_ok=True)\n\nnum_images = len(image_paths)\nnum_rows = ceil(num_images / 3)\nfig, axes = plt.subplots(num_rows, 3, figsize=(15, 5 * num_rows))\n\nfor i, path in enumerate(image_paths):\n    dicom_data = pydicom.dcmread(path)\n    img = dicom_data.pixel_array\n    row = i // 3\n    col = i % 3\n    axes[row, col].imshow(img, cmap='gray')\n    axes[row, col].axis('off')\n    axes[row, col].set_title(f\"Image {i+1}\")\n\n    # Save the image\n    image_name = os.path.basename(path)  # Get the image file name\n    image_save_path = os.path.join(selected_images_dir, f\"{image_name[:-4]}.png\")  # Save as PNG\n    plt.imsave(image_save_path, img, cmap='gray')\n\n# Remove empty subplots\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-18T13:42:04.133906Z","iopub.execute_input":"2024-05-18T13:42:04.134609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_images = min(len(image_paths), 9)\nnum_rows = (num_images + 2) // 3\nfig, axes = plt.subplots(num_rows, 3, figsize=(15, 5 * num_rows))\n\nfor i, path in enumerate(image_paths[:9]):\n    dicom_data = pydicom.dcmread(path)\n    img = dicom_to_png(dicom_data)\n    row = i // 3\n    col = i % 3\n    axes[row, col].imshow(img, cmap='gray')\n    axes[row, col].axis('off')\n    axes[row, col].set_title(f\"Image {i+1}\")\n\n# Remove empty subplots\nfor i in range(num_images, num_rows*3):\n    row = i // 3\n    col = i % 3\n    fig.delaxes(axes[row, col])\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:31:51.149296Z","iopub.execute_input":"2024-05-18T12:31:51.150187Z","iopub.status.idle":"2024-05-18T12:31:53.032625Z","shell.execute_reply.started":"2024-05-18T12:31:51.150156Z","shell.execute_reply":"2024-05-18T12:31:53.031176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import albumentations as A\n\n# Define augmentation pipeline\ntransform = A.Compose([\n    A.Rotate(limit=30, p=0.5),\n    A.HorizontalFlip(p=0.5),\n    A.VerticalFlip(p=0.5),\n    A.RandomBrightnessContrast(p=0.2),\n    A.RandomGamma(p=0.2),\n    A.GaussianBlur(p=0.2),\n    A.GaussNoise(p=0.2),\n])\n\n# Augment a single image\ndef augment_image(image):\n    augmented = transform(image=image)\n    augmented_image = augmented['image']\n    return augmented_image\n\n# Apply augmentation to DICOM images\naugmented_images = []\nfor path in image_paths[:9]:\n    dicom_data = pydicom.dcmread(path)\n    img = dicom_to_png(dicom_data)\n    augmented_img = augment_image(img)\n    augmented_images.append(augmented_img)\n\n# Display augmented images\nnum_rows = ceil(len(augmented_images) / 3)\nfig, axes = plt.subplots(num_rows, 3, figsize=(15, 5 * num_rows))\n\nfor i, img in enumerate(augmented_images):\n    row = i // 3\n    col = i % 3\n    axes[row, col].imshow(img, cmap='gray')\n    axes[row, col].axis('off')\n    axes[row, col].set_title(f\"Augmented Image {i+1}\")\n\n# Remove empty subplots\nfor i in range(len(augmented_images), num_rows*3):\n    row = i // 3\n    col = i % 3\n    fig.delaxes(axes[row, col])\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-15T23:52:36.967738Z","iopub.execute_input":"2024-05-15T23:52:36.968525Z","iopub.status.idle":"2024-05-15T23:52:39.486143Z","shell.execute_reply.started":"2024-05-15T23:52:36.968483Z","shell.execute_reply":"2024-05-15T23:52:39.484956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define additional augmentation pipeline\nadditional_transform = A.Compose([\n    A.ElasticTransform(alpha=120, sigma=120 * 0.05, alpha_affine=120 * 0.03, p=0.5),\n    A.GridDistortion(p=0.5),\n    A.RandomSizedCrop(min_max_height=(200, 256), height=256, width=256, p=0.5),\n    A.RandomBrightnessContrast(p=0.5),\n    A.CLAHE(p=0.5),\n])\n\n# Augment a single image with additional techniques\ndef augment_image_with_additional(image):\n    augmented = additional_transform(image=image)\n    augmented_image = augmented['image']\n    return augmented_image\n\n# Apply additional augmentation to DICOM images\nadditional_augmented_images = []\nfor path in image_paths[:9]:\n    dicom_data = pydicom.dcmread(path)\n    img = dicom_to_png(dicom_data)\n    augmented_img = augment_image_with_additional(img)\n    additional_augmented_images.append(augmented_img)\n\n# Display additional augmented images\nnum_rows = ceil(len(additional_augmented_images) / 3)\nfig, axes = plt.subplots(num_rows, 3, figsize=(15, 5 * num_rows))\n\nfor i, img in enumerate(additional_augmented_images):\n    row = i // 3\n    col = i % 3\n    axes[row, col].imshow(img, cmap='gray')\n    axes[row, col].axis('off')\n    axes[row, col].set_title(f\"Augmented Image {i+1}\")\n\n# Remove empty subplots\nfor i in range(len(additional_augmented_images), num_rows*3):\n    row = i // 3\n    col = i % 3\n    fig.delaxes(axes[row, col])\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-15T23:54:07.797369Z","iopub.execute_input":"2024-05-15T23:54:07.79822Z","iopub.status.idle":"2024-05-15T23:54:10.359941Z","shell.execute_reply.started":"2024-05-15T23:54:07.798166Z","shell.execute_reply":"2024-05-15T23:54:10.358758Z"},"trusted":true},"execution_count":null,"outputs":[]}]}