{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":6984590,"sourceType":"datasetVersion","datasetId":4014175},{"sourceId":153363129,"sourceType":"kernelVersion"}],"dockerImageVersionId":30615,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\ndata=pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-10T03:15:13.664408Z","iopub.execute_input":"2023-12-10T03:15:13.665405Z","iopub.status.idle":"2023-12-10T03:15:14.038794Z","shell.execute_reply.started":"2023-12-10T03:15:13.665359Z","shell.execute_reply":"2023-12-10T03:15:14.037882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a function to generate the paths based on conditions\ndef generate_paths(row):\n    if row['is_tma'] == 1:\n        return f'/kaggle/input/UBC-OCEAN/train_images/{row[\"image_id\"]}.png'\n    else:\n        return f'/kaggle/input/UBC-OCEAN/train_thumbnails/{row[\"image_id\"]}_thumbnail.png'\n\n# Apply the function to create new columns\ndata['image_path'] = data.apply(generate_paths, axis=1)\ndata","metadata":{"execution":{"iopub.status.busy":"2023-12-10T03:15:14.040365Z","iopub.execute_input":"2023-12-10T03:15:14.040729Z","iopub.status.idle":"2023-12-10T03:15:14.065528Z","shell.execute_reply.started":"2023-12-10T03:15:14.040696Z","shell.execute_reply":"2023-12-10T03:15:14.064625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_new_paths(row):\n    return f'/kaggle/input/step-2-removing-image-background/output_no_bg/{row[\"label\"]}/{row[\"image_id\"]}_no_bg.png'\n\n\n# Apply the function to create the new column 'new_path'\ndata['new_path'] = data.apply(generate_new_paths, axis=1)\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-10T03:15:14.06688Z","iopub.execute_input":"2023-12-10T03:15:14.067143Z","iopub.status.idle":"2023-12-10T03:15:14.092902Z","shell.execute_reply.started":"2023-12-10T03:15:14.067119Z","shell.execute_reply":"2023-12-10T03:15:14.091825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['new_path'][0]","metadata":{"execution":{"iopub.status.busy":"2023-12-10T03:15:14.095307Z","iopub.execute_input":"2023-12-10T03:15:14.095601Z","iopub.status.idle":"2023-12-10T03:15:14.101907Z","shell.execute_reply.started":"2023-12-10T03:15:14.095568Z","shell.execute_reply":"2023-12-10T03:15:14.100954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom PIL import Image\n\n# Get the paths for the first row\nimage_path = data['image_path'][0]\nnew_path = data['new_path'][0]\n\n# Load the images\nimage = Image.open(image_path)\nnew_image = Image.open(new_path)\n\n# Plot the images side by side\nfig, axes = plt.subplots(1, 2, figsize=(12, 6))\n\n# Plot original image\naxes[0].imshow(image)\naxes[0].set_title('Original Image')\n\n# Plot new image\naxes[1].imshow(new_image)\naxes[1].set_title('New Image')\n\n# Display the plots\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-10T03:15:14.103294Z","iopub.execute_input":"2023-12-10T03:15:14.103618Z","iopub.status.idle":"2023-12-10T03:15:15.948999Z","shell.execute_reply.started":"2023-12-10T03:15:14.103581Z","shell.execute_reply":"2023-12-10T03:15:15.94799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport zipfile\nimport os\nimport pandas as pd\n\n# Function to apply denoising to an image\ndef denoise_image(image):\n    # Replace this with your denoising logic\n    # For example, using GaussianBlur\n    denoised_image = cv2.GaussianBlur(image, (5, 5), 0)\n    return denoised_image\n\n# Specify the directory to save processed images\nprocessed_images_directory = '/kaggle/working/new_images/output_no_bg'\nos.makedirs(processed_images_directory, exist_ok=True)\n\n# Iterate over the DataFrame and process images\nfor index, row in data.iterrows():\n    # Read the image using the image path in the 'new_path' column\n    image_path = row['new_path']\n    image_name = os.path.basename(image_path)  # Extract the image name from the path\n    image = cv2.imread(image_path, 1)  # Read as color image\n\n    # Apply denoising to each channel separately\n    denoised_image = cv2.merge([denoise_image(image[:, :, 0]), denoise_image(image[:, :, 1]), denoise_image(image[:, :, 2])])\n\n    # Define the processed image path using the specified format\n    processed_image_path = os.path.join(processed_images_directory, row['label'], f\"{row['image_id']}_no_bg.png\")\n\n    # Create directories for label if they don't exist\n    os.makedirs(os.path.join(processed_images_directory, row['label']), exist_ok=True)\n\n    # Save the processed image\n    cv2.imwrite(processed_image_path, denoised_image)\n\n    # Update the 'new_path' column in the DataFrame with the processed image path\n    data.at[index, 'new_path'] = processed_image_path\n\n    # Save the processed image path in a new 'paths' column\n    data.at[index, 'paths'] = processed_image_path\n\n# Specify the path and name of the zip file to be created\nzip_file_path = '/kaggle/working/new_images.zip'\n\n# Create a ZipFile object in write mode\nwith zipfile.ZipFile(zip_file_path, 'w') as zip_file:\n    # Walk through the directory and add each file to the zip file\n    for foldername, subfolders, filenames in os.walk(processed_images_directory):\n        for filename in filenames:\n            file_path = os.path.join(foldername, filename)\n            arcname = os.path.relpath(file_path, processed_images_directory)\n            zip_file.write(file_path, arcname)\n\nprint(f'Zip file created: {zip_file_path}')\n","metadata":{"execution":{"iopub.status.busy":"2023-12-10T03:15:15.950584Z","iopub.execute_input":"2023-12-10T03:15:15.950917Z","iopub.status.idle":"2023-12-10T03:15:29.26779Z","shell.execute_reply.started":"2023-12-10T03:15:15.950888Z","shell.execute_reply":"2023-12-10T03:15:29.266938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-10T03:15:29.268915Z","iopub.execute_input":"2023-12-10T03:15:29.269646Z","iopub.status.idle":"2023-12-10T03:15:29.282308Z","shell.execute_reply.started":"2023-12-10T03:15:29.269613Z","shell.execute_reply":"2023-12-10T03:15:29.281565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom PIL import Image\nimport numpy as np\n\n# Assuming data is your DataFrame with 'image_path' and 'paths' columns\n\n# Get the paths for the first row\nimage_path = data['image_path'][0]  # Assuming 'image_path' is the column name for the original image\nnew_path = data['paths'][0]  # Assuming 'paths' is the column name for the processed image\n\n# Load the images\noriginal_image = np.array(Image.open(image_path))\nprocessed_image = np.array(Image.open(new_path))\n\n# Plot the images side by side\nfig, axes = plt.subplots(1, 2, figsize=(12, 6))\n\n# Plot original image\naxes[0].imshow(original_image)\naxes[0].set_title('Original Image')\n\n# Plot new image\naxes[1].imshow(processed_image)\naxes[1].set_title('Processed Image')\n\n# Calculate Signal-to-Noise Ratio (SNR) for both images\nsnr_original = np.mean(original_image) / np.std(original_image)\nsnr_processed = np.mean(processed_image) / np.std(processed_image)\n\n# Display SNR values\nprint(f'SNR for Original Image: {snr_original}')\nprint(f'SNR for Processed Image: {snr_processed}')\n\n# Display the plots\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-10T03:15:29.283539Z","iopub.execute_input":"2023-12-10T03:15:29.284102Z","iopub.status.idle":"2023-12-10T03:15:31.174086Z","shell.execute_reply.started":"2023-12-10T03:15:29.284068Z","shell.execute_reply":"2023-12-10T03:15:31.173233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# Assuming data is your DataFrame with 'image_id' and 'label' columns\nmask_path_column = []\n\n# Define the path to the supplemental masks\nmasks_path = '/kaggle/input/ubc-ovarian-cancer-competition-supplemental-masks'\n\n# Iterate through each row in the DataFrame\nfor index, row in data.iterrows():\n    image_id = row['image_id']\n    mask_filename = f\"{image_id}.png\"  # Assuming the mask filenames are based on image_id\n\n    # Construct the full path to the mask image\n    mask_path = os.path.join(masks_path, mask_filename)\n\n    # Check if the mask file exists\n    if os.path.exists(mask_path):\n        mask_path_column.append(mask_path)\n    else:\n        mask_path_column.append(None)  # Or any other value to indicate missing mask\n\n# Add the new column to the DataFrame\ndata['mask_path'] = mask_path_column\ndata","metadata":{"execution":{"iopub.status.busy":"2023-12-10T03:15:31.17548Z","iopub.execute_input":"2023-12-10T03:15:31.175998Z","iopub.status.idle":"2023-12-10T03:15:31.329224Z","shell.execute_reply.started":"2023-12-10T03:15:31.175964Z","shell.execute_reply":"2023-12-10T03:15:31.328212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nfrom PIL import Image\nimport pandas as pd\n\nImage.MAX_IMAGE_PIXELS = None\n\n# Assuming data is your DataFrame with 'path' and 'mask_path' columns\n\ndef process_image(image_path, patch_size=(100, 100)):\n    # Open the image\n    image = Image.open(image_path)\n\n    # Process image in patches\n    for i in range(0, image.width, patch_size[0]):\n        for j in range(0, image.height, patch_size[1]):\n            # Crop the patch\n            patch = image.crop((i, j, i + patch_size[0], j + patch_size[1]))\n\n            # Process the patch as needed\n            # For example, you can save the processed patch, apply some transformation, etc.\n            # ...\n\n            # Release memory after processing each patch\n            del patch\n            gc.collect()\n\n    # Close the image\n    image.close()\n\ndef resize_and_process_masks(row, patch_size=(100, 100)):\n    # Check if 'mask_path' is None, skip the row\n    if pd.isna(row['mask_path']):\n        return\n\n    # Open the images\n    path_image = Image.open(row['paths'])\n    mask_image = Image.open(row['mask_path'])\n\n    # Resize the mask image to match the size of the path image\n    resized_mask = mask_image.resize(path_image.size, Image.LANCZOS)  # Using LANCZOS as a replacement for ANTIALIAS\n\n    # Process masks in patches\n    for i in range(0, resized_mask.width, patch_size[0]):\n        for j in range(0, resized_mask.height, patch_size[1]):\n            # Crop the patch\n            patch = resized_mask.crop((i, j, i + patch_size[0], j + patch_size[1]))\n\n            # Release memory after processing each patch\n            del patch\n            gc.collect()\n\n    # Create the directory if it doesn't exist\n    resized_masks_dir = '/kaggle/working/resized_masks'\n    os.makedirs(resized_masks_dir, exist_ok=True)\n\n    # Get the original mask file name\n    original_mask_filename = os.path.basename(row['mask_path'])\n\n    # Update the path for resized masks in data\n    resized_mask_path = os.path.join(resized_masks_dir, original_mask_filename)\n    data.at[index, 'resized_mask_path'] = resized_mask_path\n\n    # Save or overwrite the resized mask image\n    resized_mask.save(resized_mask_path)\n\n    # Close the images\n    path_image.close()\n    mask_image.close()\n    resized_mask.close()\n\n    # Manually trigger garbage collection to free up memory\n    del path_image, mask_image, resized_mask\n    gc.collect()\n\n# Assuming 'resized_mask_path' column does not exist initially\ndata['resized_mask_path'] = ''\n\nfor index, row in data.iterrows():\n    # Resize and process masks\n    resize_and_process_masks(row)\n\n# Now, your 'resized_mask_path' column in the DataFrame is updated, and masks are resized and processed.\n","metadata":{"execution":{"iopub.status.busy":"2023-12-10T03:15:31.332226Z","iopub.execute_input":"2023-12-10T03:15:31.332559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a ZipFile for the entire directory\nzipfile_path = '/kaggle/working/resized_masks.zip'\nwith zipfile.ZipFile(zipfile_path, 'w') as zipf:\n    # Iterate through all files in the directory and add them to the zip archive\n    for root, _, files in os.walk('/kaggle/working/resized_masks'):\n        for file in files:\n            file_path = os.path.join(root, file)\n            arcname = os.path.relpath(file_path, '/kaggle/working')\n            zipf.write(file_path, arcname)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the head of the DataFrame as a CSV file\ncsv_file_path = '/kaggle/working/data_head.csv'  # Replace with your desired file path\ndata.to_csv(csv_file_path, index=False)\n\nprint(f\"Head of the DataFrame saved as CSV: {csv_file_path}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}