{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30588,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Extract non-blank tiles from TMA images only, then save it in labels-specific folders\nFast and simple method to produce relatively balanced dataset for adjusting and validating classification models before generalise it on the whole train_images dataset.","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-11-25T15:14:12.003385Z","iopub.execute_input":"2023-11-25T15:14:12.004152Z","iopub.status.idle":"2023-11-25T15:14:12.68398Z","shell.execute_reply.started":"2023-11-25T15:14:12.004115Z","shell.execute_reply":"2023-11-25T15:14:12.682495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to generate tiles from images based on provided parameters\ndef generate_tiles(image_id, label, image_dir, output_dir, tile_size, threshold_mean, threshold_std):\n    # Create label-specific output directory\n    label_output_dir = os.path.join(output_dir, label)\n    os.makedirs(label_output_dir, exist_ok=True)\n\n    # Construct the path for the image using its ID\n    image_path = os.path.join(image_dir, f\"{image_id}.png\")\n    \n    # Check if the image exists\n    if not os.path.exists(image_path):\n        print(f\"Image not found: {image_path}\")\n        return\n\n    try:\n        # Read the image using OpenCV\n        img = cv2.imread(image_path)\n        height, width, _ = img.shape\n    except Exception as e:\n        print(f\"Error reading image {image_path}: {e}\")\n        return\n\n    # Calculate the number of rows and columns for tiles\n    rows_count = height // tile_size\n    cols_count = width // tile_size\n\n    # Iterate through rows and columns to generate tiles\n    for row_idx in range(rows_count):\n        for col_idx in range(cols_count):\n            # Calculate coordinates for each tile\n            x = col_idx * tile_size\n            y = row_idx * tile_size\n\n            # Extract the tile from the image\n            tile = img[y:y+tile_size, x:x+tile_size]\n\n            # Define the name for the tiles\n            tile_name = f\"{image_id}_tile_{col_idx}_{row_idx}.png\"\n            \n            # Convert the tile to a NumPy array for mean and std calculation\n            tile_np = np.array(tile)\n\n            # Check if the tile is not mostly blank based on mean and std thresholds\n            if tile_np.mean() >= threshold_mean and tile_np.std() >= threshold_std:\n                # Define the path for the tiles\n                tile_path = os.path.join(label_output_dir, tile_name)\n                # Save the tile as a PNG files\n                cv2.imwrite(tile_path, tile)","metadata":{"execution":{"iopub.status.busy":"2023-11-25T15:17:22.480636Z","iopub.execute_input":"2023-11-25T15:17:22.481243Z","iopub.status.idle":"2023-11-25T15:17:22.493235Z","shell.execute_reply.started":"2023-11-25T15:17:22.481175Z","shell.execute_reply":"2023-11-25T15:17:22.492006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define directory paths and CSV file\nimage_dir = \"/kaggle/input/UBC-OCEAN/train_images\"\noutput_dir = \"/kaggle/working/output_tiles\"\ncsv_file_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\n\n# Read the CSV file containing image IDs and labels\ndata = pd.read_csv(csv_file_path)\n\n# Iterate through each row in the CSV file\nfor index, dataset_row in data.iterrows():\n    if dataset_row['is_tma'] == True:\n        # Extract image ID, label, and generate tiles for each entry\n        generate_tiles(dataset_row['image_id'], dataset_row['label'], image_dir, output_dir, tile_size=256, threshold_mean=170, threshold_std=15)","metadata":{"execution":{"iopub.status.busy":"2023-11-25T15:19:00.098964Z","iopub.execute_input":"2023-11-25T15:19:00.099346Z","iopub.status.idle":"2023-11-25T15:19:25.009653Z","shell.execute_reply.started":"2023-11-25T15:19:00.099317Z","shell.execute_reply":"2023-11-25T15:19:25.00835Z"},"trusted":true},"execution_count":null,"outputs":[]}]}