{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Prepared dataset for Contrails competition\n\n### Notebook goal\n\nIn this notebook I preprocessed the images and labels that were given in the competition data in order to filter out the incorrect samples. \n\nThe product datasets are available here:\n- training images: https://www.kaggle.com/datasets/bencetar/contrail-train-imgs\n- validation images: https://www.kaggle.com/datasets/bencetar/contrail-valid-imgs\n- training masks: https://www.kaggle.com/datasets/bencetar/prep-train-masks\n- validation masks: https://www.kaggle.com/datasets/bencetar/prep-valid-masks","metadata":{}},{"cell_type":"code","source":"# Import libraries\nimport os\nimport shutil\nimport json\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom scipy.ndimage import label\nfrom skimage.measure import regionprops\nimport tensorflow as tf\n\n# Directories\nTRAIN_BASE_DIR = '/kaggle/input/google-research-identify-contrails-reduce-global-warming/train'\nVALID_BASE_DIR = '/kaggle/input/google-research-identify-contrails-reduce-global-warming/validation'\nTEST_BASE_DIR = '/kaggle/input/google-research-identify-contrails-reduce-global-warming/test'\nTRAIN_DEST_DIR = '/kaggle/working/train'\nVALID_DEST_DIR = '/kaggle/working/valid'","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n##  Preprocessing masks\n\n### Steps\nFirst, I filtered out masks with no labels.\n\nSecond, I applied a funciton to filter out the masks with given criterias:\n- Objects must contain at least 10 pixels\n- Objects must be at least 3x longer than they are wide\n\nThird, double checked if the processed mask has any labels or not.\n\nFinally, the rest of these samples were counted as correct masks, and written into numpy arrays.","metadata":{}},{"cell_type":"code","source":"def filter_out_no_label(base_folder=None):\n    \"\"\" Filter out all masks where there is no label.\"\"\"\n    \n    all_id = (list(set([sample.split(\"/\")[-1] for sample in os.listdir(base_folder)])))\n    print(f'All ids: {len(all_id)}')\n    \n    corr_mask_ids = []\n    for rec_id in all_id:\n        with open(os.path.join(base_folder, rec_id, 'human_pixel_masks.npy'), 'rb') as f:\n            human_pixel_mask = np.load(f)\n\n        labeled_image, num_labels = label(human_pixel_mask)\n        if num_labels != 0:\n            corr_mask_ids.append(rec_id)\n            \n    return corr_mask_ids\n\n# ------------------------\ncorrect_train_masks = filter_out_no_label(TRAIN_BASE_DIR)\nprint(f'All correct train ids: {len(correct_train_masks)}')\n\ncorrect_valid_masks = filter_out_no_label(VALID_BASE_DIR)\nprint(f'All correct valid ids: {len(correct_valid_masks)}')","metadata":{"execution":{"iopub.status.busy":"2023-08-09T16:16:43.084828Z","iopub.execute_input":"2023-08-09T16:16:43.085572Z","iopub.status.idle":"2023-08-09T16:16:43.567627Z","shell.execute_reply.started":"2023-08-09T16:16:43.085537Z","shell.execute_reply":"2023-08-09T16:16:43.565648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ms_df = pd.Series(correct_train_masks)\nvalid_ms_df = pd.Series(correct_valid_masks)\ntrain_ms_df.to_csv('train_ids.csv')\nvalid_ms_df.to_csv('valid_ids.csv')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"uploaded_train_ids = '/kaggle/input/contrail-correct-masks/train_ids.csv'\nuploaded_valid_ids = '/kaggle/input/contrail-correct-masks/valid_ids.csv'\ntrain_ms_series = list(pd.read_csv(uploaded_train_ids, index_col=0, dtype='string')[\"0\"])\nvalid_ms_series = list(pd.read_csv(uploaded_valid_ids, index_col=0, dtype='string')[\"0\"])\nprint(len(train_ms_series), len(valid_ms_series))","metadata":{"execution":{"iopub.status.busy":"2023-08-09T16:17:04.567648Z","iopub.execute_input":"2023-08-09T16:17:04.567992Z","iopub.status.idle":"2023-08-09T16:17:04.631095Z","shell.execute_reply.started":"2023-08-09T16:17:04.567967Z","shell.execute_reply":"2023-08-09T16:17:04.630429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def filter_correct_masks(base_folder=None, dest_folder=None, corr_mask_ids=list, min_object_pixels=10, min_aspect_ratio=3):\n    \"\"\" Filter out masks and remove labels not matching the criteria:\n        1. size is bigger than 10 pixel\n        2. defined feature shape (3x as long as wide) \"\"\"\n    \n    if not os.path.exists(dest_folder):\n        os.makedirs(dest_folder)\n    \n    corr_ids_2nd_filter = []\n    for record_id in corr_mask_ids:\n        with open(os.path.join(base_folder, record_id, 'human_pixel_masks.npy'), 'rb') as f:\n            mask_image = np.load(f).astype(np.int64).squeeze(axis=2)\n    \n        # Perform connected component labeling\n        labeled_image, num_labels = label(mask_image)\n        # Calculate region properties for each labeled object\n        regions = regionprops(labeled_image)\n        # Create a boolean mask to keep track of selected objects\n        selected_objects_mask = np.zeros_like(labeled_image, dtype=np.bool)\n\n        for region in regions:\n            # Check if the object contains at least 10 pixels\n            if region.area >= min_object_pixels:\n                # Check if the region has non-zero minor_axis_length\n                if region.minor_axis_length < 1e-5:\n                    continue\n\n                # Calculate aspect ratio (length / width) of the object\n                aspect_ratio = region.major_axis_length / region.minor_axis_length\n\n                # Check if the aspect ratio is greater than or equal to 'min_aspect_ratio'\n                if aspect_ratio >= min_aspect_ratio:\n                    # Add the selected object to the mask\n                    selected_objects_mask[labeled_image == region.label] = True\n                                    \n        # Check if we have no labels in the mask after filtering\n        labeled_image, num_labels = label(selected_objects_mask)\n        if num_labels != 0:\n\n            # Append to correct ids and save to npz\n            corr_ids_2nd_filter.append(record_id)\n            np.savez_compressed(f'{dest_folder}/prep_mask_{record_id}.npz', data=labeled_image)\n\n    return corr_ids_2nd_filter\n","metadata":{"execution":{"iopub.status.busy":"2023-08-09T16:17:44.918057Z","iopub.execute_input":"2023-08-09T16:17:44.918412Z","iopub.status.idle":"2023-08-09T16:17:44.928025Z","shell.execute_reply.started":"2023-08-09T16:17:44.918382Z","shell.execute_reply":"2023-08-09T16:17:44.926408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_masks = filter_correct_masks(\n    base_folder=TRAIN_BASE_DIR,\n    dest_folder=TRAIN_DEST_DIR,\n    corr_mask_ids=train_ms_series\n)\nvalid_masks = filter_correct_masks(\n    base_folder=VALID_BASE_DIR,\n    dest_folder=VALID_DEST_DIR,\n    corr_mask_ids=valid_ms_series\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-09T16:17:47.483359Z","iopub.execute_input":"2023-08-09T16:17:47.483794Z","iopub.status.idle":"2023-08-09T16:20:31.118153Z","shell.execute_reply.started":"2023-08-09T16:17:47.483763Z","shell.execute_reply":"2023-08-09T16:20:31.116771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Should be both: 9149 train 540 valid\nprint(len(train_masks), len(valid_masks))\nprint(len(os.listdir(TRAIN_DEST_DIR)), len(os.listdir(VALID_DEST_DIR)))","metadata":{"execution":{"iopub.status.busy":"2023-08-09T16:20:31.119959Z","iopub.execute_input":"2023-08-09T16:20:31.120638Z","iopub.status.idle":"2023-08-09T16:20:31.131004Z","shell.execute_reply.started":"2023-08-09T16:20:31.120593Z","shell.execute_reply":"2023-08-09T16:20:31.129775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shutil.make_archive(\"prep_valid_masks\", \"zip\", \"/kaggle/working/valid\")","metadata":{"execution":{"iopub.status.busy":"2023-08-09T16:31:19.666318Z","iopub.execute_input":"2023-08-09T16:31:19.666662Z","iopub.status.idle":"2023-08-09T16:31:19.757719Z","shell.execute_reply.started":"2023-08-09T16:31:19.666634Z","shell.execute_reply":"2023-08-09T16:31:19.756928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing images","metadata":{}},{"cell_type":"code","source":"### (Already preprocessed and uploaded dataset: contrail_train_imgs & contrail_valid_imgs)\n\ndef normalize_range(data, bounds):\n    \"\"\"Maps data to the range [0, 1].\"\"\"\n    return (data - bounds[0]) / (bounds[1] - bounds[0])\n\ndef write_to_rgb(usable_ids=list, base_folder=None, dest_folder=None, batch_size=500):\n    \n    # Make dest folder\n    if not os.path.exists(dest_folder):\n        os.makedirs(dest_folder)\n\n    total_samples = len(usable_ids)\n    for batch_start in range(0, total_samples, batch_size):\n        batch_end = min(batch_start + batch_size, total_samples)\n        batch_ids = usable_ids[batch_start:batch_end]\n        \n        for record_id in batch_ids:\n            with open(os.path.join(base_folder, record_id, 'band_15.npy'), 'rb') as f:\n                band15= np.load(f).astype(np.float16)\n            with open(os.path.join(base_folder, record_id, 'band_14.npy'), 'rb') as f:\n                band14 = np.load(f).astype(np.float16)\n            with open(os.path.join(base_folder, record_id, 'band_11.npy'), 'rb') as f:\n                band11 = np.load(f).astype(np.float16)\n\n            r = normalize_range(band15 - band14, _TDIFF_BOUNDS)\n            g = normalize_range(band14 - band11, _CLOUD_TOP_TDIFF_BOUNDS)\n            b = normalize_range(band14, _T11_BOUNDS)\n            composite = np.clip(np.stack([r,g, b], axis=2), 0, 1)\n\n            # Save all timesteps in a single compressed npz file\n            save_pth = os.path.join(dest_folder, f'rgb_image_{record_id}.npz')\n            np.savez_compressed(save_pth, data=composite)\n            \n# -------------------------\n## Processed these in multiple parts since OOM problem\n\n#write_to_rgb(usable_ids=train_masks[:4000], base_folder=TRAIN_BASE_DIR, dest_folder=TRAIN_DEST_DIR)\nwrite_to_rgb(usable_ids=train_masks[4000:], base_folder=TRAIN_BASE_DIR, dest_folder=TRAIN_DEST_DIR)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:04:21.314157Z","iopub.execute_input":"2023-07-28T10:04:21.314702Z","iopub.status.idle":"2023-07-28T10:04:21.320998Z","shell.execute_reply.started":"2023-07-28T10:04:21.314645Z","shell.execute_reply":"2023-07-28T10:04:21.319915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shutil.make_archive('contrail_train_2', 'zip', '/kaggle/working/train')","metadata":{},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}