{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Mayo Clinic STRIP Stroke Blood Clot Origin 1:8 Pixel Resolution Reduction With Background Color Corrected Image Split Dataset\n\n## Competition\nhttps://www.kaggle.com/competitions/mayo-clinic-strip-ai\n    \nThis notebook provides a data set transform of the 1:8 size reduced color corrected background images dataset. Images are split into seperate samples by segmentation, and stored as as PNG format for a smaller loss-less format.","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:37:58.851308Z","iopub.execute_input":"2022-07-22T19:37:58.85179Z","iopub.status.idle":"2022-07-22T19:37:58.88872Z","shell.execute_reply.started":"2022-07-22T19:37:58.851685Z","shell.execute_reply":"2022-07-22T19:37:58.887205Z"}}},{"cell_type":"markdown","source":"## Setup Runtime Environment\nThis control allows for easily moving this solution out of Kaggle notebooks and into a different system.","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:43:24.551207Z","iopub.execute_input":"2022-07-22T19:43:24.551753Z","iopub.status.idle":"2022-07-22T19:43:24.584988Z","shell.execute_reply.started":"2022-07-22T19:43:24.551633Z","shell.execute_reply":"2022-07-22T19:43:24.583092Z"}}},{"cell_type":"code","source":"from pathlib import Path\n\nBASE_DIRECTORY: Path = Path(\"/kaggle/input/\")\nBASE_OUTPUT_DIRECTORY: Path = Path(\"/kaggle/working/\")\nMAYO_BASE_DIRECTORY: Path = Path(\"/kaggle/input/mayo-clinic-strip-ai/\")\n\nNOBACKGROUND_DATA_DIRECTORY: Path = BASE_DIRECTORY.joinpath(\"mayo-18-reduction-with-background-color-adjusted\", \"nobackground_data\")\nSEGMENTED_DATA_DIRECTORY: Path = BASE_OUTPUT_DIRECTORY.joinpath(\"segmented_data\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next we define our function for loading and storing the samples. Additionally internal variables storing the image are removed as soon as possible and we manually request a garbage collection.","metadata":{}},{"cell_type":"code","source":"from typing import Tuple\nfrom collections import defaultdict\nfrom pathlib import Path\nimport numpy as np\n\nfrom utils_image import load_png_image, store_image\nfrom utils_segmentation import fragement_image\n\ndef store_fragement_image(image_id: str, target_data_type: str) -> Tuple[defaultdict, defaultdict]:\n    # Define output containers\n    metadata: defaultdict = defaultdict(lambda: defaultdict(list))\n    fragment_locations: defaultdict = defaultdict(list)\n\n    # load image\n    source_image_path: Path = NOBACKGROUND_DATA_DIRECTORY.joinpath(target_data_type, f\"{image_id}.png\")\n    image: np.ndarray = load_png_image(image_path=source_image_path)\n\n    # Split the image\n    fragment_list, coord_list, fragement_metadata = fragement_image(image=image)\n\n    # Combine split operation metadata with the rest from the data set\n    metadata[source_image_path.name] = fragement_metadata\n\n    # Save the individual fragment images\n    for i, fragment in enumerate(fragment_list):\n        target_file_path: Path = SEGMENTED_DATA_DIRECTORY.joinpath(target_data_type, f\"{image_id}_{i+1}.png\")\n        store_image(image_data=fragment, target_file=target_file_path)\n\n    for i, (x1, y1, x2, y2) in enumerate(coord_list):\n        fragment_locations[\"filename\"].append(source_image_path.name)\n        fragment_locations[\"fragment_number\"].append(i)\n        fragment_locations[\"x1\"].append(x1)\n        fragment_locations[\"y1\"].append(y1)\n        fragment_locations[\"x2\"].append(x2)\n        fragment_locations[\"y2\"].append(y2)\n\n    return metadata, fragment_locations","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from typing import Dict, List, Optional, Tuple\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\n\n\ndef fragment_data_set(original_data: np.ndarray, fragmented_data: np.ndarray) -> np.ndarray:\n    final_new_df: Optional[pd.DataFrame] = None\n\n    fragment_changes: pd.DataFrame = fragmented_data[[\"file\", \"number_of_fragments\"]]\n\n    for index, row in fragment_changes.iterrows():\n        image_id: str = Path(row[\"file\"]).stem\n        frag_num: int = row[\"number_of_fragments\"]\n        print(f\"sample: {image_id}, fragment count: {frag_num}\")\n\n        # retrieve the original row\n        row_df: pd.DataFrame = original_data[original_data[\"image_id\"] == image_id].copy(deep=True)\n\n        if frag_num == 0:\n            row_df[\"image_id\"] = f\"{image_id}_0\"\n            if final_new_df is None:\n                final_new_df = row_df\n            else:\n                final_new_df = pd.concat([final_new_df, row_df])\n        else:\n            # generate new rows (to represent the fragments) and add to final data frame\n            for i in range(1, frag_num + 1):\n                new_df: pd.DataFrame = row_df.copy(deep=True)\n                new_image_id = f\"{image_id}_{i}\"\n                new_df[\"image_id\"] = new_image_id\n\n                if final_new_df is None:\n                    final_new_df = new_df\n                else:\n                    final_new_df = pd.concat([final_new_df, new_df])\n\n    return final_new_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Load the details for all three types of data (test, train, and other). We use these data to perform the preprocessing on the data set.","metadata":{}},{"cell_type":"code","source":"from typing import Dict, List\n\nimport pandas as pd\nfrom joblib import Parallel, delayed\nfrom tqdm import tqdm\n\ndata_set_metadata: Dict = {}\ndata_set_output_metadata: Dict = {}\ndata_types: List[str] = [\"train\", \"test\", \"other\"]\nfor target_data_type in data_types:\n    data_set_metadata[target_data_type] = pd.read_csv(MAYO_BASE_DIRECTORY.joinpath(f\"{target_data_type}.csv\"))[\n        \"image_id\"\n    ].tolist()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Black list filtering (This sample has difficulty being preprocessed in Kaggle, it has been excluded).\nblack_list: Dict = {\"other\": [\"2c3c06_0\"]}\nfor target_data_type in black_list.keys():\n    for image_id in black_list[target_data_type]:\n        print(f\"Removing {target_data_type}:{image_id} from dataset\")\n        data_set_metadata[target_data_type].remove(image_id)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You might not be able to use parallel executions in Kaggle instances as run out of memory. Leverage the parallelization on other platforms. Job counts have been set to 1 to help alleviant compute constraints within Kaggle.","metadata":{}},{"cell_type":"code","source":"for target_data_type in data_types:\n    data_set_output_metadata[target_data_type] = Parallel(n_jobs=2, prefer=\"threads\", verbose=10)(\n        delayed(store_fragement_image)(image_id, target_data_type)\n        for image_id in tqdm(\n            data_set_metadata[target_data_type], desc=f\"Splitting Images - {target_data_type} images\"\n        )\n    )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Store metadata for split operations\nfor target_data_type in data_types:\n    out_d = defaultdict(list)\n    loc_out = defaultdict(list)\n    # Process the Parallel result list\n    for res, loc in data_set_output_metadata[target_data_type]:\n        # Locations can be processed immediately\n        for key in loc:\n            loc_out[key].extend(loc[key])\n        # Metadata needs to be iterated and extended to final output container\n        for filename in res:\n            out_d[\"file\"].append(filename)\n            for key in res[filename]:\n                out_d[key].extend(res[filename][key])\n\n    pd.DataFrame(loc_out).to_csv(SEGMENTED_DATA_DIRECTORY.joinpath(f\"fragment_locations.{target_data_type}.csv\"), index=False)\n    pd.DataFrame(out_d).to_csv(SEGMENTED_DATA_DIRECTORY.joinpath(f\"fragment_metadata.{target_data_type}.csv\"), index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Update data set level metadata to account for more images\n# Individual rows are now multiple rows if a split occurred.\ndata_set_metadata = {}\nfor target_data_type in data_types:\n    data_set_metadata[target_data_type] = pd.read_csv(MAYO_BASE_DIRECTORY.joinpath(f\"{target_data_type}.csv\"))\n\ndata_set_fragment_metadata: Dict = {}\nfor target_data_type in data_types:\n    data_set_fragment_metadata[target_data_type] = pd.read_csv(\n        SEGMENTED_DATA_DIRECTORY.joinpath(f\"fragment_metadata.{target_data_type}.csv\")\n    )\n\n# Update data sets with new image fragments\nfor target_data_type in data_types:\n    data_set_metadata[target_data_type] = fragment_data_set(\n        data_set_metadata[target_data_type], data_set_fragment_metadata[target_data_type]\n    )\n\n# Store the updated data set metadata\nfor target_data_type in data_types:\n    pd.DataFrame(data_set_metadata[target_data_type]).to_csv(\n        SEGMENTED_DATA_DIRECTORY.joinpath(f\"{target_data_type}.fragmented.csv\"), index=False\n    )","metadata":{},"execution_count":null,"outputs":[]}]}