{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Mayo Clinic STRIP Stroke Blood Clot Origin 1:8 Pixel Resolution Reduction With Background Color and Channel Correction Dataset \n\n## Competition\nhttps://www.kaggle.com/competitions/mayo-clinic-strip-ai\n    \nThis notebook provides a data set transform of the 1:8 size reduced dataset. Images are color corrected based on background image color, and stored as as PNG format for a smaller loss-less format.","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:37:58.851308Z","iopub.execute_input":"2022-07-22T19:37:58.85179Z","iopub.status.idle":"2022-07-22T19:37:58.88872Z","shell.execute_reply.started":"2022-07-22T19:37:58.851685Z","shell.execute_reply":"2022-07-22T19:37:58.887205Z"}}},{"cell_type":"markdown","source":"## Setup Runtime Environment\nThis control allows for easily moving this solution out of Kaggle notebooks and into a different system.","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:43:24.551207Z","iopub.execute_input":"2022-07-22T19:43:24.551753Z","iopub.status.idle":"2022-07-22T19:43:24.584988Z","shell.execute_reply.started":"2022-07-22T19:43:24.551633Z","shell.execute_reply":"2022-07-22T19:43:24.583092Z"}}},{"cell_type":"code","source":"from pathlib import Path\n\nBASE_DIRECTORY: Path = Path(\"/kaggle/input/\")\nBASE_OUTPUT_DIRECTORY: Path = Path(\"/kaggle/working/\")\nMAYO_BASE_DIRECTORY: Path = Path(\"/kaggle/input/mayo-clinic-strip-ai/\")\n\nDOWNSAMPLED_DATA_DIRECTORY: Path = BASE_DIRECTORY.joinpath(\"mayo-clinic-strip-data-set-18-reduction\", \"downsampled_data\")\nNOBACKGROUND_DATA_DIRECTORY: Path = BASE_OUTPUT_DIRECTORY.joinpath(\"nobackground_data\")","metadata":{"execution":{"iopub.status.busy":"2022-07-22T21:19:06.1407Z","iopub.execute_input":"2022-07-22T21:19:06.1417Z","iopub.status.idle":"2022-07-22T21:19:06.171526Z","shell.execute_reply.started":"2022-07-22T21:19:06.141598Z","shell.execute_reply":"2022-07-22T21:19:06.170665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next we define our function for loading and storing the samples. Additionally internal variables storing the image are removed as soon as possible and we manually request a garbage collection.","metadata":{}},{"cell_type":"code","source":"from utils_image import load_png_image, store_image\nfrom utils_isolation import normalize_and_isolate\n\nimport gc\n\ndef store_background_corrected_image(image_id: str, target_data_type: str):\n    input_file_path: Path = DOWNSAMPLED_DATA_DIRECTORY.joinpath(target_data_type, f\"{image_id}.png\")\n    output_file_path: Path = NOBACKGROUND_DATA_DIRECTORY.joinpath(target_data_type, f\"{image_id}.png\")\n\n#     print(f\"{input_file_path} -> {output_file_path}\")\n\n    image_array: np.ndarray = load_png_image(image_path=input_file_path)\n    image_array = normalize_and_isolate(image_array=image_array)\n    store_image(image_data=image_array, target_file=output_file_path)\n\n    del image_array\n    gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Load the details for all three types of data (test, train, and other). We use these data to perform the preprocessing on the data set.","metadata":{}},{"cell_type":"code","source":"from typing import Dict, List\n\nimport pandas as pd\nfrom joblib import Parallel, delayed\nfrom tqdm import tqdm\n\nmetadata: Dict = {}\ndata_types: List[str] = [\"train\", \"test\", \"other\"]\nfor target_data_type in data_types:\n    metadata[target_data_type] = pd.read_csv(MAYO_BASE_DIRECTORY.joinpath(f\"{target_data_type}.csv\"))[\n        \"image_id\"\n    ].tolist()\n    \n# Black list filtering (This sample has difficulty being preprocessed in Kaggle, it has been excluded).\nblack_list: Dict = {\"other\": [\"2c3c06_0\"]}\nfor target_data_type in black_list.keys():\n    for image_id in black_list[target_data_type]:\n        print(f\"Removing {target_data_type}:{image_id} from dataset\")\n        metadata[target_data_type].remove(image_id)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You may need to adjust the number of parallel jobs used in Kaggle instances they can run out of memory.","metadata":{}},{"cell_type":"code","source":"    Parallel(n_jobs=8, prefer=\"threads\")(\n        delayed(store_background_corrected_image)(image_id, \"train\")\n        for image_id in tqdm(metadata[\"train\"], desc=\"Correcting for background color - training images\")\n    )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    Parallel(n_jobs=8, prefer=\"threads\")(\n        delayed(store_background_corrected_image)(image_id, \"other\")\n        for image_id in tqdm(metadata[\"other\"], desc=\"Correcting for background color - other images\")\n    )","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    Parallel(n_jobs=8, prefer=\"threads\")(\n        delayed(store_background_corrected_image)(image_id, \"test\")\n        for image_id in tqdm(metadata[\"test\"], desc=\"Correcting for background color - test images\")\n    )","metadata":{},"execution_count":null,"outputs":[]}]}