{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":7103246,"sourceType":"datasetVersion","datasetId":4094810}],"dockerImageVersionId":30587,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfrom glob import glob\nimport zipfile\nfrom typing import Optional\nfrom tqdm import tqdm\nimport random\nimport math\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport PIL\nfrom PIL import Image\nPIL.Image.MAX_IMAGE_PIXELS = None","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-02T18:02:01.304761Z","iopub.execute_input":"2023-12-02T18:02:01.305126Z","iopub.status.idle":"2023-12-02T18:02:02.794408Z","shell.execute_reply.started":"2023-12-02T18:02:01.305099Z","shell.execute_reply":"2023-12-02T18:02:02.792869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    csv_path = '/kaggle/input/dataset-df/label_EC_data.csv'\n    path_column = 'path'\n    tile_height = 512\n    tile_width = 512\n    tile_save_dir = 'images_tiles'\n    overlap_factor = 0.1\n    save = True\nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T18:02:02.796565Z","iopub.execute_input":"2023-12-02T18:02:02.797049Z","iopub.status.idle":"2023-12-02T18:02:02.802358Z","shell.execute_reply.started":"2023-12-02T18:02:02.797017Z","shell.execute_reply":"2023-12-02T18:02:02.801315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read metadata\ndf1 = pd.read_csv(config.csv_path)\ndf1.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T18:02:02.803567Z","iopub.execute_input":"2023-12-02T18:02:02.803905Z","iopub.status.idle":"2023-12-02T18:02:02.853362Z","shell.execute_reply.started":"2023-12-02T18:02:02.803876Z","shell.execute_reply":"2023-12-02T18:02:02.851637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport numpy as np\nimport os\nfrom tqdm import tqdm\nimport concurrent.futures\nimport time\nimport gc\nimport pandas as pd\n\nImage.MAX_IMAGE_PIXELS = None\n\ndef remove_black_background(input_image_path, output_directory, row, threshold=20, target_size=(512, 512), chunk_size=64):\n    original_img = Image.open(input_image_path)\n    resized_img = original_img.resize(target_size)\n    img = resized_img.convert(\"RGBA\")\n\n    width, height = img.size\n    img_array = np.array(img)\n\n    is_close_to_black = lambda pixel: all(value < threshold for value in pixel[:-1])\n\n    for y in range(0, height, chunk_size):\n        for x in range(0, width, chunk_size):\n            chunk = img_array[y:y+chunk_size, x:x+chunk_size, :]\n\n            mask = np.apply_along_axis(is_close_to_black, axis=-1, arr=chunk)\n            chunk[mask, 3] = (chunk[mask, 3].astype(float) * 0.5).astype(np.uint8)\n\n            img_array[y:y+chunk_size, x:x+chunk_size, :] = chunk\n\n    result_img = Image.fromarray(img_array)\n\n    # Get the label from the dataframe row\n    label = row['label']\n\n    # Create a subdirectory based on the label\n    label_subdirectory = os.path.join(output_directory, label)\n    os.makedirs(label_subdirectory, exist_ok=True)\n\n    file_name, file_extension = os.path.splitext(os.path.basename(input_image_path))\n    output_image_path = os.path.join(label_subdirectory, f\"{file_name}_no_bg{file_extension}\")\n    result_img.save(output_image_path)\n\n    # Release memory\n    del original_img, resized_img, img, img_array, result_img\n    gc.collect()\n\ndef process_image_row(row, output_directory, **kwargs):\n    input_image_path = row['path']\n    remove_black_background(input_image_path, output_directory, row=row, **kwargs)\n\ndef remove_black_background_parallel(df, output_directory, batch_size=10, max_workers=1, **kwargs):\n    total_images = df.shape[0]\n    num_batches = (total_images + batch_size - 1) // batch_size\n\n    progress_bar = tqdm(total=total_images, desc=\"Processing Images\", unit=\"image\")\n    start_time = time.time()\n\n    def update_progress(future):\n        progress_bar.update(1)\n        elapsed_time = time.time() - start_time\n        images_processed = progress_bar.n\n        images_remaining = total_images - images_processed\n        time_per_image = elapsed_time / images_processed\n        estimated_time_remaining = images_remaining * time_per_image\n        progress_bar.set_postfix({\"ETA\": f\"{estimated_time_remaining:.2f} seconds\"})\n\n    with concurrent.futures.ProcessPoolExecutor(max_workers=max_workers) as executor:\n        futures = [executor.submit(process_image_row, row, output_directory, **kwargs) for _, row in df.iterrows()]\n\n        # Use as_completed to iterate over completed futures\n        for future in concurrent.futures.as_completed(futures):\n            update_progress(future)\n\n    # Introduce a delay before closing the progress bar\n    time.sleep(2)\n    progress_bar.close()\n\n# Example usage with batch processing and limited workers\noutput_directory = \"/kaggle/working/output_no_bg\"\nremove_black_background_parallel(df1, output_directory, batch_size=10, max_workers=3, threshold=20, target_size=(512, 512), chunk_size=64)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T18:02:02.855623Z","iopub.execute_input":"2023-12-02T18:02:02.855987Z","iopub.status.idle":"2023-12-02T18:54:42.899754Z","shell.execute_reply.started":"2023-12-02T18:02:02.855957Z","shell.execute_reply":"2023-12-02T18:54:42.89629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2=pd.read_csv('/kaggle/input/dataset-df/label_MC_data.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-02T18:56:16.101922Z","iopub.execute_input":"2023-12-02T18:56:16.102483Z","iopub.status.idle":"2023-12-02T18:56:16.124709Z","shell.execute_reply.started":"2023-12-02T18:56:16.102423Z","shell.execute_reply":"2023-12-02T18:56:16.123691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"remove_black_background_parallel(df2, output_directory, batch_size=10, max_workers=2, threshold=20, target_size=(512, 512), chunk_size=64)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T19:07:03.480371Z","iopub.execute_input":"2023-12-02T19:07:03.480897Z","iopub.status.idle":"2023-12-02T19:41:58.669873Z","shell.execute_reply.started":"2023-12-02T19:07:03.480857Z","shell.execute_reply":"2023-12-02T19:41:58.665672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df3=pd.read_csv('/kaggle/input/dataset-df/label_LGSC_data.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-02T19:42:03.461669Z","iopub.execute_input":"2023-12-02T19:42:03.462753Z","iopub.status.idle":"2023-12-02T19:42:03.516515Z","shell.execute_reply.started":"2023-12-02T19:42:03.462709Z","shell.execute_reply":"2023-12-02T19:42:03.514149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"remove_black_background_parallel(df3, output_directory, batch_size=10, max_workers=2, threshold=20, target_size=(512, 512), chunk_size=64)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T19:42:07.191893Z","iopub.execute_input":"2023-12-02T19:42:07.192298Z","iopub.status.idle":"2023-12-02T20:03:36.172805Z","shell.execute_reply.started":"2023-12-02T19:42:07.192266Z","shell.execute_reply":"2023-12-02T20:03:36.171149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df4=pd.read_csv('/kaggle/input/dataset-df/label_HGSC_data.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-02T20:04:01.09616Z","iopub.execute_input":"2023-12-02T20:04:01.096597Z","iopub.status.idle":"2023-12-02T20:04:01.124339Z","shell.execute_reply.started":"2023-12-02T20:04:01.096565Z","shell.execute_reply":"2023-12-02T20:04:01.122878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"remove_black_background_parallel(df4, output_directory, batch_size=10, max_workers=2, threshold=20, target_size=(512, 512), chunk_size=64)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T20:04:06.720434Z","iopub.execute_input":"2023-12-02T20:04:06.721013Z","iopub.status.idle":"2023-12-02T22:07:17.101346Z","shell.execute_reply.started":"2023-12-02T20:04:06.720968Z","shell.execute_reply":"2023-12-02T22:07:17.094597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df5=pd.read_csv('/kaggle/input/dataset-df/label_CC_data.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-02T22:07:17.114375Z","iopub.execute_input":"2023-12-02T22:07:17.115654Z","iopub.status.idle":"2023-12-02T22:07:17.170696Z","shell.execute_reply.started":"2023-12-02T22:07:17.115531Z","shell.execute_reply":"2023-12-02T22:07:17.168227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"remove_black_background_parallel(df5, output_directory, batch_size=10, max_workers=2, threshold=20, target_size=(512, 512), chunk_size=64)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T22:07:17.172961Z","iopub.execute_input":"2023-12-02T22:07:17.174174Z","iopub.status.idle":"2023-12-02T23:14:15.009203Z","shell.execute_reply.started":"2023-12-02T22:07:17.174124Z","shell.execute_reply":"2023-12-02T23:14:15.003677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\nimport os\n\n# Define the directory to be zipped\ndirectory_to_zip = '/kaggle/working/output_no_bg'\n\n# Define the output zip file\nzip_file_path = '/kaggle/working/output_no_bg.zip'\n\n# Create a zip file\nshutil.make_archive(zip_file_path, 'zip', directory_to_zip)\n\n# Optionally, you can remove the original files/directory if needed\n# shutil.rmtree(directory_to_zip)\n\nprint(f\"Zip file created at: {zip_file_path}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-02T23:24:33.266525Z","iopub.execute_input":"2023-12-02T23:24:33.267288Z","iopub.status.idle":"2023-12-02T23:24:40.182523Z","shell.execute_reply.started":"2023-12-02T23:24:33.267225Z","shell.execute_reply":"2023-12-02T23:24:40.180741Z"},"trusted":true},"execution_count":null,"outputs":[]}]}