{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, glob, gc\nimport cv2\nimport random\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nfrom PIL import Image\nImage.MAX_IMAGE_PIXELS = None\n\nfrom collections import Counter\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-02T10:35:41.618316Z","iopub.execute_input":"2023-11-02T10:35:41.618724Z","iopub.status.idle":"2023-11-02T10:35:42.305394Z","shell.execute_reply.started":"2023-11-02T10:35:41.618688Z","shell.execute_reply":"2023-11-02T10:35:42.304279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_DIR = \"/kaggle/input/UBC-OCEAN\"\nTRAIN_DIR1 = '/kaggle/input/UBC-OCEAN/train_thumbnails'\nTRAIN_DIR2 = '/kaggle/input/ubc-resized-3000pix-tma/train_images'","metadata":{"execution":{"iopub.status.busy":"2023-11-02T10:35:42.307192Z","iopub.execute_input":"2023-11-02T10:35:42.307627Z","iopub.status.idle":"2023-11-02T10:35:42.312537Z","shell.execute_reply.started":"2023-11-02T10:35:42.307597Z","shell.execute_reply":"2023-11-02T10:35:42.311498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_file_path(image_id):\n    file = f\"{TRAIN_DIR1}/{image_id}_thumbnail.png\"\n    if not os.path.exists(file):\n        file = f\"{TRAIN_DIR2}/{image_id}.png\"\n    return file","metadata":{"execution":{"iopub.status.busy":"2023-11-02T10:35:42.31425Z","iopub.execute_input":"2023-11-02T10:35:42.314692Z","iopub.status.idle":"2023-11-02T10:35:42.326908Z","shell.execute_reply.started":"2023-11-02T10:35:42.314646Z","shell.execute_reply":"2023-11-02T10:35:42.325812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CSV files","metadata":{}},{"cell_type":"code","source":"train_images1 = sorted(glob.glob(f\"{TRAIN_DIR1}/*.png\"))\ntrain_images2 = sorted(glob.glob(f\"{TRAIN_DIR2}/*.png\"))\ntrain_images = sorted( train_images1 + train_images2 )","metadata":{"execution":{"iopub.status.busy":"2023-11-02T10:35:42.329704Z","iopub.execute_input":"2023-11-02T10:35:42.330098Z","iopub.status.idle":"2023-11-02T10:35:42.390533Z","shell.execute_reply.started":"2023-11-02T10:35:42.330028Z","shell.execute_reply":"2023-11-02T10:35:42.389442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(f\"{ROOT_DIR}/train.csv\")\ndf['file_path'] = df['image_id'].apply(get_train_file_path)\ndf = df[ df[\"file_path\"].isin(train_images) ].reset_index(drop=True)\ndf","metadata":{"execution":{"iopub.status.busy":"2023-11-02T10:35:42.392242Z","iopub.execute_input":"2023-11-02T10:35:42.393115Z","iopub.status.idle":"2023-11-02T10:35:42.468141Z","shell.execute_reply.started":"2023-11-02T10:35:42.393073Z","shell.execute_reply":"2023-11-02T10:35:42.467032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train image thumbnail","metadata":{}},{"cell_type":"code","source":"def get_cropped_images(image, th_area = 1000):\n    # Aspect ratio\n    as_ratio = image.size[0] / image.size[1]\n    \n    sxs, exs, sys, eys = [],[],[],[]\n    if as_ratio >= 1.5:\n        # Crop\n        mask = np.max( np.array(image) > 0, axis=-1 ).astype(np.uint8)\n        retval, labels = cv2.connectedComponents(mask)\n        if retval >= as_ratio:\n            x, y = np.meshgrid( np.arange(image.size[0]), np.arange(image.size[1]) )\n            for label in range(1, retval):\n                area = np.sum(labels == label)\n                if area < th_area:\n                    continue\n                xs, ys= x[ labels == label ], y[ labels == label ]\n                sx, ex = np.min(xs), np.max(xs)\n                cx = (sx + ex) // 2\n                crop_size = image.size[1]\n                sx = max(0, cx-crop_size//2)\n                ex = min(sx + crop_size - 1, image.size[0]-1)\n                sx = ex - crop_size + 1\n                sy, ey = 0, image.size[1]-1\n                sxs.append(sx)\n                exs.append(ex)\n                sys.append(sy)\n                eys.append(ey)\n        else:\n            crop_size = image.size[1]\n            for i in range(int(as_ratio)):\n                sxs.append( i * crop_size )\n                exs.append( (i+1) * crop_size - 1 )\n                sys.append( 0 )\n                eys.append( crop_size - 1 )\n    else:\n        # Not Crop (entire image)\n        sxs, exs, sys, eys = [0,],[image.size[0]-1],[0,],[image.size[1]-1]\n\n    df_crop = pd.DataFrame()\n    df_crop[\"sx\"] = sxs\n    df_crop[\"ex\"] = exs\n    df_crop[\"sy\"] = sys\n    df_crop[\"ey\"] = eys\n    return df_crop","metadata":{"execution":{"iopub.status.busy":"2023-11-02T10:35:42.469881Z","iopub.execute_input":"2023-11-02T10:35:42.470776Z","iopub.status.idle":"2023-11-02T10:35:42.486867Z","shell.execute_reply.started":"2023-11-02T10:35:42.470713Z","shell.execute_reply":"2023-11-02T10:35:42.48581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p images\ndfs = []\nfor i, row in tqdm(df.iterrows()):\n    file_thumbnail = row.file_path\n    image = Image.open(file_thumbnail)\n    df_crop = get_cropped_images(image)\n    df_crop = df_crop.drop_duplicates(subset=[\"sx\", \"ex\", \"sy\", \"ey\"]).reset_index(drop=True)\n    df_crop[\"image_id\"] = row.image_id\n    df_crop[\"label\"] = row.label\n    df_crop[\"ori_width\"] = image.size[0]\n    df_crop[\"ori_height\"] = image.size[1]\n    cropped_ids = []\n    save_paths = []\n    for j, row2 in df_crop.iterrows():\n        save_path = f\"images/{row.image_id}_{j:02d}.png\"\n        Image.fromarray(np.array(image)[row2.sy:row2.ey, row2.sx:row2.ex, :]).save( save_path )\n        cropped_ids.append(j)\n        save_paths.append(save_path)\n    df_crop[\"crop_id\"] = cropped_ids\n    df_crop[\"file_path\"] = save_paths\n    dfs.append(df_crop)\n\ndfs = pd.concat(dfs).reset_index(drop=True)\ndfs[\"weight\"] = (dfs[\"ex\"]-dfs[\"sx\"]) * (dfs[\"ey\"]-dfs[\"sy\"]) / dfs[\"ori_width\"] / dfs[\"ori_height\"]\ndfs.to_csv(\"train.csv\", index=False)\ndfs","metadata":{"execution":{"iopub.status.busy":"2023-11-02T10:35:42.488528Z","iopub.execute_input":"2023-11-02T10:35:42.488947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}