{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, glob, gc\nimport cv2\nimport random\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nfrom PIL import Image\nImage.MAX_IMAGE_PIXELS = None\n\nfrom collections import Counter\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-04T13:15:36.322917Z","iopub.execute_input":"2023-11-04T13:15:36.323328Z","iopub.status.idle":"2023-11-04T13:15:36.925513Z","shell.execute_reply.started":"2023-11-04T13:15:36.323292Z","shell.execute_reply":"2023-11-04T13:15:36.92461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_DIR = \"/kaggle/input/UBC-OCEAN\"\nTRAIN_DIR1 = '/kaggle/input/ubc-resize-images-2048pix/train_images'","metadata":{"execution":{"iopub.status.busy":"2023-11-04T13:15:36.926987Z","iopub.execute_input":"2023-11-04T13:15:36.927942Z","iopub.status.idle":"2023-11-04T13:15:36.932184Z","shell.execute_reply.started":"2023-11-04T13:15:36.927911Z","shell.execute_reply":"2023-11-04T13:15:36.931134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FILES = sorted(glob.glob(\"/kaggle/input/ubc-resize-images-2048pix-part*/train_images/*.png\"))\nID2FILE = {\n    int(os.path.basename(file).split(\".\")[0]) : file\n    for file in FILES\n}","metadata":{"execution":{"iopub.status.busy":"2023-11-04T13:15:36.93348Z","iopub.execute_input":"2023-11-04T13:15:36.933807Z","iopub.status.idle":"2023-11-04T13:15:37.029025Z","shell.execute_reply.started":"2023-11-04T13:15:36.933779Z","shell.execute_reply":"2023-11-04T13:15:37.028216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_file_path(image_id):\n    return ID2FILE[image_id]","metadata":{"execution":{"iopub.status.busy":"2023-11-04T13:15:37.030841Z","iopub.execute_input":"2023-11-04T13:15:37.03137Z","iopub.status.idle":"2023-11-04T13:15:37.035125Z","shell.execute_reply.started":"2023-11-04T13:15:37.03134Z","shell.execute_reply":"2023-11-04T13:15:37.034373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CSV files","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(f\"{ROOT_DIR}/train.csv\")\ndf['file_path'] = df['image_id'].apply(get_train_file_path)\ndf","metadata":{"execution":{"iopub.status.busy":"2023-11-04T13:15:37.03643Z","iopub.execute_input":"2023-11-04T13:15:37.036974Z","iopub.status.idle":"2023-11-04T13:15:37.091167Z","shell.execute_reply.started":"2023-11-04T13:15:37.036946Z","shell.execute_reply":"2023-11-04T13:15:37.090038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train image thumbnail","metadata":{}},{"cell_type":"code","source":"def get_cropped_images(image, th_area = 1000):\n    # Aspect ratio\n    as_ratio = image.size[0] / image.size[1]\n    \n    sxs, exs, sys, eys = [],[],[],[]\n    if as_ratio >= 1.5:\n        # Crop\n        mask = np.max( np.array(image) > 0, axis=-1 ).astype(np.uint8)\n        retval, labels = cv2.connectedComponents(mask)\n        if retval >= as_ratio:\n            x, y = np.meshgrid( np.arange(image.size[0]), np.arange(image.size[1]) )\n            for label in range(1, retval):\n                area = np.sum(labels == label)\n                if area < th_area:\n                    continue\n                xs, ys= x[ labels == label ], y[ labels == label ]\n                sx, ex = np.min(xs), np.max(xs)\n                cx = (sx + ex) // 2\n                crop_size = image.size[1]\n                sx = max(0, cx-crop_size//2)\n                ex = min(sx + crop_size - 1, image.size[0]-1)\n                sx = ex - crop_size + 1\n                sy, ey = 0, image.size[1]-1\n                sxs.append(sx)\n                exs.append(ex)\n                sys.append(sy)\n                eys.append(ey)\n        else:\n            crop_size = image.size[1]\n            for i in range(int(as_ratio)):\n                sxs.append( i * crop_size )\n                exs.append( (i+1) * crop_size - 1 )\n                sys.append( 0 )\n                eys.append( crop_size - 1 )\n    else:\n        # Not Crop (entire image)\n        sxs, exs, sys, eys = [0,],[image.size[0]-1],[0,],[image.size[1]-1]\n\n    df_crop = pd.DataFrame()\n    df_crop[\"sx\"] = sxs\n    df_crop[\"ex\"] = exs\n    df_crop[\"sy\"] = sys\n    df_crop[\"ey\"] = eys\n    return df_crop","metadata":{"execution":{"iopub.status.busy":"2023-11-04T13:15:37.092942Z","iopub.execute_input":"2023-11-04T13:15:37.093738Z","iopub.status.idle":"2023-11-04T13:15:37.106121Z","shell.execute_reply.started":"2023-11-04T13:15:37.093698Z","shell.execute_reply":"2023-11-04T13:15:37.105272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p images\ndfs = []\nfor (file_path, image_id, label) in zip(tqdm(df[\"file_path\"]), df[\"image_id\"], df[\"label\"]):\n    image = Image.open(file_path)\n    df_crop = get_cropped_images(image)\n    df_crop = df_crop.drop_duplicates(subset=[\"sx\", \"ex\", \"sy\", \"ey\"]).reset_index(drop=True)\n    df_crop[\"image_id\"] = image_id\n    df_crop[\"label\"] = label\n    df_crop[\"ori_width\"] = image.size[0]\n    df_crop[\"ori_height\"] = image.size[1]\n    cropped_ids = []\n    save_paths = []\n    for j, row2 in df_crop.iterrows():\n        save_path = f\"images/{image_id}_{j:02d}.png\"\n        Image.fromarray(np.array(image)[row2.sy:row2.ey, row2.sx:row2.ex, :]).save( save_path )\n        cropped_ids.append(j)\n        save_paths.append(save_path)\n    df_crop[\"crop_id\"] = cropped_ids\n    df_crop[\"file_path\"] = save_paths\n    dfs.append(df_crop)\n\ndfs = pd.concat(dfs).reset_index(drop=True)\ndfs[\"weight\"] = (dfs[\"ex\"]-dfs[\"sx\"]) * (dfs[\"ey\"]-dfs[\"sy\"]) / dfs[\"ori_width\"] / dfs[\"ori_height\"]\ndfs.to_csv(\"train.csv\", index=False)\ndfs","metadata":{"execution":{"iopub.status.busy":"2023-11-04T13:15:37.107852Z","iopub.execute_input":"2023-11-04T13:15:37.108636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}