{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport shutil\nimport numpy as np\nimport pandas as pd\nimport torch\nimport matplotlib.pyplot as plt\nfrom datasets import load_dataset","metadata":{"execution":{"iopub.status.busy":"2023-11-29T06:51:05.937717Z","iopub.execute_input":"2023-11-29T06:51:05.938189Z","iopub.status.idle":"2023-11-29T06:51:12.74762Z","shell.execute_reply.started":"2023-11-29T06:51:05.938153Z","shell.execute_reply":"2023-11-29T06:51:12.746552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/test.csv\")\nBASE_DIR = [\"/kaggle/input/UBC-OCEAN/train_thumbnails/\", \"/kaggle/input/UBC-OCEAN/test_thumbnails/\"]","metadata":{"execution":{"iopub.status.busy":"2023-11-29T06:51:12.749516Z","iopub.execute_input":"2023-11-29T06:51:12.750126Z","iopub.status.idle":"2023-11-29T06:51:12.775908Z","shell.execute_reply.started":"2023-11-29T06:51:12.750091Z","shell.execute_reply":"2023-11-29T06:51:12.774668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-11-29T06:52:21.819095Z","iopub.execute_input":"2023-11-29T06:52:21.819543Z","iopub.status.idle":"2023-11-29T06:52:21.841946Z","shell.execute_reply.started":"2023-11-29T06:52:21.81951Z","shell.execute_reply":"2023-11-29T06:52:21.840945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def organize_images_by_label(df: pd.DataFrame, source_dir: str) -> None:\n    image_paths=[]\n    for _, row in df.iterrows():\n        image_id = row[\"image_id\"]\n        label = row[\"label\"]\n        source_path = os.path.join(source_dir, f\"{image_id}_thumbnail.png\") \n        try:\n            image_paths.append(os.path.join(source_dir, f\"{image_id}_thumbnail.png\"))\n        except FileNotFoundError:\n            image_paths.append(1)\n            continue\n    return image_paths\n\nimage_paths = organize_images_by_label(train_df, BASE_DIR[0])\ntrain_df['image_path'] = image_paths","metadata":{"execution":{"iopub.status.busy":"2023-11-29T06:53:36.013226Z","iopub.execute_input":"2023-11-29T06:53:36.013651Z","iopub.status.idle":"2023-11-29T06:53:36.075964Z","shell.execute_reply.started":"2023-11-29T06:53:36.013619Z","shell.execute_reply":"2023-11-29T06:53:36.07456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2023-11-29T06:53:37.221628Z","iopub.execute_input":"2023-11-29T06:53:37.222199Z","iopub.status.idle":"2023-11-29T06:53:37.241157Z","shell.execute_reply.started":"2023-11-29T06:53:37.222161Z","shell.execute_reply":"2023-11-29T06:53:37.239578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_files_in_folder(folder_path):\n    try:\n        # 使用 os 模块的 listdir 方法列出文件夹中的所有文件和文件夹\n        files = os.listdir(folder_path)\n\n        # 使用 len 函数获取文件的数量\n        num_files = len(files)\n\n        print(f\"文件夹 '{folder_path}' 中的文件数量为: {num_files}\")\n    except FileNotFoundError:\n        print(f\"文件夹 '{folder_path}' 不存在\")\n    except Exception as e:\n        print(f\"发生错误: {e}\")\n\nfolder_path = '/kaggle/input/UBC-OCEAN/train_thumbnails'\ncount_files_in_folder(folder_path)","metadata":{"execution":{"iopub.status.busy":"2023-11-29T06:58:45.35428Z","iopub.execute_input":"2023-11-29T06:58:45.354886Z","iopub.status.idle":"2023-11-29T06:58:45.401638Z","shell.execute_reply.started":"2023-11-29T06:58:45.354829Z","shell.execute_reply":"2023-11-29T06:58:45.400601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.signal import find_peaks, savgol_filter\nfrom PIL import Image\nimport cv2\nimport numpy as np\nrandom_5_images = train_df[train_df['image_path']!=1].groupby('label').head(5).sort_values('label')\nrandom_5_images","metadata":{"execution":{"iopub.status.busy":"2023-11-29T06:53:52.173637Z","iopub.execute_input":"2023-11-29T06:53:52.174118Z","iopub.status.idle":"2023-11-29T06:53:52.198768Z","shell.execute_reply.started":"2023-11-29T06:53:52.174081Z","shell.execute_reply":"2023-11-29T06:53:52.197326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def smooth_signal(signal, window_length=5, polyorder=3):\n    return savgol_filter(signal, window_length, polyorder)","metadata":{"execution":{"iopub.status.busy":"2023-11-29T06:54:14.1929Z","iopub.execute_input":"2023-11-29T06:54:14.193378Z","iopub.status.idle":"2023-11-29T06:54:14.200453Z","shell.execute_reply.started":"2023-11-29T06:54:14.193342Z","shell.execute_reply":"2023-11-29T06:54:14.198285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def detect_peaks(signal, prominence=150):  # Adjust prominence as needed\n    peaks, _ = find_peaks(signal, prominence=prominence)\n    return peaks","metadata":{"execution":{"iopub.status.busy":"2023-11-29T06:54:23.68749Z","iopub.execute_input":"2023-11-29T06:54:23.688113Z","iopub.status.idle":"2023-11-29T06:54:23.694399Z","shell.execute_reply.started":"2023-11-29T06:54:23.688075Z","shell.execute_reply":"2023-11-29T06:54:23.692949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Image Cropping\nThe crop_image function processes the image to detect regions of interest (based on contours) and crops these regions out.\n\n","metadata":{}},{"cell_type":"code","source":"def crop_image(input_image):\n    image = cv2.imread(input_image)\n\n    # Convert to grayscale\n    gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n\n    # Threshold the image to create a binary image\n    _, binary = cv2.threshold(gray, 1, 255, cv2.THRESH_BINARY)\n\n    # Find the contours (regions) in the binary image\n    contours, _ = cv2.findContours(binary, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n\n    # Extract bounding boxes around each contour and save them as individual images\n    cropped_images=[]\n    for i, contour in enumerate(contours):\n        x, y, w, h = cv2.boundingRect(contour)\n        cropped_image = image[y:y+h, x:x+w]\n        if np.sum(np.array(cropped_image))>100000:\n            cropped_images.append(cropped_image)\n    return cropped_images","metadata":{"execution":{"iopub.status.busy":"2023-11-29T06:54:32.190603Z","iopub.execute_input":"2023-11-29T06:54:32.192033Z","iopub.status.idle":"2023-11-29T06:54:32.201532Z","shell.execute_reply.started":"2023-11-29T06:54:32.191988Z","shell.execute_reply":"2023-11-29T06:54:32.199723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def scale_signal(signal, min_range, max_range):\n    # Find the minimum and maximum of the signal\n    min_value = np.min(signal)\n    max_value = np.max(signal)\n    \n    # Scale the signal\n    scaled_signal = min_range + (signal - min_value) / (max_value - min_value) * (max_range - min_range)\n    \n    return scaled_signal","metadata":{"execution":{"iopub.status.busy":"2023-11-29T06:54:39.621715Z","iopub.execute_input":"2023-11-29T06:54:39.623455Z","iopub.status.idle":"2023-11-29T06:54:39.631231Z","shell.execute_reply.started":"2023-11-29T06:54:39.623405Z","shell.execute_reply":"2023-11-29T06:54:39.629762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, row in random_5_images.iterrows():\n    if os.path.isfile(row['image_path']):\n        image = Image.open(row['image_path'])\n        smoothed_signal = smooth_signal(np.mean(np.array(image)[:,:,0], axis=0))\n        peaks = detect_peaks(smoothed_signal)\n        if len(peaks)>1: #Check if there is more than one image\n            image_path = row['image_path']  \n            images = crop_image(image_path)\n            fig, axarr = plt.subplots(1, 2+len(images), figsize=(3*(2+len(images)),3))\n            fig.suptitle(\"Image ID: {}, Label: {}\".format(row['image_id'], row['label']))\n            for i, ax in enumerate(axarr.ravel()):\n                if i==0:\n                    ax.imshow(np.array(image))\n                    ax.axis('off')\n                    ax.set_title(\"Original Image\")\n                elif i==1:\n                    _range=axarr[0].get_ylim()\n                    ax.plot(scale_signal(smoothed_signal, _range[1],  _range[0]))\n                    ax.axis('off')\n                    ax.set_title(\"1D Signal of original Image\")\n                else:\n                    ax.imshow(np.array(images)[i-2])\n                    ax.axis('off')\n                    ax.set_title(\"Cropped Image {}\".format(i-2))\n            plt.tight_layout()\n            plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-29T06:54:44.11981Z","iopub.execute_input":"2023-11-29T06:54:44.120247Z","iopub.status.idle":"2023-11-29T06:55:05.003022Z","shell.execute_reply.started":"2023-11-29T06:54:44.120216Z","shell.execute_reply":"2023-11-29T06:55:05.001416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# End of the Notebook","metadata":{}}]}