{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":7071372,"sourceType":"datasetVersion","datasetId":4072329}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport shutil\n\nimport numpy as np\nimport pandas as pd\nimport torch\nimport matplotlib.pyplot as plt\n\nfrom datasets import load_dataset\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T12:50:43.115116Z","iopub.execute_input":"2023-11-27T12:50:43.115752Z","iopub.status.idle":"2023-11-27T12:50:48.782905Z","shell.execute_reply.started":"2023-11-27T12:50:43.115721Z","shell.execute_reply":"2023-11-27T12:50:48.781793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/test.csv\")\n\nBASE_DIR = [\"/kaggle/input/UBC-OCEAN/train_thumbnails/\", \"/kaggle/input/UBC-OCEAN/test_thumbnails/\"]","metadata":{"execution":{"iopub.status.busy":"2023-11-27T12:51:05.046831Z","iopub.execute_input":"2023-11-27T12:51:05.048296Z","iopub.status.idle":"2023-11-27T12:51:05.070632Z","shell.execute_reply.started":"2023-11-27T12:51:05.048251Z","shell.execute_reply":"2023-11-27T12:51:05.069316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def organize_images_by_label(df: pd.DataFrame, source_dir: str) -> None:\n    image_paths=[]\n    for _, row in df.iterrows():\n        image_id = row[\"image_id\"]\n        label = row[\"label\"]\n        source_path = os.path.join(source_dir, f\"{image_id}_thumbnail.png\") \n        try:\n            image_paths.append(os.path.join(source_dir, f\"{image_id}_thumbnail.png\"))\n        except FileNotFoundError:\n            image_paths.append(1)\n            continue\n    return image_paths\n\n\nimage_paths = organize_images_by_label(train_df, BASE_DIR[0])\ntrain_df['image_path'] = image_paths\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T12:51:20.930118Z","iopub.execute_input":"2023-11-27T12:51:20.930508Z","iopub.status.idle":"2023-11-27T12:51:20.992752Z","shell.execute_reply.started":"2023-11-27T12:51:20.930479Z","shell.execute_reply":"2023-11-27T12:51:20.991901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.signal import find_peaks, savgol_filter\nfrom PIL import Image\nimport cv2\nimport numpy as np\n\n# Randomly choose 5 images from each group\nrandom_5_images = train_df[train_df['image_path']!=1].groupby('label').head(5).sort_values('label')","metadata":{"execution":{"iopub.status.busy":"2023-11-27T12:51:37.591922Z","iopub.execute_input":"2023-11-27T12:51:37.592538Z","iopub.status.idle":"2023-11-27T12:51:38.290522Z","shell.execute_reply.started":"2023-11-27T12:51:37.592508Z","shell.execute_reply":"2023-11-27T12:51:38.289349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def smooth_signal(signal, window_length=5, polyorder=3):\n    return savgol_filter(signal, window_length, polyorder)","metadata":{"execution":{"iopub.status.busy":"2023-11-27T12:51:56.130754Z","iopub.execute_input":"2023-11-27T12:51:56.131199Z","iopub.status.idle":"2023-11-27T12:51:56.137128Z","shell.execute_reply.started":"2023-11-27T12:51:56.131165Z","shell.execute_reply":"2023-11-27T12:51:56.135901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def detect_peaks(signal, prominence=150):  # Adjust prominence as needed\n    peaks, _ = find_peaks(signal, prominence=prominence)\n    return peaks","metadata":{"execution":{"iopub.status.busy":"2023-11-27T12:52:06.820765Z","iopub.execute_input":"2023-11-27T12:52:06.821204Z","iopub.status.idle":"2023-11-27T12:52:06.826904Z","shell.execute_reply.started":"2023-11-27T12:52:06.821173Z","shell.execute_reply":"2023-11-27T12:52:06.825607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_image(input_image):\n    image = cv2.imread(input_image)\n\n    # Convert to grayscale\n    gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n\n    # Threshold the image to create a binary image\n    _, binary = cv2.threshold(gray, 1, 255, cv2.THRESH_BINARY)\n\n    # Find the contours (regions) in the binary image\n    contours, _ = cv2.findContours(binary, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n\n    # Extract bounding boxes around each contour and save them as individual images\n    cropped_images=[]\n    for i, contour in enumerate(contours):\n        x, y, w, h = cv2.boundingRect(contour)\n        cropped_image = image[y:y+h, x:x+w]\n        if np.sum(np.array(cropped_image))>100000:\n            cropped_images.append(cropped_image)\n    return cropped_images","metadata":{"execution":{"iopub.status.busy":"2023-11-27T12:52:12.970861Z","iopub.execute_input":"2023-11-27T12:52:12.971318Z","iopub.status.idle":"2023-11-27T12:52:12.979568Z","shell.execute_reply.started":"2023-11-27T12:52:12.971275Z","shell.execute_reply":"2023-11-27T12:52:12.978385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Signal Scaling\nscale_signal scales a signal to a specified range. This can be useful for visualization purposes.","metadata":{}},{"cell_type":"code","source":"def scale_signal(signal, min_range, max_range):\n    # Find the minimum and maximum of the signal\n    min_value = np.min(signal)\n    max_value = np.max(signal)\n    \n    # Scale the signal\n    scaled_signal = min_range + (signal - min_value) / (max_value - min_value) * (max_range - min_range)\n    \n    return scaled_signal","metadata":{"execution":{"iopub.status.busy":"2023-11-27T12:52:19.710241Z","iopub.execute_input":"2023-11-27T12:52:19.71065Z","iopub.status.idle":"2023-11-27T12:52:19.717318Z","shell.execute_reply.started":"2023-11-27T12:52:19.710613Z","shell.execute_reply":"2023-11-27T12:52:19.716138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\n\n'''i = 0\nwhile i < len(bright_images) - 1:\n    cv2.imwrite(f'path_to/image_{i}.png', bright_images[i])\n    i += 1'''\n\nfor i, row in random_5_images.iterrows():\n    if os.path.isfile(row['image_path']):\n        image = Image.open(row['image_path'])\n        smoothed_signal = smooth_signal(np.mean(np.array(image)[:,:,0], axis=0))\n        peaks = detect_peaks(smoothed_signal)\n        if len(peaks)>1: #Check if there is more than one image\n            image_path = row['image_path']\n            print(image_path)\n            thumb = image_path.split(\"/\")[5]\n            thumb_img = thumb.split(\".\")[0]\n            print(thumb_img)\n            images = crop_image(image_path)\n            for i in images:\n                #pil_image = Image.fromarray(i[0],'RGB')\n                cv2.imwrite(f'/kaggle/working/thumb_img.png',i[0])\n            fig, axarr = plt.subplots(1, 2+len(images), figsize=(3*(2+len(images)),3))\n            fig.suptitle(\"Image ID: {}, Label: {}\".format(row['image_id'], row['label']))\n            for i, ax in enumerate(axarr.ravel()):\n                if i==0:\n                    ax.imshow(np.array(image))\n                    ax.axis('off')\n                    ax.set_title(\"Original Image\")\n                elif i==1:\n                    _range=axarr[0].get_ylim()\n                    ax.plot(scale_signal(smoothed_signal, _range[1],  _range[0]))\n                    ax.axis('off')\n                    ax.set_title(\"1D Signal of original Image\")\n                else:\n                    ax.imshow(np.array(images)[i-2])\n                    ax.axis('off')\n                    ax.set_title(\"Cropped Image {}\".format(i-2))\n            plt.tight_layout()\n            plt.show()\n            \n#         else:\n#             # Use this if you want to visualize non-multiple images\n#             image_path = row['image_path']  \n#             images = crop_image(image_path)\n#             fig, axarr = plt.subplots(1, 2, figsize=(10*(2),10))\n        \n#             for i, ax in enumerate(axarr.ravel()):\n#                 if i==0:\n#                     ax.imshow(np.array(image))\n#                 elif i==1:\n#                     ax.plot(smoothed_signal)\n#             plt.tight_layout()\n#             plt.show()\n#         ax.imshow(image)\n#         ax.set_title(len(peaks))\n#         ax.axis('off')  # To hide axis labels\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T14:01:11.818647Z","iopub.execute_input":"2023-11-27T14:01:11.819105Z","iopub.status.idle":"2023-11-27T14:01:29.279614Z","shell.execute_reply.started":"2023-11-27T14:01:11.81907Z","shell.execute_reply":"2023-11-27T14:01:29.278362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# End of the Notebook","metadata":{}}]}