{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfrom glob import glob\nimport zipfile\nfrom typing import Optional\nfrom tqdm import tqdm\nimport random\nimport math\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport PIL\nfrom PIL import Image\nPIL.Image.MAX_IMAGE_PIXELS = None","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-02T05:24:40.265458Z","iopub.execute_input":"2023-12-02T05:24:40.265952Z","iopub.status.idle":"2023-12-02T05:24:42.445298Z","shell.execute_reply.started":"2023-12-02T05:24:40.265908Z","shell.execute_reply":"2023-12-02T05:24:42.443864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Read metadata\ndf = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\n\n# Convert 'image_id' values to strings\ndf['image_id'] = df['image_id'].astype(str)\n\n# Assuming that the 'image_id' column contains the unique identifiers for the images\ndf['path'] = '/kaggle/input/UBC-OCEAN/train_images/' + df['image_id'] + '.png'\n\n# Display the DataFrame with the new 'path' column\ndf.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-02T05:29:08.118603Z","iopub.execute_input":"2023-12-02T05:29:08.119037Z","iopub.status.idle":"2023-12-02T05:29:08.144391Z","shell.execute_reply.started":"2023-12-02T05:29:08.118997Z","shell.execute_reply":"2023-12-02T05:29:08.143551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Separate the data based on unique labels\nlabel_groups = df.groupby('label')\n\n# Save each group to a CSV file\noutput_directory = '/kaggle/working/labeled_data'\n\n# Create the output directory if it doesn't exist\nos.makedirs(output_directory, exist_ok=True)\n\nfor label, group in label_groups:\n    # Define the output path for each group\n    output_path = os.path.join(output_directory, f'label_{label}_data.csv')\n    \n    # Save the group to the CSV file\n    group.to_csv(output_path, index=False)\n\n    print(f\"Saved data for label {label} to {output_path}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-02T05:32:13.884713Z","iopub.execute_input":"2023-12-02T05:32:13.885161Z","iopub.status.idle":"2023-12-02T05:32:13.917583Z","shell.execute_reply.started":"2023-12-02T05:32:13.885128Z","shell.execute_reply":"2023-12-02T05:32:13.916194Z"},"trusted":true},"execution_count":null,"outputs":[]}]}