{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        mm=print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-07T13:11:38.733879Z","iopub.execute_input":"2023-12-07T13:11:38.734432Z","iopub.status.idle":"2023-12-07T13:11:38.764819Z","shell.execute_reply.started":"2023-12-07T13:11:38.734385Z","shell.execute_reply":"2023-12-07T13:11:38.763298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filenames = mm.tolist()\n\nsplit_1 = int(0.8 * len(filenames))\nsplit_2 = int(0.9 * len(filenames))\ntrain_filenames = filenames[:split_1]\ndev_filenames = filenames[split_1:split_2]\ntest_filenames = filenames[split_2:]","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:13:13.459723Z","iopub.execute_input":"2023-12-07T13:13:13.461374Z","iopub.status.idle":"2023-12-07T13:13:13.506832Z","shell.execute_reply.started":"2023-12-07T13:13:13.461323Z","shell.execute_reply":"2023-12-07T13:13:13.504841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import cv2 as cv\n# import numpy as np\n# import matplotlib.pyplot as plt\n# from matplotlib import gridspec\n# import matplotlib.image as img\n\n\n# #Reading the image\n# img = cv.imread('/kaggle/input/UBC-OCEAN/test_images')\n# #img= cv.cvtColor(img, cv.COLOR_BGR2GRAY)\n# plt.figure(figsize=(15, 5))#(15inch, 5inch)= (15*80 pixels, 5*80 pixels)\n# plt.title('input image')\n# plt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T12:11:02.342617Z","iopub.execute_input":"2023-12-07T12:11:02.343235Z","iopub.status.idle":"2023-12-07T12:11:02.758826Z","shell.execute_reply.started":"2023-12-07T12:11:02.343188Z","shell.execute_reply":"2023-12-07T12:11:02.757138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\nimport cv2 as cv\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom matplotlib import gridspec\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T12:08:30.099349Z","iopub.execute_input":"2023-12-07T12:08:30.101025Z","iopub.status.idle":"2023-12-07T12:08:30.129263Z","shell.execute_reply.started":"2023-12-07T12:08:30.100965Z","shell.execute_reply":"2023-12-07T12:08:30.12764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Reading the image\n#img = view_random_image(target_dir= \"/kaggle/input/UBC-OCEAN/train_thumbnails\")\n\nimg = cv.imread('/kaggle/input/UBC-OCEAN/test_thumbnails/41_thumbnail.png')\n#img= cv.cvtColor(img, cv.COLOR_BGR2GRAY)\nplt.figure(figsize=(15, 5))#(15inch, 5inch)= (15*80 pixels, 5*80 pixels)\nplt.title('input image')\nplt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T12:23:24.685831Z","iopub.execute_input":"2023-12-07T12:23:24.686677Z","iopub.status.idle":"2023-12-07T12:23:26.48212Z","shell.execute_reply.started":"2023-12-07T12:23:24.686636Z","shell.execute_reply":"2023-12-07T12:23:26.480924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img.shape","metadata":{"execution":{"iopub.status.busy":"2023-12-07T12:17:31.870388Z","iopub.execute_input":"2023-12-07T12:17:31.871895Z","iopub.status.idle":"2023-12-07T12:17:31.879255Z","shell.execute_reply.started":"2023-12-07T12:17:31.871838Z","shell.execute_reply":"2023-12-07T12:17:31.877948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Train_csv = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\n\nTrain_csv","metadata":{"execution":{"iopub.status.busy":"2023-12-07T12:17:35.925985Z","iopub.execute_input":"2023-12-07T12:17:35.926558Z","iopub.status.idle":"2023-12-07T12:17:35.977834Z","shell.execute_reply.started":"2023-12-07T12:17:35.926497Z","shell.execute_reply":"2023-12-07T12:17:35.976417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Train_csv['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-12-07T12:17:40.10696Z","iopub.execute_input":"2023-12-07T12:17:40.107508Z","iopub.status.idle":"2023-12-07T12:17:40.125361Z","shell.execute_reply.started":"2023-12-07T12:17:40.107464Z","shell.execute_reply":"2023-12-07T12:17:40.123636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Train_csv[\"is_tma\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-12-07T12:17:46.354293Z","iopub.execute_input":"2023-12-07T12:17:46.354781Z","iopub.status.idle":"2023-12-07T12:17:46.368348Z","shell.execute_reply.started":"2023-12-07T12:17:46.354744Z","shell.execute_reply":"2023-12-07T12:17:46.366748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#img_normalized = cv.normalize(img, None, 0, 1.0,cv.NORM_MINMAX, dtype=cv.CV_32f)\nimg_normalized = cv.normalize(img, None, 0, 255,cv.NORM_MINMAX, dtype=cv.CV_8U)\nplt.figure(figsize=(15, 5))\nplt.title('Normalized Image')\nplt.imshow(img_normalized)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T12:17:50.712322Z","iopub.execute_input":"2023-12-07T12:17:50.712898Z","iopub.status.idle":"2023-12-07T12:17:52.371367Z","shell.execute_reply.started":"2023-12-07T12:17:50.712858Z","shell.execute_reply":"2023-12-07T12:17:52.370244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train='/kaggle/input/UBC-OCEAN/train_images'\ntrain_f=os.listdir(train)\nprint(len(train_f))","metadata":{"execution":{"iopub.status.busy":"2023-12-07T12:17:58.39179Z","iopub.execute_input":"2023-12-07T12:17:58.392294Z","iopub.status.idle":"2023-12-07T12:17:58.401342Z","shell.execute_reply.started":"2023-12-07T12:17:58.392257Z","shell.execute_reply":"2023-12-07T12:17:58.399617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train='/kaggle/input/UBC-OCEAN/train_thumbnails'\ntrain_f=os.listdir(train)\nprint(len(train_f))","metadata":{"execution":{"iopub.status.busy":"2023-12-07T12:18:33.555831Z","iopub.execute_input":"2023-12-07T12:18:33.556355Z","iopub.status.idle":"2023-12-07T12:18:33.566028Z","shell.execute_reply.started":"2023-12-07T12:18:33.556318Z","shell.execute_reply":"2023-12-07T12:18:33.564501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test='/kaggle/input/UBC-OCEAN/test_thumbnails'\ntest_f=os.listdir(test)\nprint(len(test_f))","metadata":{"execution":{"iopub.status.busy":"2023-12-07T12:18:41.806865Z","iopub.execute_input":"2023-12-07T12:18:41.807388Z","iopub.status.idle":"2023-12-07T12:18:41.816512Z","shell.execute_reply.started":"2023-12-07T12:18:41.807347Z","shell.execute_reply":"2023-12-07T12:18:41.814823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nnb_samples = 6\nn, m = len(np.unique(df_train['label'])), nb_samples,\nfig, axarr = plt.subplots(nrows=n, ncols=m, figsize=(m * 2, n * 2))\nfor ilb, (lb, df_) in enumerate(df_train.groupby('label')):\n    img_ids = list(df_['image_id'])\n    for i in range(m):\n        if i == 0:\n            axarr[ilb, i].set_title(f\"{lb} #{len(df_)}\")\n        ls_imgs = glob.glob(os.path.join(DATASET_IMAGES, str(img_ids[i]), \"*.png\"))\n        img_path = ls_imgs[0]\n        img = plt.imread(img_path)\n        mask = np.sum(img[..., :3], axis=2) == 0\n        img[mask, :] = 255\n        axarr[ilb, i].imshow(img)\n        # axarr[ilb, i].set_xticks([])\n        # axarr[ilb, i].set_yticks([])\n_= plt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2023-12-07T12:19:01.821142Z","iopub.execute_input":"2023-12-07T12:19:01.821721Z","iopub.status.idle":"2023-12-07T12:19:01.872354Z","shell.execute_reply.started":"2023-12-07T12:19:01.821683Z","shell.execute_reply":"2023-12-07T12:19:01.870699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### import os \nimport shutil \nfrom sklearn.model_selection import train_test_split \n\n# Define the paths to our data \ndata_dir = \"/kaggle/input/UBC-OCEAN\"\ntrain_dir = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\nvalid_dir = \"/kaggle/working/valid\"\ntest_dir =\"/kaggle/input/UBC-OCEAN/test_thumbnails\"\n# Create the train and valid directories if they don't exist\nos.makedirs(train_dir, exist_ok=True)\nos.makedirs(valid_dir, exist_ok=True)\n\n# Perform the split\nfor class_name in os.listdir(data_dir):\n    class_path = os.path.join(data_dir, class_name)\n    train_path = os.path.join(train_dir, class_name)\n    valid_path = os.path.join(valid_dir, class_name)\n\n    os.makedirs(train_path, exist_ok=True)\n    os.makedirs(valid_path, exist_ok=True)\n\n    images = os.listdir(class_path)\n    train_images, valid_images = train_test_split(images, test_size=0.2, random_state=42)\n\n    for img in train_images:\n        shutil.copy(os.path.join(class_path, img), os.path.join(train_path, img))\n\n    for img in valid_images:\n        shutil.copy(os.path.join(class_path, img), os.path.join(valid_path, img))","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:46:26.84328Z","iopub.execute_input":"2023-12-07T13:46:26.844187Z","iopub.status.idle":"2023-12-07T13:46:26.915657Z","shell.execute_reply.started":"2023-12-07T13:46:26.844137Z","shell.execute_reply":"2023-12-07T13:46:26.914459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\n\n# Define the source and destination directories\nsource_dir = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\ndestination_dir = \"/kaggle/working/train_thumbnails_classified\"\n\n# Iterate over rows in train_df and move images to their respective class directories\nfor index, row in train_df.iterrows():\n    image_id = row[\"image_id\"]\n    class_name = row[\"label\"]\n    \n    # Construct source and destination paths with correct file extension\n    source_path = os.path.join(source_dir, f\"{image_id}_thumbnail.png\")\n    destination_path = os.path.join(destination_dir, class_name, f\"{image_id}_thumbnail.png\")\n    \n    # Check if the source file exists\n    if os.path.exists(source_path):\n        # Check if the destination directory exists, create if not\n        os.makedirs(os.path.dirname(destination_path), exist_ok=True)\n        \n        # Print for debugging\n        print(f\"Copying from {source_path} to {destination_path}\")\n        \n        # Use shutil.copy to copy the file\n        shutil.copy(source_path, destination_path)\n    else:\n        print(f\"Source file {source_path} not found. Skipping.\")\n\nprint(\"Images successfully classified.\")","metadata":{"execution":{"iopub.status.busy":"2023-12-07T12:35:18.362416Z","iopub.execute_input":"2023-12-07T12:35:18.363347Z","iopub.status.idle":"2023-12-07T12:35:18.420547Z","shell.execute_reply.started":"2023-12-07T12:35:18.363294Z","shell.execute_reply":"2023-12-07T12:35:18.418815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport pandas as pd\ndf = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\nx = df.iloc[:,0:4].values\ny = df.iloc[:,0:1].values\ntrain_x, val_x, train_y, val_y = train_test_split(x, y, random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:53:56.159622Z","iopub.execute_input":"2023-12-07T13:53:56.160182Z","iopub.status.idle":"2023-12-07T13:53:56.178184Z","shell.execute_reply.started":"2023-12-07T13:53:56.160147Z","shell.execute_reply":"2023-12-07T13:53:56.176786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"val_y shape: {val_y.shape}\")\nprint(f\"train_x shape: {train_x.shape}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:54:18.269333Z","iopub.execute_input":"2023-12-07T13:54:18.270171Z","iopub.status.idle":"2023-12-07T13:54:18.278174Z","shell.execute_reply.started":"2023-12-07T13:54:18.270111Z","shell.execute_reply":"2023-12-07T13:54:18.277072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = cv.imread('/kaggle/input/UBC-OCEAN/train_thumbnails/10077_thumbnail.png')\n#img= cv.cvtColor(img, cv.COLOR_BGR2GRAY)\nplt.figure(figsize=(15, 5))#(15inch, 5inch)= (15*80 pixels, 5*80 pixels)\nplt.title('input image')\nplt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T13:56:21.579792Z","iopub.execute_input":"2023-12-07T13:56:21.580292Z","iopub.status.idle":"2023-12-07T13:56:23.663836Z","shell.execute_reply.started":"2023-12-07T13:56:21.580259Z","shell.execute_reply":"2023-12-07T13:56:23.662378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}