{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom skimage import io\nimport os\nimport cv2\nimport random\nimport os\nimport glob","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-07T07:41:17.68092Z","iopub.execute_input":"2023-11-07T07:41:17.681306Z","iopub.status.idle":"2023-11-07T07:41:19.15498Z","shell.execute_reply.started":"2023-11-07T07:41:17.681274Z","shell.execute_reply":"2023-11-07T07:41:19.153682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')\ntest_data","metadata":{"execution":{"iopub.status.busy":"2023-11-07T07:41:19.157213Z","iopub.execute_input":"2023-11-07T07:41:19.158663Z","iopub.status.idle":"2023-11-07T07:41:19.193062Z","shell.execute_reply.started":"2023-11-07T07:41:19.158618Z","shell.execute_reply":"2023-11-07T07:41:19.191966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data= pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2023-11-07T07:41:19.194687Z","iopub.execute_input":"2023-11-07T07:41:19.195384Z","iopub.status.idle":"2023-11-07T07:41:19.223532Z","shell.execute_reply.started":"2023-11-07T07:41:19.195345Z","shell.execute_reply":"2023-11-07T07:41:19.222413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(os.listdir('/kaggle/input/UBC-OCEAN/train_images')))\nprint(len(os.listdir('/kaggle/input/UBC-OCEAN/train_thumbnails')))","metadata":{"execution":{"iopub.status.busy":"2023-11-07T07:41:19.228383Z","iopub.execute_input":"2023-11-07T07:41:19.228714Z","iopub.status.idle":"2023-11-07T07:41:19.334248Z","shell.execute_reply.started":"2023-11-07T07:41:19.228685Z","shell.execute_reply":"2023-11-07T07:41:19.332977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def append_ext(fn):\n    return str(fn) + \".png\"\n\ndef append_ext_thum(fn):\n    return str(fn) + \"_thumbnail.png\"\n\n\ntrain_data[\"image_id_path\"]=train_data[\"image_id\"].apply(append_ext)\ntrain_data[\"image_id_path_thum\"]=train_data[\"image_id\"].apply(append_ext_thum)\n\n\ntest_data[\"image_id_path\"]=test_data[\"image_id\"].apply(append_ext)\ntest_data[\"image_id_path_thum\"]=test_data[\"image_id\"].apply(append_ext_thum)","metadata":{"execution":{"iopub.status.busy":"2023-11-07T07:41:19.336013Z","iopub.execute_input":"2023-11-07T07:41:19.336799Z","iopub.status.idle":"2023-11-07T07:41:19.350032Z","shell.execute_reply.started":"2023-11-07T07:41:19.336759Z","shell.execute_reply":"2023-11-07T07:41:19.34894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data","metadata":{"execution":{"iopub.status.busy":"2023-11-07T07:41:19.351217Z","iopub.execute_input":"2023-11-07T07:41:19.352076Z","iopub.status.idle":"2023-11-07T07:41:19.374591Z","shell.execute_reply.started":"2023-11-07T07:41:19.352038Z","shell.execute_reply":"2023-11-07T07:41:19.373302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2023-11-07T07:41:19.376453Z","iopub.execute_input":"2023-11-07T07:41:19.377136Z","iopub.status.idle":"2023-11-07T07:41:19.393456Z","shell.execute_reply.started":"2023-11-07T07:41:19.377096Z","shell.execute_reply":"2023-11-07T07:41:19.39237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate pixels based on image width and height\ntrain_data['total_pixels'] = train_data['image_width'] * train_data['image_height']\n\n# Display the updated DataFrame with the new column\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T07:41:19.394989Z","iopub.execute_input":"2023-11-07T07:41:19.395319Z","iopub.status.idle":"2023-11-07T07:41:19.414443Z","shell.execute_reply.started":"2023-11-07T07:41:19.395284Z","shell.execute_reply":"2023-11-07T07:41:19.413352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_labels = ['CC', 'EC', 'HGSC', 'LGSC', 'MC','Other']\ntrain_data['is_tma'] = train_data['is_tma'].astype('int8')\ntrain_data['label'] = train_data['label'].replace({'CC':0, 'EC':1, 'HGSC':2, 'LGSC':3, 'MC':4,'Other':5})\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T07:41:19.41583Z","iopub.execute_input":"2023-11-07T07:41:19.416836Z","iopub.status.idle":"2023-11-07T07:41:19.438139Z","shell.execute_reply.started":"2023-11-07T07:41:19.416807Z","shell.execute_reply":"2023-11-07T07:41:19.436954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom PIL import Image\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom skimage.io import imshow\n\nImage.MAX_IMAGE_PIXELS = 10000000000\n\ndef process_image(img_id, label, tma):\n    if tma == 0:\n        img_name = f\"{img_id}_thumbnail.png\"\n        image_path = f\"/kaggle/input/UBC-OCEAN/train_thumbnails/{img_name}\"\n    else:\n        img_name = f\"{img_id}.png\"\n        image_path = f\"/kaggle/input/UBC-OCEAN/train_images/{img_name}\"\n    \n    if os.path.exists(image_path):\n        image = Image.open(image_path)\n        image = image.resize((512, 512))\n        image = np.array(image)\n        return image, label\n\n# Process and display the first 60 images in a 10x6 grid\nfig, axes = plt.subplots(10, 6, figsize=(20, 20))\n\nimage_count = 0  # Counter to track the number of displayed images\nfor img_id, label, tma in zip(train_data['image_id'], train_data['label'], train_data['is_tma']):\n    if image_count >= 60:  # Limiting to the first 60 images\n        break\n    \n    result = process_image(img_id, label, tma)\n    if result is not None:\n        image, label = result\n        row = image_count // 6\n        col = image_count % 6\n        axes[row, col].imshow(image)\n        axes[row, col].set_title(f\"Label: {label}\")\n        axes[row, col].axis('off')  # Turn off axis\n        image_count += 1\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-07T07:41:19.441836Z","iopub.execute_input":"2023-11-07T07:41:19.442441Z","iopub.status.idle":"2023-11-07T07:41:46.886174Z","shell.execute_reply.started":"2023-11-07T07:41:19.442411Z","shell.execute_reply":"2023-11-07T07:41:46.885267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# Assuming 'train_data' and 'process_image' function are defined\n\n# Process and extract features of all images\nimage_features = []\nimage_labels = []\n\nfor img_id, label, tma in zip(train_data['image_id'], train_data['label'], train_data['is_tma']):\n    result = process_image(img_id, label, tma)\n    if result is not None:\n        image, label = result\n        image_features.append(image.flatten())  # Extract features and flatten the image\n        image_labels.append(label)\n\n# Convert lists to NumPy arrays\nimage_features = np.array(image_features)\nimage_labels = np.array(image_labels)\n\n# Create 'image_features' and 'image_labels' columns in train_data\ntrain_data['image_features'] = list(image_features)\ntrain_data['image_labels'] = list(image_labels)","metadata":{"execution":{"iopub.status.busy":"2023-11-07T07:41:46.887261Z","iopub.execute_input":"2023-11-07T07:41:46.887754Z","iopub.status.idle":"2023-11-07T07:44:42.807042Z","shell.execute_reply.started":"2023-11-07T07:41:46.887725Z","shell.execute_reply":"2023-11-07T07:44:42.806023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create 'image_features' and 'image_labels' columns in train_data\ntrain_data['image_features'] = list(image_features)\ntrain_data['image_labels'] = list(image_labels)\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2023-11-07T07:44:42.808289Z","iopub.execute_input":"2023-11-07T07:44:42.80864Z","iopub.status.idle":"2023-11-07T07:44:42.849974Z","shell.execute_reply.started":"2023-11-07T07:44:42.808611Z","shell.execute_reply":"2023-11-07T07:44:42.848933Z"},"trusted":true},"execution_count":null,"outputs":[]}]}