{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"import stuff\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport tensorflow as tf\nimport torch  \nfrom torchvision import transforms as T\nfrom pathlib import Path\n\nfrom PIL import Image \nImage.MAX_IMAGE_PIXELS = None","metadata":{"execution":{"iopub.status.busy":"2023-10-08T15:47:03.237666Z","iopub.execute_input":"2023-10-08T15:47:03.23801Z","iopub.status.idle":"2023-10-08T15:47:06.318627Z","shell.execute_reply.started":"2023-10-08T15:47:03.237975Z","shell.execute_reply":"2023-10-08T15:47:06.317648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_dir_t = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\ntest_data_dir_t = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"\n\ntrain_data_dir = \"/kaggle/input/UBC-OCEAN/train_images\"\ntest_data_dir = \"/kaggle/input/UBC-OCEAN/test_images\"","metadata":{"execution":{"iopub.status.busy":"2023-10-08T15:47:06.320437Z","iopub.execute_input":"2023-10-08T15:47:06.321901Z","iopub.status.idle":"2023-10-08T15:47:06.327255Z","shell.execute_reply.started":"2023-10-08T15:47:06.321869Z","shell.execute_reply":"2023-10-08T15:47:06.325894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_dir_train = '/kaggle/working/train/'\noutput_dir_test = '/kaggle/working/test/'","metadata":{"execution":{"iopub.status.busy":"2023-10-08T15:47:06.328418Z","iopub.execute_input":"2023-10-08T15:47:06.329239Z","iopub.status.idle":"2023-10-08T15:47:06.338561Z","shell.execute_reply.started":"2023-10-08T15:47:06.329203Z","shell.execute_reply":"2023-10-08T15:47:06.337589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"read in the data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-08T15:47:06.341372Z","iopub.execute_input":"2023-10-08T15:47:06.341973Z","iopub.status.idle":"2023-10-08T15:47:06.363071Z","shell.execute_reply.started":"2023-10-08T15:47:06.341941Z","shell.execute_reply":"2023-10-08T15:47:06.362351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"get the labels","metadata":{}},{"cell_type":"code","source":"unique_labels = df['label'].unique()\nprint(unique_labels)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T15:47:06.364514Z","iopub.execute_input":"2023-10-08T15:47:06.365074Z","iopub.status.idle":"2023-10-08T15:47:06.379527Z","shell.execute_reply.started":"2023-10-08T15:47:06.365042Z","shell.execute_reply":"2023-10-08T15:47:06.378152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"check if the thumbnail exsits based on the larger dataset","metadata":{}},{"cell_type":"code","source":"def t_exists(row):\n    image_id_with_thumbnail = str(row['image_id']) + '_thumbnail.png'\n    return image_id_with_thumbnail in image_files_train_t","metadata":{"execution":{"iopub.status.busy":"2023-10-08T15:47:06.381433Z","iopub.execute_input":"2023-10-08T15:47:06.382393Z","iopub.status.idle":"2023-10-08T15:47:06.389798Z","shell.execute_reply.started":"2023-10-08T15:47:06.382306Z","shell.execute_reply":"2023-10-08T15:47:06.388866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"some simple helpers to get the image paths","metadata":{}},{"cell_type":"code","source":"def get_train_file_path(image_id):\n    return train_data_dir + \"/\" + str(image_id) + '.png'\n\ndef get_test_file_path(image_id):\n    return test_data_dir + \"/\" + str(image_id) + '.png'\n\n\n\ndef get_train_file_path_t(image_id):\n    return train_data_dir_t + \"/\" + str(image_id) + '_thumbnail.png'\n\ndef get_test_file_path_t(image_id):\n    return test_data_dir_t + \"/\" + str(image_id) + '_thumbnail.png'\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-10-08T15:47:06.391241Z","iopub.execute_input":"2023-10-08T15:47:06.391818Z","iopub.status.idle":"2023-10-08T15:47:06.401025Z","shell.execute_reply.started":"2023-10-08T15:47:06.391789Z","shell.execute_reply":"2023-10-08T15:47:06.399994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"get array of images for larger images and thumbnails","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/UBC-OCEAN/train.csv')\ntest = pd.read_csv('../input/UBC-OCEAN/sample_submission.csv')\n\n\n# full resolutions\nimage_files_train = [filename for _, _, filenames in os.walk(train_data_dir) for filename in filenames]\nimage_files_test = [filename for _, _, filenames in os.walk(test_data_dir) for filename in filenames]\n\n# thumbnails\nimage_files_train_t = [filename for _, _, filenames in os.walk(train_data_dir_t) for filename in filenames]\nimage_files_test_t = [filename for _, _, filenames in os.walk(test_data_dir_t) for filename in filenames]\n\n\ntrain['file_path'] = train['image_id'].apply(get_train_file_path)\ntest['file_path'] = test['image_id'].apply(get_test_file_path)\n\ntrain['file_path_t'] = train['image_id'].apply(get_train_file_path_t)\ntest['file_path_t'] = test['image_id'].apply(get_test_file_path_t)\n\ntrain['t_exists'] = train.apply(t_exists, axis=1)\n\n\ndisplay(train.head())\ndisplay(test.head())","metadata":{"execution":{"iopub.status.busy":"2023-10-08T15:47:06.402547Z","iopub.execute_input":"2023-10-08T15:47:06.403236Z","iopub.status.idle":"2023-10-08T15:47:06.578814Z","shell.execute_reply.started":"2023-10-08T15:47:06.403205Z","shell.execute_reply":"2023-10-08T15:47:06.577722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"create directories for the resized images","metadata":{}},{"cell_type":"code","source":"base_dir = '/kaggle/working/train'\n\n# Create label-specific directories\nfor label in unique_labels:\n    label_dir = os.path.join(base_dir, str(label))\n    Path(label_dir).mkdir(parents=True, exist_ok=True)\n\nPath('/kaggle/working/test').mkdir(parents=True, exist_ok=True)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-08T15:47:06.580722Z","iopub.execute_input":"2023-10-08T15:47:06.581549Z","iopub.status.idle":"2023-10-08T15:47:06.589335Z","shell.execute_reply.started":"2023-10-08T15:47:06.58151Z","shell.execute_reply":"2023-10-08T15:47:06.588177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"resize images for the classification model","metadata":{}},{"cell_type":"code","source":"# Loop through the DataFrame rows\nfor index, row in train.iterrows():\n    if row['t_exists']:  # Check if 't_exists' is True\n        image_path_t = row['file_path_t']\n        \n        # Load the image\n        image = Image.open(image_path_t)\n        \n        # Define the resize transformation\n        resize_transform = T.Resize(size=(224, 224))\n        \n        # Use the transformation on the input image\n        image = resize_transform(image)\n        \n        # Get the label for this image\n        label = row['label'] \n        \n        # Create the target directory if it doesn't exist\n        output_dir_label = os.path.join(output_dir_train, str(label))\n        os.makedirs(output_dir_label, exist_ok=True)\n        \n        # Construct the output path for the resized image\n        file_name = os.path.basename(image_path_t)\n        output_path = os.path.join(output_dir_label, file_name)\n        \n        # Save the resized image to the target directory\n        image.save(output_path)\n\nprint(\"Resizing and organizing images complete.\")\n","metadata":{"execution":{"iopub.status.busy":"2023-10-08T15:47:06.593039Z","iopub.execute_input":"2023-10-08T15:47:06.593829Z","iopub.status.idle":"2023-10-08T15:49:15.604621Z","shell.execute_reply.started":"2023-10-08T15:47:06.593795Z","shell.execute_reply":"2023-10-08T15:49:15.603582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"read the images into the model and set up some params","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\nfrom tensorflow.keras.applications import resnet_v2\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Define the data augmentation for training data\ntrain_aug = ImageDataGenerator(\n    rotation_range=15,\n    zoom_range=0.2,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    shear_range=0.1,\n    horizontal_flip=True,\n    vertical_flip=True,\n    fill_mode=\"nearest\")\n\n# Use the augmented data generator for training data\ntrain_generator = train_aug.flow_from_directory(\n    '/kaggle/working/train',\n    target_size=(224, 224),\n    batch_size=32)","metadata":{"execution":{"iopub.status.busy":"2023-10-08T15:49:15.605856Z","iopub.execute_input":"2023-10-08T15:49:15.60689Z","iopub.status.idle":"2023-10-08T15:49:15.628767Z","shell.execute_reply.started":"2023-10-08T15:49:15.606858Z","shell.execute_reply":"2023-10-08T15:49:15.628052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"train the model","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.optimizers import SGD\nfrom keras.layers import Dropout\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D  \nfrom tensorflow.keras.models import Model\n\n# Create model\nbase_model = EfficientNetB0(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\nx = Dense(1024, activation='relu')(x)\nx = Dropout(0.5)(x)  # Add dropout layer with 50% dropout rate\npredictions = Dense(5, activation='softmax')(x)\n\nmodel = Model(inputs=base_model.input, outputs=predictions)\n\n# Compile model\nopt = SGD(lr=0.01, momentum=0.9) \nmodel.compile(optimizer=opt, loss='categorical_crossentropy', metrics=['accuracy'])\n\n\n# Train the model\nmodel.fit(\n    train_generator,\n    steps_per_epoch=len(train_generator),\n    epochs=100)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-08T15:49:15.629931Z","iopub.execute_input":"2023-10-08T15:49:15.630332Z","iopub.status.idle":"2023-10-08T16:04:51.144911Z","shell.execute_reply.started":"2023-10-08T15:49:15.630285Z","shell.execute_reply":"2023-10-08T16:04:51.143713Z"},"trusted":true},"execution_count":null,"outputs":[]}]}