{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":992,"sourceType":"modelInstanceVersion","modelInstanceId":846}],"dockerImageVersionId":30588,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom PIL import Image\nimport numpy as np\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\n\n# Load the metadata from CSV files\ntrain_metadata = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ntest_metadata = pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')\n\n\n# Filter for rows where \"is_tma\" is False\ntrain_metadata = train_metadata[train_metadata['is_tma'] == False]\n\n\n# Split the training data into training and validation sets\ntrain_data, val_data = train_test_split(train_metadata, test_size=0.2, random_state=42)\n\n# Define the paths to your image files\ntrain_image_paths = ['/kaggle/input/UBC-OCEAN/train_thumbnails/' + str(img_id) + '_thumbnail.png' for img_id in train_data['image_id']]\nval_image_paths = ['/kaggle/input/UBC-OCEAN/train_thumbnails/' + str(img_id) + '_thumbnail.png' for img_id in val_data['image_id']]\ntest_image_paths = ['/kaggle/input/UBC-OCEAN/test_thumbnails/' + str(img_id) + '_thumbnail.png' for img_id in test_metadata['image_id']]\n\n# Define a function to load and preprocess images\ndef load_and_preprocess_image(image_path):\n    img = Image.open(image_path)\n    img = img.resize((224, 224))  # Resize to desired dimensions\n    img = np.array(img)  # Convert to numpy array\n    # Apply any further preprocessing steps if needed\n    return img\n\n# Apply preprocessing to all image paths\ntrain_images = [load_and_preprocess_image(path) for path in train_image_paths]\nval_images = [load_and_preprocess_image(path) for path in val_image_paths]\ntest_images = [load_and_preprocess_image(path) for path in test_image_paths]\n\n# Convert labels to one-hot encoding\nfrom sklearn.preprocessing import LabelEncoder\n\nlabel_encoder = LabelEncoder()\ntrain_labels = label_encoder.fit_transform(train_data['label'])\nval_labels = label_encoder.transform(val_data['label'])\n# Note: Keep label_encoder for later use in decoding predictions\n\n\n# Define a data generator for augmentation\ntrain_datagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    vertical_flip=True,\n    fill_mode='nearest'\n)\n\n\n\n# Define a data generator function\ndef data_generator(images, labels, batch_size, data_augmentation=True):\n    while True:\n        # Generate random indices for the batch\n        indices = np.random.choice(len(images), size=batch_size, replace=False)\n        batch_images = []\n        batch_labels = []\n        \n        for idx in indices:\n            # Load and preprocess the image\n            image = load_and_preprocess_image(images[idx])\n            label = labels[idx]\n            \n            # Apply data augmentation (if enabled)\n            if data_augmentation:\n                image = train_datagen.random_transform(image)\n            \n            batch_images.append(image)\n            batch_labels.append(label)\n        \n        yield np.array(batch_images), np.array(batch_labels)\n\n# Define batch size\nbatch_size = 32\n\n# Create data generators for training and validation sets\ntrain_generator = data_generator(train_image_paths, train_labels, batch_size)\nval_generator = data_generator(val_image_paths, val_labels, batch_size, data_augmentation=False)\n\n# Check the data generator\nbatch_images, batch_labels = next(train_generator)\n\n# Define the CNN architecture\nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(224, 224, 3)),\n    MaxPooling2D((2, 2)),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Conv2D(128, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Flatten(),\n    Dense(128, activation='relu'),\n    Dropout(0.5),\n    Dense(6, activation='softmax')  # Assuming 6 classes for the subtypes\n])\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n\n# Convert images and labels to numpy arrays\n\n\ntrain_images = np.array(train_images)\nval_images = np.array(val_images)\ntest_images = np.array(test_images)\ntrain_labels = np.array(train_labels)\nval_labels = np.array(val_labels)\n\n# Normalize pixel values to be between 0 and 1\ntrain_images = train_images / 255.0\nval_images = val_images / 255.0\ntest_images = test_images / 255.0\n\n# Train the model\nhistory = model.fit(train_images, train_labels, epochs=20, validation_data=(val_images, val_labels))\n\n\n# Make predictions on the test set\npredictions = model.predict(test_images)\n\n# Convert the predicted probabilities to class labels\npredicted_labels = [np.argmax(prediction) for prediction in predictions]\n\n# Decode the predicted labels using the label_encoder\npredicted_subtypes = label_encoder.inverse_transform(predicted_labels)\n\n# Print the predicted subtypes\nprint(predicted_subtypes)\n# Create a DataFrame with the image IDs and predicted subtypes\nsubmission_df = pd.DataFrame({'image_id': test_metadata['image_id'], 'predicted_subtype': predicted_subtypes})\n\n# Save the DataFrame to a CSV file\nsubmission_df.to_csv('submission.csv', index=False)\n","metadata":{"_uuid":"6b8d5fcd-450f-486e-88f8-b7b7466576de","_cell_guid":"c4e69d2f-76b5-4389-98dc-1f7dabaf171a","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-11-29T17:46:51.682522Z","iopub.execute_input":"2023-11-29T17:46:51.682814Z","iopub.status.idle":"2023-11-29T17:50:27.975894Z","shell.execute_reply.started":"2023-11-29T17:46:51.682787Z","shell.execute_reply":"2023-11-29T17:50:27.974961Z"},"trusted":true},"execution_count":null,"outputs":[]}]}