{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport cv2 \nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom keras.utils import to_categorical\nfrom keras.models import Sequential\nfrom keras.models import load_model\nfrom keras.layers import Conv2D, MaxPooling2D\nfrom keras.layers import Activation, Dropout, Flatten, Dense\nimport os\nimport joblib","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-17T15:36:27.623402Z","iopub.execute_input":"2023-11-17T15:36:27.624167Z","iopub.status.idle":"2023-11-17T15:36:43.799531Z","shell.execute_reply.started":"2023-11-17T15:36:27.624121Z","shell.execute_reply":"2023-11-17T15:36:43.798499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train_data preprocessing and Training the model","metadata":{}},{"cell_type":"code","source":"train_data=pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2023-11-17T15:36:43.801411Z","iopub.execute_input":"2023-11-17T15:36:43.802013Z","iopub.status.idle":"2023-11-17T15:36:43.852962Z","shell.execute_reply.started":"2023-11-17T15:36:43.80198Z","shell.execute_reply":"2023-11-17T15:36:43.8518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We have a DataFrame named train_data with columns image_id, label, is_tma, and image_path\n\nIMG_SIZE = 256\n\n# Define a function to generate the paths based on conditions\ndef generate_paths_and_resize(row):\n    # Generate the image path\n    if row['is_tma'] == 1:\n        image_path = f'/kaggle/input/UBC-OCEAN/train_images/{row[\"image_id\"]}.png'\n    else:\n        image_path = f'/kaggle/input/UBC-OCEAN/train_thumbnails/{row[\"image_id\"]}_thumbnail.png'\n    \n    # Read and resize the image\n    img = cv2.imread(image_path)\n    img_resized = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    \n    # Normalize the resized image array\n    img_normalized = img_resized / 255.0\n    \n    return image_path, img_normalized\n\n# Apply the function to create new columns\ntrain_data[['image_path', 'normalized_image_array']] = train_data.apply(generate_paths_and_resize, axis=1, result_type='expand')\n\n# Use LabelEncoder to encode the 'label' column\nlabel_encoder = LabelEncoder()\ntrain_data['label_encoded'] = label_encoder.fit_transform(train_data['label'])\n\n# Display the updated DataFrame\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-17T15:36:43.858831Z","iopub.execute_input":"2023-11-17T15:36:43.859145Z","iopub.status.idle":"2023-11-17T15:39:08.977029Z","shell.execute_reply.started":"2023-11-17T15:36:43.859118Z","shell.execute_reply":"2023-11-17T15:39:08.975885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = np.array(train_data['normalized_image_array'].tolist())\nX = X.reshape(-1, IMG_SIZE, IMG_SIZE, 3)\nY = train_data['label_encoded'].values","metadata":{"execution":{"iopub.status.busy":"2023-11-17T15:39:08.978275Z","iopub.execute_input":"2023-11-17T15:39:08.978608Z","iopub.status.idle":"2023-11-17T15:39:09.378621Z","shell.execute_reply.started":"2023-11-17T15:39:08.978579Z","shell.execute_reply":"2023-11-17T15:39:09.377363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, Y_train, Y_test = train_test_split(X, Y, test_size = 0.2, random_state = 4)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T15:39:09.379782Z","iopub.execute_input":"2023-11-17T15:39:09.380157Z","iopub.status.idle":"2023-11-17T15:39:09.78574Z","shell.execute_reply.started":"2023-11-17T15:39:09.380124Z","shell.execute_reply":"2023-11-17T15:39:09.784679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(Y_train.shape)\nprint(Y_test.shape)\nprint(X_train.shape)\nprint(X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T15:39:09.787116Z","iopub.execute_input":"2023-11-17T15:39:09.787569Z","iopub.status.idle":"2023-11-17T15:39:09.794104Z","shell.execute_reply.started":"2023-11-17T15:39:09.787527Z","shell.execute_reply":"2023-11-17T15:39:09.79301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_classes = np.unique(Y_train)\nnb_classes = len(unique_classes)# Convert the target labels to one-hot encoded format\nY_train = to_categorical(Y_train, nb_classes)\nY_test = to_categorical(Y_test, nb_classes)\nY_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-17T15:39:09.79558Z","iopub.execute_input":"2023-11-17T15:39:09.795891Z","iopub.status.idle":"2023-11-17T15:39:09.810004Z","shell.execute_reply.started":"2023-11-17T15:39:09.795862Z","shell.execute_reply":"2023-11-17T15:39:09.80891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We have X_train, X_test, Y_train, and Y_test defined from the previous step\nbatch_size = 20\nnb_classes = 5  # Update this to the correct number of classes\nnb_epochs = 12\nimg_rows, img_columns = 256, 256  # Adjusted to match your resized image size\nimg_channel = 3\nnb_filters = 32\nnb_pool = 2\nnb_conv = 3\n\n# Define the model\nmodel = Sequential()\n\nmodel.add(Conv2D(nb_filters, (nb_conv, nb_conv), input_shape=(img_rows, img_columns, img_channel)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(nb_pool, nb_pool)))\n\nmodel.add(Flatten())\nmodel.add(Dense(100))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.9))\nmodel.add(Dense(nb_classes))\nmodel.add(Activation('softmax'))\n\n# Compile the model\nmodel.compile(loss='categorical_crossentropy', optimizer='adam', metrics=['accuracy'])\n\n# Train the model\nmodel.fit(X_train, Y_train, batch_size=batch_size, epochs=nb_epochs, verbose=1, validation_data=(X_test, Y_test))","metadata":{"execution":{"iopub.status.busy":"2023-11-17T15:39:09.812375Z","iopub.execute_input":"2023-11-17T15:39:09.812819Z","iopub.status.idle":"2023-11-17T15:45:35.139084Z","shell.execute_reply.started":"2023-11-17T15:39:09.81278Z","shell.execute_reply":"2023-11-17T15:45:35.136961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score = model.evaluate(X_test, Y_test, verbose = 0 )\nprint(\"Test Score: \", score[0])\nprint(\"Test accuracy: \", score[1])","metadata":{"execution":{"iopub.status.busy":"2023-11-17T15:45:35.144385Z","iopub.execute_input":"2023-11-17T15:45:35.144864Z","iopub.status.idle":"2023-11-17T15:45:36.586191Z","shell.execute_reply.started":"2023-11-17T15:45:35.144828Z","shell.execute_reply":"2023-11-17T15:45:36.585214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluating the Test Data ","metadata":{}},{"cell_type":"code","source":"# Load the test data\ntest_data = pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')\n\n# Define the preprocess_test_data function\ndef preprocess_test_data(row):\n    # Check if the image exists in the test_images directory\n    image_path = f'/kaggle/input/UBC-OCEAN/test_images/{row[\"image_id\"]}.png'\n    thumbnail_path = f'/kaggle/input/UBC-OCEAN/test_thumbnails/{row[\"image_id\"]}_thumbnail.png'\n\n    if os.path.exists(image_path):\n        img = cv2.imread(image_path)\n    else:\n        img = cv2.imread(thumbnail_path)\n\n    # Resize the image\n    img_resized = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n\n    # Normalize the resized image array\n    img_normalized = img_resized / 255.0\n\n    return img_normalized\n\n# Apply the preprocessing function to the test data\ntest_data['normalized_image_array'] = test_data.apply(preprocess_test_data, axis=1)\n\n# Convert the normalized image arrays to a NumPy array\nX_test_new = np.array(test_data['normalized_image_array'].tolist())\nX_test_new = X_test_new.reshape(-1, IMG_SIZE, IMG_SIZE, 3)\n\n# Make predictions on the test data\npredictions = model.predict(X_test_new)\n\n# Convert predictions to class labels\npredicted_labels_categorical = label_encoder.inverse_transform(np.argmax(predictions, axis=1))","metadata":{"execution":{"iopub.status.busy":"2023-11-17T15:45:36.587398Z","iopub.execute_input":"2023-11-17T15:45:36.588122Z","iopub.status.idle":"2023-11-17T15:45:58.404097Z","shell.execute_reply.started":"2023-11-17T15:45:36.588085Z","shell.execute_reply":"2023-11-17T15:45:58.40311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a DataFrame for the submission file\nsubmission_df = pd.DataFrame({'image_id': test_data['image_id'], 'label': predicted_labels_categorical})\n# Save the DataFrame to a CSV file without including the index\nsubmission_df.to_csv('submission.csv', index=False)\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2023-11-17T15:46:12.811985Z","iopub.execute_input":"2023-11-17T15:46:12.812375Z","iopub.status.idle":"2023-11-17T15:46:12.828329Z","shell.execute_reply.started":"2023-11-17T15:46:12.812344Z","shell.execute_reply":"2023-11-17T15:46:12.827108Z"},"trusted":true},"execution_count":null,"outputs":[]}]}