{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nfrom tqdm.auto import tqdm\ntqdm.pandas()\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nimport PIL.Image as Image\nfrom sklearn.model_selection import train_test_split\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-06T00:32:20.310189Z","iopub.execute_input":"2023-11-06T00:32:20.310673Z","iopub.status.idle":"2023-11-06T00:32:22.689953Z","shell.execute_reply.started":"2023-11-06T00:32:20.310623Z","shell.execute_reply":"2023-11-06T00:32:22.688938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Loading the Dataset**","metadata":{}},{"cell_type":"code","source":"# Reading the Train and Test content\ntrain = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ntest = pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')\ntrain_thumbnails = os.listdir(\"/kaggle/input/UBC-OCEAN/train_thumbnails\")\n\n# Define the path to the folder containing the images\nimage_folder = \"/kaggle/input/UBC-OCEAN/train_images/\"\n\n# List all PNG files in the folder\npng_files = [f for f in os.listdir(image_folder) if f.endswith('.png')]\n\n# Create the image paths\ntrain['image_path'] = image_folder + pd.Series(png_files).astype(str)\n\n# Count the number of PNG files\nnum_png_files = len(png_files)\n\n# Print the number of PNG files\nprint(f\"Number of PNG files in the folder: {num_png_files}\")\n\n# Print the shapes of the train and test dataframes\nprint(f'train.shape: {train.shape}')\nprint(f'test.shape: {test.shape}')\n\n# Display the first 10 rows of the train dataframe\ntrain.head(10)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-06T00:32:27.674202Z","iopub.execute_input":"2023-11-06T00:32:27.674749Z","iopub.status.idle":"2023-11-06T00:32:28.004753Z","shell.execute_reply.started":"2023-11-06T00:32:27.674715Z","shell.execute_reply":"2023-11-06T00:32:28.003639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Check for any missing values**","metadata":{}},{"cell_type":"code","source":"# Check for missing data\nmissing_data = train.isnull().sum()\nprint(missing_data)","metadata":{"execution":{"iopub.status.busy":"2023-11-05T18:09:12.460756Z","iopub.execute_input":"2023-11-05T18:09:12.461589Z","iopub.status.idle":"2023-11-05T18:09:12.470485Z","shell.execute_reply.started":"2023-11-05T18:09:12.461553Z","shell.execute_reply":"2023-11-05T18:09:12.469657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Exploratory Data Analysis (EDA)**","metadata":{}},{"cell_type":"code","source":"# Function to perform EDA for a column\ndef perform_eda(column_name):\n  \"\"\"\n  Perform Exploratory Data Analysis (EDA) for a given column.\n\n  Args:\n    column_name (str): The name of the column to perform EDA on.\n\n  Returns:\n    None\n  \"\"\"\n\n  print(f\"Exploratory Data Analysis for '{column_name}':\")\n\n  # Check the data type of the column\n  column_type = train[column_name].dtype\n\n  # Perform EDA based on the data type of the column\n  if column_type == 'object':\n    # Categorical data\n    print(\"Unique values:\")\n    print(train[column_name].unique())\n    print(\"\\n\")\n\n    # Value counts\n    print(\"Value counts:\")\n    print(train[column_name].value_counts())\n    print(\"\\n\")\n\n    # Create a bar chart of the value counts\n    plt.figure(figsize=(10, 5))\n    sns.barplot(x=train[column_name].value_counts().index, y=train[column_name].value_counts().values)\n    plt.xticks(rotation=90)\n    plt.xlabel(column_name)\n    plt.ylabel('Count')\n    plt.title(f'Distribution of {column_name}')\n    plt.show()\n\n  elif column_type == 'float64' or column_type == 'int64':\n    # Numerical data\n    print(\"Summary statistics:\")\n    print(train[column_name].describe())\n    print(\"\\n\")\n\n    # Create a histogram of the data\n    plt.figure(figsize=(10, 5))\n    sns.histplot(train[column_name], bins=20, kde=True)\n    plt.xlabel(column_name)\n    plt.ylabel('Density')\n    plt.title(f'Distribution of {column_name}')\n    plt.show()\n\n  elif column_type == 'bool':\n    # Boolean data\n    print(\"Unique values:\")\n    print(train[column_name].unique())\n    print(\"\\n\")\n\n    # Value counts\n    print(\"Value counts:\")\n    print(train[column_name].value_counts())\n    print(\"\\n\")\n\n    # Create a pie chart of the value counts\n    plt.figure(figsize=(10, 5))\n    plt.pie(train[column_name].value_counts().values, labels=train[column_name].value_counts().index, autopct=\"%1.1f%%\")\n    plt.title(f'Distribution of {column_name}')\n    plt.show()\n\n  else:\n    print(\"Unsupported data type:\", column_type)\n\n  print(\"\\n\" + \"=\" * 50 + \"\\n\")\n\n# Perform EDA for all columns in the DataFrame\nfor column_name in train.columns:\n  perform_eda(column_name)","metadata":{"execution":{"iopub.status.busy":"2023-10-30T05:20:09.567199Z","iopub.execute_input":"2023-10-30T05:20:09.568074Z","iopub.status.idle":"2023-10-30T05:20:17.853531Z","shell.execute_reply.started":"2023-10-30T05:20:09.56804Z","shell.execute_reply":"2023-10-30T05:20:17.852592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Check for Correlation**","metadata":{}},{"cell_type":"code","source":"corr_matrix = train[['image_width', 'image_height']].corr()\n\n# Print the correlation matrix\nprint(corr_matrix)","metadata":{"execution":{"iopub.status.busy":"2023-11-05T23:37:41.823136Z","iopub.execute_input":"2023-11-05T23:37:41.824108Z","iopub.status.idle":"2023-11-05T23:37:41.839707Z","shell.execute_reply.started":"2023-11-05T23:37:41.824071Z","shell.execute_reply":"2023-11-05T23:37:41.838529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ****Scatter Plot Of Image Width vs. Image Height ****","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nplt.scatter(train['image_width'], train['image_height'], alpha=0.9)\nplt.title('Scatter Plot of Image Width vs. Image Height')\nplt.xlabel('Image Width (pixels)')\nplt.ylabel('Image Height (pixels)')\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-30T06:38:17.859237Z","iopub.execute_input":"2023-10-30T06:38:17.859647Z","iopub.status.idle":"2023-10-30T06:38:18.151158Z","shell.execute_reply.started":"2023-10-30T06:38:17.859617Z","shell.execute_reply":"2023-10-30T06:38:18.150331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Label Mapping**","metadata":{}},{"cell_type":"code","source":"labels = ['CC', 'EC', 'HGSC', 'LGSC', 'MC']\ntrain['label'] = train['label'].map({label: index for index, label in enumerate(labels)})\ntrain.head()\ntrain.info()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-06T00:32:49.57003Z","iopub.execute_input":"2023-11-06T00:32:49.570423Z","iopub.status.idle":"2023-11-06T00:32:49.594544Z","shell.execute_reply.started":"2023-11-06T00:32:49.57039Z","shell.execute_reply":"2023-11-06T00:32:49.593531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Image PreProcessing**","metadata":{}},{"cell_type":"code","source":"# Set the maximum number of pixels for large images\nImage.MAX_IMAGE_PIXELS = 10000000000\n\n# Define patch size and overlap\npatch_size = (128, 128)\noverlap = 10\n\nimage_data = []\nimage_label = []\nempty_img = 0  # Initialized empty_img as a global variable\n\n# Function to check if a patch is empty\ndef is_empty_patch(patch):\n    return np.sum(patch) < 100 # threshold as needed\n\ndef process_image(img_id, label, tma):\n    global empty_img  # Declaring empty_img as a global variable\n    \n    if tma == 0:\n        img_name = f\"{img_id}_thumbnail.png\"\n        image_path = f\"/kaggle/input/UBC-OCEAN/train_thumbnails/{img_name}\"\n    else:\n        img_name = f\"{img_id}.png\"\n        image_path = f\"/kaggle/input/UBC-OCEAN/train_images/{img_name}\"\n    \n    large_image = Image.open(image_path)\n    \n    for y in range(0, large_image.height, patch_size[0] - overlap):\n        for x in range(0, large_image.width, patch_size[1] - overlap):\n            patch = large_image.crop((x, y, x + patch_size[1], y + patch_size[0]))\n            image = np.array(patch)\n            \n            if is_empty_patch(image):\n                empty_img += 1\n            else:\n                image_data.append(image)\n                image_label.append(label)\n\n\nfor img_id, label, tma in zip(train['image_id'], train['label'], train['is_tma']):\n    process_image(img_id, label, tma)\n\nprint(f\"Total images processed: {len(image_data)}\")\nprint(f\"Total empty patches: {empty_img}\")\nset(image_label)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-06T00:32:54.041483Z","iopub.execute_input":"2023-11-06T00:32:54.04228Z","iopub.status.idle":"2023-11-06T00:35:25.298111Z","shell.execute_reply.started":"2023-11-06T00:32:54.042226Z","shell.execute_reply":"2023-11-06T00:35:25.297104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Visualizing Preprocessed Images**","metadata":{}},{"cell_type":"code","source":"start_index = 500\nend_index = 670\nrows, cols = 5, 10  # Define the number of rows and columns\n\n# Create a figure with a specific size\nfig, axes = plt.subplots(rows, cols, figsize=(40, 50))\n\nfor i, ax in enumerate(axes.flat):\n    if i >= (end_index - start_index):\n        ax.axis('off')  # Turn off empty subplots\n    else:\n        image_index = start_index + i\n        ax.imshow(image_data[image_index])\n        ax.set_title(f\"Label: {labels[image_label[image_index]]}\", fontsize=10)\n        ax.axis('off')  # Turn off axis labels\n\nplt.tight_layout() \nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-30T18:26:17.741286Z","iopub.execute_input":"2023-10-30T18:26:17.741667Z","iopub.status.idle":"2023-10-30T18:26:28.233278Z","shell.execute_reply.started":"2023-10-30T18:26:17.741637Z","shell.execute_reply":"2023-10-30T18:26:28.231322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Converting into Numpy Array for Model Building**","metadata":{}},{"cell_type":"code","source":"X = np.array(image_data) \nY = np.array(image_label)\nprint(X.shape)\nprint(Y.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-06T00:35:48.08975Z","iopub.execute_input":"2023-11-06T00:35:48.090687Z","iopub.status.idle":"2023-11-06T00:35:51.734693Z","shell.execute_reply.started":"2023-11-06T00:35:48.090649Z","shell.execute_reply":"2023-11-06T00:35:51.733606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Splitting Dataset using Sklearn**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Split the dataset into training (80%) and testing (20%) sets\nX_train, X_test, Y_train, Y_test = train_test_split(X[:6000], Y[:6000], test_size=0.2, random_state=42, shuffle =True)\nprint(X_train.shape)\nprint(X_test.shape)\nprint(Y_train.shape)\nprint(Y_test.shape)\ntrain_x_scaled= X_train/255\ntest_x_scaled = X_test/255\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-06T00:35:53.910808Z","iopub.execute_input":"2023-11-06T00:35:53.911192Z","iopub.status.idle":"2023-11-06T00:35:54.712885Z","shell.execute_reply.started":"2023-11-06T00:35:53.91116Z","shell.execute_reply":"2023-11-06T00:35:54.711944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PreTrained Model using EfficientNetB0 & Balancing the Class Weight","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Dense\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import LearningRateScheduler, EarlyStopping\nimport numpy as np\n\n# Convert Y_train to a NumPy array\nY_train_array = np.array(Y_train)\n\n# Loading the pre-trained EfficientNetB0 model\nbase_model = EfficientNetB0(include_top=False, weights='imagenet', input_shape=(128, 128, 3))\n\n# Unfreeze some of the later layers\nfor layer in base_model.layers[-20:]:\n    layer.trainable = True\n\n# Adding custom top layers for classification\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\nx = Dense(1024, activation='relu')(x)\npredictions = Dense(5, activation='softmax')(x)\n\n# Create the final model\nmodel = Model(inputs=base_model.input, outputs=predictions)\n\n# Manually calculate class weights based on class distribution\nclass_weights = {}\ntotal_samples = len(Y_train_array)\nunique_classes = np.unique(Y_train_array)\nfor c in unique_classes:\n    class_count = len(Y_train_array[Y_train_array == c])\n    class_weights[c] = total_samples / (len(unique_classes) * class_count)\n\n# Implement a learning rate schedule\ndef learning_rate_scheduler(epoch, lr):\n    if epoch < 5:\n        return lr\n    else:\n        return lr * 0.95\n\n# Compile the model with manually calculated class weights and custom learning rate\nmodel.compile(loss='sparse_categorical_crossentropy', optimizer=Adam(learning_rate=0.0001), metrics=['accuracy'])\n\n# Augmenting training data\ndatagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    horizontal_flip=True,\n    vertical_flip=True,\n    preprocessing_function=None  \n)\n\ndatagen.fit(X_train)\n\n# Train the model with learning rate scheduler\nmodel.fit(datagen.flow(X_train, Y_train_array, batch_size=64), validation_data=(X_test, Y_test),\n          steps_per_epoch=len(X_train) // 64, epochs=10, class_weight=class_weights,\n          callbacks=[LearningRateScheduler(learning_rate_scheduler), EarlyStopping(patience=5, restore_best_weights=True)])\n\n# Evaluate the model\nloss, accuracy = model.evaluate(X_test, Y_test)\nprint(f'Validation Loss: {loss:.4f}, Accuracy: {accuracy:.4f}')\n","metadata":{"execution":{"iopub.status.busy":"2023-11-06T00:40:45.37401Z","iopub.execute_input":"2023-11-06T00:40:45.374983Z","iopub.status.idle":"2023-11-06T00:44:32.669319Z","shell.execute_reply.started":"2023-11-06T00:40:45.374944Z","shell.execute_reply":"2023-11-06T00:44:32.668312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Evaluation on Test and Train Data**","metadata":{}},{"cell_type":"code","source":"loss ,acc = model.evaluate(X_train, Y_train)\nprint(\"Accuracy on Train Data:\",acc)\nprint()\nloss ,acc = model.evaluate(X_test, Y_test)\nprint(\"Accuracy on Test Data:\",acc)","metadata":{"execution":{"iopub.status.busy":"2023-11-06T00:44:37.237661Z","iopub.execute_input":"2023-11-06T00:44:37.238592Z","iopub.status.idle":"2023-11-06T00:44:41.257917Z","shell.execute_reply.started":"2023-11-06T00:44:37.238547Z","shell.execute_reply":"2023-11-06T00:44:41.256981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Calculate Prediction**","metadata":{}},{"cell_type":"code","source":"Y_pred = model.predict(X_test)\nY_pred_test = [np.argmax(i) for i in Y_pred]","metadata":{"execution":{"iopub.status.busy":"2023-11-06T00:44:44.637252Z","iopub.execute_input":"2023-11-06T00:44:44.638199Z","iopub.status.idle":"2023-11-06T00:44:46.893355Z","shell.execute_reply.started":"2023-11-06T00:44:44.638162Z","shell.execute_reply":"2023-11-06T00:44:46.892549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize to compare Results from Actual and Predicted  ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(24,40))\nfor i in range(60):\n    plt.subplot(10,6,i+1)\n    plt.imshow(X_test[i])\n    plt.title(f\"Actual Label:{labels[Y_test[i]]}\\nPredicted Label:{labels[Y_pred_test[i]]}\")\n    plt.axis(\"off\")\n","metadata":{"execution":{"iopub.status.busy":"2023-11-05T23:14:09.71466Z","iopub.execute_input":"2023-11-05T23:14:09.715051Z","iopub.status.idle":"2023-11-05T23:14:17.90732Z","shell.execute_reply.started":"2023-11-05T23:14:09.715017Z","shell.execute_reply":"2023-11-05T23:14:17.906325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Calculating Balanced Accuracy Score**","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import balanced_accuracy_score\n\ndef compute_metrics(pred):\n    labels = pred.label_ids\n    preds = pred.predictions.argmax(-1)\n    balanced_acc = balanced_accuracy_score(labels, preds)\n    return {\"balanced_accuracy\": balanced_acc}\n","metadata":{"execution":{"iopub.status.busy":"2023-11-06T00:44:51.857307Z","iopub.execute_input":"2023-11-06T00:44:51.858287Z","iopub.status.idle":"2023-11-06T00:44:51.863466Z","shell.execute_reply.started":"2023-11-06T00:44:51.858235Z","shell.execute_reply":"2023-11-06T00:44:51.862423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"actual_labels = [labels[Y_test[i]] for i in range(len(Y_test))]\npredicted_labels = [labels[Y_pred_test[i]] for i in range(len(Y_pred_test))]\n\n# Calculate balanced accuracy\nbalanced_accuracy = balanced_accuracy_score(actual_labels, predicted_labels)\nprint(\"Balanced Accuracy:\", balanced_accuracy)","metadata":{"execution":{"iopub.status.busy":"2023-11-06T00:44:53.402016Z","iopub.execute_input":"2023-11-06T00:44:53.402864Z","iopub.status.idle":"2023-11-06T00:44:53.417055Z","shell.execute_reply.started":"2023-11-06T00:44:53.402826Z","shell.execute_reply":"2023-11-06T00:44:53.4162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/UBC-OCEAN/sample_submission.csv\")\nsample_submission['label'] = labels[Y_pred_test[i]]\n\n# Save the updated DataFrame to a CSV file\nsample_submission.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-05T23:14:35.63067Z","iopub.execute_input":"2023-11-05T23:14:35.631515Z","iopub.status.idle":"2023-11-05T23:14:35.644936Z","shell.execute_reply.started":"2023-11-05T23:14:35.631475Z","shell.execute_reply":"2023-11-05T23:14:35.643933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-05T23:15:28.670209Z","iopub.execute_input":"2023-11-05T23:15:28.670679Z","iopub.status.idle":"2023-11-05T23:15:28.681978Z","shell.execute_reply.started":"2023-11-05T23:15:28.670642Z","shell.execute_reply":"2023-11-05T23:15:28.680977Z"},"trusted":true},"execution_count":null,"outputs":[]}]}