{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30588,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import Input\nfrom tensorflow.keras.applications import ResNet50,ResNet101,ResNet152\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D\nfrom tensorflow.keras.layers import Input, Conv2D, MaxPooling2D, Dense, Flatten, Dropout\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\nimport numpy as np\nimport os\nfrom PIL import Image\nfrom shutil import copyfile  # Import the copyfile function\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-12-07T07:13:29.597943Z","iopub.execute_input":"2023-12-07T07:13:29.598309Z","iopub.status.idle":"2023-12-07T07:13:29.605087Z","shell.execute_reply.started":"2023-12-07T07:13:29.598279Z","shell.execute_reply":"2023-12-07T07:13:29.604073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the CSV file\ntrain_dataframe = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ntrain_dataframe.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:28:43.099553Z","iopub.execute_input":"2023-12-07T06:28:43.100288Z","iopub.status.idle":"2023-12-07T06:28:43.113674Z","shell.execute_reply.started":"2023-12-07T06:28:43.100254Z","shell.execute_reply":"2023-12-07T06:28:43.112639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\n\ndef create_image_path_dataframe(csv_data, images_dir):\n    image_paths = []\n    for index, row in csv_data.iterrows():\n        image_id = row['image_id']\n        image_label = row['label']  # Change this according to your columns\n        image_filename = f\"{image_id}_thumbnail.png\"  # Assuming image filenames are based on image_id with png extension\n        image_path = os.path.join(images_dir, image_filename)\n        \n        if os.path.exists(image_path):\n            image_paths.append({\n                'image_id': image_id,\n                'image_path': image_path,\n                'label': image_label,\n                'image_width': row['image_width'],\n                'image_height': row['image_height'],\n                'is_tma': row['is_tma']\n                # Add more columns as needed\n            })\n    \n    # Create a DataFrame from the matched information\n    image_path_df = pd.DataFrame(image_paths)\n    return image_path_df\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:36:01.552227Z","iopub.execute_input":"2023-12-07T06:36:01.553019Z","iopub.status.idle":"2023-12-07T06:36:01.560184Z","shell.execute_reply.started":"2023-12-07T06:36:01.552987Z","shell.execute_reply":"2023-12-07T06:36:01.55913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"main_dataset = create_image_path_dataframe(train_dataframe, train_images_directory)\nmain_dataset.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:36:46.479947Z","iopub.execute_input":"2023-12-07T06:36:46.480833Z","iopub.status.idle":"2023-12-07T06:36:46.824319Z","shell.execute_reply.started":"2023-12-07T06:36:46.480799Z","shell.execute_reply":"2023-12-07T06:36:46.823157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Distribution of Image Labels in the Dataset**","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# Calculate label counts from the main_dataset\nlabel_counts_main_dataset = main_dataset['label'].value_counts()\n\n# Define custom colors for the plot\ncustom_colors = ['#FF9999', '#66B2FF', '#99FF99', '#FFCC99', '#c2c2f0']\n\n# Plotting the label distribution for the main dataset\nplt.figure(figsize=(10, 6))\nbars = label_counts_main_dataset.plot(kind='bar', color=custom_colors)\n\nplt.title('Label Distribution in Dataset', fontname='serif', fontsize=18, weight='bold', pad=20)\nplt.xlabel('Labels', fontname='serif', fontsize=14, weight='bold', labelpad=10)\nplt.ylabel('Count', fontname='serif', fontsize=14, weight='bold', labelpad=10)\nplt.xticks(rotation=45, fontname='serif', fontsize=12, weight='bold')\nplt.yticks(fontname='serif', fontsize=12, weight='bold')\n# Annotate the bars with their respective counts\nfor i, count in enumerate(label_counts_main_dataset):\n    plt.text(i, count + 10, str(count), ha='center', va='bottom', fontname='serif', fontsize=10, weight='bold')\n\n# Customize grid lines\nplt.grid(axis='y', linestyle='--', alpha=0.7)\n\n# Remove spines\nplt.gca().spines['top'].set_visible(False)\nplt.gca().spines['right'].set_visible(False)\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T07:09:28.287803Z","iopub.execute_input":"2023-12-07T07:09:28.288554Z","iopub.status.idle":"2023-12-07T07:09:28.570793Z","shell.execute_reply.started":"2023-12-07T07:09:28.288521Z","shell.execute_reply":"2023-12-07T07:09:28.569964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Creating Image Data Generators for Training and Validation**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Splitting the dataset into 80% for training and 20% for testing\ntrain_data, test_data = train_test_split(main_dataset, test_size=0.2, random_state=42)\n\n\n# Image dimensions and other parameters\nimg_width, img_height = 224, 224\nbatch_size = 32\n\n# Data generators for RGB images\ntrain_datagen = ImageDataGenerator(rescale=1./255)\nvalidation_datagen = ImageDataGenerator(rescale=1./255)\n\n# Train and validation generators\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_data,\n    x_col='image_path',\n    y_col='label',\n    target_size=(img_width, img_height),\n    batch_size=batch_size,\n    class_mode='categorical',\n    shuffle=True  # Set to True for training\n)\n\nvalidation_generator = validation_datagen.flow_from_dataframe(\n    dataframe=test_data,\n    x_col='image_path',\n    y_col='label',\n    target_size=(img_width, img_height),\n    batch_size=batch_size,\n    class_mode='categorical',\n    shuffle=False  # Set to False for validation/testing\n)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:38:16.885895Z","iopub.execute_input":"2023-12-07T06:38:16.88671Z","iopub.status.idle":"2023-12-07T06:38:17.127635Z","shell.execute_reply.started":"2023-12-07T06:38:16.886672Z","shell.execute_reply":"2023-12-07T06:38:17.126639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Checking Index Numbers From Train and Validation Dataset**","metadata":{}},{"cell_type":"code","source":"train_class_indices = train_generator.class_indices\nprint(\"Train Dataset Indexing:\", class_indices)\n\ntest_class_indices = validation_generator.class_indices\nprint(\"Validation Dataset Indexing:\", class_indices)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T07:05:15.613528Z","iopub.execute_input":"2023-12-07T07:05:15.613931Z","iopub.status.idle":"2023-12-07T07:05:15.619363Z","shell.execute_reply.started":"2023-12-07T07:05:15.613902Z","shell.execute_reply":"2023-12-07T07:05:15.618496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Display 5 images from the Train Dataset**","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Display 5 images with their respective details from the training generator\nfig, axes = plt.subplots(nrows=1, ncols=5, figsize=(15, 3))\n\nfor idx in range(5):\n    batch = train_generator.next()  # Get a batch of images from the generator\n    image = batch[0][idx]  # Fetch an image from the batch\n    label = batch[1][idx]  # Fetch the corresponding label\n    image_id = train_data.iloc[idx]['image_id']  # Fetch image ID from the DataFrame\n    actual_label = train_data.iloc[idx]['label']  # Fetch actual label from the DataFrame\n    \n    ax = axes[idx]\n    ax.imshow(image)\n    ax.set_title(f\"ID: {image_id}\\nNumerical Label: {label.argmax()}\\nCategorical Label: {actual_label}\\nShape: {image.shape}\", fontdict={'family':'serif'})\n    ax.axis('off')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:57:49.994116Z","iopub.execute_input":"2023-12-07T06:57:49.994803Z","iopub.status.idle":"2023-12-07T06:58:16.085307Z","shell.execute_reply.started":"2023-12-07T06:57:49.994769Z","shell.execute_reply":"2023-12-07T06:58:16.0845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Display 5 images from the Validation Dataset**","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Display 5 images with their respective details from the validation generator\nfig, axes = plt.subplots(nrows=1, ncols=5, figsize=(15, 3))\n\nfor idx in range(5):\n    batch = validation_generator.next()  # Get a batch of images from the generator\n    image = batch[0][idx]  # Fetch an image from the batch\n    label = batch[1][idx]  # Fetch the corresponding label\n    image_id = test_data.iloc[idx]['image_id']  # Fetch image ID from the DataFrame\n    actual_label = test_data.iloc[idx]['label']  # Fetch actual label from the DataFrame\n    \n    ax = axes[idx]\n    ax.imshow(image)\n    ax.set_title(f\"ID: {image_id}\\nNumerical Label: {label.argmax()}\\nCategorical Label: {actual_label}\\nShape: {image.shape}\", fontdict={'family':'serif'})\n    ax.axis('off')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T06:58:16.087065Z","iopub.execute_input":"2023-12-07T06:58:16.087696Z","iopub.status.idle":"2023-12-07T06:58:37.362022Z","shell.execute_reply.started":"2023-12-07T06:58:16.087659Z","shell.execute_reply":"2023-12-07T06:58:37.361132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Check for GPU availability**","metadata":{}},{"cell_type":"code","source":"# Check for GPU availability\nprint(\"GPU is\", \"available\" if tf.config.list_physical_devices('GPU') else \"NOT available\")\n\n# Set TensorFlow to use the GPU device\nif tf.config.list_physical_devices('GPU'):\n    tf.config.experimental.set_memory_growth(tf.config.list_physical_devices('GPU')[0], True)\n    print(\"GPU device configured\")\nelse:\n    print(\"No GPU device found\")","metadata":{"execution":{"iopub.status.busy":"2023-12-07T07:09:57.946704Z","iopub.execute_input":"2023-12-07T07:09:57.947606Z","iopub.status.idle":"2023-12-07T07:09:58.321642Z","shell.execute_reply.started":"2023-12-07T07:09:57.947548Z","shell.execute_reply":"2023-12-07T07:09:58.320622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **ResNet152 as Feature Extractor**","metadata":{}},{"cell_type":"code","source":"BATCH_SIZE = 8\nIMG_WIDTH = 224\nIMG_HEIGHT = 224\n\ndef create_model(summary=True):\n    # apply transfer learning\n    new_input = Input(shape=(IMG_WIDTH, IMG_HEIGHT, 3))\n    model = ResNet152(weights='imagenet', include_top=False, input_tensor=new_input)\n    # add new classifier layers\n    flat1 = Flatten()(model.layers[-1].output)\n    output = Dense(5, activation='softmax')(flat1)\n    # define new model\n    model = Model(inputs=model.inputs, outputs=output)\n    model.compile(optimizer=Adam(learning_rate=1e-4), loss='categorical_crossentropy', metrics=['accuracy'])\n    if summary:\n        print(model.summary())\n    return model\n  \nmodel_dir = '/kaggle/working/Ovarian_Cancer_Subtype_Classification/model'\n\nif not os.path.exists(model_dir):\n    os.makedirs(model_dir)\n\ncheckpoint_path = model_dir + '/cp.ckpt'\ncheckpoint_dir = os.path.dirname(checkpoint_path)\n\n# Create a callback that saves the model's weights\ncp_callback = tf.keras.callbacks.ModelCheckpoint(filepath=checkpoint_path,\n                                                 save_weights_only=True,\n                                                 verbose=1)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T08:09:07.536762Z","iopub.execute_input":"2023-12-07T08:09:07.53746Z","iopub.status.idle":"2023-12-07T08:09:07.546619Z","shell.execute_reply.started":"2023-12-07T08:09:07.537429Z","shell.execute_reply":"2023-12-07T08:09:07.545765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = create_model()","metadata":{"execution":{"iopub.status.busy":"2023-12-07T08:09:10.761432Z","iopub.execute_input":"2023-12-07T08:09:10.762181Z","iopub.status.idle":"2023-12-07T08:09:16.917451Z","shell.execute_reply.started":"2023-12-07T08:09:10.762143Z","shell.execute_reply":"2023-12-07T08:09:16.916469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_generator,\n    steps_per_epoch=10,\n    epochs=5,\n    validation_data=validation_generator,\n    validation_steps=10,\n    callbacks=[cp_callback]\n)","metadata":{"execution":{"iopub.status.busy":"2023-12-07T08:10:14.704733Z","iopub.execute_input":"2023-12-07T08:10:14.705499Z","iopub.status.idle":"2023-12-07T08:17:32.581161Z","shell.execute_reply.started":"2023-12-07T08:10:14.705458Z","shell.execute_reply":"2023-12-07T08:17:32.580321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom matplotlib.lines import Line2D\nfrom matplotlib.legend_handler import HandlerLine2D\nimport numpy as np\n\n# Plot training & validation loss values\nplt.figure(figsize=(8, 6))\n\n# Plot Loss\ntrain_loss, = plt.plot(history.history['loss'], label='Train Loss', color='blue')\nval_loss, = plt.plot(history.history['val_loss'], label='Validation Loss', color='orange')\ntrain_accuracy, = plt.plot(history.history['accuracy'], label='Train Accuracy',  color='green')\nval_accuracy, = plt.plot(history.history['val_accuracy'], label='Validation Accuracy', color='red')\n\n# Set x-axis label with specified font properties\nplt.xlabel('No. of Epochs', fontdict={'family': 'Serif', 'weight': 'bold', 'size': 12})\n\n# Set x-axis ticks font properties\nplt.xticks(np.linspace(0, len(history.history['loss']), num=6), fontname='Serif', weight='bold')\n\n# Set y-axis ticks font properties\nplt.yticks(np.linspace(0.2, 1, num=5), fontname='Serif', weight='bold')\n\n# Set the x-axis and y-axis limits\nplt.xlim(0, len(history.history['loss']))\nplt.ylim(0, 1)\n\n# Define custom legend lines with desired line properties\nlegend_lines = [\n    Line2D([0], [0], color='blue', lw=3),          # Train Loss\n    Line2D([0], [0], color='orange', lw=3),       # Validation Loss\n    Line2D([0], [0], color='green', lw=3),        # Train Accuracy\n    Line2D([0], [0], color='red', lw=3)           # Validation Accuracy\n]\n\n# Place legend outside the graph by adjusting bbox_to_anchor and specifying it to be outside the axes\nplt.legend(legend_lines, ['Train Loss', 'Validation Loss', 'Train Accuracy', 'Validation Accuracy'],\n           loc='upper center', bbox_to_anchor=(0.5, 1.1), ncol=4,\n           prop={'family': 'Serif', 'weight': 'bold', 'size': 8}, frameon=False,\n           handler_map={Line2D: HandlerLine2D(numpoints=5)})\n\n# Adjust padding between x-axis label and x-axis ticks\nplt.gca().xaxis.labelpad = 10  # Change the value as needed to adjust the space\n\n# Remove top and right spines\nplt.gca().spines['top'].set_visible(False)\nplt.gca().spines['right'].set_visible(False)\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T08:18:06.989693Z","iopub.execute_input":"2023-12-07T08:18:06.990637Z","iopub.status.idle":"2023-12-07T08:18:07.19884Z","shell.execute_reply.started":"2023-12-07T08:18:06.990592Z","shell.execute_reply":"2023-12-07T08:18:07.197836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Sample Submission**","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Directory containing test images\ntest_images_directory = '/kaggle/input/UBC-OCEAN/test_thumbnails/'\n\n# Function to extract image IDs from filenames\ndef extract_image_id(file_name):\n    return file_name.split('_')[0]\n\n# Get a list of test image filenames\ntest_image_files = os.listdir(test_images_directory)\n\n# Extract image IDs from filenames\nimage_ids = [extract_image_id(file) for file in test_image_files]\n\n# Create an empty list to store image IDs and predicted labels\nresults_data = []\n\n# Predict labels for each image and add to the list\nfor file_name, image_id in zip(test_image_files, image_ids):\n    image_path = os.path.join(test_images_directory, file_name)\n    \n    # Load and preprocess the image\n    img = image.load_img(image_path, target_size=(224, 224))\n    img_array = image.img_to_array(img)\n    img_array = np.expand_dims(img_array, axis=0)\n    img_array = img_array / 255.0  # Normalize the image data\n    \n    # Perform prediction on the image\n    prediction = model.predict(img_array)\n    \n    # Get the predicted label index\n    predicted_label_index = np.argmax(prediction, axis=1)[0]\n    \n    # Define the mapping of numeric labels to categorical labels\n    label_mapping = {0: 'CC', 1: 'EC', 2: 'HGSC', 3: 'LGSC', 4: 'MC'}\n    \n    # Convert the predicted label index to categorical label\n    predicted_label_categorical = label_mapping[predicted_label_index]\n    \n    # Append image ID and predicted label to the list\n    results_data.append({'image_id': image_id, 'label': predicted_label_categorical})\n\n# Convert the list of dictionaries to a DataFrame\nresults_df = pd.DataFrame(results_data)\n\n# Save the DataFrame to a CSV file\nresults_df.to_csv('/kaggle/working/submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T08:24:21.582982Z","iopub.execute_input":"2023-12-07T08:24:21.583901Z","iopub.status.idle":"2023-12-07T08:24:21.839673Z","shell.execute_reply.started":"2023-12-07T08:24:21.58387Z","shell.execute_reply":"2023-12-07T08:24:21.838849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\n# Manually specify the checkpoint file name\ncheckpoint_dir = '/kaggle/working/Ovarian_Cancer_Subtype_Classification/model'\n\n# Load the latest checkpoint file\nlatest_checkpoint = tf.train.latest_checkpoint(checkpoint_dir)\n\nif latest_checkpoint is not None:\n    # Create a new model instance\n    loaded_model = create_model(summary=True)\n\n    # Load the previously saved weights\n    loaded_model.load_weights(latest_checkpoint)\nelse:\n    print(\"No checkpoint file found in the specified directory.\")\n","metadata":{"execution":{"iopub.status.busy":"2023-12-07T08:18:29.623115Z","iopub.execute_input":"2023-12-07T08:18:29.623739Z","iopub.status.idle":"2023-12-07T08:18:41.762417Z","shell.execute_reply.started":"2023-12-07T08:18:29.623705Z","shell.execute_reply":"2023-12-07T08:18:41.761619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}