{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":6974585,"sourceType":"datasetVersion","datasetId":4007605},{"sourceId":7219155,"sourceType":"datasetVersion","datasetId":4178227}],"dockerImageVersionId":30580,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# UBC Ovarian Cancer Subtype Classification and Outlier Detection","metadata":{}},{"cell_type":"code","source":"try:   \n    import numpy as np\n    import pandas as pd\n    import cv2\n    import matplotlib.pyplot as plt\n    import seaborn as sns\n    import re\n    import os\n    import warnings\n    import shutil\n    from skimage import io, color\n    from sklearn.model_selection import train_test_split\n    from sklearn.svm import SVC\n    from sklearn import metrics\n    from sklearn.preprocessing import StandardScaler, LabelEncoder\n    from imblearn.combine import SMOTEENN\n    from imblearn.over_sampling import SMOTE\n    from tensorflow.keras.utils import to_categorical\n    from tensorflow.keras.preprocessing.image import ImageDataGenerator\n    from tensorflow.keras.applications import ResNet50V2\n    from tensorflow.keras.callbacks import EarlyStopping\n    from tensorflow.keras import layers, models, optimizers\n    from PIL import Image\n\n    warnings.simplefilter(action='ignore', category=FutureWarning)\n    warnings.filterwarnings(\"ignore\", category=UserWarning, message=\".*low contrast image.*\")\nexcept ImportError as import_error:\n    print(f\"Import Error: {import_error}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-18T10:59:53.898062Z","iopub.execute_input":"2023-12-18T10:59:53.898522Z","iopub.status.idle":"2023-12-18T10:59:53.908638Z","shell.execute_reply.started":"2023-12-18T10:59:53.898486Z","shell.execute_reply":"2023-12-18T10:59:53.907498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data exploration and analysis","metadata":{}},{"cell_type":"code","source":"try:    \n    path_to_data = '/kaggle/input/UBC-OCEAN/'\n    df = pd.read_csv(path_to_data + 'train.csv')\n\n    print(\"Dataset Overview:\")\n    print(df.info())\n\n    print(\"\\nSummary Statistics:\")\n    print(df.describe())\n\n    print(\"\\nMissing Values:\")\n    print(df.isnull().sum())\n    \nexcept Exception as data_exploration_error:\n    print(f\"Data Exploration Error: {data_exploration_error}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-18T10:59:53.910821Z","iopub.execute_input":"2023-12-18T10:59:53.911151Z","iopub.status.idle":"2023-12-18T10:59:53.939041Z","shell.execute_reply.started":"2023-12-18T10:59:53.911114Z","shell.execute_reply":"2023-12-18T10:59:53.938112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Class distribution","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.countplot(x='label', data=df)\nplt.title('Distribution of Target Classes')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-18T10:59:53.940409Z","iopub.execute_input":"2023-12-18T10:59:53.94072Z","iopub.status.idle":"2023-12-18T10:59:54.184116Z","shell.execute_reply.started":"2023-12-18T10:59:53.940695Z","shell.execute_reply":"2023-12-18T10:59:54.183247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Distribution of image dimentions","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nsns.scatterplot(x='image_width', y='image_height', hue='label', data=df)\nplt.title('Distribution of Image Dimensions')\nplt.xlabel('Image Width (pixels)')\nplt.ylabel('Image Height (pixels)')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-18T10:59:54.185159Z","iopub.execute_input":"2023-12-18T10:59:54.185416Z","iopub.status.idle":"2023-12-18T10:59:54.617583Z","shell.execute_reply.started":"2023-12-18T10:59:54.185393Z","shell.execute_reply":"2023-12-18T10:59:54.616706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### TMA","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 5))\nsns.countplot(x='is_tma', data=df)\nplt.title('Distribution of Tissue Microarray (TMA)')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-18T10:59:54.620125Z","iopub.execute_input":"2023-12-18T10:59:54.620561Z","iopub.status.idle":"2023-12-18T10:59:54.774076Z","shell.execute_reply.started":"2023-12-18T10:59:54.620534Z","shell.execute_reply":"2023-12-18T10:59:54.773225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Samples","metadata":{}},{"cell_type":"code","source":"sample_images = df.sample(5)\nplt.figure(figsize=(15, 8))\nfor i, (_, row) in enumerate(sample_images.iterrows(), 1):\n    plt.subplot(2, 3, i)\n   \n    img_path = path_to_data + f'train_thumbnails/{row[\"image_id\"]}_thumbnail.png'\n    if os.path.exists(img_path):\n        img = plt.imread(img_path)\n        plt.imshow(img)\n        plt.title(f'Class: {row[\"label\"]}')\n        plt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-18T10:59:54.775269Z","iopub.execute_input":"2023-12-18T10:59:54.775734Z","iopub.status.idle":"2023-12-18T11:00:01.966428Z","shell.execute_reply.started":"2023-12-18T10:59:54.7757Z","shell.execute_reply":"2023-12-18T11:00:01.965499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Image tiling","metadata":{}},{"cell_type":"code","source":"# not recommended\nImage.MAX_IMAGE_PIXELS = None","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:00:01.96765Z","iopub.execute_input":"2023-12-18T11:00:01.967945Z","iopub.status.idle":"2023-12-18T11:00:01.97228Z","shell.execute_reply.started":"2023-12-18T11:00:01.967919Z","shell.execute_reply":"2023-12-18T11:00:01.9715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Defining a fuction to tile the image with overlap, eliminate black images and generate a new csv file for labeling :\n","metadata":{}},{"cell_type":"code","source":"def tile_generator(image_path, tile_size, overlap, label, black_threshold=70):\n    image = io.imread(image_path)\n    image_name = os.path.splitext(os.path.basename(image_path))[0]\n\n    # tile the image with overlap\n    step_size = int(tile_size * (1 - overlap))\n    for i in range(0, image.shape[0] - tile_size + 1, step_size):\n        for j in range(0, image.shape[1] - tile_size + 1, step_size):\n            tile = image[i:i+tile_size, j:j+tile_size, :]\n\n            # eliminate black images\n            grayscale_tile = color.rgb2gray(tile)\n            black_percentage = (grayscale_tile < 0.5).sum() / grayscale_tile.size * 100\n\n            if black_percentage < black_threshold:\n                tile_filename = f'{image_name}_tile_{i}_{j}.png'\n                yield tile, label, tile_filename","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:00:01.973353Z","iopub.execute_input":"2023-12-18T11:00:01.973665Z","iopub.status.idle":"2023-12-18T11:00:01.983316Z","shell.execute_reply.started":"2023-12-18T11:00:01.973638Z","shell.execute_reply":"2023-12-18T11:00:01.982485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tile_and_save(image_path, tile_size, overlap, output_dir, label, csv_output_path, black_threshold=70):\n    generator = tile_generator(image_path, tile_size, overlap, label, black_threshold)\n    \n    csv_data = []\n\n    for idx, (tile, label, tile_filename) in enumerate(generator):\n        tile_filename = os.path.join(output_dir, tile_filename)\n        io.imsave(tile_filename, tile)\n        csv_data.append([tile_filename, label])\n\n    # save the csv file\n    csv_df = pd.DataFrame(csv_data, columns=[\"tile_path\", \"label\"])\n    csv_df.to_csv(csv_output_path, mode='a', header=not os.path.exists(csv_output_path), index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:00:01.984276Z","iopub.execute_input":"2023-12-18T11:00:01.984555Z","iopub.status.idle":"2023-12-18T11:00:01.993321Z","shell.execute_reply.started":"2023-12-18T11:00:01.984532Z","shell.execute_reply":"2023-12-18T11:00:01.992495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Actula tiling\n\n*modify directory paths as needed*","metadata":{}},{"cell_type":"code","source":"# set parameters\nimages_directory = \"/kaggle/input/UBC-OCEAN/train_thumbnails\" # \"/kaggle/input/UBC-OCEAN/train_images\" \noutput_directory = \"/kaggle/working/tiles\"\n\n# clear output directory or create it if it doesn't exsist\nif os.path.exists(output_directory):\n    shutil.rmtree(output_directory)\nif not os.path.exists(output_directory):\n    os.makedirs(output_directory)\n\n# csv file, create if it doesn't exsist\ncsv_output_path = \"/kaggle/working/tiles_labels.csv\"\nif os.path.exists(csv_output_path):\n    with open(csv_output_path, 'w') as file:\n        # Write an empty string to clear the content\n        file.write(\"\")\n    columns=[\"tile_path\", \"label\"]\n    df = pd.DataFrame(columns=columns)\n    df.to_csv(csv_output_path, index=False)\n    \nif not os.path.exists(csv_output_path):\n    columns=[\"tile_path\", \"label\"]\n    df = pd.DataFrame(columns=columns)\n    df.to_csv(csv_output_path, index=False)\n    \n# tiling parameters\ntile_size = 224 \noverlap = 0.2\ndata = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\") \n\n\n# go through train.csv and tile each image\ntry:\n    for index, row in data.iterrows():\n        image_id = row[\"image_id\"]\n        label = row[\"label\"]\n        image_path = os.path.join(images_directory, f\"{image_id}_thumbnail.png\") # change to \"{image_id}.png\" ?\n        try:\n            if os.path.exists(image_path):\n                tile_and_save(image_path, tile_size, overlap, output_directory, label, csv_output_path)\n            print(f\"LOG: image : {image_id}, has been tiled!\")\n        except Exception as tile_exception:\n            print(f\"Error while tiling image {image_id}: {tile_exception}\")\n        # end loop early for testing, delete line below \n        if(index == 5): break # ONLY x IMAGES\nexcept Exception as tiling_error:\n    print(f\"Tiling Error: {tiling_error}\")    \n# display head of new csv file \nnew_csv_data = pd.read_csv(csv_output_path)\nprint(\"NEW CSV FILE CREATED : \")\nprint(new_csv_data.head())\n","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:00:01.994543Z","iopub.execute_input":"2023-12-18T11:00:01.99486Z","iopub.status.idle":"2023-12-18T11:00:14.167644Z","shell.execute_reply.started":"2023-12-18T11:00:01.994836Z","shell.execute_reply":"2023-12-18T11:00:14.166696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Display examples","metadata":{}},{"cell_type":"code","source":"# displaying some of the tiles \noutput_directory = \"/kaggle/working/tiles\"\n\nnum_displayed_tiles = 4\nnum_rows = (num_displayed_tiles + 1) // 2\nnum_cols = 2\n\nplt.figure(figsize=(12, 12))\n\n# list of image files in the directory\nimage_files = [f for f in os.listdir(output_directory) if f.endswith('.png')]\n\nfor i in range(num_displayed_tiles):\n    if i < len(image_files):\n        tile_filename = os.path.join(output_directory, image_files[i])\n        tile = io.imread(tile_filename)\n        plt.subplot(num_rows, num_cols, i + 1)  \n        plt.imshow(tile)\n        plt.axis('off')\n        plt.title(f'Tile {i + 1}')\n    else:\n        # display an empty plot, if there are less then 4 images\n        plt.subplot(num_rows, num_cols, i + 1)\n        plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:00:14.169008Z","iopub.execute_input":"2023-12-18T11:00:14.169367Z","iopub.status.idle":"2023-12-18T11:00:15.50048Z","shell.execute_reply.started":"2023-12-18T11:00:14.169336Z","shell.execute_reply":"2023-12-18T11:00:15.499204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Class distribution check","metadata":{}},{"cell_type":"code","source":"tiles_labels_df = pd.read_csv(\"/kaggle/working/tiles_labels.csv\")\nclass_counts = tiles_labels_df['label'].value_counts()\nprint(class_counts)","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:00:15.501653Z","iopub.execute_input":"2023-12-18T11:00:15.501956Z","iopub.status.idle":"2023-12-18T11:00:15.511557Z","shell.execute_reply.started":"2023-12-18T11:00:15.50193Z","shell.execute_reply":"2023-12-18T11:00:15.510655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data preparation\n","metadata":{}},{"cell_type":"markdown","source":"### Load and Preprocess Data","metadata":{}},{"cell_type":"code","source":"tiles_labels_df = pd.read_csv(\"/kaggle/working/tiles_labels.csv\")\n\n# Pixel normalization\nX_images = [cv2.imread(img_path) / 255.0 for img_path in tiles_labels_df['tile_path']]\nX_images = np.array(X_images)\ny_labels = tiles_labels_df['label']\n\n# Label encoding\nlabel_encoder = LabelEncoder()\ny_encoded = label_encoder.fit_transform(y_labels)\n\n# Train-test split\nX_train, X_test, y_train, y_test = train_test_split(\n    X_images, y_encoded, test_size=0.2, random_state=42, stratify=y_encoded\n)","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:00:15.512636Z","iopub.execute_input":"2023-12-18T11:00:15.512878Z","iopub.status.idle":"2023-12-18T11:00:17.81922Z","shell.execute_reply.started":"2023-12-18T11:00:15.512857Z","shell.execute_reply":"2023-12-18T11:00:17.81821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_values1 = np.unique(y_encoded)\nprint(\"Unique values in y_encoded:\", unique_values1)\nunique_values2 = np.unique(y_labels)\nprint(\"correspond to:\", unique_values2)","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:00:17.823693Z","iopub.execute_input":"2023-12-18T11:00:17.82399Z","iopub.status.idle":"2023-12-18T11:00:17.830345Z","shell.execute_reply.started":"2023-12-18T11:00:17.823965Z","shell.execute_reply":"2023-12-18T11:00:17.829437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Before using SMOTE :**","metadata":{}},{"cell_type":"code","source":"unique_classes, class_counts = np.unique(y_train, return_counts=True)\n\nprint(\"Number of different classes:\", len(unique_classes))\nprint(\"Instances for each class:\")\nfor class_label, count in zip(unique_classes, class_counts):\n    print(f\"Class {class_label}: {count} instances\")\n\nmax_num_samples = np.max(class_counts)\nprint(f\"max num of samples = {max_num_samples}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:00:17.831399Z","iopub.execute_input":"2023-12-18T11:00:17.831706Z","iopub.status.idle":"2023-12-18T11:00:17.842058Z","shell.execute_reply.started":"2023-12-18T11:00:17.831661Z","shell.execute_reply":"2023-12-18T11:00:17.841088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Oversampling with SMOTE\nX_train_flattened = X_train.reshape(X_train.shape[0], -1)\nsmote = SMOTE(sampling_strategy='auto', random_state=42)\nX_train_resampled, y_train_resampled = smote.fit_resample(X_train_flattened, y_train)\n\n# Reshape back to the original format\nX_train_resampled = X_train_resampled.reshape(-1, 224, 224, 3)","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:00:17.842988Z","iopub.execute_input":"2023-12-18T11:00:17.84324Z","iopub.status.idle":"2023-12-18T11:00:19.767666Z","shell.execute_reply.started":"2023-12-18T11:00:17.843213Z","shell.execute_reply":"2023-12-18T11:00:19.766854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**After using SMOTE :**","metadata":{}},{"cell_type":"code","source":"unique_classes1, class_counts1 = np.unique(y_train_resampled, return_counts=True)\n\nprint(\"Number of different classes:\", len(unique_classes1))\nprint(\"Instances for each class:\")\nfor class_label, count in zip(unique_classes1, class_counts1):\n    print(f\"Class {class_label}: {count} instances\")\n","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:00:19.768772Z","iopub.execute_input":"2023-12-18T11:00:19.769067Z","iopub.status.idle":"2023-12-18T11:00:19.775374Z","shell.execute_reply.started":"2023-12-18T11:00:19.769042Z","shell.execute_reply":"2023-12-18T11:00:19.774495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Delete unnecessary variables","metadata":{}},{"cell_type":"code","source":"del X_images, y_labels, df","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:00:19.776757Z","iopub.execute_input":"2023-12-18T11:00:19.777059Z","iopub.status.idle":"2023-12-18T11:00:19.787602Z","shell.execute_reply.started":"2023-12-18T11:00:19.777033Z","shell.execute_reply":"2023-12-18T11:00:19.786707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data augmentation","metadata":{}},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(\n    rotation_range=15,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    shear_range=0.1,\n    zoom_range=0.1,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\ntrain_datagen.fit(X_train_resampled)\n\naugmented_images_per_original = 2\naugmented_data = []\nfor X_batch, y_batch in train_datagen.flow(\n    X_train_resampled, y_train_resampled, batch_size=augmented_images_per_original\n):\n    augmented_data.append((X_batch, y_batch))\n    if len(augmented_data) >= len(X_train_resampled):\n        break\n\nX_train_augmented = np.concatenate([X_train_resampled] + [data[0] for data in augmented_data])\ny_train_augmented = np.concatenate([y_train_resampled] * (augmented_images_per_original + 1))\n\naugmented_indices = np.arange(len(X_train_augmented))\nnp.random.shuffle(augmented_indices)\nX_train_augmented = X_train_augmented[augmented_indices]\ny_train_augmented = y_train_augmented[augmented_indices]","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:00:19.788658Z","iopub.execute_input":"2023-12-18T11:00:19.788935Z","iopub.status.idle":"2023-12-18T11:00:37.051009Z","shell.execute_reply.started":"2023-12-18T11:00:19.788897Z","shell.execute_reply":"2023-12-18T11:00:37.050232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ResNet","metadata":{}},{"cell_type":"code","source":"weights_path = '/kaggle/input/weights/resnet50v2_weights_tf_dim_ordering_tf_kernels_notop.h5'\nbase_model = ResNet50V2(\n    include_top=False, weights=weights_path, input_shape=(224, 224, 3)\n)\n\nmodel = models.Sequential([\n    base_model,\n    layers.GlobalAveragePooling2D(),\n    layers.Dense(64, activation='relu'),\n    layers.Dense(len(unique_classes), activation='softmax') \n])\n\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\ntrain_generator = train_datagen.flow(X_train_augmented, y_train_augmented, batch_size=16)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:00:37.052114Z","iopub.execute_input":"2023-12-18T11:00:37.052396Z","iopub.status.idle":"2023-12-18T11:00:44.382956Z","shell.execute_reply.started":"2023-12-18T11:00:37.052371Z","shell.execute_reply":"2023-12-18T11:00:44.381885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"early_stopping = EarlyStopping(monitor='val_loss', patience=3, restore_best_weights=True)\ntry:\n    model.fit(\n        train_generator,\n        epochs=2, #100\n        validation_data=(X_test, y_test),\n        # callbacks=[early_stopping],\n        verbose=1\n    )\nexcept Exception as model_training_error:\n    print(f\"Model Training Error: {model_training_error}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:00:44.384186Z","iopub.execute_input":"2023-12-18T11:00:44.384482Z","iopub.status.idle":"2023-12-18T11:02:19.683319Z","shell.execute_reply.started":"2023-12-18T11:00:44.384436Z","shell.execute_reply":"2023-12-18T11:02:19.682382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Preparing test data","metadata":{}},{"cell_type":"code","source":"def get_number_from_filename(filename):\n    match = re.match(r'^(\\d+)_', filename)\n    if match:\n        return int(match.group(1))\n    else:\n        return None\n\nimage_dir_test = '/kaggle/input/UBC-OCEAN/test_thumbnails'\nimages_test = []\nnumbers_test = []\n\nimage_files = [f for f in os.listdir(image_dir_test) if os.path.isfile(os.path.join(image_dir_test, f))]\ntry:\n    for image_file in image_files:\n        try:\n            image_path = os.path.join(image_dir_test, image_file)\n            img = cv2.imread(image_path)\n            img = cv2.resize(img, (224, 224))\n            img_array = np.array(img)\n            number = get_number_from_filename(image_file)\n\n            images_test.append(img_array)\n            numbers_test.append(number)\n\n        except Exception as e:\n            print(f\"An error occurred treating the file : {image_file}: {str(e)}\")\nexcept Exception as test_data_prep_error:\n    print(f\"Test Data Preparation Error: {test_data_prep_error}\")\n    \nX_test = np.array(images_test)\nX_test_normalized = X_test / 255.0","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:02:19.684942Z","iopub.execute_input":"2023-12-18T11:02:19.685639Z","iopub.status.idle":"2023-12-18T11:02:19.894586Z","shell.execute_reply.started":"2023-12-18T11:02:19.685601Z","shell.execute_reply":"2023-12-18T11:02:19.893744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prediction","metadata":{}},{"cell_type":"code","source":"# Model Prediction\ny_pred = model.predict(X_test_normalized)\ny_pred_categories = label_encoder.inverse_transform([np.argmax(y) for y in y_pred])","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:02:19.895833Z","iopub.execute_input":"2023-12-18T11:02:19.896156Z","iopub.status.idle":"2023-12-18T11:02:21.394761Z","shell.execute_reply.started":"2023-12-18T11:02:19.89613Z","shell.execute_reply":"2023-12-18T11:02:21.394008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission","metadata":{}},{"cell_type":"code","source":"# Create Submission DataFrame\nsubmission_df = pd.DataFrame(\n    {'image_id': numbers_test, 'label': y_pred_categories},\n    columns=['image_id', 'label']\n)\n\n# Save Submission CSV\nsubmission_df.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-18T11:02:21.395867Z","iopub.execute_input":"2023-12-18T11:02:21.396149Z","iopub.status.idle":"2023-12-18T11:02:21.403155Z","shell.execute_reply.started":"2023-12-18T11:02:21.396125Z","shell.execute_reply":"2023-12-18T11:02:21.402183Z"},"trusted":true},"execution_count":null,"outputs":[]}]}