{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":9900,"sourceType":"datasetVersion","datasetId":6209},{"sourceId":6774553,"sourceType":"datasetVersion","datasetId":3898019},{"sourceId":146934283,"sourceType":"kernelVersion"},{"sourceId":150735049,"sourceType":"kernelVersion"},{"sourceId":150771473,"sourceType":"kernelVersion"},{"sourceId":150840073,"sourceType":"kernelVersion"}],"dockerImageVersionId":30580,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!ls /kaggle/input/pyvips-python-and-deb-package-gpu\n# intall the deb packages\n!yes | dpkg -i --force-depends /kaggle/input/pyvips-python-and-deb-package-gpu/linux_packages/archives/*.deb\n# install the python wrapper\n!pip install pyvips -f /kaggle/input/pyvips-python-and-deb-package-gpu/python_packages/ --no-index","metadata":{"scrolled":true,"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-11-16T19:00:53.683929Z","iopub.execute_input":"2023-11-16T19:00:53.68435Z","iopub.status.idle":"2023-11-16T19:02:09.20648Z","shell.execute_reply.started":"2023-11-16T19:00:53.684288Z","shell.execute_reply":"2023-11-16T19:02:09.205235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%matplotlib inline\n\nimport os\nimport pyvips\nimport numpy as np\nfrom PIL import Image\nfrom sklearn.preprocessing import LabelEncoder\nimport matplotlib.pyplot as plt\nimport pandas as pd\n\nimport os, glob\nimport numpy as np\nimport random\nfrom PIL import Image\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\n\nimport torch\n\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nimport tensorflow_addons as tfa\nimport cv2\nfrom collections import Counter\nimport pickle \nos.environ['VIPS_CONCURRENCY'] = '4'\nos.environ['VIPS_DISC_THRESHOLD'] = '24gb'\n\n\n# Later on, loading the model from the file\nwith open('/kaggle/input/patch-extraction-based-on-k-means-clustering/model-kmeans.pkl', 'rb') as file:\n    loaded_model = pickle.load(file)\n    \nfeatures = np.load(\"/kaggle/input/patch-extraction-based-on-k-means-clustering/features.npy\")\n\ncluster = loaded_model.predict(features)\n\ntrain_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/test.csv\")\n\n\nBASE_DIR = [\"/kaggle/input/UBC-OCEAN/train_thumbnails/\", \"/kaggle/input/UBC-OCEAN/test_thumbnails/\"]\nTRAIN_DIR= \"/kaggle/input/UBC-OCEAN/train_images\"\nTEST_DIR= \"/kaggle/input/UBC-OCEAN/test_images\"","metadata":{"execution":{"iopub.status.busy":"2023-11-16T19:02:09.209157Z","iopub.execute_input":"2023-11-16T19:02:09.209613Z","iopub.status.idle":"2023-11-16T19:02:41.976119Z","shell.execute_reply.started":"2023-11-16T19:02:09.209574Z","shell.execute_reply":"2023-11-16T19:02:41.974529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/input/robust-feature-extraction-resnet-k-means-t-sne/feature_vectors.csv')\ndata['cluster']=cluster","metadata":{"execution":{"iopub.status.busy":"2023-11-16T19:02:41.977694Z","iopub.execute_input":"2023-11-16T19:02:41.978066Z","iopub.status.idle":"2023-11-16T19:04:20.025649Z","shell.execute_reply.started":"2023-11-16T19:02:41.978034Z","shell.execute_reply":"2023-11-16T19:04:20.02438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"indices=train_df['image_id'].values\nrandom.shuffle(indices)\ntrain_indices=indices[:int(len(indices)*0.8)]\ntest_indices=indices[int(len(indices)*0.8):]\ndata=data[data['cluster'].isin([1,3])]","metadata":{"execution":{"iopub.status.busy":"2023-11-16T19:09:18.521895Z","iopub.execute_input":"2023-11-16T19:09:18.522413Z","iopub.status.idle":"2023-11-16T19:09:19.037217Z","shell.execute_reply.started":"2023-11-16T19:09:18.522378Z","shell.execute_reply":"2023-11-16T19:09:19.036058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()\n\ntrain = data[data['image_id'].isin(train_indices)]\nlabel_encoder.fit(train['class'])\ntrain['class'] = label_encoder.fit_transform(train['class'])\n\ntest = data[data['image_id'].isin(test_indices)]\ntest['class'] = label_encoder.fit_transform(test['class'])\n\ncolumns = [f'pixel{i}' for i in range(1, 2049)]\n\nX_train, y_train = train[columns].values, train['class'].values\nX_test, y_test = test[columns].values, test['class'].values\n\nprint(\"Training X shape {}, y shape {}\".format(X_train.shape, y_train.shape))\nprint(\"Testing X shape {}, y shape {}\".format(X_test.shape, y_test.shape))\n\nX, y = data[columns].values, label_encoder.fit_transform(data['class'])\nprint(\"All X shape {}, y shape {}\".format(X.shape, y.shape))\n","metadata":{"execution":{"iopub.status.busy":"2023-11-16T19:13:10.082267Z","iopub.execute_input":"2023-11-16T19:13:10.082689Z","iopub.status.idle":"2023-11-16T19:13:11.570369Z","shell.execute_reply.started":"2023-11-16T19:13:10.08266Z","shell.execute_reply":"2023-11-16T19:13:11.569148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import xgboost as xgb\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import accuracy_score\n\n# Assuming you have training and testing datasets as follows:\n\n# Initialize the XGBClassifier\nxgbmodel = XGBClassifier({'colsample_bytree': 0.7, #Hyperparameters chosen based on the below CV method\n                          'learning_rate': 0.2, \n                          'max_depth': 5, \n                          'n_estimators': 300, \n                          'subsample': 0.8}, \n                         tree_method = \"hist\", device = \"cuda\")\n\n# Fit the model with training data\nxgbmodel.fit(X, y)\n\n# Predict the labels for the test set\ny_pred = xgbmodel.predict(X_test)\n\n# Calculate the accuracy\naccuracy = accuracy_score(y_test, y_pred)\nprint(f\"Model Accuracy: {accuracy * 100:.2f}%\")\n","metadata":{"execution":{"iopub.status.busy":"2023-11-16T19:13:20.8739Z","iopub.execute_input":"2023-11-16T19:13:20.874357Z","iopub.status.idle":"2023-11-16T19:14:05.964277Z","shell.execute_reply.started":"2023-11-16T19:13:20.874316Z","shell.execute_reply":"2023-11-16T19:14:05.963027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import GridSearchCV\n\n# # Define the parameter grid\n# param_grid = {\n#     'max_depth': [3, 5, 7],\n#     'learning_rate': [0.01, 0.1, 0.2],\n#     'n_estimators': [100, 200, 300],\n#     'subsample': [0.7, 0.8, 0.9],\n#     'colsample_bytree': [0.7, 0.8, 0.9],\n# }\n\n# # Initialize the XGBClassifier\n# xgb_clf = XGBClassifier(tree_method = \"hist\", device = \"cuda\")\n\n# # Initialize the GridSearchCV object\n# grid_search = GridSearchCV(estimator=xgb_clf, param_grid=param_grid, scoring='accuracy', cv=3, n_jobs=-1)\n\n# # Fit the grid search to the data\n# grid_search.fit(X_train, y_train)\n\n# # Get the best parameters and the best score\n# best_parameters = grid_search.best_params_\n# best_score = grid_search.best_score_\n\n# print(f\"Best Parameters: {best_parameters}\")\n# print(f\"Best Score: {best_score}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-16T03:15:00.215422Z","iopub.execute_input":"2023-11-16T03:15:00.215734Z","iopub.status.idle":"2023-11-16T03:15:00.220573Z","shell.execute_reply.started":"2023-11-16T03:15:00.215708Z","shell.execute_reply":"2023-11-16T03:15:00.219532Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_class=list(label_encoder.inverse_transform(y_pred))\ntest.loc[:, 'class'] = _class","metadata":{"execution":{"iopub.status.busy":"2023-11-16T19:20:09.322893Z","iopub.execute_input":"2023-11-16T19:20:09.323352Z","iopub.status.idle":"2023-11-16T19:20:09.344858Z","shell.execute_reply.started":"2023-11-16T19:20:09.323285Z","shell.execute_reply":"2023-11-16T19:20:09.343206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# Suppose you have an encoded variable 'encoded_labels'\nencoded_labels = y_pred # Example encoded labels\n\n# Inverse transform the encoded labels back to original class names\noriginal_class_names = label_encoder.inverse_transform(encoded_labels)\n\ntest.loc[:, 'class']=label_encoder.inverse_transform(y_test)\ntest.loc[:, 'pred']=label_encoder.inverse_transform(y_pred)","metadata":{"execution":{"iopub.status.busy":"2023-11-16T19:20:09.482464Z","iopub.execute_input":"2023-11-16T19:20:09.483805Z","iopub.status.idle":"2023-11-16T19:20:09.496152Z","shell.execute_reply.started":"2023-11-16T19:20:09.483755Z","shell.execute_reply":"2023-11-16T19:20:09.494796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_mode = test.groupby('image_id')['class'].agg(lambda x: x.value_counts().idxmax())\npred_mode = test.groupby('image_id')['pred'].agg(lambda x: x.value_counts().idxmax())\n\n# Combine the results into a new DataFrame\nresult_df = pd.DataFrame({\n    'max_class_count': class_mode,\n    'max_pred_count': pred_mode\n}).reset_index()","metadata":{"execution":{"iopub.status.busy":"2023-11-16T19:20:09.65099Z","iopub.execute_input":"2023-11-16T19:20:09.651451Z","iopub.status.idle":"2023-11-16T19:20:09.77016Z","shell.execute_reply.started":"2023-11-16T19:20:09.651417Z","shell.execute_reply":"2023-11-16T19:20:09.768905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_mode = train.groupby('image_id')['class'].agg(lambda x: x.value_counts().idxmax())\n\n# Combine the results into a new DataFrame\nresult_df = pd.DataFrame({\n    'max_class_count': class_mode,\n}).reset_index()","metadata":{"execution":{"iopub.status.busy":"2023-11-16T19:20:10.561268Z","iopub.execute_input":"2023-11-16T19:20:10.561741Z","iopub.status.idle":"2023-11-16T19:20:10.729405Z","shell.execute_reply.started":"2023-11-16T19:20:10.561682Z","shell.execute_reply":"2023-11-16T19:20:10.728359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications.resnet50 import ResNet50, preprocess_input\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Model\n\nimport numpy as np\n\n# Load the ResNet50 model pre-trained on ImageNet data, without the top classification layer\nmodel = ResNet50(weights=None, include_top=False, pooling='avg', input_shape=(512, 512, 3))\nmodel.load_weights(\"/kaggle/input/dk-resnet50/resnet50.h5\")\n\n# Set up your data generator - adjust based on your data\ndatagen = ImageDataGenerator(preprocessing_function=preprocess_input)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-16T19:20:10.739234Z","iopub.execute_input":"2023-11-16T19:20:10.73968Z","iopub.status.idle":"2023-11-16T19:20:20.661069Z","shell.execute_reply.started":"2023-11-16T19:20:10.739647Z","shell.execute_reply":"2023-11-16T19:20:20.659992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Method 1: https://www.kaggle.com/code/jirkaborovec/cancer-subtype-lightning-torch-inference-tiles\nimport os\nimport pyvips\nimport numpy as np\nimport random\nfrom PIL import Image\nimport shutil\nfrom collections import Counter\n\n\nos.environ['VIPS_CONCURRENCY'] = '4'\nos.environ['VIPS_DISC_THRESHOLD'] = '15gb'\n\ntmp_dir=\"/tmp\"\n\nif os.path.exists(tmp_dir):\n    pass\nelse:\n    os.makedirs(tmp_dir)\n\ndef extract_image_tiles(\n    p_img, folder, size: int = 2048, scale: float = 0.5,\n    drop_thr: float = 0.6, white_thr: int = 240, max_samples: int = 50\n) -> list:\n    name, _ = os.path.splitext(os.path.basename(p_img))\n    im = pyvips.Image.new_from_file(p_img)\n    w = h = size\n    # https://stackoverflow.com/a/47581978/4521646\n    idxs = [(y, y + h, x, x + w) for y in range(0, im.height, h) for x in range(0, im.width, w)]\n    # random subsample\n    max_samples = max_samples if isinstance(max_samples, int) else int(len(idxs) * max_samples)\n    random.shuffle(idxs)\n    files = []\n    for y, y_, x, x_ in idxs:\n        # https://libvips.github.io/pyvips/vimage.html#pyvips.Image.crop\n        tile = im.crop(x, y, min(w, im.width - x), min(h, im.height - y)).numpy()[..., :3]\n        if tile.shape[:2] != (h, w):\n            tile_ = tile\n            tile_size = (h, w) if tile.ndim == 2 else (h, w, tile.shape[2])\n            tile = np.zeros(tile_size, dtype=tile.dtype)\n            tile[:tile_.shape[0], :tile_.shape[1], ...] = tile_\n        black_bg = np.sum(tile, axis=2) == 0\n        tile[black_bg, :] = 255\n        mask_bg = np.mean(tile, axis=2) > white_thr\n        if np.sum(mask_bg) >= (np.prod(mask_bg.shape) * drop_thr):\n            #print(f\"skip almost empty tile: {k:06}_{int(x_ / w)}-{int(y_ / h)}\")\n            continue\n        p_img = os.path.join(folder, f\"{int(x_ / w)}-{int(y_ / h)}.png\")\n        # print(tile.shape, tile.dtype, tile.min(), tile.max())\n        new_size = int(size * scale), int(size * scale)\n        Image.fromarray(tile).resize(new_size, Image.LANCZOS).save(p_img)\n        files.append(p_img)\n        # need to set counter check as some empty tiles could be skipped earlier\n        if len(files) >= max_samples:\n            break\n    return files\n\ndef extract_prune_tiles(\n    path_img: str, folder: str, size: int = 2048, scale: float = 0.25,\n    drop_thr: float = 0.6, max_samples: int = 50\n) -> str:\n    print(f\"processing: {path_img}\")\n    name, _ = os.path.splitext(os.path.basename(path_img))\n    folder = os.path.join(folder, name)\n    os.makedirs(folder, exist_ok=True)\n    tiles = extract_image_tiles(\n        path_img, folder, size=size, scale=scale,\n        drop_thr=drop_thr, max_samples=max_samples)\n    return folder\n\n_image_id=[]\n_label=[]\nfor _, row in test_df.iterrows():\n#     try:\n    row = dict(row)\n    # prepare data - cut and load tiles\n    folder_tiles = extract_prune_tiles(os.path.join(TEST_DIR, f\"{str(row['image_id'])}.png\"),\n        tmp_dir, size=2048, scale=0.25)\n    generator = datagen.flow_from_directory(\n        '/tmp',\n        target_size=(512, 512),\n        batch_size=64,\n        class_mode=None,  # This ensures the generator does not return labels\n        shuffle=False)\n    _features = model.predict(generator, steps=len(generator))\n    columns = [f'pixel{i}' for i in range(1, 2049)]\n    df = pd.DataFrame(_features, columns=columns)\n    df['cluster'] = loaded_model.predict(_features)\n    df=df[df['cluster'].isin([1,3])]\n    results = xgbmodel.predict(df[columns].values)\n    most_common_element = Counter(label_encoder.inverse_transform(results)).most_common(1)[0][0]\n    _image_id.append(row['image_id'])\n    _label.append(most_common_element)\n    shutil.rmtree(os.path.join(tmp_dir, str(row['image_id'])))\n#     except Exception as e:\n    _image_id.append(row['image_id'])\n    _label.append(random.choice(['CC', 'EC', 'HGSC', 'LGSC', 'MC']))\n    shutil.rmtree(os.path.join(tmp_dir, str(row['image_id'])))\n        \n    ","metadata":{"execution":{"iopub.status.busy":"2023-11-16T19:23:09.948663Z","iopub.execute_input":"2023-11-16T19:23:09.949354Z","iopub.status.idle":"2023-11-16T19:24:25.911578Z","shell.execute_reply.started":"2023-11-16T19:23:09.949291Z","shell.execute_reply":"2023-11-16T19:24:25.910433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Method 2: https://www.kaggle.com/code/jirkaborovec/cancer-subtype-lit-torch-infer-tiles-parallel\n# import os\n# import pyvips\n# import numpy as np\n# import random\n# from PIL import Image\n\n# def extract_image_tiles(\n#     p_img, size: int = 2048, scale: float = 0.5,\n#     drop_thr: float = 0.6, white_thr: int = 245, max_samples: int = 50\n# ) -> list:\n#     im = pyvips.Image.new_from_file(p_img)\n#     w = h = size\n#     # https://stackoverflow.com/a/47581978/4521646\n#     idxs = [(y, y + h, x, x + w) for y in range(0, im.height, h) for x in range(0, im.width, w)]\n#     # random subsample\n#     max_samples = max_samples if isinstance(max_samples, int) else int(len(idxs) * max_samples)\n#     random.shuffle(idxs)\n#     images = []\n#     for y, y_, x, x_ in idxs:\n#         # https://libvips.github.io/pyvips/vimage.html#pyvips.Image.crop\n#         tile = im.crop(x, y, min(w, im.width - x), min(h, im.height - y)).numpy()[..., :3]\n#         if tile.shape[:2] != (h, w):\n#             tile_ = tile\n#             tile_size = (h, w) if tile.ndim == 2 else (h, w, tile.shape[2])\n#             tile = np.zeros(tile_size, dtype=tile.dtype)\n#             tile[:tile_.shape[0], :tile_.shape[1], ...] = tile_\n#         black_bg = np.sum(tile, axis=2) == 0\n#         tile[black_bg, :] = 255\n#         mask_bg = np.mean(tile, axis=2) > white_thr\n#         if np.sum(mask_bg) >= (np.prod(mask_bg.shape) * drop_thr):\n#             #print(f\"skip almost empty tile: {k:06}_{int(x_ / w)}-{int(y_ / h)}\")\n#             continue\n#         # print(tile.shape, tile.dtype, tile.min(), tile.max())\n#         new_size = int(size * scale), int(size * scale)\n#         images.append(np.array(\n#             Image.fromarray(tile).resize(new_size, Image.LANCZOS)\n#         ))\n#         # need to set counter check as some empty tiles could be skipped earlier\n#         if len(images) >= max_samples:\n#             break\n#     return images","metadata":{"execution":{"iopub.status.busy":"2023-11-16T03:15:07.028265Z","iopub.execute_input":"2023-11-16T03:15:07.02877Z","iopub.status.idle":"2023-11-16T03:15:07.040602Z","shell.execute_reply.started":"2023-11-16T03:15:07.028743Z","shell.execute_reply":"2023-11-16T03:15:07.039753Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import torch\n# from PIL import Image\n# from torch.utils.data import Dataset\n\n# class TilesImageDataset(Dataset):\n\n#     def __init__(\n#         self,\n#         img_path: str,\n#         size: int = 2048,\n#         scale: float = 0.25,\n#         drop_thr: float = 0.6,\n#         max_samples: int = 30,\n#         transforms = None\n#     ):\n#         assert os.path.isfile(img_path)\n#         self.transforms = transforms\n#         self.imgs = extract_image_tiles(\n#             img_path, size=size, scale=scale,\n#             drop_thr=drop_thr, max_samples=max_samples)\n\n#     def __getitem__(self, idx: int) -> tuple:\n#         img = self.imgs[idx]\n#         # filter background\n#         mask = np.sum(img, axis=2) == 0\n#         img[mask, :] = 255\n#         if np.max(img) < 1.5:\n#             img = np.clip(img * 255, 0, 255).astype(np.uint8)\n#         # augmentation\n#         if self.transforms:\n#             img = self.transforms(Image.fromarray(img))\n#         #print(f\"img dim: {img.shape}\")\n#         return img\n\n#     def __len__(self) -> int:\n#         return len(self.imgs)","metadata":{"execution":{"iopub.status.busy":"2023-11-16T03:15:07.041553Z","iopub.execute_input":"2023-11-16T03:15:07.041839Z","iopub.status.idle":"2023-11-16T03:15:07.052759Z","shell.execute_reply.started":"2023-11-16T03:15:07.041815Z","shell.execute_reply":"2023-11-16T03:15:07.051777Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# _image_id=[]\n# _label=[]\n# for _, row in test_df.iterrows():\n#     try:\n#         row = dict(row)\n#         dataset = TilesImageDataset(os.path.join(TEST_DIR, f\"{str(row['image_id'])}.png\"))\n#         _features=[]\n#         for i in range(len(dataset)):    \n#             _features.append(model.predict(np.expand_dims(dataset[i], axis=0), verbose=0))\n#         columns = [f'pixel{i}' for i in range(1, 2049)]\n#         df = pd.DataFrame(np.squeeze(_features, axis=1), columns=columns)\n#         df['cluster'] = loaded_model.predict(np.squeeze(_features, axis=1))\n#         df=df[df['cluster'].isin([1,3])]\n#         results = xgbmodel.predict(df[columns].values)\n#         most_common_element = Counter(label_encoder.inverse_transform(results)).most_common(1)[0][0]\n#         _image_id.append(row['image_id'])\n#         _label.append(most_common_element)\n#     except:\n#         print(\"Exception\")\n#         _image_id.append(row['image_id'])\n#         _label.append(random.choice(['CC', 'EC', 'HGSC', 'LGSC', 'MC']))\n    \n   ","metadata":{"execution":{"iopub.status.busy":"2023-11-16T18:52:52.63455Z","iopub.execute_input":"2023-11-16T18:52:52.634933Z","iopub.status.idle":"2023-11-16T18:52:53.000387Z","shell.execute_reply.started":"2023-11-16T18:52:52.6349Z","shell.execute_reply":"2023-11-16T18:52:52.999026Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df= pd.DataFrame({\n    'image_id': _image_id,\n    'label': _label\n})\n\ndf.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-16T19:24:32.179251Z","iopub.execute_input":"2023-11-16T19:24:32.179708Z","iopub.status.idle":"2023-11-16T19:24:32.189887Z","shell.execute_reply.started":"2023-11-16T19:24:32.179672Z","shell.execute_reply":"2023-11-16T19:24:32.188812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-11-16T19:24:32.505417Z","iopub.execute_input":"2023-11-16T19:24:32.50583Z","iopub.status.idle":"2023-11-16T19:24:33.725193Z","shell.execute_reply.started":"2023-11-16T19:24:32.505798Z","shell.execute_reply":"2023-11-16T19:24:33.723477Z"},"trusted":true},"execution_count":null,"outputs":[]}]}