{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":6774400,"sourceType":"datasetVersion","datasetId":3895136},{"sourceId":6774553,"sourceType":"datasetVersion","datasetId":3898019},{"sourceId":6889036,"sourceType":"datasetVersion","datasetId":3957431},{"sourceId":6947868,"sourceType":"datasetVersion","datasetId":3958714},{"sourceId":6957918,"sourceType":"datasetVersion","datasetId":3886036},{"sourceId":7128234,"sourceType":"datasetVersion","datasetId":4110911},{"sourceId":7130245,"sourceType":"datasetVersion","datasetId":3965936},{"sourceId":7167155,"sourceType":"datasetVersion","datasetId":4140444}],"dockerImageVersionId":30587,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### import torch\n\nfrom IPython.display import clear_output\n\n\n                                                   \n\n\n\nfrom PIL import Image\nImage.MAX_IMAGE_PIXELS = None\nimport pandas as pd \nimport numpy as np\nimport gc\nimport math\n\nfrom collections import Counter   ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-02T06:11:07.278889Z","iopub.execute_input":"2023-12-02T06:11:07.279278Z","iopub.status.idle":"2023-12-02T06:11:12.086404Z","shell.execute_reply.started":"2023-12-02T06:11:07.279243Z","shell.execute_reply":"2023-12-02T06:11:12.084952Z"}}},{"cell_type":"code","source":"import torch\nfrom IPython.display import clear_output\n\nfrom PIL import Image \nImage.MAX_IMAGE_PIXELS = None \nimport pandas as pd\nimport numpy as np\nimport gc\nimport math\n\nfrom collections import Counter","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:14:38.7551Z","iopub.execute_input":"2023-12-10T10:14:38.755655Z","iopub.status.idle":"2023-12-10T10:14:42.870546Z","shell.execute_reply.started":"2023-12-10T10:14:38.755596Z","shell.execute_reply":"2023-12-10T10:14:42.869315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --no-index --no-deps /kaggle/input/wheel-file/archive/kaggle/working/wheelfile/*.whl ","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:14:42.872637Z","iopub.execute_input":"2023-12-10T10:14:42.873118Z","iopub.status.idle":"2023-12-10T10:15:08.589692Z","shell.execute_reply.started":"2023-12-10T10:14:42.873093Z","shell.execute_reply":"2023-12-10T10:15:08.587756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom efficientnet.tfkeras import EfficientNetB0\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Dense\nfrom tensorflow.keras.callbacks import ModelCheckpoint","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:15:08.591758Z","iopub.execute_input":"2023-12-10T10:15:08.593411Z","iopub.status.idle":"2023-12-10T10:15:25.842754Z","shell.execute_reply.started":"2023-12-10T10:15:08.593356Z","shell.execute_reply":"2023-12-10T10:15:25.84111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/input/pyvips-python-and-deb-package\n# intall the deb packages\n!yes | dpkg -i --force-depends /kaggle/input/pyvips-python-and-deb-package/linux_packages/archives/*.deb\n# install the python wrapper\n!pip install pyvips -f /kaggle/input/pyvips-python-and-deb-package/python_packages/ --no-index\n!pip list | grep pyvips\n\n\nfrom IPython import display\ndisplay.clear_output()","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:15:25.847444Z","iopub.execute_input":"2023-12-10T10:15:25.848899Z","iopub.status.idle":"2023-12-10T10:17:17.255932Z","shell.execute_reply.started":"2023-12-10T10:15:25.848837Z","shell.execute_reply":"2023-12-10T10:17:17.255039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_SIZE = [256, 256] # at this size, a GPU will run out of memory. Use the TPU\nEPOCHS = 10\nBATCH_SIZE = 16 \n\n\nNUM_TRAINING_IMAGES = 500000\nNUM_TEST_IMAGES = 7382\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\nAUTO = tf.data.experimental.AUTOTUNE\nprint(STEPS_PER_EPOCH)","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:17:17.257566Z","iopub.execute_input":"2023-12-10T10:17:17.258179Z","iopub.status.idle":"2023-12-10T10:17:17.267643Z","shell.execute_reply.started":"2023-12-10T10:17:17.258152Z","shell.execute_reply":"2023-12-10T10:17:17.265048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pyvips\nimport numpy as np\nimport random\nfrom PIL import Image\nfrom IPython import display\nfrom tqdm import tqdm\ndef extract_image_tiles(\n    p_img,label, folder, size: int = 2048, scale: float = 0.5,\n    drop_thr: float = 0.6, white_thr: int = 240, max_samples: int = 50\n) -> list:\n    name, _ = os.path.splitext(os.path.basename(p_img))\n    im = pyvips.Image.new_from_file(p_img)\n    w = h = size\n    print(f\"processing: {p_img}\")\n    # https://stackoverflow.com/a/47581978/4521646\n    idxs = [(y, y + h, x, x + w) for y in range(0, im.height, h) for x in range(0, im.width, w)]\n    # random subsample\n    max_samples = max_samples if isinstance(max_samples, int) else int(len(idxs) * max_samples)\n    random.shuffle(idxs)\n    files = []\n    imageeslist=[]\n    for y, y_, x, x_ in idxs:        # https://libvips.github.io/pyvips/vimage.html#pyvips.Image.crop\n\n    #for y, y_, x, x_ in tqdm(idxs, total=len(idxs)):        # https://libvips.github.io/pyvips/vimage.html#pyvips.Image.crop\n        tile = im.crop(x, y, min(w, im.width - x), min(h, im.height - y)).numpy()[..., :3]\n        if tile.shape[:2] != (h, w):\n            tile_ = tile\n            tile_size = (h, w) if tile.ndim == 2 else (h, w, tile.shape[2])\n            tile = np.zeros(tile_size, dtype=tile.dtype)\n            tile[:tile_.shape[0], :tile_.shape[1], ...] = tile_\n        black_bg = np.sum(tile, axis=2) == 0\n        tile[black_bg, :] = 255\n        img = np.dot(tile[..., :3], [0.2989, 0.5870, 0.1140]).astype(np.uint8)\n        white_pixels = np.sum(img>220)\n#         mask_bg = np.mean(tile, axis=2) > white_thr\n        if np.sum(white_pixels) >= (np.prod(img.shape) * drop_thr):\n#             display.clear_output()\n#             plt.imshow(tile)\n#             plt.show()\n            continue\n        imageeslist.append(tile)\n#         p_img = os.path.join(folder, f\"label-{label}-{int(x_ / w)}-{int(y_ / h)}.png\")\n#         # print(tile.shape, tile.dtype, tile.min(), tile.max())\n# #         new_size = int(size * scale), int(size * scale)\n#          #Image.fromarray(tile).resize(new_size, Image.LANCZOS).save(p_img)\n#         Image.fromarray(tile).save(p_img)\n\n#         files.append(p_img)\n#         # need to set counter check as some empty tiles could be skipped earlier\n        if len(imageeslist) >= max_samples:\n            break\n    return imageeslist","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:17:17.269726Z","iopub.execute_input":"2023-12-10T10:17:17.270365Z","iopub.status.idle":"2023-12-10T10:17:17.676832Z","shell.execute_reply.started":"2023-12-10T10:17:17.27031Z","shell.execute_reply":"2023-12-10T10:17:17.67482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# imagees_list = extract_prune_tiles(\"/kaggle/input/UBC-OCEAN/test_images/41.png\",0, IMAGES_FOLDER, size=256, scale=1,drop_thr=0.4,white_thr=222,max_samples=10)","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:17:17.678516Z","iopub.execute_input":"2023-12-10T10:17:17.678943Z","iopub.status.idle":"2023-12-10T10:17:17.684334Z","shell.execute_reply.started":"2023-12-10T10:17:17.67891Z","shell.execute_reply":"2023-12-10T10:17:17.683106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, glob\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport re\n\nDATASET_FOLDER = \"/kaggle/input/UBC-OCEAN/\"\nIMAGES_FOLDER = \"./test_tiles\"\n\nos.environ['VIPS_CONCURRENCY'] = '4'\nos.environ['VIPS_DISC_THRESHOLD'] = '15gb'\nimport torch\nfrom PIL import Image\nfrom torch.utils.data import Dataset\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:17:17.685717Z","iopub.execute_input":"2023-12-10T10:17:17.686052Z","iopub.status.idle":"2023-12-10T10:17:17.701625Z","shell.execute_reply.started":"2023-12-10T10:17:17.686021Z","shell.execute_reply":"2023-12-10T10:17:17.7005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_prune_tiles(\n    path_img: str,label:str, folder: str, size: int = 2048, scale: float = 0.25,\n    drop_thr: float = 0.6,white_thr: int=0.5, max_samples: int = 1000\n) -> str:\n    print(f\"processing: {path_img}\")\n    name, _ = os.path.splitext(os.path.basename(path_img))\n    folder = os.path.join(folder, name)\n    os.makedirs(folder, exist_ok=True)\n    tiles = extract_image_tiles(\n        path_img,label, folder, size=size, scale=scale,\n        drop_thr=drop_thr,white_thr=225 ,max_samples=max_samples)\n    return tiles","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:17:17.703929Z","iopub.execute_input":"2023-12-10T10:17:17.704566Z","iopub.status.idle":"2023-12-10T10:17:17.73701Z","shell.execute_reply.started":"2023-12-10T10:17:17.704526Z","shell.execute_reply":"2023-12-10T10:17:17.735437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/sample_submission.csv\")\n# test_df=pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:17:17.741493Z","iopub.execute_input":"2023-12-10T10:17:17.742335Z","iopub.status.idle":"2023-12-10T10:17:17.763351Z","shell.execute_reply.started":"2023-12-10T10:17:17.74229Z","shell.execute_reply":"2023-12-10T10:17:17.761849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:17:17.774131Z","iopub.execute_input":"2023-12-10T10:17:17.774594Z","iopub.status.idle":"2023-12-10T10:17:17.782983Z","shell.execute_reply.started":"2023-12-10T10:17:17.774568Z","shell.execute_reply":"2023-12-10T10:17:17.780737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, glob\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport re\n\nDATASET_FOLDER = \"/kaggle/input/UBC-OCEAN/\"\nIMAGES_FOLDER = \"./test_tiles\"\n\nos.environ['VIPS_CONCURRENCY'] = '4'\nos.environ['VIPS_DISC_THRESHOLD'] = '15gb'","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:17:17.785387Z","iopub.execute_input":"2023-12-10T10:17:17.786643Z","iopub.status.idle":"2023-12-10T10:17:17.797936Z","shell.execute_reply.started":"2023-12-10T10:17:17.786587Z","shell.execute_reply":"2023-12-10T10:17:17.79651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df=pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ntest_df=pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')\n# submission_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/sample_submission.csv\")\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:17:17.800448Z","iopub.execute_input":"2023-12-10T10:17:17.801355Z","iopub.status.idle":"2023-12-10T10:17:17.873723Z","shell.execute_reply.started":"2023-12-10T10:17:17.801305Z","shell.execute_reply":"2023-12-10T10:17:17.871808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n\n\n# for idx,row in test_df.iterrows():\n#     path=f\"/kaggle/input/UBC-OCEAN/test_images/{row['image_id']}.png\"\n#     label=row['image_id']\n#     folder_tiles = extract_prune_tiles(path,label, IMAGES_FOLDER, size=256, scale=1,drop_thr=0.4,white_thr=222)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:17:17.875669Z","iopub.execute_input":"2023-12-10T10:17:17.876251Z","iopub.status.idle":"2023-12-10T10:17:17.881361Z","shell.execute_reply.started":"2023-12-10T10:17:17.876219Z","shell.execute_reply":"2023-12-10T10:17:17.87991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Defining model","metadata":{}},{"cell_type":"code","source":"def normalize(image):\n    mean = [0.485, 0.456, 0.406]\n    std = [0.229, 0.224, 0.225]\n    image = (image - mean) / std\n    return image","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:17:17.882869Z","iopub.execute_input":"2023-12-10T10:17:17.883408Z","iopub.status.idle":"2023-12-10T10:17:17.89999Z","shell.execute_reply.started":"2023-12-10T10:17:17.883314Z","shell.execute_reply":"2023-12-10T10:17:17.898617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"infer code","metadata":{}},{"cell_type":"code","source":"# device='cuda' if torch.cuda.is_available() else 'cpu'\n# model=ResNet(Bottleneck, [3, 4, 6, 3], 6).to(device)\n# num_features = model.fc.in_features\n# model.fc = nn.Linear(num_features, 6) \n# state_dict=torch.load(\"/kaggle/input/first-resnet-model/resnetv4.pth\",map_location=torch.device('cpu'))\n# model.load_state_dict(state_dict[\"model_state_dict\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:17:17.902373Z","iopub.execute_input":"2023-12-10T10:17:17.902759Z","iopub.status.idle":"2023-12-10T10:17:17.924773Z","shell.execute_reply.started":"2023-12-10T10:17:17.902734Z","shell.execute_reply":"2023-12-10T10:17:17.923231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from torchvision import transforms\n","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:17:17.926049Z","iopub.execute_input":"2023-12-10T10:17:17.927533Z","iopub.status.idle":"2023-12-10T10:17:17.951757Z","shell.execute_reply.started":"2023-12-10T10:17:17.927483Z","shell.execute_reply":"2023-12-10T10:17:17.946538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df=pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\").sample(2)\n# test_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:17:17.95439Z","iopub.execute_input":"2023-12-10T10:17:17.954971Z","iopub.status.idle":"2023-12-10T10:17:17.9672Z","shell.execute_reply.started":"2023-12-10T10:17:17.954926Z","shell.execute_reply":"2023-12-10T10:17:17.965995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom efficientnet.tfkeras import EfficientNetB0\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Dense\nfrom tensorflow.keras.callbacks import ModelCheckpoint\n\n   \n# pretrained_model = EfficientNetB0( weights=,include_top=False, input_shape=(256,256, 3))\n    \n#     # Freeze the layers of the pre-trained model for transfer learning\n# pretrained_model.trainable = True\n    \n#     # Build your model using the EfficientNetB0 base\n# model = tf.keras.Sequential([\n#         pretrained_model,\n#         GlobalAveragePooling2D(),\n#         Dense(5)  # Assuming you have 6 classes\n#     ])\n        \n        \n# checkpoint_callback = ModelCheckpoint(filepath='weights.{epoch:02d}-{batch:04d}.h5', save_freq=10000)\n# model.compile(\n#         optimizer=tf.keras.optimizers.Adam(),\n#         loss='sparse_categorical_crossentropy',\n#         metrics=['sparse_categorical_accuracy']\n#     )\nmodel_path =\"/kaggle/input/1024-effbo-model/best_tpu_trained_0-001lrmodelv2.h5\"\nmodel=tf.keras.models.load_model(model_path)\n\n# historical = model.fit(training_dataset, \n#                        steps_per_epoch=STEPS_PER_EPOCH, \n#                        epochs=EPOCHS,\n#                        callbacks=[checkpoint_callback])","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:29:43.752561Z","iopub.execute_input":"2023-12-10T10:29:43.752986Z","iopub.status.idle":"2023-12-10T10:29:49.620848Z","shell.execute_reply.started":"2023-12-10T10:29:43.75296Z","shell.execute_reply":"2023-12-10T10:29:49.618675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n\nimagelist=[]\nnewlist=[]\nanswer=[]\ndef extract_features(image):\n    image = tf.image.resize(image, [256, 256])\n    image = tf.cast(image, tf.float32)\n    \n    image = normalize(image)\n    result=model.predict(image,verbose=0)\n    \n        \n#         probabilities = torch.nn.functional.softmax(output, dim=1)  # Apply softmax\n        \n    return result\n\n\nfor idx,row in test_df.iterrows():\n    \n    desired_path=f\"/kaggle/input/UBC-OCEAN/test_images/{row['image_id']}.png\"\n    if os.path.exists(desired_path):\n        path = desired_path\n    else:\n        #path=f\"/kaggle/input/UBC-OCEAN/test_thumbnails/{row['image_id']}.png\"\n        path=f\"/kaggle/input/UBC-OCEAN/train_images/{row['image_id']}.png\"\n    label=row['image_id']\n    imagees_list=extract_image_tiles( path,label, IMAGES_FOLDER, size=256, scale=1,drop_thr=0.4,white_thr=222,max_samples=10)\n    \n\n    print(\"images finished processing\")\n    results=[] \n    for idx1 in range(len(imagees_list)):\n            imagepath=imagees_list[idx1]\n            image_array = np.expand_dims(imagepath, axis=0)  # Add batch dimension\n            \n\n            probabilities = extract_features(image_array)\n           \n            result=np.argmax(probabilities)\n            \n\n           \n            \n\n            \n            results.append(int(result))\n            del image_array\n            del imagepath\n            del probabilities\n            \n#     print(results)\n    element_counts = Counter(results)\n    del results\n    del imagees_list\n    most_common_element = element_counts.most_common(1)[0][0]\n    answer.append(most_common_element)\n    imagelist.append(row['image_id'])\n    gc.collect()\n    \n\n#     newlist.append(new_data)\n#     if idx > 3:\n#         break\n\n \nnew_data = {\n             'image_id': imagelist,\n             'label': answer\n                    }  \n  \nsubmission_df= pd.DataFrame(new_data)\nlabel_to_be_replace=['CC','EC','HGSC','LGSC','MC','Others']\nlabel_to_be_replaced=[0,1,2,3,4,5]\nsubmission_df['label'].replace(label_to_be_replaced,label_to_be_replace,inplace=True)\n        # Save the DataFrame to a CSV file named 'submission.csv'\nsubmission_df.to_csv('submission.csv', index=False)\n        \n        # Print the DataFrame\nprint(submission_df)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:39:14.003334Z","iopub.execute_input":"2023-12-10T10:39:14.003913Z","iopub.status.idle":"2023-12-10T10:39:35.398822Z","shell.execute_reply.started":"2023-12-10T10:39:14.003873Z","shell.execute_reply":"2023-12-10T10:39:35.395888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-12-10T10:17:22.517818Z","iopub.status.idle":"2023-12-10T10:17:22.519004Z","shell.execute_reply.started":"2023-12-10T10:17:22.518774Z","shell.execute_reply":"2023-12-10T10:17:22.518798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%whos","metadata":{"execution":{"iopub.status.busy":"2023-12-06T07:27:36.459868Z","iopub.execute_input":"2023-12-06T07:27:36.463094Z","iopub.status.idle":"2023-12-06T07:27:36.486476Z","shell.execute_reply.started":"2023-12-06T07:27:36.462985Z","shell.execute_reply":"2023-12-06T07:27:36.484615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}