{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":6844262,"sourceType":"datasetVersion","datasetId":3934666},{"sourceId":6979608,"sourceType":"datasetVersion","datasetId":4009169},{"sourceId":7092658,"sourceType":"datasetVersion","datasetId":4087402},{"sourceId":7542374,"sourceType":"datasetVersion","datasetId":4391381,"isSourceIdPinned":false},{"sourceId":7594698,"sourceType":"datasetVersion","datasetId":3931100,"isSourceIdPinned":false},{"sourceId":7594925,"sourceType":"datasetVersion","datasetId":3939049}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Phikon + ABMIL Png Inference for Ovarian Cancer","metadata":{}},{"cell_type":"markdown","source":"\n\nThe featuere extraction, model training and inference code are located in the dataset \"[clam-for-ovarian-cancer](https://www.kaggle.com/datasets/dantee/clam-for-ovarian-cancer)\" later referenced as CLAM. The code uses ABMIL instead of CLAM now, I just didn't get around to rename the dataset.\n\n1. Feature extraction via [Phikon](https://huggingface.co/owkin/phikon) in the function CLAM.extract_png_features . The weight for Phikon can be found in the dataset [phikon-weights](https://www.kaggle.com/datasets/dantee/phikon-weights)\n2. Training code of the [ABMIL](https://github.com/AMLab-Amsterdam/AttentionDeepMIL) model in CLAM.main. This notebook does not run the training. It rather uses the weights from my local training run, located in the dataset .\n3. Inference code of the [ABMIL](https://github.com/AMLab-Amsterdam/AttentionDeepMIL) model is the CLAM.eval.eval_utils and uses the model weights in the dataset [traind-clam-for-ovarian-cancer](https://www.kaggle.com/datasets/dantee/phikon-weights). There are 5 models each trained on 80% of my training data as described in the [3rd place model summary](https://www.kaggle.com/competitions/UBC-OCEAN/discussion/465527).\n   ","metadata":{}},{"cell_type":"code","source":"%load_ext autoreload\n%autoreload 2","metadata":{"execution":{"iopub.status.busy":"2024-02-09T09:21:22.693513Z","iopub.execute_input":"2024-02-09T09:21:22.693903Z","iopub.status.idle":"2024-02-09T09:21:22.726338Z","shell.execute_reply.started":"2024-02-09T09:21:22.693872Z","shell.execute_reply":"2024-02-09T09:21:22.725356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nif os.path.isdir('/kaggle/input'):\n    !cp /kaggle/input/library-fastkaggle/fastkaggle-0.0.8-py3-none-any.whl /kaggle/working \n    !pip install  /kaggle/working/fastkaggle-0.0.8-py3-none-any.whl -q\nfrom fastkaggle import setup_comp, iskaggle\n!pip install /kaggle/input/library-histomicstk/wheelhouse/histomicstk-1.3.0-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl --no-index --find-links /kaggle/input/library-histomicstk/wheelhouse -q\n!pip install /kaggle/input/xformers-wheel/xformers/xformers-0.0.22.post7+cu118-cp310-cp310-manylinux2014_x86_64.whl --no-index --find-links /kaggle/input/xformers-wheel/xformers","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-02-09T09:21:22.728165Z","iopub.execute_input":"2024-02-09T09:21:22.728467Z","iopub.status.idle":"2024-02-09T09:24:48.145345Z","shell.execute_reply.started":"2024-02-09T09:21:22.728441Z","shell.execute_reply":"2024-02-09T09:24:48.14422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"comp_slug = 'UBC-OCEAN'\ncomp_path = setup_comp(comp_slug)\ncomp_path","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-02-09T09:24:48.146918Z","iopub.execute_input":"2024-02-09T09:24:48.147517Z","iopub.status.idle":"2024-02-09T09:24:48.173385Z","shell.execute_reply.started":"2024-02-09T09:24:48.147488Z","shell.execute_reply":"2024-02-09T09:24:48.172212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport sys\nimport re\nimport json\nfrom pathlib import Path\nimport time\n# import psutil\nimport gc\nimport ctypes\nimport pandas as pd\nimport numpy as np\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\n\nimport torch\nimport torch.nn.functional as F","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-02-09T09:24:48.176349Z","iopub.execute_input":"2024-02-09T09:24:48.176711Z","iopub.status.idle":"2024-02-09T09:24:50.960356Z","shell.execute_reply.started":"2024-02-09T09:24:48.176682Z","shell.execute_reply":"2024-02-09T09:24:50.959456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.environ['VIPS_CONCURRENCY'] = '4'\nos.environ['VIPS_DISC_THRESHOLD'] = '15gb'","metadata":{"execution":{"iopub.status.busy":"2024-02-09T09:24:50.965024Z","iopub.execute_input":"2024-02-09T09:24:50.9654Z","iopub.status.idle":"2024-02-09T09:24:51.003913Z","shell.execute_reply.started":"2024-02-09T09:24:50.965364Z","shell.execute_reply":"2024-02-09T09:24:51.002876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make Juptyter allow multi-processing in the data loader\n# also seems to save memory\nfrom multiprocessing import set_start_method\ntry:\n    set_start_method('spawn')\nexcept RuntimeError as ex:\n    print(ex)","metadata":{"execution":{"iopub.status.busy":"2024-02-09T09:24:51.005304Z","iopub.execute_input":"2024-02-09T09:24:51.005724Z","iopub.status.idle":"2024-02-09T09:24:51.041232Z","shell.execute_reply.started":"2024-02-09T09:24:51.005692Z","shell.execute_reply":"2024-02-09T09:24:51.040455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# torch.set_num_threads(4)\n\nif iskaggle:\n    clam_path = '/kaggle/input/clam-for-ovarian-cancer'    \nelse:\n    clam_path = '../clam'\nsys.path.append(clam_path)\n\nfrom CLAM.extract_png_features import extract_png_features\nfrom CLAM.datasets.dataset_generic import Generic_WSI_Classification_Dataset\nfrom CLAM.utils.eval_utils import initiate_model","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-02-09T09:24:51.042268Z","iopub.execute_input":"2024-02-09T09:24:51.042531Z","iopub.status.idle":"2024-02-09T09:25:00.837162Z","shell.execute_reply.started":"2024-02-09T09:24:51.042507Z","shell.execute_reply":"2024-02-09T09:25:00.836265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ckpt_folder = Path('/kaggle/input/trained-clam-for-ovarian-cancer')\nwith open(ckpt_folder/'settings.json', 'r') as file:\n    settings = json.load(file)\nprint(settings)\n\nmodel = settings['FEATURE_EXTRACT_MODEL']\nuse_fp16 = settings['USE_FP16_FOR_FEATURE_EXTRACTION']\ntile_size = settings['TILE_SIZE']\nmodel_size = settings['MODEL_SIZE']\ntma_megapixel_threshold = settings['TMA_MEGAPIXEL_THRESHOLD']\n    \n# Kaggle Submission\nif len(os.listdir(comp_path/'test_images')) != 1:\n    is_submission = True\n    test_meta = pd.read_csv(comp_path/'test.csv')\n    train_test = 'test'\n    test_folder = comp_path/f'test_images'\n# Kaggle Test Run\nelse:\n    is_submission = False\n    train_test = 'train'\n    test_meta = pd.read_csv(comp_path/f'{train_test}.csv')\n    # test_meta = test_meta[test_meta['image_id'].isin([39728, 39872, 39880, 29084, 44232, 34247, 42125, 5264])]\n    # test_meta = test_meta[test_meta['image_id'].isin([45630, 36678, 8713, 14424])] # four largest images\n    test_meta = test_meta.sample(frac=0.05)\n    test_folder = comp_path/f'{train_test}_images'\nproject_root = Path('/tmp')\nmodel_path = '/kaggle/input/phikon-weights'\n\nfile_sizes = []\nformatted_sizes = []\nfor image_id in test_meta['image_id']:\n    size = os.path.getsize(test_folder/f'{image_id}.png')\n    file_sizes.append(size)\n    formatted_sizes.append(f\"{size / 1024**3:.2f} GB\")\ntest_meta['file_gb'] = np.array(file_sizes) / 1024 **3\ntest_meta['formatted_file_size'] = formatted_sizes\ntest_meta['n_mega_pixels'] = (test_meta['image_width'] * test_meta['image_height'] // 1e6).astype(int)\nn_classes = 6","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-02-09T09:44:11.396044Z","iopub.execute_input":"2024-02-09T09:44:11.397025Z","iopub.status.idle":"2024-02-09T09:44:11.706678Z","shell.execute_reply.started":"2024-02-09T09:44:11.396995Z","shell.execute_reply":"2024-02-09T09:44:11.705626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rows = []\n# for file in os.listdir(test_folder):\n#     size = os.path.getsize(test_folder/file)\n#     rows.append({'file_name': file, 'size': size, 'readable_size': f\"{size / 1024**3:.2f} GB\"})\n\n# file_sizes = pd.DataFrame(rows)\n# file_sizes.sort_values('size', ascending=False)[:20]","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-02-08T14:23:09.273457Z","iopub.execute_input":"2024-02-08T14:23:09.273802Z","iopub.status.idle":"2024-02-08T14:23:09.354898Z","shell.execute_reply.started":"2024-02-08T14:23:09.273776Z","shell.execute_reply":"2024-02-08T14:23:09.353984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# file_sizes[file_sizes['file_name'] == '1020.png']","metadata":{"execution":{"iopub.status.busy":"2024-02-08T14:23:09.497621Z","iopub.execute_input":"2024-02-08T14:23:09.497903Z","iopub.status.idle":"2024-02-08T14:23:09.564577Z","shell.execute_reply.started":"2024-02-08T14:23:09.49788Z","shell.execute_reply":"2024-02-08T14:23:09.563749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def print_memory():\n#     ram_info = psutil.virtual_memory()\n#     print(f\"Available: {ram_info.available / 1024 / 1024 / 1024:.2f} GB\")","metadata":{"execution":{"iopub.status.busy":"2024-02-08T14:23:09.793676Z","iopub.execute_input":"2024-02-08T14:23:09.79395Z","iopub.status.idle":"2024-02-08T14:23:09.860776Z","shell.execute_reply.started":"2024-02-08T14:23:09.793928Z","shell.execute_reply":"2024-02-08T14:23:09.859893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_meta.shape[0]","metadata":{"execution":{"iopub.status.busy":"2024-02-09T09:31:36.313486Z","iopub.execute_input":"2024-02-09T09:31:36.314532Z","iopub.status.idle":"2024-02-09T09:31:36.397615Z","shell.execute_reply.started":"2024-02-09T09:31:36.314494Z","shell.execute_reply":"2024-02-09T09:31:36.396305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 512\nstart = time.time()\nextract_png_features(test_meta,\n                     comp_path,\n                     project_root,\n                     train_or_test=train_test,\n                     tma_megapixel_threshold=tma_megapixel_threshold,\n                     model=model,\n                     model_path=model_path,\n                     use_fp16=use_fp16,\n                     num_workers=4,\n                     prefetch_factor=2,\n                     tile_size=tile_size,\n                     batch_size=batch_size,\n                     print_every_batches=5,\n                     tissue_threshold=0.05,\n                     gc_after_batch=False,\n                     print_memory=False,\n                     skip_existing=False)\nprint(f'Finished in {(time.time()-start) // 60:.0f} min {(time.time()-start) % 60:.1f} s')","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-02-09T09:32:37.430466Z","iopub.execute_input":"2024-02-09T09:32:37.431149Z","iopub.status.idle":"2024-02-09T09:38:03.701156Z","shell.execute_reply.started":"2024-02-09T09:32:37.431118Z","shell.execute_reply":"2024-02-09T09:38:03.699025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Model Checkpoints","metadata":{"tags":[]}},{"cell_type":"code","source":"labels = pd.read_csv(ckpt_folder/'label_mapping.csv', header=None)\nlabels.columns=['label', 'idx']\nlabels","metadata":{"execution":{"iopub.status.busy":"2024-02-09T09:44:17.043369Z","iopub.execute_input":"2024-02-09T09:44:17.04375Z","iopub.status.idle":"2024-02-09T09:44:17.155382Z","shell.execute_reply.started":"2024-02-09T09:44:17.043722Z","shell.execute_reply":"2024-02-09T09:44:17.15442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_out=0.65\nuse_inst_predictions=True\n\nre_checkpoint = re.compile(r's_\\d+_checkpoint.pt')\nckpt_files = [path for path in os.listdir(ckpt_folder) if re_checkpoint.match(path)]\nlabel_dict = {row['label']: row['idx'] for i, row in labels.iterrows()}\nmodels = [initiate_model(label_dict, \n                         'abmil', \n                         os.path.join(ckpt_folder, ckpt_file),\n                         model_size=model_size,\n                         drop_out=drop_out,\n                         feature_dim=768,\n                         use_inst_predictions=use_inst_predictions) for ckpt_file in ckpt_files]","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-02-09T09:44:17.492781Z","iopub.execute_input":"2024-02-09T09:44:17.493677Z","iopub.status.idle":"2024-02-09T09:44:17.790497Z","shell.execute_reply.started":"2024-02-09T09:44:17.49364Z","shell.execute_reply":"2024-02-09T09:44:17.789633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_meta","metadata":{"execution":{"iopub.status.busy":"2024-02-09T09:44:18.639614Z","iopub.execute_input":"2024-02-09T09:44:18.640482Z","iopub.status.idle":"2024-02-09T09:44:18.754201Z","shell.execute_reply.started":"2024-02-09T09:44:18.640451Z","shell.execute_reply":"2024-02-09T09:44:18.753157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bag_weight = 0.7\n\ndefault_prediction = 'HGSC'\npredictions = {}\n\nnot_found_count = 0\nfor i, row in test_meta.iterrows():\n    image_id = row['image_id']\n    is_tma = (row['n_mega_pixels'] <= 50)\n    feature_path = project_root/'features'/f'{image_id}.pt'\n    try:\n        features = torch.load(feature_path, map_location='cuda')\n    except FileNotFoundError:\n        not_found_count += 1\n        continue\n\n    probs = []\n    for model in models:\n        model.eval()\n        with torch.no_grad():\n            result = model(features, bag_weight, is_tma)[0]\n        probs.append(F.softmax(result, dim=1).cpu().numpy())\n    model_avg = np.array(probs).mean(axis=0)\n\n    predictions[image_id] = labels['label'].iloc[model_avg.argmax()]","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-02-09T09:44:19.629762Z","iopub.execute_input":"2024-02-09T09:44:19.630632Z","iopub.status.idle":"2024-02-09T09:44:19.990509Z","shell.execute_reply.started":"2024-02-09T09:44:19.630602Z","shell.execute_reply":"2024-02-09T09:44:19.9895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = pd.DataFrame(predictions.items(), columns=['image_id', 'label'])\nif not_found_count > 0:\n    print(f'Could not find {not_found_count} feature files.')","metadata":{"execution":{"iopub.status.busy":"2024-02-09T09:44:21.090344Z","iopub.execute_input":"2024-02-09T09:44:21.090746Z","iopub.status.idle":"2024-02-09T09:44:21.195658Z","shell.execute_reply.started":"2024-02-09T09:44:21.090715Z","shell.execute_reply":"2024-02-09T09:44:21.19461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Replacing {predictions['label'].isnull().sum()} null labels\")\npredictions['label'].fillna(default_prediction, inplace=True)\nprint(f\"Replacing {(~(predictions['label'].isin(labels['label']))).sum()} invalid labels\")\npredictions[~predictions['label'].isin(labels['label'])] = default_prediction\nprint(f\"Fill default prediction for {(~test_meta['image_id'].isin(predictions['image_id'])).sum()} misisng image_ids.\")\nmissing_ids = test_meta[~test_meta['image_id'].isin(predictions['image_id'])][['image_id']]\nmissing_ids['label'] = default_prediction\npredictions = pd.concat([predictions, missing_ids])\npredictions.sort_values('image_id').to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-09T09:44:21.231285Z","iopub.execute_input":"2024-02-09T09:44:21.231686Z","iopub.status.idle":"2024-02-09T09:44:21.37604Z","shell.execute_reply.started":"2024-02-09T09:44:21.231656Z","shell.execute_reply":"2024-02-09T09:44:21.374969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cat submission.csv","metadata":{"execution":{"iopub.status.busy":"2024-02-09T09:44:32.444768Z","iopub.execute_input":"2024-02-09T09:44:32.445169Z","iopub.status.idle":"2024-02-09T09:44:33.686979Z","shell.execute_reply.started":"2024-02-09T09:44:32.44514Z","shell.execute_reply":"2024-02-09T09:44:33.68581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# while \"submission.csv\" not in os.listdir(\"/kaggle/working\"):\n#     predictions.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-03T10:35:45.444411Z","iopub.execute_input":"2023-12-03T10:35:45.445251Z","iopub.status.idle":"2023-12-03T10:35:45.451589Z","shell.execute_reply.started":"2023-12-03T10:35:45.445206Z","shell.execute_reply":"2023-12-03T10:35:45.450494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Only for local use","metadata":{}},{"cell_type":"code","source":"if not iskaggle:\n    pred = pd.DataFrame(predictions)\n    pred.columns = ['image_id', 'pred_label']\n    comp = pred.merge(tma_test, on='image_id')\n    acc = (comp['pred_label'] == comp['label']).mean()\n    print(acc)\n    display(comp)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-11-23T17:07:22.397675Z","iopub.status.idle":"2023-11-23T17:07:22.39806Z","shell.execute_reply.started":"2023-11-23T17:07:22.397852Z","shell.execute_reply":"2023-11-23T17:07:22.397868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# upload libraries\nupload = True\nif not iskaggle and upload:\n    from fastkaggle import create_libs_datasets\n    local_lib_path = Path('./clam_pip_libraries')\n    username = 'dantee'\n    create_libs_datasets(libs, local_lib_path, username)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-11-23T17:07:22.399067Z","iopub.status.idle":"2023-11-23T17:07:22.399403Z","shell.execute_reply.started":"2023-11-23T17:07:22.399234Z","shell.execute_reply":"2023-11-23T17:07:22.399255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### <span style=\"color:red\">⚠ Set Kernel to Python 3 before pushing the notebook ⚠</span>","metadata":{"tags":[]}},{"cell_type":"code","source":"upload = True\nif not iskaggle and upload:\n    from fastkaggle import iskaggle, push_notebook\n    push_notebook('dantee',\n                  'clam-png-inference-for-ovarian-cancer',\n                  'CLAM Png Inference for Ovarian Cancer',\n                  'CLAM Png Inference for Ovarian Cancer.ipynb',\n                  competition='UBC-OCEAN',\n                  private=True,\n                  gpu=True,\n                  internet=False,\n                  linked_datasets=[\n                      'dantee/clam-for-ovarian-cancer',\n                      'dantee/trained-clam-for-ovarian-cancer',\n                      'gunesevitan/libvips-pyvips-installation-and-getting-started',\n                      'mhiro2/pytorch-pretrained-models',\n                  ])","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-11-23T17:07:22.400654Z","iopub.status.idle":"2023-11-23T17:07:22.401034Z","shell.execute_reply.started":"2023-11-23T17:07:22.400831Z","shell.execute_reply":"2023-11-23T17:07:22.400848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}