{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Introduction\n* clustering dataset V2\n* cosine similarity with T5 embedding","metadata":{}},{"cell_type":"markdown","source":"## Setup","metadata":{}},{"cell_type":"code","source":"!pip install faiss-gpu==1.7.2","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:28.555704Z","iopub.execute_input":"2023-07-17T02:20:28.556143Z","iopub.status.idle":"2023-07-17T02:20:39.542072Z","shell.execute_reply.started":"2023-07-17T02:20:28.556098Z","shell.execute_reply":"2023-07-17T02:20:39.540898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GLOBAL_SEED = 42\n\nimport os\nos.environ['PYTHONHASHSEED'] = str(GLOBAL_SEED)\n\nimport numpy as np # linear algebra\nfrom numpy import random as np_rnd\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport random as rnd\nimport pickle\nimport gc\n\nfrom Bio import SeqIO\nfrom Bio.SeqUtils.ProtParam import ProteinAnalysis\nfrom collections import Counter\n\nfrom sklearn.preprocessing import MinMaxScaler, StandardScaler, OneHotEncoder\nfrom sklearn.decomposition import PCA\nfrom sklearn import linear_model as lm\nfrom sklearn import metrics as skl_metrics\n\nfrom scipy.stats import rankdata\nimport faiss\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-17T02:20:39.544566Z","iopub.execute_input":"2023-07-17T02:20:39.54498Z","iopub.status.idle":"2023-07-17T02:20:39.553134Z","shell.execute_reply.started":"2023-07-17T02:20:39.544938Z","shell.execute_reply":"2023-07-17T02:20:39.552193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    debug = False\n    seq_backbone = \"t5\"\n    seq_embed_size = {\n        \"esm2\": 1280,\n        \"t5\": 1024,\n        \"protbert\": 1024,\n    }","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:39.554388Z","iopub.execute_input":"2023-07-17T02:20:39.555283Z","iopub.status.idle":"2023-07-17T02:20:39.568352Z","shell.execute_reply.started":"2023-07-17T02:20:39.555249Z","shell.execute_reply":"2023-07-17T02:20:39.567195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=42):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    # python random\n    rnd.seed(seed)\n    # numpy random\n    np_rnd.seed(seed)\n    # RAPIDS random\n    try:\n        cupy.random.seed(seed)\n    except:\n        pass\n    # tf random\n    try:\n        tf_rnd.set_seed(seed)\n    except:\n        pass\n    # pytorch random\n    try:\n        torch.manual_seed(seed)\n        torch.cuda.manual_seed(seed)\n        torch.backends.cudnn.deterministic = True\n    except:\n        pass\n\ndef pickleIO(obj, src, op=\"r\"):\n    if op==\"w\":\n        with open(src, op + \"b\") as f:\n            pickle.dump(obj, f)\n    elif op==\"r\":\n        with open(src, op + \"b\") as f:\n            tmp = pickle.load(f)\n        return tmp\n    else:\n        print(\"unknown operation\")\n        return obj","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:39.570988Z","iopub.execute_input":"2023-07-17T02:20:39.571444Z","iopub.status.idle":"2023-07-17T02:20:39.581841Z","shell.execute_reply.started":"2023-07-17T02:20:39.571412Z","shell.execute_reply":"2023-07-17T02:20:39.580926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading data & Building Nearest Neighbor Seacher","metadata":{}},{"cell_type":"code","source":"# this is max searching value on faiss-gpu\ntopN = 2048","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:39.583043Z","iopub.execute_input":"2023-07-17T02:20:39.584003Z","iopub.status.idle":"2023-07-17T02:20:39.591642Z","shell.execute_reply.started":"2023-07-17T02:20:39.583969Z","shell.execute_reply":"2023-07-17T02:20:39.590672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_npz_data = np.load(\"/kaggle/input/cafa-create-dataset-for-clustering-v2/df_full.npz\")\n# l2 normalization\ntrain_feature = train_npz_data[CFG.seq_backbone] / np.linalg.norm(train_npz_data[CFG.seq_backbone], axis=1).reshape(-1, 1)\ntrain_label = train_npz_data[\"label\"]\ntrain_feature.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:39.594714Z","iopub.execute_input":"2023-07-17T02:20:39.594999Z","iopub.status.idle":"2023-07-17T02:20:42.499926Z","shell.execute_reply.started":"2023-07-17T02:20:39.594969Z","shell.execute_reply":"2023-07-17T02:20:42.498997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res = faiss.StandardGpuResources()\nfaiss_searcher = faiss.IndexFlatIP(train_feature.shape[-1])\n# transform to gpu\nfaiss_searcher = faiss.index_cpu_to_gpu(res, 0, faiss_searcher)\nfaiss_searcher.add(train_feature)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:42.501158Z","iopub.execute_input":"2023-07-17T02:20:42.501905Z","iopub.status.idle":"2023-07-17T02:20:42.823662Z","shell.execute_reply.started":"2023-07-17T02:20:42.50187Z","shell.execute_reply":"2023-07-17T02:20:42.822675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test sample\nscores, indicies = faiss_searcher.search(train_feature[:3], k=topN)\n# normalizing cosine similarity 0-1\nscores = (scores + 1) / 2","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:42.825216Z","iopub.execute_input":"2023-07-17T02:20:42.825569Z","iopub.status.idle":"2023-07-17T02:20:42.834332Z","shell.execute_reply.started":"2023-07-17T02:20:42.825537Z","shell.execute_reply":"2023-07-17T02:20:42.833332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:42.835844Z","iopub.execute_input":"2023-07-17T02:20:42.836215Z","iopub.status.idle":"2023-07-17T02:20:42.844909Z","shell.execute_reply.started":"2023-07-17T02:20:42.836183Z","shell.execute_reply":"2023-07-17T02:20:42.844006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"indicies","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:42.849204Z","iopub.execute_input":"2023-07-17T02:20:42.849942Z","iopub.status.idle":"2023-07-17T02:20:42.857868Z","shell.execute_reply.started":"2023-07-17T02:20:42.849917Z","shell.execute_reply":"2023-07-17T02:20:42.85689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_feature","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:42.859598Z","iopub.execute_input":"2023-07-17T02:20:42.860281Z","iopub.status.idle":"2023-07-17T02:20:42.86593Z","shell.execute_reply.started":"2023-07-17T02:20:42.860248Z","shell.execute_reply":"2023-07-17T02:20:42.864849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{}},{"cell_type":"code","source":"test_npz_data = np.load(\"/kaggle/input/cafa-create-dataset-for-clustering-v2/df_test.npz\")\n# l2 normalization\ntest_feature = test_npz_data[CFG.seq_backbone] / np.linalg.norm(test_npz_data[CFG.seq_backbone], axis=1).reshape(-1, 1)\ntest_feature.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:42.867214Z","iopub.execute_input":"2023-07-17T02:20:42.867956Z","iopub.status.idle":"2023-07-17T02:20:44.635666Z","shell.execute_reply.started":"2023-07-17T02:20:42.867903Z","shell.execute_reply":"2023-07-17T02:20:44.634686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nscores, indicies = faiss_searcher.search(test_feature[:10] if CFG.debug else test_feature, k=topN)\n# normalizing cosine similarity 0-1\nscores = (scores + 1) / 2","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:44.636975Z","iopub.execute_input":"2023-07-17T02:20:44.637422Z","iopub.status.idle":"2023-07-17T02:20:44.64947Z","shell.execute_reply.started":"2023-07-17T02:20:44.637388Z","shell.execute_reply":"2023-07-17T02:20:44.648167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Ranking","metadata":{}},{"cell_type":"code","source":"majority_terms = pickleIO(None, \"/kaggle/input/cafa-create-dataset-for-clustering-v2/majority_terms.pkl\", \"r\")","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:44.650752Z","iopub.execute_input":"2023-07-17T02:20:44.651165Z","iopub.status.idle":"2023-07-17T02:20:44.658734Z","shell.execute_reply.started":"2023-07-17T02:20:44.651132Z","shell.execute_reply":"2023-07-17T02:20:44.657695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_vector = np.zeros((len(test_feature), len(majority_terms)), dtype=\"float32\")\nfor i in tqdm(range(len(scores))):\n    \n    # calculating score\n    output_vector[i] = (train_label[indicies[i]] * scores[i].reshape(-1, 1)).mean(axis=0)\n    # final score normalizing 0-1\n    output_vector[i] = output_vector[i] / output_vector[i].max()\n\n    if CFG.debug:\n        if i >= 100:\n            break","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:44.66029Z","iopub.execute_input":"2023-07-17T02:20:44.660708Z","iopub.status.idle":"2023-07-17T02:20:44.799251Z","shell.execute_reply.started":"2023-07-17T02:20:44.660675Z","shell.execute_reply":"2023-07-17T02:20:44.797972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_vector.min(), output_vector.max()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:44.800677Z","iopub.execute_input":"2023-07-17T02:20:44.801579Z","iopub.status.idle":"2023-07-17T02:20:45.293219Z","shell.execute_reply.started":"2023-07-17T02:20:44.801543Z","shell.execute_reply":"2023-07-17T02:20:45.292356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pickleIO(output_vector, \"./t5_cs_output.pkl\", \"w\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"submission = pickleIO(None, \"/kaggle/input/cafa-create-dataset-for-clustering-v2/df_test_meta.pkl\", \"r\")\nsubmission","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:45.294653Z","iopub.execute_input":"2023-07-17T02:20:45.294986Z","iopub.status.idle":"2023-07-17T02:20:45.463349Z","shell.execute_reply.started":"2023-07-17T02:20:45.294954Z","shell.execute_reply":"2023-07-17T02:20:45.461572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for uniprot_id, output in tqdm(zip(submission[\"uniprot_id\"], output_vector), total=len(submission)):\n    df_tmp = pd.DataFrame()\n    df_tmp[\"term\"] = majority_terms.copy().values\n    df_tmp[\"prob\"] = output.round(3)\n    df_tmp = df_tmp[df_tmp[\"prob\"] > 0.0]\n    df_tmp[\"uniprot_id\"] = uniprot_id\n    df_tmp = df_tmp[[\"uniprot_id\", \"term\", \"prob\"]]\n    df_tmp = df_tmp.sort_values(\"prob\", ascending=False).reset_index(drop=True)\n    df_tmp.to_csv(\"submission.tsv\", mode='a', header=False, sep=\"\\t\", index=False)\n    \n    if CFG.debug:\n        break","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:45.465453Z","iopub.execute_input":"2023-07-17T02:20:45.466284Z","iopub.status.idle":"2023-07-17T02:20:45.51141Z","shell.execute_reply.started":"2023-07-17T02:20:45.466236Z","shell.execute_reply":"2023-07-17T02:20:45.509912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head submission.tsv","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:45.514066Z","iopub.execute_input":"2023-07-17T02:20:45.515154Z","iopub.status.idle":"2023-07-17T02:20:46.737806Z","shell.execute_reply.started":"2023-07-17T02:20:45.515117Z","shell.execute_reply":"2023-07-17T02:20:46.73666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tail submission.tsv","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:20:46.739253Z","iopub.execute_input":"2023-07-17T02:20:46.739707Z","iopub.status.idle":"2023-07-17T02:20:47.843846Z","shell.execute_reply.started":"2023-07-17T02:20:46.739665Z","shell.execute_reply":"2023-07-17T02:20:47.842736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}