{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This is the code for calculating embeddings from the t5 dataset https://www.kaggle.com/datasets/sergeifironov/t5embeds. Unfortunately, it is impossible to run it on Kaggle resources, even with a batch size of 1 you need A100 for evaluation.","metadata":{}},{"cell_type":"code","source":"!pip install obonet\n!pip install pyvis","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport json\nfrom typing import Dict\nfrom collections import Counter\n\nimport random\nimport obonet\nimport pandas as pd\nimport numpy as np\nfrom Bio import SeqIO\n\nfrom transformers import T5Tokenizer, T5EncoderModel\nimport torch\ndevice = torch.device('cuda:0' if torch.cuda.is_available() else 'cpu')\n\n# Load the tokenizer\ntokenizer = T5Tokenizer.from_pretrained('Rostlab/prot_t5_xl_half_uniref50-enc', do_lower_case=False) #.to(device)\n\n# Load the model\nmodel = T5EncoderModel.from_pretrained(\"Rostlab/prot_t5_xl_half_uniref50-enc\").to(device);\n\n# only GPUs support half-precision currently; if you want to run on CPU use full-precision (not recommended, much slower)\n#model.full() if device=='cpu' else model.half()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/cafa-5-protein-function-prediction'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\npath = Path('/kaggle/input/cafa-5-protein-function-prediction')\n!head {path}/'Test (Targets)/testsuperset-taxon-list.tsv'","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head {path}/'Test (Targets)/testsuperset.fasta'","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\ndef get_embeddings(seq):\n    sequence_examples = [\" \".join(list(re.sub(r\"[UZOB]\", \"X\", seq)))]\n\n    ids = tokenizer.batch_encode_plus(sequence_examples, add_special_tokens=True, padding=\"longest\")\n\n    input_ids = torch.tensor(ids['input_ids']).to(device)\n    attention_mask = torch.tensor(ids['attention_mask']).to(device)\n\n    # generate embeddings\n    with torch.no_grad():\n        embedding_repr = model(input_ids=input_ids,\n                               attention_mask=attention_mask)\n\n    # extract residue embeddings for the first ([0,:]) sequence in the batch and remove padded & special tokens ([0,:7]) \n    emb_0 = embedding_repr.last_hidden_state[0]\n    emb_0_per_protein = emb_0.mean(dim=0)\n    \n    return emb_0_per_protein\n\nget_embeddings('MTMDKSELVQKAKLAEQAERYDDMAAAMKAVTEQGHELSNEERNLLSVAYKNVVGARRSS')\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fn = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta'\nprint(\"Sequence example:\\n\\n\", next(iter(SeqIO.parse(fn, \"fasta\"))))\nsequences = SeqIO.parse(fn, \"fasta\")\nnum_sequences = sum(1 for seq in sequences)\nprint()\nprint(\"Number of sequences in train:\", num_sequences)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tqdm\nfn = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta'\n\nsequences = SeqIO.parse(fn, \"fasta\")\n\nids = []\nembeds = np.zeros((num_sequences, 1024))\ni = 0\nfor seq in tqdm.tqdm(sequences):\n    ids.append(seq.id)\n    embeds[i] = get_embeddings(str(seq.seq)).detach().cpu().numpy()\n    i += 1\n    break #remove it for full calculation\n        \nnp.save('train_embeds.npy', embeds)\nnp.save('train_ids.npy', np.array(ids))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fn = '/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta'\n\nsequences = SeqIO.parse(fn, \"fasta\")\nnum_sequences = sum(1 for seq in sequences)\nprint(\"Number of sequences in test:\", num_sequences)\nsequences = SeqIO.parse(fn, \"fasta\")\n\n\nids = []\nembeds = np.zeros((num_sequences, 1024))\ni = 0\nfor seq in tqdm.tqdm(sequences):\n    ids.append(seq.id)\n    embeds[i] = get_embeddings(str(seq.seq)).detach().cpu().numpy()\n    i += 1\n    break #remove it for full calculation\n    \nnp.save('test_embeds.npy', embeds)\nnp.save('test_ids.npy', np.array(ids))","metadata":{},"execution_count":null,"outputs":[]}]}