{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-04T12:47:13.920832Z","iopub.execute_input":"2023-08-04T12:47:13.921263Z","iopub.status.idle":"2023-08-04T12:47:13.935193Z","shell.execute_reply.started":"2023-08-04T12:47:13.921211Z","shell.execute_reply":"2023-08-04T12:47:13.934047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader\n\nimport tqdm","metadata":{"execution":{"iopub.status.busy":"2023-08-04T12:13:00.884062Z","iopub.execute_input":"2023-08-04T12:13:00.884681Z","iopub.status.idle":"2023-08-04T12:13:03.718846Z","shell.execute_reply.started":"2023-08-04T12:13:00.884647Z","shell.execute_reply":"2023-08-04T12:13:03.717917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract Amino acid sequences from fasta files...\n\ndef load_fasta_dict(dir_fasta):\n\n    with open(dir_fasta, mode='r') as f: # file open\n        lines = f.readlines()\n    dict_id_to_seq = {}\n    for l in lines:\n        l = l.strip()    \n        if l.startswith('>'):\n            current_id = l.split()[0][1:]\n            dict_id_to_seq[current_id] = ''\n        else:\n            dict_id_to_seq[current_id] += l\n\n\n    return dict_id_to_seq # {id : seq}","metadata":{"execution":{"iopub.status.busy":"2023-08-04T12:13:03.720619Z","iopub.execute_input":"2023-08-04T12:13:03.721131Z","iopub.status.idle":"2023-08-04T12:13:03.727357Z","shell.execute_reply.started":"2023-08-04T12:13:03.7211Z","shell.execute_reply":"2023-08-04T12:13:03.72648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract Amino acid sequences from fasta files\n\ndir_train_fasta = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta'\n\ndict_id_to_seq = load_fasta_dict(dir_train_fasta)\n\nprint('=' * 36)\nprint('| TOTAL NUMS OF FASTA | %10d |' % len(dict_id_to_seq))\nprint('=' * 36)\n\nprint(f'Example EntryID : {list(dict_id_to_seq.keys())[0]}')\nprint(f'Example Sequence: {list(dict_id_to_seq.values())[0][:50]}...')","metadata":{"execution":{"iopub.status.busy":"2023-08-04T12:13:03.728643Z","iopub.execute_input":"2023-08-04T12:13:03.72921Z","iopub.status.idle":"2023-08-04T12:13:06.490547Z","shell.execute_reply.started":"2023-08-04T12:13:03.729174Z","shell.execute_reply":"2023-08-04T12:13:06.489445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sequence lengths\n\nseq_lengths = np.array([len(seq) for seq in dict_id_to_seq.values()])\n\nprint('=' * 40)\nprint('| ----- AA sequences ----- |   length   |' % seq_lengths.max())\n\nprint('| Max     of AAs sequences | %10d |' % seq_lengths.max())\nprint('| Min     of AAs sequences | %10d |' % seq_lengths.min())\nprint('| 25%', 'Per of AAs sequences | %10d |' % np.percentile(seq_lengths, q=25))\nprint('| 50%', 'Per of AAs sequences | %10d |' % np.percentile(seq_lengths, q=50))\nprint('| 75%', 'Per of AAs sequences | %10d |' % np.percentile(seq_lengths, q=75))\nprint('=' * 40)","metadata":{"execution":{"iopub.status.busy":"2023-08-04T12:13:07.463716Z","iopub.execute_input":"2023-08-04T12:13:07.464063Z","iopub.status.idle":"2023-08-04T12:13:07.528284Z","shell.execute_reply.started":"2023-08-04T12:13:07.464036Z","shell.execute_reply":"2023-08-04T12:13:07.527287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 3))\nplt.subplot(1, 4, 1)\nplt.hist(seq_lengths, bins=100)\n\nplt.subplot(1, 4, 2)\ndata = [l for l in seq_lengths if l >= 1024]\nplt.hist(data, bins=100)\nplt.title(f'{len(data) / len(seq_lengths) *100:.2f}%_lengths>=1024')\n\nplt.subplot(1, 4, 3)\ndata = [l for l in seq_lengths if l >= 2048]\nplt.hist(data, bins=100)\nplt.title(f'{len(data) / len(seq_lengths) *100:.2f}%_lengths>=2048')\n\n\nplt.subplot(1, 4, 4)\ndata = [l for l in seq_lengths if l >= 3072]\nplt.hist(data, bins=100)\nplt.title(f'{len(data) / len(seq_lengths) *100:.2f}%_lengths>=3072')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-04T12:13:09.339003Z","iopub.execute_input":"2023-08-04T12:13:09.339738Z","iopub.status.idle":"2023-08-04T12:13:10.616155Z","shell.execute_reply.started":"2023-08-04T12:13:09.339699Z","shell.execute_reply":"2023-08-04T12:13:10.61512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Constructing DataFrame (columns | ID | Amino Acid Seq | Labels(targets))\n\ndef load_GO_term(dir_term, dict_id_to_seq:dict):\n    df_term = pd.read_csv(dir_term, sep='\\t') # df_term is the dataset including Target labels...\n    df_labels = pd.DataFrame({\n        'EntryID' : df_term.EntryID.values,\n        'Target' : df_term.term.values,\n        'Seq' : [dict_id_to_seq[id] if id in dict_id_to_seq.keys() else 0 for id in df_term.EntryID.values],\n    })\n\n    return df_labels","metadata":{"execution":{"iopub.status.busy":"2023-08-04T12:13:12.892366Z","iopub.execute_input":"2023-08-04T12:13:12.892734Z","iopub.status.idle":"2023-08-04T12:13:12.898737Z","shell.execute_reply.started":"2023-08-04T12:13:12.892703Z","shell.execute_reply":"2023-08-04T12:13:12.897733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dir_term = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv'\n\ndf_term = pd.read_csv(dir_term, sep='\\t')\ndf_term.head() # original dataset...","metadata":{"execution":{"iopub.status.busy":"2023-08-04T12:13:13.379932Z","iopub.execute_input":"2023-08-04T12:13:13.380449Z","iopub.status.idle":"2023-08-04T12:13:16.677156Z","shell.execute_reply.started":"2023-08-04T12:13:13.380408Z","shell.execute_reply":"2023-08-04T12:13:16.67616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_labels = load_GO_term(dir_term, dict_id_to_seq)\ndf_labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-04T12:13:16.678762Z","iopub.execute_input":"2023-08-04T12:13:16.679085Z","iopub.status.idle":"2023-08-04T12:13:20.788067Z","shell.execute_reply.started":"2023-08-04T12:13:16.679059Z","shell.execute_reply":"2023-08-04T12:13:20.78731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_labels.to_csv('train_dataset.csv')","metadata":{"execution":{"iopub.status.busy":"2023-08-04T12:52:49.036271Z","iopub.execute_input":"2023-08-04T12:52:49.036608Z","iopub.status.idle":"2023-08-04T12:54:35.069693Z","shell.execute_reply.started":"2023-08-04T12:52:49.03658Z","shell.execute_reply":"2023-08-04T12:54:35.0685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's tokenized the Amino acid sequences and label encodig....\n\nclass Tokenizer:\n    '''\n    \"Tokenizers\" tokenize amino acid sequences and label them numerically.\n    '''\n    def __init__(self, maxlen=30000, padding=True):\n        self.maxlen = maxlen\n        self.padding = padding\n        self.amino_acids = ['<PAD>', '<Unk>','H', 'C', 'S','A','V','Q','O','P','T','Y','F','D','M','X','B','Z','U','G','L','I','W','R','E','K','N'] # amino acid vocab\n        self.dict_amino_acids_to_index = {aa : i for i, aa in enumerate(self.amino_acids)}\n        self.dict_index_to_amino_acids = {i : aa for i, aa in enumerate(self.amino_acids)}\n\n    def __call__(self, seq):\n        seq_embded = self.amino_acid_to_index(seq) \n\n        if self.padding:\n            if len(seq_embded) <= self.maxlen:\n                tmp = np.zeros(shape=(self.maxlen, ))\n                tmp[:len(seq_embded)] = seq_embded\n                seq_embded = tmp.copy()\n            else:\n                seq_embded = seq_embded[:self.maxlen]\n        \n        return seq_embded\n    \n    def amino_acid_to_index(self, seq_fasta: str) -> list:\n        '''\n        Converting amino acids to label encoded sequences\n        '''\n        seq_embeded = []\n        for aa in seq_fasta:\n            if aa not in self.amino_acids:\n                aa = self.amino_acids[1] # <Unk>\n            seq_embeded.append(self.dict_amino_acids_to_index[aa])  \n\n        return seq_embeded\n    \n    def index_to_amino_acid(self, seq_embeded: list) -> str:\n        '''\n        Converting label encoded sequnces to amino acids\n        '''\n        return ''.join([self.dict_index_to_amino_acids[i] for i in seq_embeded])","metadata":{"execution":{"iopub.status.busy":"2023-08-04T12:13:20.789682Z","iopub.execute_input":"2023-08-04T12:13:20.79012Z","iopub.status.idle":"2023-08-04T12:13:20.804611Z","shell.execute_reply.started":"2023-08-04T12:13:20.790082Z","shell.execute_reply":"2023-08-04T12:13:20.803535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example\n\nexample_id = list(dict_id_to_seq.keys())[0]\nexample_seq = list(dict_id_to_seq.values())[0]\n\ntokenizer = Tokenizer(maxlen=50)\nencoded_seq= tokenizer(example_seq)\nencoded_seq_to_amino_acid = tokenizer.index_to_amino_acid(encoded_seq)\n\n\nprint(f'Example EntryID           : {example_id}')\nprint(f'Example Sequence          : {example_seq[:20]}...')\nprint(f'Example Labeled seq       : {encoded_seq[:20]}...')\nprint(f'Example Convert seq to AA : {encoded_seq_to_amino_acid[:20]}...')","metadata":{"execution":{"iopub.status.busy":"2023-08-04T12:13:20.806773Z","iopub.execute_input":"2023-08-04T12:13:20.807281Z","iopub.status.idle":"2023-08-04T12:13:20.831491Z","shell.execute_reply.started":"2023-08-04T12:13:20.807222Z","shell.execute_reply":"2023-08-04T12:13:20.830405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-04T12:20:04.965676Z","iopub.execute_input":"2023-08-04T12:20:04.966646Z","iopub.status.idle":"2023-08-04T12:20:04.977194Z","shell.execute_reply.started":"2023-08-04T12:20:04.966599Z","shell.execute_reply":"2023-08-04T12:20:04.97611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Multilabel problem...\n\ndef selected_label_proportion(df_labels, criteria):\n    df_label_count = df_labels.Target.value_counts()\n    df_selected_label = list(df_label_count[df_label_count > criteria].index)\n    df_label_selected = df_labels[df_labels.Target.isin(df_selected_label)]    \n    return len(df_label_selected) / len(df_labels) * 100, df_selected_label\n    \n\nprint(f'numbs of total labels | count_criteria > {0:3d} : {selected_label_proportion(df_labels, 0)[0]:.2f}% | num_labels : {len(selected_label_proportion(df_labels, 0)[1])}')\nprint(f'numbs of total labels | count_criteria > {200:3d} : {selected_label_proportion(df_labels, 200)[0]:.2f}% | num_labels : {len(selected_label_proportion(df_labels, 200)[1])}')\nprint(f'numbs of total labels | count_criteria > {400:3d} : {selected_label_proportion(df_labels, 400)[0]:.2f}% | num_labels : {len(selected_label_proportion(df_labels, 400)[1])}')\nprint(f'numbs of total labels | count_criteria > {600:3d} : {selected_label_proportion(df_labels, 600)[0]:.2f}% | num_labels : {len(selected_label_proportion(df_labels, 600)[1])}')\n","metadata":{"execution":{"iopub.status.busy":"2023-08-04T12:24:09.512735Z","iopub.execute_input":"2023-08-04T12:24:09.5131Z","iopub.status.idle":"2023-08-04T12:24:18.671327Z","shell.execute_reply.started":"2023-08-04T12:24:09.513072Z","shell.execute_reply":"2023-08-04T12:24:18.670088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I selected kinds of labels 1145... ()\n\n_, selected_labels = selected_label_proportion(df_labels, 600)\nselected_labels[:10]","metadata":{"execution":{"iopub.status.busy":"2023-08-04T12:33:26.174171Z","iopub.execute_input":"2023-08-04T12:33:26.174588Z","iopub.status.idle":"2023-08-04T12:33:27.291847Z","shell.execute_reply.started":"2023-08-04T12:33:26.174553Z","shell.execute_reply":"2023-08-04T12:33:27.290601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Final Create CAFADataset...\n            \nclass CAFADataset(torch.utils.data.Dataset):\n    def __init__(self, dir_dataset, maxlen=2047, is_train=False, selected_label_list=[], padding=True, ):\n\n        self.dir_dataset = dir_dataset\n        self.tokenizer = Tokenizer(maxlen, padding)\n        \n        # read preprocessed file (after load_rawdataset.py)\n        self.df_raw_data = pd.read_csv(dir_dataset) # columns -> EntryID, Seq, Tax\n        self.df_raw_data = self.df_raw_data.dropna(axis=0)\n        \n        self.is_train = is_train\n        self.dict_id_to_seq = {id : seq for id, seq in zip(self.df_raw_data.EntryID, self.df_raw_data.Seq)} # {id : seq}       \n       \n        # EntryID set...\n        id_set = set(self.df_raw_data.EntryID.values)\n\n\n        self.df_tmp = pd.DataFrame({\n            'EntryID' : list(id_set),\n            'Seq' : [self.dict_id_to_seq[id] for id in list(id_set)],\n        })\n\n        # Ids\n        self.ids = self.df_tmp.EntryID.values\n\n        # Sequence labeling using Tokenizer\n        self.seq_embedding = [torch.tensor(self.tokenizer(seq), dtype=torch.int32) for seq in self.df_tmp.Seq]\n        \n        if is_train:\n             # Multi Label class...\n            self.target = self.get_labels(selected_label_list)\n        \n\n    def __len__(self, ):\n        return len(self.ids)  \n\n    def __getitem__(self, idx):\n        \n        if self.is_train:\n            # train\n            return self.seq_embedding[idx], torch.tensor(self.target.iloc[idx], dtype=torch.int64)  # Type....\n        else:\n            # test\n            return self.seq_embedding[idx], self.ids[idx],   \n   \n    def get_labels(self, selected_label_list):\n\n        targets = {}\n        for _target in tqdm.tqdm(selected_label_list, desc='multilabeling...'):\n\n            # selection of \"df_id with _target\"\n            df_id_target = self.df_raw_data[self.df_raw_data.Target == _target].EntryID\n            \n            # New_multi labels columns\n            new_label = list((self.df_tmp['EntryID'].isin(df_id_target)).astype(np.int64))\n            targets[_target] = new_label\n\n        targets = pd.DataFrame.from_dict(targets)\n        return targets       \n","metadata":{"execution":{"iopub.status.busy":"2023-08-04T12:55:53.757274Z","iopub.execute_input":"2023-08-04T12:55:53.757639Z","iopub.status.idle":"2023-08-04T12:55:53.770702Z","shell.execute_reply.started":"2023-08-04T12:55:53.75761Z","shell.execute_reply":"2023-08-04T12:55:53.769182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CAFAdataset\ntrain_dataset = CAFADataset(dir_dataset='train_dataset.csv', maxlen=2048, is_train=True, selected_label_list=selected_labels)","metadata":{"execution":{"iopub.status.busy":"2023-08-04T12:55:55.268833Z","iopub.execute_input":"2023-08-04T12:55:55.269196Z","iopub.status.idle":"2023-08-04T13:06:17.682763Z","shell.execute_reply.started":"2023-08-04T12:55:55.269168Z","shell.execute_reply":"2023-08-04T13:06:17.681579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Samples\n\nsamples = next(iter(train_dataset))\n\nprint('Sequence')\nprint(f'Sequence shape, {samples[0].shape}')\nprint(f'Sequence :', samples[0])\n\nprint('\\nlabels')\nprint(f'Sequence shape, {samples[1].shape}')\nprint(f'Labels :', samples[1])\n","metadata":{"execution":{"iopub.status.busy":"2023-08-04T13:11:21.19834Z","iopub.execute_input":"2023-08-04T13:11:21.198729Z","iopub.status.idle":"2023-08-04T13:11:21.20784Z","shell.execute_reply.started":"2023-08-04T13:11:21.198697Z","shell.execute_reply":"2023-08-04T13:11:21.206868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}