{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import Ridge\nfrom sklearn.metrics import roc_auc_score\n\nimport numpy as np\nimport pandas as pd\n\nfrom tqdm import tqdm\ntqdm.pandas()\n\n# PCA for dimensionality reduction\nfrom sklearn.decomposition import PCA","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-28T16:03:45.837001Z","iopub.execute_input":"2023-04-28T16:03:45.837787Z","iopub.status.idle":"2023-04-28T16:03:47.300914Z","shell.execute_reply.started":"2023-04-28T16:03:45.83773Z","shell.execute_reply":"2023-04-28T16:03:47.298764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Assigning labels","metadata":{}},{"cell_type":"code","source":"DATA_DIR = '/kaggle/input/cafa-5-protein-function-prediction'\nMAX_LABELS = 1500","metadata":{"execution":{"iopub.status.busy":"2023-04-28T16:04:43.578707Z","iopub.execute_input":"2023-04-28T16:04:43.579185Z","iopub.status.idle":"2023-04-28T16:04:43.584613Z","shell.execute_reply.started":"2023-04-28T16:04:43.579147Z","shell.execute_reply":"2023-04-28T16:04:43.583109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms = pd.read_csv(os.path.join(DATA_DIR, 'Train', 'train_terms.tsv'), sep='\\t')\n\nterms = train_terms.groupby(['aspect', 'term'])['term'].count().reset_index(name='frequency')\nprint(terms.groupby('aspect')['term'].nunique())","metadata":{"execution":{"iopub.status.busy":"2023-04-28T16:04:43.873464Z","iopub.execute_input":"2023-04-28T16:04:43.874787Z","iopub.status.idle":"2023-04-28T16:04:49.401292Z","shell.execute_reply.started":"2023-04-28T16:04:43.874735Z","shell.execute_reply":"2023-04-28T16:04:49.39981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fractions = (terms.groupby('aspect')['term'].nunique() / terms['term'].nunique() * MAX_LABELS).apply(round)\nprint(fractions)\n\nselected_terms = set()\nfor aspect, number in fractions.items():\n    selection = terms.loc[(terms.aspect == aspect)]\n    selection = selection.nlargest(number, columns='frequency', keep='first')\n    selected_terms.update(selection.term.to_list())","metadata":{"execution":{"iopub.status.busy":"2023-04-28T16:04:57.93457Z","iopub.execute_input":"2023-04-28T16:04:57.935016Z","iopub.status.idle":"2023-04-28T16:04:57.993991Z","shell.execute_reply.started":"2023-04-28T16:04:57.934967Z","shell.execute_reply":"2023-04-28T16:04:57.992379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(selected_terms)","metadata":{"execution":{"iopub.status.busy":"2023-04-28T16:05:20.943437Z","iopub.execute_input":"2023-04-28T16:05:20.943919Z","iopub.status.idle":"2023-04-28T16:05:20.950926Z","shell.execute_reply.started":"2023-04-28T16:05:20.943874Z","shell.execute_reply":"2023-04-28T16:05:20.949432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def assign_labels(annotations, selected_terms=selected_terms):\n    \n    intersection = selected_terms.intersection(annotations)\n    labels = np.isin(np.array(list(selected_terms)), np.array(list(intersection)))\n    \n    return list(labels.astype('int'))\n\nannotations = train_terms.groupby('EntryID')['term'].apply(set)\nlabels = annotations.progress_apply(assign_labels)\n\nlabels.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-28T16:05:32.699546Z","iopub.execute_input":"2023-04-28T16:05:32.700048Z","iopub.status.idle":"2023-04-28T16:05:53.676203Z","shell.execute_reply.started":"2023-04-28T16:05:32.700007Z","shell.execute_reply":"2023-04-28T16:05:53.674855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading train embeddings","metadata":{}},{"cell_type":"code","source":"train_ids = np.load('/kaggle/input/cafa-5-ems-2-embeddings-numpy/train_ids.npy')\n\nx_train = np.load('/kaggle/input/cafa-5-ems-2-embeddings-numpy/train_embeddings.npy')\ny_train = np.array(labels[train_ids].to_list())","metadata":{"execution":{"iopub.status.busy":"2023-04-28T16:05:57.934869Z","iopub.execute_input":"2023-04-28T16:05:57.935332Z","iopub.status.idle":"2023-04-28T16:06:05.530673Z","shell.execute_reply.started":"2023-04-28T16:05:57.935291Z","shell.execute_reply":"2023-04-28T16:06:05.529264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"x_train, x_valid, y_train, y_valid = train_test_split(x_train, y_train, shuffle=True, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-04-28T16:06:21.502546Z","iopub.execute_input":"2023-04-28T16:06:21.503028Z","iopub.status.idle":"2023-04-28T16:06:21.973194Z","shell.execute_reply.started":"2023-04-28T16:06:21.502985Z","shell.execute_reply":"2023-04-28T16:06:21.971979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reduce dimensionality\npca = PCA(n_components=512)\nx_train = pca.fit_transform(x_train)\nx_valid = pca.transform(x_valid)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Ridge(random_state=42).fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-04-28T16:06:52.900838Z","iopub.execute_input":"2023-04-28T16:06:52.901265Z","iopub.status.idle":"2023-04-28T16:06:55.658736Z","shell.execute_reply.started":"2023-04-28T16:06:52.901229Z","shell.execute_reply":"2023-04-28T16:06:55.656918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_hat = model.predict(x_valid)\n\nscores = pd.DataFrame(columns=selected_terms, index=['roc_auc'])\n\nfor i, term in enumerate(selected_terms):\n    score = roc_auc_score(y_valid[:, i], y_hat[:, i])\n    scores[term] = score\n\nscores.mean(axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-04-28T16:07:04.817323Z","iopub.execute_input":"2023-04-28T16:07:04.817735Z","iopub.status.idle":"2023-04-28T16:07:06.388253Z","shell.execute_reply.started":"2023-04-28T16:07:04.8177Z","shell.execute_reply":"2023-04-28T16:07:06.387071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"test_ids = np.load('/kaggle/input/cafa-5-ems-2-embeddings-numpy/test_ids.npy')\nx_test = np.load('/kaggle/input/cafa-5-ems-2-embeddings-numpy/test_embeddings.npy')","metadata":{"execution":{"iopub.status.busy":"2023-04-28T16:07:31.612338Z","iopub.execute_input":"2023-04-28T16:07:31.612771Z","iopub.status.idle":"2023-04-28T16:07:37.874243Z","shell.execute_reply.started":"2023-04-28T16:07:31.612734Z","shell.execute_reply":"2023-04-28T16:07:37.87302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = pca.transform(x_test)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(x_test)\n\nsub = pd.DataFrame(data=predictions, columns=selected_terms, index=test_ids)\nsub = sub.T.unstack().reset_index(name='prediction')\nsub = sub.loc[sub['prediction'] > 0]\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-28T16:07:53.410645Z","iopub.execute_input":"2023-04-28T16:07:53.412245Z","iopub.status.idle":"2023-04-28T16:07:57.487826Z","shell.execute_reply.started":"2023-04-28T16:07:53.412189Z","shell.execute_reply":"2023-04-28T16:07:57.486338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.tsv', sep='\\t', index=False, header=False)","metadata":{},"execution_count":null,"outputs":[]}]}