{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import Ridge\nfrom sklearn.metrics import roc_auc_score\n\nimport numpy as np\nimport pandas as pd\n\nfrom tqdm import tqdm\ntqdm.pandas()\n\nfrom keras.models import Sequential\nfrom keras.layers import Dense\n# measure roc auc score metric \nfrom tensorflow.keras.metrics import AUC","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":1.290929,"end_time":"2023-04-28T16:11:40.612349","exception":false,"start_time":"2023-04-28T16:11:39.32142","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-28T17:43:59.704109Z","iopub.execute_input":"2023-06-28T17:43:59.704586Z","iopub.status.idle":"2023-06-28T17:43:59.722943Z","shell.execute_reply.started":"2023-06-28T17:43:59.704542Z","shell.execute_reply":"2023-06-28T17:43:59.721041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Assigning labels","metadata":{"papermill":{"duration":0.004422,"end_time":"2023-04-28T16:11:40.621902","exception":false,"start_time":"2023-04-28T16:11:40.61748","status":"completed"},"tags":[]}},{"cell_type":"code","source":"DATA_DIR = '/kaggle/input/cafa-5-protein-function-prediction'\nMAX_LABELS = 500","metadata":{"papermill":{"duration":0.01492,"end_time":"2023-04-28T16:11:40.6416","exception":false,"start_time":"2023-04-28T16:11:40.62668","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-28T17:43:59.725642Z","iopub.execute_input":"2023-06-28T17:43:59.731033Z","iopub.status.idle":"2023-06-28T17:43:59.746243Z","shell.execute_reply.started":"2023-06-28T17:43:59.730976Z","shell.execute_reply":"2023-06-28T17:43:59.74523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms = pd.read_csv(os.path.join(DATA_DIR, 'Train', 'train_terms.tsv'), sep='\\t')\n\nterms = train_terms.groupby(['aspect', 'term'])['term'].count().reset_index(name='frequency')\nprint(terms.groupby('aspect')['term'].nunique())","metadata":{"papermill":{"duration":5.386139,"end_time":"2023-04-28T16:11:46.032438","exception":false,"start_time":"2023-04-28T16:11:40.646299","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-28T17:43:59.748143Z","iopub.execute_input":"2023-06-28T17:43:59.749025Z","iopub.status.idle":"2023-06-28T17:44:05.592365Z","shell.execute_reply.started":"2023-06-28T17:43:59.748979Z","shell.execute_reply":"2023-06-28T17:44:05.591114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fractions = (terms.groupby('aspect')['term'].nunique() / terms['term'].nunique() * MAX_LABELS).apply(round)\nprint(fractions)\n\nselected_terms = set()\nfor aspect, number in fractions.items():\n    selection = terms.loc[(terms.aspect == aspect)]\n    selection = selection.nlargest(number, columns='frequency', keep='first')\n    selected_terms.update(selection.term.to_list())","metadata":{"papermill":{"duration":0.07447,"end_time":"2023-04-28T16:11:46.112388","exception":false,"start_time":"2023-04-28T16:11:46.037918","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-28T17:44:05.594012Z","iopub.execute_input":"2023-06-28T17:44:05.594804Z","iopub.status.idle":"2023-06-28T17:44:05.653645Z","shell.execute_reply.started":"2023-06-28T17:44:05.594762Z","shell.execute_reply":"2023-06-28T17:44:05.652471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(selected_terms)","metadata":{"papermill":{"duration":0.016051,"end_time":"2023-04-28T16:11:46.133324","exception":false,"start_time":"2023-04-28T16:11:46.117273","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-28T17:44:05.657012Z","iopub.execute_input":"2023-06-28T17:44:05.657309Z","iopub.status.idle":"2023-06-28T17:44:05.667797Z","shell.execute_reply.started":"2023-06-28T17:44:05.65728Z","shell.execute_reply":"2023-06-28T17:44:05.666673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def assign_labels(annotations, selected_terms=selected_terms):\n    \n    intersection = selected_terms.intersection(annotations)\n    labels = np.isin(np.array(list(selected_terms)), np.array(list(intersection)))\n    \n    return list(labels.astype('int'))\n\nannotations = train_terms.groupby('EntryID')['term'].apply(set)\nlabels = annotations.progress_apply(assign_labels)\n\nlabels.head()","metadata":{"papermill":{"duration":141.942432,"end_time":"2023-04-28T16:14:08.08098","exception":false,"start_time":"2023-04-28T16:11:46.138548","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-28T17:44:05.669132Z","iopub.execute_input":"2023-06-28T17:44:05.669568Z","iopub.status.idle":"2023-06-28T17:45:18.73452Z","shell.execute_reply.started":"2023-06-28T17:44:05.669534Z","shell.execute_reply":"2023-06-28T17:45:18.733449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading train embeddings","metadata":{"papermill":{"duration":0.082986,"end_time":"2023-04-28T16:14:08.247278","exception":false,"start_time":"2023-04-28T16:14:08.164292","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_ids = np.load('/kaggle/input/t5embeds/train_ids.npy')\n\nx_train = np.load('/kaggle/input/t5embeds/train_embeds.npy')\ny_train = np.array(labels[train_ids].to_list())","metadata":{"papermill":{"duration":20.577499,"end_time":"2023-04-28T16:14:28.907666","exception":false,"start_time":"2023-04-28T16:14:08.330167","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-28T17:45:18.737121Z","iopub.execute_input":"2023-06-28T17:45:18.738083Z","iopub.status.idle":"2023-06-28T17:45:35.890413Z","shell.execute_reply.started":"2023-06-28T17:45:18.738043Z","shell.execute_reply":"2023-06-28T17:45:35.889282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{"papermill":{"duration":0.083305,"end_time":"2023-04-28T16:14:29.076391","exception":false,"start_time":"2023-04-28T16:14:28.993086","status":"completed"},"tags":[]}},{"cell_type":"code","source":"x_train, x_valid, y_train, y_valid = train_test_split(x_train, y_train, shuffle=True, random_state=42)","metadata":{"papermill":{"duration":1.175161,"end_time":"2023-04-28T16:14:30.339394","exception":false,"start_time":"2023-04-28T16:14:29.164233","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-28T17:45:35.891977Z","iopub.execute_input":"2023-06-28T17:45:35.892372Z","iopub.status.idle":"2023-06-28T17:45:36.45479Z","shell.execute_reply.started":"2023-06-28T17:45:35.892316Z","shell.execute_reply":"2023-06-28T17:45:36.453557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build a simple MLP model in Keras with ReLU activation and nothing else\nnfeats = x_train.shape[1]\nnlabels = y_train.shape[1]\nmodel = Sequential()\nmodel.add(Dense(256, activation='relu', input_dim=nfeats))\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dense(nlabels, activation='sigmoid'))\nmodel.compile(loss='binary_crossentropy',\n                optimizer='adam',\n                metrics=[AUC()])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T17:45:36.456307Z","iopub.execute_input":"2023-06-28T17:45:36.456985Z","iopub.status.idle":"2023-06-28T17:45:39.395986Z","shell.execute_reply.started":"2023-06-28T17:45:36.456941Z","shell.execute_reply":"2023-06-28T17:45:39.395132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(x_train, y_train, epochs=15, batch_size=128, validation_data=(x_valid, y_valid))","metadata":{"papermill":{"duration":6.259531,"end_time":"2023-04-28T16:14:36.681742","exception":false,"start_time":"2023-04-28T16:14:30.422211","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-28T17:45:39.397535Z","iopub.execute_input":"2023-06-28T17:45:39.397912Z","iopub.status.idle":"2023-06-28T17:47:04.277603Z","shell.execute_reply.started":"2023-06-28T17:45:39.397875Z","shell.execute_reply":"2023-06-28T17:47:04.276469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_hat = model.predict(x_valid)\n\nscores = pd.DataFrame(columns=list(selected_terms), index=['roc_auc'])\n\nfor i, term in enumerate(selected_terms):\n    score = roc_auc_score(y_valid[:, i], y_hat[:, i])\n    scores[term] = score\n\nscores.mean(axis=1)","metadata":{"papermill":{"duration":20.422772,"end_time":"2023-04-28T16:14:57.256109","exception":false,"start_time":"2023-04-28T16:14:36.833337","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-28T17:47:04.27933Z","iopub.execute_input":"2023-06-28T17:47:04.279804Z","iopub.status.idle":"2023-06-28T17:47:13.842945Z","shell.execute_reply.started":"2023-06-28T17:47:04.279773Z","shell.execute_reply":"2023-06-28T17:47:13.841564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{"papermill":{"duration":0.08303,"end_time":"2023-04-28T16:14:57.422033","exception":false,"start_time":"2023-04-28T16:14:57.339003","status":"completed"},"tags":[]}},{"cell_type":"code","source":"test_ids = np.load('/kaggle/input/t5embeds/test_ids.npy')\nx_test = np.load('/kaggle/input/t5embeds/test_embeds.npy')","metadata":{"papermill":{"duration":6.254099,"end_time":"2023-04-28T16:15:03.759386","exception":false,"start_time":"2023-04-28T16:14:57.505287","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-28T17:47:13.844677Z","iopub.execute_input":"2023-06-28T17:47:13.845094Z","iopub.status.idle":"2023-06-28T17:47:25.961602Z","shell.execute_reply.started":"2023-06-28T17:47:13.845062Z","shell.execute_reply":"2023-06-28T17:47:25.960015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del x_train, y_train, x_valid, y_valid, labels\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T17:47:25.964558Z","iopub.execute_input":"2023-06-28T17:47:25.965127Z","iopub.status.idle":"2023-06-28T17:47:27.518532Z","shell.execute_reply.started":"2023-06-28T17:47:25.965092Z","shell.execute_reply":"2023-06-28T17:47:27.517439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(x_test)\ndel x_test\ngc.collect()\n\nchunk_size = 5_000\nchunks = [range(i, min(i + chunk_size, len(predictions))) for i in range(0, len(predictions), chunk_size)]\n\nfinal_sub = pd.DataFrame()  # Create an empty DataFrame to hold the final result\n\nprint(f\"processing {len(chunks)} chunks of {chunk_size} predictions each\")\n\nfor chunk in chunks:\n    print(f\"processing chunk {chunk}\")\n    sub = pd.DataFrame(data=predictions[chunk], columns=list(selected_terms), index=test_ids[chunk])\n    sub = sub.T.unstack().reset_index(name='prediction')\n    sub = sub.loc[sub['prediction'] > 0]\n    final_sub = pd.concat([final_sub, sub])  # Concatenate current chunk DataFrame to the final DataFrame\n\nfinal_sub.head()","metadata":{"papermill":{"duration":48.344071,"end_time":"2023-04-28T16:15:52.186228","exception":false,"start_time":"2023-04-28T16:15:03.842157","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-28T17:47:27.523872Z","iopub.execute_input":"2023-06-28T17:47:27.524631Z","iopub.status.idle":"2023-06-28T17:48:39.735154Z","shell.execute_reply.started":"2023-06-28T17:47:27.524593Z","shell.execute_reply":"2023-06-28T17:48:39.733954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_sub.to_csv('submission.tsv', sep='\\t', index=False, header=False)","metadata":{"papermill":{"duration":296.790114,"end_time":"2023-04-28T16:20:49.060174","exception":false,"start_time":"2023-04-28T16:15:52.27006","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-28T17:48:39.737472Z","iopub.execute_input":"2023-06-28T17:48:39.738308Z","iopub.status.idle":"2023-06-28T17:53:46.348811Z","shell.execute_reply.started":"2023-06-28T17:48:39.738265Z","shell.execute_reply":"2023-06-28T17:53:46.347686Z"},"trusted":true},"execution_count":null,"outputs":[]}]}