{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":41875,"databundleVersionId":5521661,"sourceType":"competition"},{"sourceId":6055430,"sourceType":"datasetVersion","datasetId":3464728},{"sourceId":6065104,"sourceType":"datasetVersion","datasetId":3471058},{"sourceId":6068990,"sourceType":"datasetVersion","datasetId":3473533},{"sourceId":6170376,"sourceType":"datasetVersion","datasetId":3540453},{"sourceId":6232763,"sourceType":"datasetVersion","datasetId":3580425},{"sourceId":6233248,"sourceType":"datasetVersion","datasetId":3580735},{"sourceId":6315473,"sourceType":"datasetVersion","datasetId":3633917},{"sourceId":6315842,"sourceType":"datasetVersion","datasetId":3634124},{"sourceId":6315891,"sourceType":"datasetVersion","datasetId":3634152}],"dockerImageVersionId":30527,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-17T02:28:57.050404Z","iopub.execute_input":"2023-08-17T02:28:57.050946Z","iopub.status.idle":"2023-08-17T02:28:57.075514Z","shell.execute_reply.started":"2023-08-17T02:28:57.050909Z","shell.execute_reply":"2023-08-17T02:28:57.07462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install obonet","metadata":{"execution":{"iopub.status.busy":"2023-08-17T01:47:27.494082Z","iopub.execute_input":"2023-08-17T01:47:27.494751Z","iopub.status.idle":"2023-08-17T01:47:42.088121Z","shell.execute_reply.started":"2023-08-17T01:47:27.494715Z","shell.execute_reply":"2023-08-17T01:47:42.086483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_preprocessed = pd.read_csv('/kaggle/input/proprocessed-abstract-train/preprocessed_data.csv')","metadata":{"execution":{"iopub.status.busy":"2023-08-17T01:47:42.091257Z","iopub.execute_input":"2023-08-17T01:47:42.09207Z","iopub.status.idle":"2023-08-17T01:48:04.645367Z","shell.execute_reply.started":"2023-08-17T01:47:42.092014Z","shell.execute_reply":"2023-08-17T01:48:04.643934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preprocessed = pd.read_csv('/kaggle/input/preprocessed-data-testdi/preprocessed_data_test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-08-17T01:48:04.647468Z","iopub.execute_input":"2023-08-17T01:48:04.648004Z","iopub.status.idle":"2023-08-17T01:48:21.474624Z","shell.execute_reply.started":"2023-08-17T01:48:04.647965Z","shell.execute_reply":"2023-08-17T01:48:21.473229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_preprocessed","metadata":{"execution":{"iopub.status.busy":"2023-08-17T01:48:21.477763Z","iopub.execute_input":"2023-08-17T01:48:21.478291Z","iopub.status.idle":"2023-08-17T01:48:21.496139Z","shell.execute_reply.started":"2023-08-17T01:48:21.478258Z","shell.execute_reply":"2023-08-17T01:48:21.494483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preprocessed","metadata":{"execution":{"iopub.status.busy":"2023-08-17T01:48:21.498123Z","iopub.execute_input":"2023-08-17T01:48:21.498548Z","iopub.status.idle":"2023-08-17T01:48:21.518317Z","shell.execute_reply.started":"2023-08-17T01:48:21.498485Z","shell.execute_reply":"2023-08-17T01:48:21.51683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import obonet\nimport re\nimport nltk\nfrom nltk.corpus import stopwords\nimport string\n# Specify the path to your OBO file\nobo_file_path = '/kaggle/input/cafa-5-protein-function-prediction/Train/go-basic.obo'\n\n# Read the OBO file\ngraph = obonet.read_obo(obo_file_path)\n\nterms = [data.get('name') for _, data in graph.nodes.items()]\n\n# Combine all terms into a single string\nall_terms_text = ' '.join(terms)\n\n# Clean the text\ncleaned_text = re.sub(r'\\W', ' ', all_terms_text.lower())  # Remove non-alphanumeric characters\ncleaned_text = re.sub(r'\\s+', ' ', cleaned_text)  # Remove extra whitespaces\n\n# Tokenize the text\ntokens = nltk.word_tokenize(cleaned_text)\n\n# Remove stopwords\nstopwords = set(stopwords.words('english'))\nfiltered_words = [word for word in tokens if word.lower() not in stopwords and word not in string.punctuation and not word.isdigit() and len(word)>2]\n\n# Remove duplicate words\nfiltered_words_all = list(set(filtered_words))\n\n#print(filtered_words)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-17T01:48:21.520555Z","iopub.execute_input":"2023-08-17T01:48:21.521306Z","iopub.status.idle":"2023-08-17T01:48:46.258198Z","shell.execute_reply.started":"2023-08-17T01:48:21.521259Z","shell.execute_reply":"2023-08-17T01:48:46.256815Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_go_vocab = pd.DataFrame(filtered_words_all, columns=[\"vocab\"])\ndf_go_vocab.to_csv('/kaggle/working/Go_vocab_terms.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-08-17T01:48:46.25993Z","iopub.execute_input":"2023-08-17T01:48:46.260285Z","iopub.status.idle":"2023-08-17T01:48:46.299636Z","shell.execute_reply.started":"2023-08-17T01:48:46.260257Z","shell.execute_reply":"2023-08-17T01:48:46.298435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_go_vocab","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:41:02.878095Z","iopub.execute_input":"2023-08-17T02:41:02.878611Z","iopub.status.idle":"2023-08-17T02:41:02.894609Z","shell.execute_reply.started":"2023-08-17T02:41:02.878572Z","shell.execute_reply":"2023-08-17T02:41:02.892997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set_a = set(test_preprocessed['Protein ID'])\nset_b = set(train_preprocessed['Protein ID'])\n\n# Find elements in set_a but not in set_b\ndiff = set_a - set_b\n\n# Convert the result back to a list (if desired)\ndiff_list = list(diff)\n\n# Print the result\nprint(len(diff_list))","metadata":{"execution":{"iopub.status.busy":"2023-08-17T01:48:46.316484Z","iopub.execute_input":"2023-08-17T01:48:46.316978Z","iopub.status.idle":"2023-08-17T01:48:46.439413Z","shell.execute_reply.started":"2023-08-17T01:48:46.316936Z","shell.execute_reply":"2023-08-17T01:48:46.438209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_test_preprocessed = pd.concat([train_preprocessed, test_preprocessed], axis=0, ignore_index=True)\nprint(train_test_preprocessed)","metadata":{"execution":{"iopub.status.busy":"2023-08-17T01:48:46.444007Z","iopub.execute_input":"2023-08-17T01:48:46.444817Z","iopub.status.idle":"2023-08-17T01:48:46.487674Z","shell.execute_reply.started":"2023-08-17T01:48:46.444771Z","shell.execute_reply":"2023-08-17T01:48:46.486269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_test_preprocessed.drop_duplicates(subset=['Protein ID'], inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-17T01:48:46.489027Z","iopub.execute_input":"2023-08-17T01:48:46.489417Z","iopub.status.idle":"2023-08-17T01:48:46.59922Z","shell.execute_reply.started":"2023-08-17T01:48:46.489383Z","shell.execute_reply":"2023-08-17T01:48:46.597617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_test_preprocessed","metadata":{"execution":{"iopub.status.busy":"2023-08-17T01:48:46.60187Z","iopub.execute_input":"2023-08-17T01:48:46.602409Z","iopub.status.idle":"2023-08-17T01:48:46.625955Z","shell.execute_reply.started":"2023-08-17T01:48:46.602369Z","shell.execute_reply":"2023-08-17T01:48:46.6236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"import pandas as pd\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom scipy.sparse import save_npz\nabstracts_df = train_test_preprocessed\nprint('creating_tfidf')\n\n# Create the TfidfVectorizer\nvectorizer = TfidfVectorizer(vocabulary=filtered_words_all)\nvectorizer.fit(abstracts_df['Preprocessed Abstract'])\n","metadata":{"execution":{"iopub.status.busy":"2023-08-17T01:55:36.346125Z","iopub.execute_input":"2023-08-17T01:55:36.346561Z","iopub.status.idle":"2023-08-17T01:59:48.516323Z","shell.execute_reply.started":"2023-08-17T01:55:36.346505Z","shell.execute_reply":"2023-08-17T01:59:48.514641Z"}}},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-08-17T01:59:48.518722Z","iopub.execute_input":"2023-08-17T01:59:48.519982Z","iopub.status.idle":"2023-08-17T02:06:29.617819Z","shell.execute_reply.started":"2023-08-17T01:59:48.519933Z","shell.execute_reply":"2023-08-17T02:06:29.616427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install joblib","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:06:29.621916Z","iopub.execute_input":"2023-08-17T02:06:29.622451Z","iopub.status.idle":"2023-08-17T02:06:45.563073Z","shell.execute_reply.started":"2023-08-17T02:06:29.622403Z","shell.execute_reply":"2023-08-17T02:06:45.561636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib\n","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:09:56.991276Z","iopub.execute_input":"2023-08-17T02:09:56.991901Z","iopub.status.idle":"2023-08-17T02:09:56.997772Z","shell.execute_reply.started":"2023-08-17T02:09:56.991861Z","shell.execute_reply":"2023-08-17T02:09:56.996789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#joblib.dump(vectorizer, '/kaggle/working/vectorizer.pkl')\n","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:10:41.090554Z","iopub.execute_input":"2023-08-17T02:10:41.090981Z","iopub.status.idle":"2023-08-17T02:10:41.280846Z","shell.execute_reply.started":"2023-08-17T02:10:41.090948Z","shell.execute_reply":"2023-08-17T02:10:41.279537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorizer = joblib.load('vectorizer.pkl')","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:40:05.021428Z","iopub.execute_input":"2023-08-17T02:40:05.021971Z","iopub.status.idle":"2023-08-17T02:40:05.137057Z","shell.execute_reply.started":"2023-08-17T02:40:05.021933Z","shell.execute_reply":"2023-08-17T02:40:05.135972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"tfidf_train_vectors = vectorizer.transform(train_preprocessed['Preprocessed Abstract'])\ntfidf_test_vectors = vectorizer.transform(test_preprocessed['Preprocessed Abstract'])\n","metadata":{}},{"cell_type":"markdown","source":"# Convert the TF-IDF vectors to an array\ntrain_vectors = tfidf_train_vectors.toarray()\ntest_vectors = tfidf_test_vectors.toarray()\n\n# Update the abstracts dataframe with the embedding vectors\ntrain_preprocessed['Embedding Vector'] = list(train_vectors)\ntest_preprocessed['Embedding Vector'] = list(test_vectors)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-08-17T00:03:16.914901Z","iopub.execute_input":"2023-08-17T00:03:16.915376Z"}}},{"cell_type":"code","source":"# Save the TF-IDF sparse matrices\n#save_npz('/kaggle/working/tfidf_go_sparse_matrix_train.npz', tfidf_train_vectors)\n#save_npz('/kaggle/working/tfidf_go_sparse_matrix_test.npz', tfidf_test_vectors)\n\n# Save the protein IDs for both dataframes\ntrain_preprocessed['Protein ID'].to_csv('/kaggle/working/protein_ids_train.csv', index=False)\ntest_preprocessed['Protein ID'].to_csv('/kaggle/working/protein_ids_test.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:18:31.73888Z","iopub.execute_input":"2023-08-17T02:18:31.739449Z","iopub.status.idle":"2023-08-17T02:19:15.187008Z","shell.execute_reply.started":"2023-08-17T02:18:31.739407Z","shell.execute_reply":"2023-08-17T02:19:15.185893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ids_np = np.load('/kaggle/input/train-labels-sorted-zong/Ankh_train_labels_sorted.npy')","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:19:15.188812Z","iopub.execute_input":"2023-08-17T02:19:15.189412Z","iopub.status.idle":"2023-08-17T02:19:15.200381Z","shell.execute_reply.started":"2023-08-17T02:19:15.189361Z","shell.execute_reply":"2023-08-17T02:19:15.199279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ids_df = pd.DataFrame(train_ids_np, columns = ['ID'])","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:19:15.20197Z","iopub.execute_input":"2023-08-17T02:19:15.202381Z","iopub.status.idle":"2023-08-17T02:19:15.23752Z","shell.execute_reply.started":"2023-08-17T02:19:15.20235Z","shell.execute_reply":"2023-08-17T02:19:15.236442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ids_df\n","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:30:01.351501Z","iopub.execute_input":"2023-08-17T02:30:01.352022Z","iopub.status.idle":"2023-08-17T02:30:01.373148Z","shell.execute_reply.started":"2023-08-17T02:30:01.351985Z","shell.execute_reply":"2023-08-17T02:30:01.370948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embed_ids = pd.read_csv('/kaggle/input/abstract-embeds-final/protein_ids_train_sparse.csv')","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:30:02.99279Z","iopub.execute_input":"2023-08-17T02:30:02.993229Z","iopub.status.idle":"2023-08-17T02:30:03.13799Z","shell.execute_reply.started":"2023-08-17T02:30:02.993197Z","shell.execute_reply":"2023-08-17T02:30:03.136514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embed_ids","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:30:03.644467Z","iopub.execute_input":"2023-08-17T02:30:03.646885Z","iopub.status.idle":"2023-08-17T02:30:03.660444Z","shell.execute_reply.started":"2023-08-17T02:30:03.646841Z","shell.execute_reply":"2023-08-17T02:30:03.659005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.sparse import load_npz\nsparse_matrix_train = load_npz('/kaggle/input/abstract-embeds-final2/tfidf_go_sparse_matrix_train_try2.npz')\n\ndf_npz = pd.DataFrame.sparse.from_spmatrix(sparse_matrix_train)\ndf_npz['Protein_ID'] = embed_ids['Protein ID'].values\n\n# Convert series_train_protein_ids to DataFrame\ndf_ids = train_ids_df\n\n# Reset the index of df_npz to allow the merge\n#df_npz_reset = df_npz.reset_index()\n\n# Perform the merge with a left join\ndf_train = pd.merge(df_ids, df_npz, how='left', left_on='ID', right_on = 'Protein_ID')\n\n# Fill NaN values with zeros\ndf_train[df_npz.columns] = df_train[df_npz.columns].fillna(0)\n\n# Set 'ID' as the index of the DataFrame again\ndf_train.set_index('Protein_ID', inplace=True)\ndf_train = df_train.drop('ID', axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:30:11.491035Z","iopub.execute_input":"2023-08-17T02:30:11.491504Z","iopub.status.idle":"2023-08-17T02:30:18.465441Z","shell.execute_reply.started":"2023-08-17T02:30:11.491469Z","shell.execute_reply":"2023-08-17T02:30:18.463131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:30:18.466445Z","iopub.status.idle":"2023-08-17T02:30:18.467001Z","shell.execute_reply.started":"2023-08-17T02:30:18.466709Z","shell.execute_reply":"2023-08-17T02:30:18.466739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:30:18.469106Z","iopub.status.idle":"2023-08-17T02:30:18.470512Z","shell.execute_reply.started":"2023-08-17T02:30:18.470171Z","shell.execute_reply":"2023-08-17T02:30:18.470207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Get the feature names (words) from the original TfidfVectorizer\nfeature_names = vectorizer.get_feature_names_out()\n\n# Select the row from df_train that corresponds to the given Protein ID\nprotein_id = 'A0A009IHW8'\nrow = df_train.loc[protein_id]\n\n# Get the TF-IDF values for that row\ntfidf_values = row[:-1]  # Assuming the last column is 'Protein_ID'\n\n# Extract the words that have non-zero TF-IDF values\nnon_zero_indices = tfidf_values[tfidf_values != 0].index\nnon_zero_words = [feature_names[i] for i in non_zero_indices]\n\nprint(f\"Words with non-zero TF-IDF values for Protein ID '{protein_id}': {non_zero_words}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:30:18.471683Z","iopub.status.idle":"2023-08-17T02:30:18.472854Z","shell.execute_reply.started":"2023-08-17T02:30:18.472495Z","shell.execute_reply":"2023-08-17T02:30:18.472525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_non_zero_words(protein_id, df, vectorizer):\n    # Get the feature names (words) from the original TfidfVectorizer\n    feature_names = vectorizer.get_feature_names_out()\n\n    # Select the row from df that corresponds to the given Protein ID\n    row = df.loc[protein_id]\n\n    # Get the TF-IDF values for that row\n    tfidf_values = row[:-1]  # Assuming the last column is 'Protein_ID'\n\n    # Extract the words and their TF-IDF values that have non-zero TF-IDF values\n    non_zero_indices = tfidf_values[tfidf_values != 0].index\n    non_zero_words = [(feature_names[i], tfidf_values[i]) for i in non_zero_indices]\n\n    # Sort the words in descending order of their TF-IDF values\n    sorted_words = sorted(non_zero_words, key=lambda x: x[1], reverse=True)\n\n    return sorted_words\n\n# Example usage:\nprotein_id = 'Q5K4F8'\nsorted_words = get_non_zero_words(protein_id, df_train, vectorizer)","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:30:18.474548Z","iopub.status.idle":"2023-08-17T02:30:18.475568Z","shell.execute_reply.started":"2023-08-17T02:30:18.475303Z","shell.execute_reply":"2023-08-17T02:30:18.475329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test_IDS","metadata":{"execution":{"iopub.status.busy":"2023-08-17T00:45:44.845981Z","iopub.execute_input":"2023-08-17T00:45:44.846608Z","iopub.status.idle":"2023-08-17T00:45:44.852719Z","shell.execute_reply.started":"2023-08-17T00:45:44.846569Z","shell.execute_reply":"2023-08-17T00:45:44.851402Z"}}},{"cell_type":"code","source":"#test_ids = np.load('/kaggle/input/test-ids-sorted-zong/t5_labels_sorted.npy')\n#test_ids_df =  pd.DataFrame(test_ids, columns = ['ID'])","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:30:51.43907Z","iopub.execute_input":"2023-08-17T02:30:51.440336Z","iopub.status.idle":"2023-08-17T02:30:51.547259Z","shell.execute_reply.started":"2023-08-17T02:30:51.440285Z","shell.execute_reply":"2023-08-17T02:30:51.545757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test_ids_df","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:30:51.849671Z","iopub.execute_input":"2023-08-17T02:30:51.850105Z","iopub.status.idle":"2023-08-17T02:30:51.86518Z","shell.execute_reply.started":"2023-08-17T02:30:51.850065Z","shell.execute_reply":"2023-08-17T02:30:51.863949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#embed_ids_test = pd.read_csv('/kaggle/input/abstract-embeds-final/protein_ids_test_sparse.csv')","metadata":{"execution":{"iopub.status.busy":"2023-08-17T01:32:20.472279Z","iopub.execute_input":"2023-08-17T01:32:20.472708Z","iopub.status.idle":"2023-08-17T01:32:20.569125Z","shell.execute_reply.started":"2023-08-17T01:32:20.472677Z","shell.execute_reply":"2023-08-17T01:32:20.567823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"from scipy.sparse import load_npz\nsparse_matrix_test = load_npz('/kaggle/input/abstract-embeds-final2/tfidf_go_sparse_matrix_test_try2.npz')\n\ndf_npz = pd.DataFrame.sparse.from_spmatrix(sparse_matrix_test)\ndf_npz['Protein_ID'] = embed_ids_test['Protein ID'].values\n\n# Convert series_test_protein_ids to DataFrame\ndf_ids = test_ids_df\n\n# Reset the index of df_npz to allow the merge\n#df_npz_reset = df_npz.reset_index()\n\n# Perform the merge with a left join\ndf_test = pd.merge(df_ids, df_npz, how='left', left_on='ID', right_on = 'Protein_ID')\n\n# Fill NaN values with zeros\ndf_test[df_npz.columns] = df_test[df_npz.columns].fillna(0)\n\n#Index 631 and 632 are duplicates. So drop one. Both are protein A0A1D6E0S8 and have the same embedding vector as well.\ndf_test = df_test.drop(632)\n\n# Set 'ID' as the index of the DataFrame again\ndf_test.set_index('Protein_ID', inplace=True)\ndf_test = df_test.drop('ID', axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:31:11.638896Z","iopub.execute_input":"2023-08-17T02:31:11.639372Z","iopub.status.idle":"2023-08-17T02:34:10.905706Z","shell.execute_reply.started":"2023-08-17T02:31:11.639339Z","shell.execute_reply":"2023-08-17T02:34:10.904316Z"}}},{"cell_type":"code","source":"#df_test","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:34:10.909169Z","iopub.execute_input":"2023-08-17T02:34:10.90983Z","iopub.status.idle":"2023-08-17T02:34:11.712979Z","shell.execute_reply.started":"2023-08-17T02:34:10.909775Z","shell.execute_reply":"2023-08-17T02:34:11.711617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df_test","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:34:11.714659Z","iopub.execute_input":"2023-08-17T02:34:11.715047Z","iopub.status.idle":"2023-08-17T02:34:11.906339Z","shell.execute_reply.started":"2023-08-17T02:34:11.715013Z","shell.execute_reply":"2023-08-17T02:34:11.904806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sorted_words = get_non_zero_words('Q60590', df_test, vectorizer)","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:40:13.048861Z","iopub.execute_input":"2023-08-17T02:40:13.049346Z","iopub.status.idle":"2023-08-17T02:40:13.337592Z","shell.execute_reply.started":"2023-08-17T02:40:13.049302Z","shell.execute_reply":"2023-08-17T02:40:13.336045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(sorted_words)","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:43:00.143665Z","iopub.execute_input":"2023-08-17T02:43:00.144156Z","iopub.status.idle":"2023-08-17T02:43:00.152797Z","shell.execute_reply.started":"2023-08-17T02:43:00.144121Z","shell.execute_reply":"2023-08-17T02:43:00.151255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save('/kaggle/working/abstract_embeds_train_sorted_all.npy', df_train.values)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#np.save('/kaggle/working/abstract_embeds_test_sorted_all.npy', df_test.values)","metadata":{"execution":{"iopub.status.busy":"2023-08-17T02:37:35.684253Z","iopub.execute_input":"2023-08-17T02:37:35.684777Z","iopub.status.idle":"2023-08-17T02:37:43.550163Z","shell.execute_reply.started":"2023-08-17T02:37:35.684741Z","shell.execute_reply":"2023-08-17T02:37:43.54749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}