{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ?\n\nDimensional reductions and visualizations for protein emebeddings . \n\nIn first try we do not see something interesting - nor clusters , neither something else. \n","metadata":{}},{"cell_type":"markdown","source":"Versions and outcomes :\n\n    1 T5 embeddings - train part - nothing interesting seen \n    2 same for Test part - similar \n    3 ProtBert from RostLab - first 70 000 samples from train - generated here: \n    https://www.kaggle.com/code/alexandervc/protbert-embedding-starter?scriptVersionId=126811755\n    Again clusters not seen \n    4 ProtT5 embedder  human proteins - from the the authors - \n    Umap shows many small clusters - may be just bad params of umap\n    PCA0,1 shows something like 2-3 clusters but not very clearly \n","metadata":{}},{"cell_type":"markdown","source":"# Preparations and load data","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-23T20:33:26.070267Z","iopub.execute_input":"2023-04-23T20:33:26.070732Z","iopub.status.idle":"2023-04-23T20:33:26.086553Z","shell.execute_reply.started":"2023-04-23T20:33:26.070691Z","shell.execute_reply":"2023-04-23T20:33:26.085201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/t5embeds/train_ids.npy'\nv = np.load(fn)\nprint(v.shape)\nv","metadata":{"execution":{"iopub.status.busy":"2023-04-23T20:33:26.088975Z","iopub.execute_input":"2023-04-23T20:33:26.089422Z","iopub.status.idle":"2023-04-23T20:33:26.107735Z","shell.execute_reply.started":"2023-04-23T20:33:26.089383Z","shell.execute_reply":"2023-04-23T20:33:26.106409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfn = '/kaggle/input/protein-embeddings-1/reduced_embeddings_file.npy'\n# fn = '/kaggle/input/protein-embeddings-1/embed_protbert_train_clip_1200_first_70000_prot.csv'\n# fn = '/kaggle/input/t5embeds/train_embeds.npy'\n# fn = '/kaggle/input/t5embeds/test_embeds.npy'\n\nprint(fn)\nif '.csv' in fn:\n    df = pd.read_csv(fn, index_col = 0)\n    X = df.values\nelif '.npy' in fn:\n    X = np.load(fn)\nprint(X.shape)\nX","metadata":{"execution":{"iopub.status.busy":"2023-04-23T20:33:26.109552Z","iopub.execute_input":"2023-04-23T20:33:26.112748Z","iopub.status.idle":"2023-04-23T20:33:26.193773Z","shell.execute_reply.started":"2023-04-23T20:33:26.11269Z","shell.execute_reply":"2023-04-23T20:33:26.192359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PCA ","metadata":{}},{"cell_type":"code","source":"%%time \n\nimport numpy as np\nfrom sklearn.decomposition import PCA\n\npca = PCA(n_components=10)\nr = pca.fit_transform(X)\n\n# print(pca.explained_variance_ratio_)\n\n# print(pca.singular_values_)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T20:33:26.197088Z","iopub.execute_input":"2023-04-23T20:33:26.198393Z","iopub.status.idle":"2023-04-23T20:33:27.471344Z","shell.execute_reply.started":"2023-04-23T20:33:26.198339Z","shell.execute_reply":"2023-04-23T20:33:27.469921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\ni,j = 0,1\nfor i, j in [(0,1),(0,2),(0,3),(1,2),(1,3),(2,3)]:\n    sns.scatterplot(x = r[:,i], y = r[:,j])\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T20:33:27.472759Z","iopub.execute_input":"2023-04-23T20:33:27.473865Z","iopub.status.idle":"2023-04-23T20:33:28.962262Z","shell.execute_reply.started":"2023-04-23T20:33:27.473824Z","shell.execute_reply":"2023-04-23T20:33:28.960837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# UMAP ","metadata":{}},{"cell_type":"code","source":"%%time \nimport umap\nreducer = umap.UMAP()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-23T20:33:28.964252Z","iopub.execute_input":"2023-04-23T20:33:28.965084Z","iopub.status.idle":"2023-04-23T20:33:28.972978Z","shell.execute_reply.started":"2023-04-23T20:33:28.965032Z","shell.execute_reply":"2023-04-23T20:33:28.971244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \nr = reducer.fit_transform(X)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T20:33:28.9748Z","iopub.execute_input":"2023-04-23T20:33:28.975923Z","iopub.status.idle":"2023-04-23T20:33:46.234386Z","shell.execute_reply.started":"2023-04-23T20:33:28.975869Z","shell.execute_reply":"2023-04-23T20:33:46.233049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, j in [(0,1)]: # ,(0,2),(0,3),(1,2),(1,3),(2,3)]:\n    sns.scatterplot(x = r[:,i], y = r[:,j])\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T20:33:46.237319Z","iopub.execute_input":"2023-04-23T20:33:46.238204Z","iopub.status.idle":"2023-04-23T20:33:46.518551Z","shell.execute_reply.started":"2023-04-23T20:33:46.238141Z","shell.execute_reply":"2023-04-23T20:33:46.517589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}