{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-23T15:01:17.775442Z","iopub.execute_input":"2023-04-23T15:01:17.775717Z","iopub.status.idle":"2023-04-23T15:01:17.820128Z","shell.execute_reply.started":"2023-04-23T15:01:17.775688Z","shell.execute_reply":"2023-04-23T15:01:17.819043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/uniprot-description-only/uniprot_description_only.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-23T15:01:17.822069Z","iopub.execute_input":"2023-04-23T15:01:17.82289Z","iopub.status.idle":"2023-04-23T15:01:19.393412Z","shell.execute_reply.started":"2023-04-23T15:01:17.82285Z","shell.execute_reply":"2023-04-23T15:01:19.39225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T15:01:19.399707Z","iopub.execute_input":"2023-04-23T15:01:19.402473Z","iopub.status.idle":"2023-04-23T15:01:19.426829Z","shell.execute_reply.started":"2023-04-23T15:01:19.402424Z","shell.execute_reply":"2023-04-23T15:01:19.425726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-23T15:01:19.432916Z","iopub.execute_input":"2023-04-23T15:01:19.435494Z","iopub.status.idle":"2023-04-23T15:01:19.447329Z","shell.execute_reply.started":"2023-04-23T15:01:19.43545Z","shell.execute_reply":"2023-04-23T15:01:19.445965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"description\"].replace('', np.nan, inplace=True)\n\ndf[\"description\"].replace('nan', np.nan, inplace=True)\n\ndf = df[df[\"description\"].notna()]\n\ndf = df[df[\"description\"].notnull()]","metadata":{"execution":{"iopub.status.busy":"2023-04-23T15:01:19.452125Z","iopub.execute_input":"2023-04-23T15:01:19.45247Z","iopub.status.idle":"2023-04-23T15:01:19.561783Z","shell.execute_reply.started":"2023-04-23T15:01:19.452441Z","shell.execute_reply":"2023-04-23T15:01:19.560509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-23T15:01:19.567263Z","iopub.execute_input":"2023-04-23T15:01:19.569092Z","iopub.status.idle":"2023-04-23T15:01:19.581281Z","shell.execute_reply.started":"2023-04-23T15:01:19.569045Z","shell.execute_reply":"2023-04-23T15:01:19.580019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T15:01:19.585944Z","iopub.execute_input":"2023-04-23T15:01:19.586738Z","iopub.status.idle":"2023-04-23T15:01:19.604674Z","shell.execute_reply.started":"2023-04-23T15:01:19.586697Z","shell.execute_reply":"2023-04-23T15:01:19.603444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n!pip install Cython\n!pip install umap-learn\n!pip install --upgrade git+https://github.com/scikit-learn-contrib/hdbscan.git","metadata":{"execution":{"iopub.status.busy":"2023-04-23T15:01:19.606171Z","iopub.execute_input":"2023-04-23T15:01:19.611994Z","iopub.status.idle":"2023-04-23T15:03:04.63782Z","shell.execute_reply.started":"2023-04-23T15:01:19.611948Z","shell.execute_reply":"2023-04-23T15:03:04.636519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-04-23T15:03:04.63945Z","iopub.execute_input":"2023-04-23T15:03:04.643171Z","iopub.status.idle":"2023-04-23T15:03:04.649006Z","shell.execute_reply.started":"2023-04-23T15:03:04.643135Z","shell.execute_reply":"2023-04-23T15:03:04.647936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corpus = df['description'].tolist()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T15:03:04.65614Z","iopub.execute_input":"2023-04-23T15:03:04.656444Z","iopub.status.idle":"2023-04-23T15:03:04.684323Z","shell.execute_reply.started":"2023-04-23T15:03:04.656417Z","shell.execute_reply":"2023-04-23T15:03:04.683311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n!pip install sentence_transformers","metadata":{"execution":{"iopub.status.busy":"2023-04-23T15:03:04.685938Z","iopub.execute_input":"2023-04-23T15:03:04.686333Z","iopub.status.idle":"2023-04-23T15:03:15.999865Z","shell.execute_reply.started":"2023-04-23T15:03:04.686294Z","shell.execute_reply":"2023-04-23T15:03:15.998584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sentence_transformers import SentenceTransformer, util","metadata":{"execution":{"iopub.status.busy":"2023-04-23T15:03:16.002659Z","iopub.execute_input":"2023-04-23T15:03:16.003442Z","iopub.status.idle":"2023-04-23T15:03:20.719627Z","shell.execute_reply.started":"2023-04-23T15:03:16.003395Z","shell.execute_reply":"2023-04-23T15:03:20.718555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = SentenceTransformer('all-distilroberta-v1')","metadata":{"execution":{"iopub.status.busy":"2023-04-23T15:03:20.721163Z","iopub.execute_input":"2023-04-23T15:03:20.722088Z","iopub.status.idle":"2023-04-23T15:03:29.334213Z","shell.execute_reply.started":"2023-04-23T15:03:20.722047Z","shell.execute_reply":"2023-04-23T15:03:29.332875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nembeddings = model.encode(corpus, show_progress_bar=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T15:03:29.339346Z","iopub.execute_input":"2023-04-23T15:03:29.341726Z","iopub.status.idle":"2023-04-23T15:08:14.727786Z","shell.execute_reply.started":"2023-04-23T15:03:29.341679Z","shell.execute_reply":"2023-04-23T15:08:14.726553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import umap.umap_ as umap","metadata":{"execution":{"iopub.status.busy":"2023-04-23T15:09:20.68371Z","iopub.execute_input":"2023-04-23T15:09:20.684444Z","iopub.status.idle":"2023-04-23T15:09:39.052703Z","shell.execute_reply.started":"2023-04-23T15:09:20.684406Z","shell.execute_reply":"2023-04-23T15:09:39.051585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\numap_embedding = umap.UMAP(\n                                 \n    n_neighbors=50,\n    min_dist=0.0,\n    n_components=2,\n    random_state=23,\n    repulsion_strength=1.0,).fit_transform(embeddings)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T15:12:05.023124Z","iopub.execute_input":"2023-04-23T15:12:05.024018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import hdbscan","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nlabels = hdbscan.HDBSCAN(\nmin_samples=50,\nmin_cluster_size=50,\n).fit_predict(umap_embedding)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from bokeh.plotting import figure, show, output_notebook\nfrom bokeh.models import HoverTool, ColumnDataSource\nfrom bokeh.palettes import Spectral10\nfrom bokeh.models import LinearColorMapper\nfrom bokeh.models import ColorBar\n\noutput_notebook()\n\n%matplotlib inline","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = pd.DataFrame(umap_embedding, columns=('x', 'y'))\ntemp['group'] = [int(x) for x in labels]\ntemp['topic'] = df.gene_value\n\ngroup = list(temp['group'].unique())\n\n\ndatasource = ColumnDataSource(temp)\n\nyfig = figure(\n    plot_width=600,\n    plot_height=600,\n    tools=('pan, wheel_zoom, reset')\n)\n\nyfig.add_tools(HoverTool(tooltips=\"\"\"\n<div>\n    <div>\n        <span style='font-size: 16px; color: #224499'>Group: </span>\n        <span style='font-size: 18px'>@group</span>\n    </div>\n    <div>\n        <span style='font-size: 16px; color: #224499'>Topic: </span>\n        <span style='font-size: 18px'>@topic</span>\n    </div>\n</div>\n\"\"\"))\n\nexp_cmap = LinearColorMapper(palette='Turbo256', low=min(group), high=max(group))\n    \nyfig.circle(\n    'x',\n    'y',\n    source=datasource,\n    color={'field': 'group', 'transform': exp_cmap},\n    line_alpha=0.6,\n    fill_alpha=0.6,\n    size=4\n)\n\nbar = ColorBar(color_mapper=exp_cmap, location=(0,0))\nyfig.add_layout(bar, \"left\")\n\nshow(yfig)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_corpus = pd.DataFrame(corpus, columns = ['description'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clustered_list = pd.DataFrame(np.column_stack([df_corpus, labels]), \n                               columns=['sentence', 'cluster'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"key_list = list(clustered_list['sentence'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict_lookup = dict(zip(df['description'], df['gene_value']))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clustered_list['gene_value'] = [dict_lookup[item] for item in key_list]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clustered_list.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clustered_list['cluster'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clustered_list_filter = clustered_list.loc[clustered_list['gene_value'].str.contains('CD44')]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clustered_list_filter_word","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cluster_worldcloud(cluster_num):\n    comment_words = ''\n    stopwords = set(STOPWORDS)\n\n \n    for val in clustered_list[clustered_list['cluster']==cluster_num].sentence.values:\n      \n      \n        val = str(val)\n\n      \n        tokens = val.split()\n      \n      \n        for i in range(len(tokens)):\n            tokens[i] = tokens[i].lower()\n      \n        comment_words += \" \".join(tokens)+\" \"\n\n    wordcloud = WordCloud( background_color = 'white',\n                          stopwords = stopwords, \n                          width = 2048, height = 1080).generate(comment_words)\n\n\n                     \n    plt.figure(figsize= (20,10) )\n    plt.imshow(wordcloud, interpolation='bilinear')\n    plt.axis(\"off\")\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from wordcloud import WordCloud, STOPWORDS\nimport seaborn as sns\nimport nltk\nimport nltk\nnltk.download('stopwords')\nfrom nltk.corpus import stopwords\nimport matplotlib.pyplot as plt\n%matplotlib inline\nsns.set()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cluster_worldcloud(4845)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_emb = pd.DataFrame({'description': corpus, 'embedding': None})\ndf_emb['embedding'] = embeddings.tolist()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_emb_columns = pd.DataFrame(df_emb['embedding'].values.tolist())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_emb_new = pd.concat([df_emb, df_emb_columns], axis=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"key_list = list(df_emb_new['description'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict_lookup = dict(zip(df['description'], df['gene_value']))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_emb_new['gene_value'] = [dict_lookup[item] for item in key_list]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_emb_new[\"gene_value\"].replace('', np.nan, inplace=True)\n\ndf_emb_new[\"gene_value\"].replace('nan', np.nan, inplace=True)\n\ndf_emb_new = df_emb_new[df_emb_new[\"gene_value\"].notna()]\n\ndf_emb_new = df_emb_new[df_emb_new[\"gene_value\"].notnull()]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_emb_new = df_emb_new.drop(columns=['description', 'embedding'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = df_emb_new.columns.tolist()\n\ncols = cols[-1:] + cols[:-1]\n\ndf_emb_new = df_emb_new[cols]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_emb_new.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_emb_new.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_emb_new.to_csv('./sentence_transformers_embeddings.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}