{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# numpy for numerical computing\nimport numpy as np\n\n# pandas for DataFrames\nimport pandas as pd\n\n# plotly for visualization of the reduced embeddings\nimport plotly.express as px\n\n# to calculate the PCAs \nfrom sklearn.decomposition import PCA\n\n# SVD for dimensionality reduction\nfrom sklearn.decomposition import TruncatedSVD\n\n# for K means clustering\nfrom sklearn.cluster import KMeans","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# path to the t5 embeddings\ntrain_embeddings_path = '/kaggle/input/t5embeds/train_embeds.npy'\ntest_embeddings_path = '/kaggle/input/t5embeds/test_embeds.npy'\ntrain_ids_path = '/kaggle/input/t5embeds/train_ids.npy'\ntest_ids_path = '/kaggle/input/t5embeds/test_ids.npy'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read the ids and embeddings into numpy arrays\ntrain_ids = np.load(train_ids_path)\ntest_ids = np.load(test_ids_path)\ntrain_embeddings = np.load(train_embeddings_path)\ntest_embeddings = np.load(test_embeddings_path)\ntrain_ids.shape, test_ids.shape, train_embeddings.shape, test_embeddings.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we want to reduce the embeddings to 2 dimensions for visualization\npca = PCA(n_components=2)\ntrain_embeddings_reduced = pca.fit_transform(train_embeddings)\ntest_embeddings_reduced = pca.transform(test_embeddings)\n# shape\ntrain_embeddings_reduced.shape, test_embeddings_reduced.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualize the reduced embeddings\n# label train and test to distinguish them\ntrain_labels = np.repeat('train', train_embeddings_reduced.shape[0])\ntest_labels = np.repeat('test', test_embeddings_reduced.shape[0])\n# combine the embeddings and labels\nembeddings = np.concatenate([train_embeddings_reduced, test_embeddings_reduced])\nlabels = np.concatenate([train_labels, test_labels])\n# create a DataFrame\ndf = pd.DataFrame(embeddings, columns=['x', 'y'])\ndf['labels'] = labels\n# plot\nfig = px.scatter(df, x='x', y='y', color='labels', title='T5 Embeddings Reduced to 2 Dimensions', opacity=0.2)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we have the GO annotations for the training set\n# read the annotations into a DataFrame\ntrain_annotations = pd.read_csv('/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv', sep='\\t')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# accumalate for each EntryID the GO terms in a list\ntrain_annotations = train_annotations.groupby('EntryID')['term'].apply(list).reset_index()\ntrain_annotations.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# count the number of GO terms for each EntryID\ntrain_annotations['num_terms'] = train_annotations['term'].apply(lambda x: len(x))\ntrain_annotations.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make a dictonary using EntryID and num_terms\ntrain_annotations_dict = dict(zip(train_annotations['EntryID'], train_annotations['num_terms']))\n# map the dictonary to the train_ids and use 0 as default value for train -1 for test\n# this will be used to color the plot\ntrain_labels = np.array([train_annotations_dict.get(x, 0) for x in train_ids])\ntest_labels = np.repeat(-1, test_ids.shape[0])\n# combine the embeddings and labels\nembeddings = np.concatenate([train_embeddings_reduced, test_embeddings_reduced])\nlabels = np.concatenate([train_labels, test_labels])\n# create a DataFrame\ndf = pd.DataFrame(embeddings, columns=['x', 'y'])\ndf['labels'] = labels\n# plot, use a rainbow color scheme\nfig = px.scatter(df, x='x', y='y', color='labels', title='T5 Embeddings Reduced to 2 Dimensions', opacity=0.2, color_continuous_scale='rainbow')\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# determine the most frequent GO terms\n# flatten the list of lists\nterms = [item for sublist in train_annotations['term'].values for item in sublist]\n# count the number of occurences for each GO term\nterm_counts = pd.Series(terms).value_counts()\n# make a boolean column for the annotations and see if the term is in the top 1000\ntrain_annotations['top_1000'] = train_annotations['term'].apply(lambda x: any([term in x for term in term_counts.index[:1000]]))\ntrain_annotations.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use the top 1000 terms to color the plot\n# make a dictonary using EntryID and top_1000\ntrain_annotations_dict = dict(zip(train_annotations['EntryID'], train_annotations['top_1000']))\n# map the dictonary, use -1 as default value\ntrain_labels = np.array([train_annotations_dict.get(x, -1) for x in train_ids])\ntest_labels = np.repeat(-1, test_ids.shape[0])\n# combine the embeddings and labels\nembeddings = np.concatenate([train_embeddings_reduced, test_embeddings_reduced])\nlabels = np.concatenate([train_labels, test_labels])\n# create a DataFrame\ndf = pd.DataFrame(embeddings, columns=['x', 'y'])\ndf['labels'] = labels\n# plot, use different color for each label\nfig = px.scatter(df, x='x', y='y', color='labels', title='T5 Embeddings Reduced to 2 Dimensions', opacity=0.1, color_discrete_sequence=px.colors.qualitative.Pastel)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use SVD to reduce the embeddings to 2 dimensions\nsvd = TruncatedSVD(n_components=2)\ntrain_embeddings_reduced = svd.fit_transform(train_embeddings)\ntest_embeddings_reduced = svd.transform(test_embeddings)\n# shape\ntrain_embeddings_reduced.shape, test_embeddings_reduced.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualize the reduced embeddings, labelling which is in train and which in test as above\ntrain_labels = np.repeat('train', train_embeddings_reduced.shape[0])\ntest_labels = np.repeat('test', test_embeddings_reduced.shape[0])\nembeddings = np.concatenate([train_embeddings_reduced, test_embeddings_reduced])\nlabels = np.concatenate([train_labels, test_labels])\ndf = pd.DataFrame(embeddings, columns=['x', 'y'])\ndf['labels'] = labels\nfig = px.scatter(df, x='x', y='y', color='labels', title='T5 Embeddings Reduced to 2 Dimensions', opacity=0.2)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reduce to 2 dimensions with PCA, cluster the reduced embeddings with k-means and color the plot with clusters as labels\npca = PCA(n_components=2)\ntrain_embeddings_reduced = pca.fit_transform(train_embeddings)\ntest_embeddings_reduced = pca.transform(test_embeddings)\n# cluster the reduced embeddings with k-means\nkmeans = KMeans(n_init=10, n_clusters=10, random_state=42)\ntrain_clusters = kmeans.fit_predict(train_embeddings_reduced)\ntest_clusters = kmeans.predict(test_embeddings_reduced)\n# combine the embeddings and labels\nembeddings = np.concatenate([train_embeddings_reduced, test_embeddings_reduced])\nlabels = np.concatenate([train_clusters, test_clusters])\n# create a DataFrame\ndf = pd.DataFrame(embeddings, columns=['x', 'y'])\ndf['labels'] = labels\n# plot\nfig = px.scatter(df, x='x', y='y', color='labels', title='T5 Embeddings Reduced to 2 Dimensions', opacity=0.2)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# count how many entries of train and test are in each cluster\ntrain_clusters_counts = pd.Series(train_clusters).value_counts()\ntest_clusters_counts = pd.Series(test_clusters).value_counts()\n# visualize the distribution of train and test entries in the clusters seperately\n# make a DataFrame\ndf = pd.DataFrame({'train': train_clusters_counts, 'test': test_clusters_counts})\n# plot\nfig = px.bar(df, x=df.index, y=['train', 'test'], title='Distribution of Train and Test Entries in the Clusters', barmode='group')\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# it seems that the clusters are not very useful for seperating train and test after reduction by PCA","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# same as above but with SVD\nsvd = TruncatedSVD(n_components=2)\ntrain_embeddings_reduced = svd.fit_transform(train_embeddings)\ntest_embeddings_reduced = svd.transform(test_embeddings)\n# cluster the reduced embeddings with k-means\nkmeans = KMeans(n_init=10, n_clusters=10, random_state=42)\ntrain_clusters = kmeans.fit_predict(train_embeddings_reduced)\ntest_clusters = kmeans.predict(test_embeddings_reduced)\n# combine the embeddings and labels\nembeddings = np.concatenate([train_embeddings_reduced, test_embeddings_reduced])\nlabels = np.concatenate([train_clusters, test_clusters])\n# create a DataFrame\ndf = pd.DataFrame(embeddings, columns=['x', 'y'])\ndf['labels'] = labels\n# plot\nfig = px.scatter(df, x='x', y='y', color='labels', title='T5 Embeddings Reduced to 2 Dimensions', opacity=0.2)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# count how many entries of train and test are in each cluster\ntrain_clusters_counts = pd.Series(train_clusters).value_counts()\ntest_clusters_counts = pd.Series(test_clusters).value_counts()\n# visualize the distribution of train and test entries in the clusters seperately\n# make a DataFrame\ndf = pd.DataFrame({'train': train_clusters_counts, 'test': test_clusters_counts})\n# plot\nfig = px.bar(df, x=df.index, y=['train', 'test'], title='Distribution of Train and Test Entries in the Clusters', barmode='group')\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# similar as for pca we see that the clusters are not very useful for seperating train and test after reduction by SVD","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make a PCA reduction to 3 dimensions\npca = PCA(n_components=3)\ntrain_embeddings_reduced = pca.fit_transform(train_embeddings)\ntest_embeddings_reduced = pca.transform(test_embeddings)\n# shape\ntrain_embeddings_reduced.shape, test_embeddings_reduced.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# label with train and test and plot\ntrain_labels = np.repeat('train', train_embeddings_reduced.shape[0])\ntest_labels = np.repeat('test', test_embeddings_reduced.shape[0])\nembeddings = np.concatenate([train_embeddings_reduced, test_embeddings_reduced])\nlabels = np.concatenate([train_labels, test_labels])\ndf = pd.DataFrame(embeddings, columns=['x', 'y', 'z'])\ndf['labels'] = labels\nfig = px.scatter_3d(df, x='x', y='y', z='z', color='labels', title='T5 Embeddings Reduced to 3 Dimensions', opacity=0.2)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}