{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\n!pip install obonet\n!pip install taxoniq","metadata":{"execution":{"iopub.status.busy":"2023-05-03T16:20:12.766239Z","iopub.execute_input":"2023-05-03T16:20:12.767056Z","iopub.status.idle":"2023-05-03T16:21:00.733827Z","shell.execute_reply.started":"2023-05-03T16:20:12.766998Z","shell.execute_reply":"2023-05-03T16:21:00.732173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-03T16:21:00.73716Z","iopub.execute_input":"2023-05-03T16:21:00.73772Z","iopub.status.idle":"2023-05-03T16:21:00.756391Z","shell.execute_reply.started":"2023-05-03T16:21:00.737662Z","shell.execute_reply":"2023-05-03T16:21:00.754573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_file ='/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv'\ndf1=pd.read_csv(data_file,sep='\\t')\ndframe1=pd.DataFrame(df1)\nlist(dframe1)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T16:21:00.758001Z","iopub.execute_input":"2023-05-03T16:21:00.758354Z","iopub.status.idle":"2023-05-03T16:21:04.846885Z","shell.execute_reply.started":"2023-05-03T16:21:00.758322Z","shell.execute_reply":"2023-05-03T16:21:04.845768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(dframe1.info())","metadata":{"execution":{"iopub.status.busy":"2023-05-03T16:21:04.849399Z","iopub.execute_input":"2023-05-03T16:21:04.84983Z","iopub.status.idle":"2023-05-03T16:21:04.878862Z","shell.execute_reply.started":"2023-05-03T16:21:04.849796Z","shell.execute_reply":"2023-05-03T16:21:04.877586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(dframe1.describe())","metadata":{"execution":{"iopub.status.busy":"2023-05-03T16:21:04.880113Z","iopub.execute_input":"2023-05-03T16:21:04.880448Z","iopub.status.idle":"2023-05-03T16:21:06.969938Z","shell.execute_reply.started":"2023-05-03T16:21:04.880411Z","shell.execute_reply":"2023-05-03T16:21:06.968369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set_res = set(dframe1['term']) \nprint(\"The unique elements of the input list using set():\\n\") \nlist_res = (list(set_res))\n\nprint ('Unique values---->>> ', len(list_res), '  <<<----------' )","metadata":{"execution":{"iopub.status.busy":"2023-05-03T16:21:06.972067Z","iopub.execute_input":"2023-05-03T16:21:06.973574Z","iopub.status.idle":"2023-05-03T16:21:07.646348Z","shell.execute_reply.started":"2023-05-03T16:21:06.973339Z","shell.execute_reply":"2023-05-03T16:21:07.645117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import networkx\nimport obonet\n\nobofile ='/kaggle/input/cafa-5-protein-function-prediction/Train/go-basic.obo'\ngraph = obonet.read_obo(obofile)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T17:10:28.751845Z","iopub.execute_input":"2023-05-03T17:10:28.752273Z","iopub.status.idle":"2023-05-03T17:10:37.627258Z","shell.execute_reply.started":"2023-05-03T17:10:28.752236Z","shell.execute_reply":"2023-05-03T17:10:37.625988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"One way to access the elements of a dictionary, but is not good enough","metadata":{}},{"cell_type":"code","source":"# Number of nodes\nnodes = len(graph)\nprint(\"Number of nodes = \",nodes)\n#print(graph.nodes)\n\n# Check if the ontology is a DAG\n# Number of edges\ngraph.number_of_edges()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T16:21:19.066628Z","iopub.execute_input":"2023-05-03T16:21:19.067486Z","iopub.status.idle":"2023-05-03T16:21:19.216602Z","shell.execute_reply.started":"2023-05-03T16:21:19.067397Z","shell.execute_reply":"2023-05-03T16:21:19.215163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*I think it is better to work with * \n**Lists**","metadata":{}},{"cell_type":"code","source":"keys =list()\ni=0\nvalues =[]\nfor key in graph:\n    #print(key) this conserve the dictionary structure, but we can make it strings \n    keys.append(key)\n    print(keys[i]) ##uncomment to see the results\n    values.append([])\n    j=0\n    for value in graph[key]:\n        #print(value) this also conserve the dictionary structure, but we can make it strings\n        values[i].append (value)\n        print (\"----> \", values[i][j]) #uncomment to see the results\n        j += 1\n    i += 1\n#print(keys)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T18:12:49.160936Z","iopub.execute_input":"2023-05-03T18:12:49.161404Z","iopub.status.idle":"2023-05-03T18:12:51.042161Z","shell.execute_reply.started":"2023-05-03T18:12:49.161344Z","shell.execute_reply":"2023-05-03T18:12:51.04084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"# Reading the Protein's aminoacid sequence, fasta file\nfrom Bio import SeqIO\n\nfasta_sequences = SeqIO.parse(open(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta\"),'fasta')\nfor fasta in fasta_sequences:\n    name, sequence = fasta.id, str(fasta.seq)\n    print (\"Protein ID --->\", name)","metadata":{}},{"cell_type":"code","source":"# Reading the Protein's aminoacid sequence, fasta file\nfrom Bio import SeqIO\n\nfasta_sequences = SeqIO.parse(open(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta\"),'fasta')\nfor fasta in fasta_sequences:\n    name, sequence = fasta.id, str(fasta.seq)\n    print (\"Protein ID --->\", name)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T17:59:49.037895Z","iopub.execute_input":"2023-05-03T17:59:49.039074Z","iopub.status.idle":"2023-05-03T17:59:53.291683Z","shell.execute_reply.started":"2023-05-03T17:59:49.039011Z","shell.execute_reply":"2023-05-03T17:59:53.290442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i =0 \nwhile i < nodes - 1: # Source node\n    j= i+ 1\n    while j < nodes:# Target node\n#        values\n        k = 0\n        while k < len (values[i]):\n            l=0 \n            while l < len (values[j]):\n            #using if doSomething_GA with values [i][k] and values [j][l]\n                l +=1\n            k += 1\n        j + =1\n    i +=1\n    ","metadata":{"execution":{"iopub.status.busy":"2023-05-03T16:21:19.220994Z","iopub.execute_input":"2023-05-03T16:21:19.22146Z","iopub.status.idle":"2023-05-03T16:21:19.36599Z","shell.execute_reply.started":"2023-05-03T16:21:19.221407Z","shell.execute_reply":"2023-05-03T16:21:19.364204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n \n# Checks if s1 is a substring of s2\n# Python3 program to check if\n# a string is substring of other.\n \n \ndef isSubstring(s1, s2):\n    if s1 in s2:\n        return s2.index(s1)\n    return -1\n \n \n# Driver Code\nif __name__ == \"__main__\":\n    s1 = \"for\"\n    s2 = \"geeksforgeeks\"\n    res = isSubstring(s1, s2)\n    if res == -1:\n        print(\"Not present\")\n    else:\n        print(\"Present at index \" + str(res))\n \n# This code is contributed by phasing17","metadata":{"execution":{"iopub.status.busy":"2023-05-03T16:21:19.586326Z","iopub.status.idle":"2023-05-03T16:21:19.586735Z","shell.execute_reply.started":"2023-05-03T16:21:19.586539Z","shell.execute_reply":"2023-05-03T16:21:19.586561Z"},"trusted":true},"execution_count":null,"outputs":[]}]}