{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Published on April 18, 2023. By Marília Prata. mpwolke","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-18T22:27:08.076262Z","iopub.execute_input":"2023-04-18T22:27:08.077784Z","iopub.status.idle":"2023-04-18T22:27:09.461151Z","shell.execute_reply.started":"2023-04-18T22:27:08.077726Z","shell.execute_reply":"2023-04-18T22:27:09.459735Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Bioinformatics\n\nBioinformatics is defined as the application of tools of computation and analysis to the capture and interpretation of biological data.","metadata":{}},{"cell_type":"markdown","source":"![](https://omgenomics.com/assets/bioinformatics-data-science-venn-diagrams.png)https://omgenomics.com/what-is-bioinformatics/","metadata":{}},{"cell_type":"markdown","source":"#Fasta file\n\nThat's not fast. It simply stopped my Notebook","metadata":{}},{"cell_type":"code","source":"#https://stackoverflow.com/questions/29805642/learning-to-parse-a-fasta-file-with-python\n\ndef read_fasta(fp):\n        name, seq = None, []\n        for line in fp:\n            line = line.rstrip()\n            if line.startswith(\">\"):\n                if name: yield (name, ''.join(seq))\n                name, seq = line, []\n            else:\n                seq.append(line)\n        if name: yield (name, ''.join(seq))\n\nwith open('../input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta') as fp:\n    for name, seq in read_fasta(fp):\n        print(name, seq)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T22:49:04.050144Z","iopub.execute_input":"2023-04-18T22:49:04.050627Z"},"_kg_hide-output":true,"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#IA txt file","metadata":{}},{"cell_type":"code","source":"path_of_file = '../input/cafa-5-protein-function-prediction/IA.txt'\ntext = open(path_of_file, 'r').read()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T22:44:04.205231Z","iopub.execute_input":"2023-04-18T22:44:04.205888Z","iopub.status.idle":"2023-04-18T22:44:04.227917Z","shell.execute_reply.started":"2023-04-18T22:44:04.205839Z","shell.execute_reply":"2023-04-18T22:44:04.226668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text[:3000]","metadata":{"execution":{"iopub.status.busy":"2023-04-18T22:44:20.639869Z","iopub.execute_input":"2023-04-18T22:44:20.640901Z","iopub.status.idle":"2023-04-18T22:44:20.648702Z","shell.execute_reply.started":"2023-04-18T22:44:20.640846Z","shell.execute_reply":"2023-04-18T22:44:20.647271Z"},"_kg_hide-output":true,"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/cafa-5-protein-function-prediction/Train/train_taxonomy.tsv', sep='\\t', error_bad_lines=False)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T22:29:36.47519Z","iopub.execute_input":"2023-04-18T22:29:36.475676Z","iopub.status.idle":"2023-04-18T22:29:36.63566Z","shell.execute_reply.started":"2023-04-18T22:29:36.475636Z","shell.execute_reply":"2023-04-18T22:29:36.634471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#To open testsuperset-taxon-list TSV file\n\nAdd encoding= 'unicode_escape'","metadata":{}},{"cell_type":"code","source":"lis = pd.read_csv('../input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset-taxon-list.tsv', sep='\\t', error_bad_lines=False, encoding= 'unicode_escape')\nlis.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T23:39:35.245506Z","iopub.execute_input":"2023-04-18T23:39:35.245999Z","iopub.status.idle":"2023-04-18T23:39:35.26708Z","shell.execute_reply.started":"2023-04-18T23:39:35.245955Z","shell.execute_reply":"2023-04-18T23:39:35.265654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/cafa-5-protein-function-prediction/sample_submission.tsv', sep='\\t', error_bad_lines=False)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T22:32:32.310848Z","iopub.execute_input":"2023-04-18T22:32:32.311424Z","iopub.status.idle":"2023-04-18T22:32:32.591922Z","shell.execute_reply.started":"2023-04-18T22:32:32.311365Z","shell.execute_reply":"2023-04-18T22:32:32.590696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ter = pd.read_csv('../input/cafa-5-protein-function-prediction/Train/train_terms.tsv', sep='\\t', error_bad_lines=False )\nter.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T22:37:29.761701Z","iopub.execute_input":"2023-04-18T22:37:29.762154Z","iopub.status.idle":"2023-04-18T22:37:33.396164Z","shell.execute_reply.started":"2023-04-18T22:37:29.762117Z","shell.execute_reply":"2023-04-18T22:37:33.394848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install goatools","metadata":{"execution":{"iopub.status.busy":"2023-04-18T23:08:16.436453Z","iopub.execute_input":"2023-04-18T23:08:16.437679Z","iopub.status.idle":"2023-04-18T23:08:51.610457Z","shell.execute_reply.started":"2023-04-18T23:08:16.437624Z","shell.execute_reply":"2023-04-18T23:08:51.608759Z"},"_kg_hide-output":true,"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import the OBO parser from GOATools\nfrom goatools import obo_parser","metadata":{"execution":{"iopub.status.busy":"2023-04-18T23:09:35.859253Z","iopub.execute_input":"2023-04-18T23:09:35.859797Z","iopub.status.idle":"2023-04-18T23:09:36.145636Z","shell.execute_reply.started":"2023-04-18T23:09:35.859738Z","shell.execute_reply":"2023-04-18T23:09:36.144682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install wget","metadata":{"execution":{"iopub.status.busy":"2023-04-18T23:09:53.779133Z","iopub.execute_input":"2023-04-18T23:09:53.780247Z","iopub.status.idle":"2023-04-18T23:10:07.307349Z","shell.execute_reply.started":"2023-04-18T23:09:53.780182Z","shell.execute_reply":"2023-04-18T23:10:07.305131Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import wget","metadata":{"execution":{"iopub.status.busy":"2023-04-18T23:10:48.50209Z","iopub.execute_input":"2023-04-18T23:10:48.502646Z","iopub.status.idle":"2023-04-18T23:10:48.514836Z","shell.execute_reply.started":"2023-04-18T23:10:48.502579Z","shell.execute_reply":"2023-04-18T23:10:48.513539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Alexander Chervov https://www.kaggle.com/code/alexandervc/gene-ontology-python-tutorial\n\nobo_path = \"../input/cafa-5-protein-function-prediction/Train/go-basic.obo\"\n\n#go_obo_url = 'http://purl.obolibrary.org/obo/go/go-basic.obo'\ndata_folder = os.getcwd() + '/data'\n\n# Check if we have the ./data directory already\nif(not os.path.isfile(data_folder)):\n    # Emulate mkdir -p (no error if folder exists)\n    try:\n        os.mkdir(data_folder)\n    except OSError as e:\n        if(e.errno != 17):\n            raise e\nelse:\n    raise Exception('Data path (' + data_folder + ') exists as a file. '\n                   'Please rename, remove or change the desired location of the data path.')\n\n# Check if the file exists already\nif(not os.path.isfile(data_folder+'/go-basic.obo')):\n    go_obo = wget.download(go_obo_url, data_folder+'/go-basic.obo')\nelse:\n    go_obo = data_folder+'/go-basic.obo'","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Andrey Shtrauss ttps://www.kaggle.com/code/shtrausslearning/biopython-bioinformatics-basics\n\nfrom Bio.Seq import Seq\nfrom Bio import SeqIO, SearchIO\nfrom Bio.SeqRecord import SeqRecord\n\nn = 'ATGACGGATCAGCCGCAAGCGGAATTGGCGTTTACGTACGATGCGCCGTAA'  # nucleotide sequence\naa = 'MMMELQHQRLMALAGQLQLESLISAAPALSQQAVDQEWSYMDFLEHLLHE' # protein sequence\n\nseq_n = Seq(n)\nseq_aa = Seq(aa)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T23:23:28.411895Z","iopub.execute_input":"2023-04-18T23:23:28.412393Z","iopub.status.idle":"2023-04-18T23:23:28.418798Z","shell.execute_reply.started":"2023-04-18T23:23:28.41235Z","shell.execute_reply":"2023-04-18T23:23:28.417716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Andrey Shtrauss ttps://www.kaggle.com/code/shtrausslearning/biopython-bioinformatics-basics\n\nprint(seq_n.reverse_complement()) # possible\nprint(seq_aa.reverse_complement()) # not actually possible","metadata":{"execution":{"iopub.status.busy":"2023-04-18T23:23:52.704885Z","iopub.execute_input":"2023-04-18T23:23:52.705341Z","iopub.status.idle":"2023-04-18T23:23:52.711897Z","shell.execute_reply.started":"2023-04-18T23:23:52.705298Z","shell.execute_reply":"2023-04-18T23:23:52.710479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I didn't run that cause I got an error that I couldn't fix\nprint('FASTA Annotations:')\nread_seq1 = SeqIO.read('../input/cafa-5-protein-function-prediction/Train/train_sequences.fasta','fasta')\nprint(read_seq1.annotations)  # annotations","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#I will learn a lot in this competition. Even to load the files\n\n![](https://pbs.twimg.com/media/Ew-RMIfXEAIvHfr.jpg)https://twitter.com/wetlabsucks\n\n#Maybe I'll ask Chervov and Shtrauss some tips : )\n\nAfter 1h.22, that's all I could make.","metadata":{}},{"cell_type":"markdown","source":"#Acknowledgements:\n\nAlexander Chervov https://www.kaggle.com/code/alexandervc/gene-ontology-python-tutorial\n\nAndrey Shtrauss ttps://www.kaggle.com/code/shtrausslearning/biopython-bioinformatics-basics","metadata":{}}]}