{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport time\nt0start = time.time()\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-10T18:42:27.759873Z","iopub.execute_input":"2023-06-10T18:42:27.760403Z","iopub.status.idle":"2023-06-10T18:42:28.582292Z","shell.execute_reply.started":"2023-06-10T18:42:27.760351Z","shell.execute_reply":"2023-06-10T18:42:28.580592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CD-hit ","metadata":{}},{"cell_type":"markdown","source":"## Installation and command line use \n\nThanks to Arthur Zalevsky ! \nWe will download sources and compile","metadata":{}},{"cell_type":"code","source":"!wget https://github.com/weizhongli/cdhit/releases/download/V4.8.1/cd-hit-v4.8.1-2019-0228.tar.gz\n!tar -xf cd-hit-v4.8.1-2019-0228.tar.gz\n\nos.chdir(\"/kaggle/working/cd-hit-v4.8.1-2019-0228\")\n!make -j 4\nos.chdir(\"/kaggle/working\")","metadata":{"execution":{"iopub.status.busy":"2023-06-10T18:43:03.31204Z","iopub.execute_input":"2023-06-10T18:43:03.312719Z","iopub.status.idle":"2023-06-10T18:43:13.894997Z","shell.execute_reply.started":"2023-06-10T18:43:03.312618Z","shell.execute_reply":"2023-06-10T18:43:13.89361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check launch - just show help information:\nif 0:\n    !/kaggle/working/cd-hit-v4.8.1-2019-0228/cd-hit -h","metadata":{"execution":{"iopub.status.busy":"2023-06-10T20:04:31.560283Z","iopub.execute_input":"2023-06-10T20:04:31.56081Z","iopub.status.idle":"2023-06-10T20:04:31.568536Z","shell.execute_reply.started":"2023-06-10T20:04:31.560765Z","shell.execute_reply":"2023-06-10T20:04:31.566908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n# To increase sensitivity we should decrease \"-c\" and \"-n\" - that increases computation time \n# \n# possible configurations:\n# -c 0.63 -n 5  # 1min 49s  #  79408  clusters\n# -c 0.6  -n 4  # 27 minutes # 76864  clusters\n!/kaggle/working/cd-hit-v4.8.1-2019-0228/cd-hit -i \"/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta\" -o \"cdhit_output\" -c 0.55 -n 3","metadata":{"execution":{"iopub.status.busy":"2023-06-10T20:22:45.461869Z","iopub.execute_input":"2023-06-10T20:22:45.462391Z","iopub.status.idle":"2023-06-10T20:24:35.0832Z","shell.execute_reply.started":"2023-06-10T20:22:45.462339Z","shell.execute_reply":"2023-06-10T20:24:35.081032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head \"cdhit_output.clstr\"\n!tail \"cdhit_output.clstr\"","metadata":{"execution":{"iopub.status.busy":"2023-06-10T18:51:49.658603Z","iopub.execute_input":"2023-06-10T18:51:49.6591Z","iopub.status.idle":"2023-06-10T18:51:51.876731Z","shell.execute_reply.started":"2023-06-10T18:51:49.659054Z","shell.execute_reply":"2023-06-10T18:51:51.874795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Ways which does not work: ","metadata":{}},{"cell_type":"markdown","source":"## Fail 1 - pip install form github ","metadata":{}},{"cell_type":"code","source":"if 0:\n    !pip install git+https://github.com/sdivye92/cd_hit_py.git\n    '''\n    Collecting git+https://github.com/sdivye92/cd_hit_py.git\n      Cloning https://github.com/sdivye92/cd_hit_py.git to /tmp/pip-req-build-m9089act\n      Running command git clone --filter=blob:none --quiet https://github.com/sdivye92/cd_hit_py.git /tmp/pip-req-build-m9089act\n      Resolved https://github.com/sdivye92/cd_hit_py.git to commit a10c566479cea1b3b81ede7c8eca2a2e9d09cbac\n    ERROR: git+https://github.com/sdivye92/cd_hit_py.git does not appear to be a Python project: neither 'setup.py' nor 'pyproject.toml' found.\n    '''    \n    \n    from cd_hit import CD_HIT\n    '''\n    ModuleNotFoundError                       Traceback (most recent call last)\n    /tmp/ipykernel_28/4208656010.py in <module>\n    ----> 1 from cd_hit import CD_HIT\n\n    ModuleNotFoundError: No module named 'cd_hit'\n    '''\n","metadata":{"execution":{"iopub.status.busy":"2023-06-10T14:44:32.759311Z","iopub.execute_input":"2023-06-10T14:44:32.759838Z","iopub.status.idle":"2023-06-10T14:44:32.769152Z","shell.execute_reply.started":"2023-06-10T14:44:32.759793Z","shell.execute_reply":"2023-06-10T14:44:32.767556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fail 2 - git clone works , but import no ","metadata":{}},{"cell_type":"code","source":"!git clone https://github.com/sdivye92/cd_hit_py.git\n\nif 0 :\n    from cd_hit import CD_HIT\n    '''\n    ---------------------------------------------------------------------------\n    ModuleNotFoundError                       Traceback (most recent call last)\n    /tmp/ipykernel_28/1731773599.py in <module>\n          1 get_ipython().system('git clone https://github.com/sdivye92/cd_hit_py.git' class=\"ansi-blue-fg\">)\n          2 \n    ----> 3 from cd_hit import CD_HIT\n          4 \n\n    ModuleNotFoundError: No module named 'cd_hit'    \n    '''","metadata":{"execution":{"iopub.status.busy":"2023-06-10T15:26:10.088681Z","iopub.execute_input":"2023-06-10T15:26:10.090114Z","iopub.status.idle":"2023-06-10T15:26:11.1913Z","shell.execute_reply.started":"2023-06-10T15:26:10.090043Z","shell.execute_reply":"2023-06-10T15:26:11.1896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fail 3 - via conda install - install works, but import does not ","metadata":{}},{"cell_type":"code","source":"if 0:\n    # ChatGPT motivated way: \n    # !conda info --envs\n    # !conda activate base\n    # !which conda\n\n    !conda install --yes -c bioconda cd-hit # Run conda install will \"yes\" to all asked questions , it works okay\n    !conda list # we see cd_hit in the list, but still it import will not see it\n    !conda info --envs # check we have only one environment - base \n\n    from cd_hit import CD_HIT \n    # fail:\n    '''\n    ---------------------------------------------------------------------------\n    ModuleNotFoundError                       Traceback (most recent call last)\n    /tmp/ipykernel_27/2494723855.py in <module>\n          7 get_ipython().system('conda info --envs')\n          8 \n    ----> 9 from cd_hit import CD_HIT\n\n    ModuleNotFoundError: No module named 'cd_hit'\n    '''","metadata":{"execution":{"iopub.status.busy":"2023-06-10T18:32:20.442382Z","iopub.execute_input":"2023-06-10T18:32:20.442857Z","iopub.status.idle":"2023-06-10T18:34:40.47905Z","shell.execute_reply.started":"2023-06-10T18:32:20.442816Z","shell.execute_reply":"2023-06-10T18:34:40.476996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# MMseqs ","metadata":{}},{"cell_type":"code","source":"%%time\nif 0:\n    print('Install from static binary:')\n    #!wget https://mmseqs.com/latest/mmseqs-linux-avx2.tar.gz # Download the appropriate static binary package for your system based on the available instruction sets. For systems supporting AVX2, use the following command:\n    !wget https://mmseqs.com/latest/mmseqs-linux-sse41.tar.gz # For systems supporting SSE4.1, use this command:\n    #!wget https://mmseqs.com/latest/mmseqs-linux-sse2.tar.gz # For very old systems with only support for SSE2, use this command:    \n    !tar xvzf mmseqs-linux-*.tar.gz # Extract the downloaded archive:\n    !export PATH=$(pwd)/mmseqs/bin/:$PATH # Add the MMseqs2 binary directory to your PATH environment variable by running the following command:\n\n    print('Create database:')\n    !./mmseqs/bin/mmseqs createdb \"/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta\" \"sequenceDB\" # Create database from fasta   \n    print('Clustering:')\n    !./mmseqs/bin/mmseqs easy-linclust \"/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta\" clusterRes tmp # easy-linclust clusters the entries of a FASTA/FASTQ file. The runtime scales linearly with input size. This mode is recommended for huge datasets.    \n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('%.1f seconds passed total '%(time.time()-t0start) )","metadata":{"execution":{"iopub.status.busy":"2023-06-10T14:43:01.81282Z","iopub.execute_input":"2023-06-10T14:43:01.81333Z","iopub.status.idle":"2023-06-10T14:43:01.821077Z","shell.execute_reply.started":"2023-06-10T14:43:01.81328Z","shell.execute_reply":"2023-06-10T14:43:01.819868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}