From 491b98955a1536d3bdc9193e1f4943c84432e83a Mon Sep 17 00:00:00 2001 From: Nikolai Romashchenko Date: Fri, 18 Sep 2026 17:02:49 +0200 Subject: [PATCH 1/4] Updated README Updated README.md to enhance clarity and instructions for installation and usage. --- README.md | 68 +++++++++++++++++++++++++++++-------------------------- 1 file changed, 36 insertions(+), 32 deletions(-) diff --git a/README.md b/README.md index 9f9f5d4..bd71497 100755 --- a/README.md +++ b/README.md @@ -1,54 +1,60 @@ # HogProf - - HogProf is an extensible and tunable approach to phylogenetic profiling using orthology data. It is powered by minhash based datastructures and computationally efficient. - - Still under major development and may change +HogProf is an extensible and tunable approach to phylogenetic profiling using orthology data. It is powered by minhash-based data structures and computationally efficient. # Features - - Using orthoxoml files and a taxonomy calculated enhanced phylogenies of each family - These are transformed into minhash signatures and a locally sensitive hashing forest object for search and comparison of profiles - - Taxonomic levels and evolutionary event types ( presence, loss, duplication ) can have custom weight in profile construction + - Taxonomic levels and evolutionary event types (presence, loss, duplication) can have custom weight in profile construction - Optimization of weights using machine learning If you run into any problems feel free to contact me at [dmoi@unil.ch](dmoi@unil.ch) # Quickstart +## Install from PyPI (recommended) -to install from github -``` -$ git clone https://github.com/DessimozLab/HogProf.git -$ pip install -r pipreqs.txt . +```bash +pip install hogprof ``` -or to install from pypi -``` -$ pip install hogprof + +## Install from sources +```bash +git clone https://github.com/DessimozLab/HogProf.git +cd HogProf +pip install . ``` -lets get a current version of the OMA hdf5 file and GAF. This will alow us to use the HOGs and study the functional enrichment of our search results. +# Usage +## Example: using the OMA database +Let's get a current version of the OMA hdf5 file and GAF. This will aloww us to use the HOGs and study the functional enrichment of our search results. **Careful, this download is heavy (hundreds of GB)**: + +```bash +mkdir YourOmaDirectory +cd YourOmaDirectory +wget https://omabrowser.org/All/OmaServer.h5 +wget https://omabrowser.org/All/oma-go.txt.gz ``` -$ cd ../.. -$ mkdir YourOmaDirectory -$ cd YourOmaDirectory -$ wget https://omabrowser.org/All/OmaServer.h5 -$ wget https://omabrowser.org/All/oma-go.txt.gz + +The latest release (May.2026) is more than 300GB in size. Alternatively, you can use earlier releases, for example: +```bash +wget https://omabrowser.org/All.Jul2024/OmaServer.h5 ``` +which is 180GB in size. For other releases, check `https://omabrowser.org/oma/archives/`, select release and search for "OMA Browser database (as hdf5)". + We also need to make a location to store our pyprofiler databases +```bash +cd .. +mkdir YourHogProfDirectory ``` -$ cd .. -$ mkdir YourHogProfDirectory -``` - -Ok. We're ready! Now let's compile a database containing all HOGs and our desired taxonomic levels using default settings. Launch the lshbuilder. -dbtypes available on the command line are : all , plants , archaea, bacteria , eukarya , protists , fungi , metazoa and vertebrates. These will use the NCBI taxonomy as a tree to annotate events in different gene family's histories. +Let's now compile a database containing all HOGs and our desired taxonomic levels using default settings. Launch the lshbuilder as shown below. +`dbtypes` available on the command line are: `all`, `plants`, `archaea`, `bacteria`, `eukarya`, `protists`, `fungi`, `metazoa`, and `vertebrates`. These will use the NCBI taxonomy as a tree to annotate events in different gene family's histories. If you are using an OMA release before 2022 you will need to use the NCBI tree. This is the default tree used by HogProf. - - ``` $python lshbuilder.py --outpath YourHogProfDirectory --dbtype all --OMA YourOmaDirectory/OmaServer.h5 --nthreads numberOfCPUcores @@ -56,19 +62,17 @@ $python lshbuilder.py --outpath YourHogProfDirectory --dbtype all --OMA YourOmaD If you are using OMA releases after 2022 we will also be downloading the OMA taxonomic tree since it is more accurate than the NCBI tree. This will be used to annotate the events in the gene family histories. -``` +```bash wget https://omabrowser.org/All/speciestree.nwk - ``` -Ok now we're ready to use the OMA tree to build our database. - -``` +We are ready to use the OMA tree to build our database. -$python lshbuilder.py --outpath YourHogProfDirectory --dbtype all --OMA YourOmaDirectory/OmaServer.h5 --nthreads numberOfCPUcores --mastertree YourOmaDirectory/speciestree.nwk --reformat_names True +```bash +lshbuilder --outpath YourHogProfDirectory --dbtype all --OMA YourOmaDirectory/OmaServer.h5 --nthreads numberOfCPUcores --mastertree YourOmaDirectory/speciestree.nwk --reformat_names True ``` This should build a taxonomic tree for the genomes contained in the release and then calculate enhanced phylogenies for all HOGs in OMA. -Once the database is completed it can be interogated using a profiler object. Construction and usage of this object should be done using a python script or notebook. This shown in the example notebook searchenrich.ipynb found in the examples. Please feel free to modify it to suit the needs of your own research. +Once the database is completed it can be interogated using a profiler object. Construction and usage of this object should be done using a python script or notebook. This shown in the example notebook `searchenrich.ipynb` found in the examples. Please feel free to modify it to suit the needs of your own research. From b21ca74adaa3fdea9bdb4c300fedf83c89286b61 Mon Sep 17 00:00:00 2001 From: Nikolai Romashchenko Date: Fri, 18 Sep 2026 17:03:52 +0200 Subject: [PATCH 2/4] Updated README --- README.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index bd71497..292024f 100755 --- a/README.md +++ b/README.md @@ -7,7 +7,7 @@ HogProf is an extensible and tunable approach to phylogenetic profiling using or - Taxonomic levels and evolutionary event types (presence, loss, duplication) can have custom weight in profile construction - Optimization of weights using machine learning -If you run into any problems feel free to contact me at [dmoi@unil.ch](dmoi@unil.ch) +If you run into any problems, feel free to contact me at [dmoi@unil.ch](dmoi@unil.ch) # Quickstart ## Install from PyPI (recommended) @@ -56,7 +56,7 @@ Let's now compile a database containing all HOGs and our desired taxonomic level If you are using an OMA release before 2022 you will need to use the NCBI tree. This is the default tree used by HogProf. ``` -$python lshbuilder.py --outpath YourHogProfDirectory --dbtype all --OMA YourOmaDirectory/OmaServer.h5 --nthreads numberOfCPUcores +lshbuilder --outpath YourHogProfDirectory --dbtype all --OMA YourOmaDirectory/OmaServer.h5 --nthreads numberOfCPUcores ``` From 5e44e523ecb5e3664a9937bc994669c984acfa4f Mon Sep 17 00:00:00 2001 From: AthinaGav Date: Wed, 23 Sep 2026 15:06:00 +0200 Subject: [PATCH 3/4] added species_lim --- src/HogProf/lshbuilder.py | 23 +++++++++++++++++------ 1 file changed, 17 insertions(+), 6 deletions(-) diff --git a/src/HogProf/lshbuilder.py b/src/HogProf/lshbuilder.py index d74e97d..2d90309 100755 --- a/src/HogProf/lshbuilder.py +++ b/src/HogProf/lshbuilder.py @@ -41,7 +41,8 @@ class LSHBuilder: with a list of taxonomic codes for all the species in your db """ - def __init__(self,h5_oma=None,fileglob = None, taxa=None,masterTree=None, saving_name=None , numperm = 256, treeweights= None , taxfilter = None, taxmask= None , lossonly = False, duplonly = False, verbose = False , use_taxcodes = False , datetime = datetime.now() , reformat_names = False): + def __init__(self,h5_oma=None,fileglob = None, taxa=None,masterTree=None, saving_name=None , numperm = 256, treeweights= None , taxfilter = None, taxmask= None , lossonly = False, duplonly = False, verbose = False , use_taxcodes = False , datetime = datetime.now() , reformat_names = False, + limit_species = 10): """ Initializes the LSHBuilder class with the specified parameters and sets up the necessary objects. @@ -57,6 +58,8 @@ def __init__(self,h5_oma=None,fileglob = None, taxa=None,masterTree=None, saving - taxfilter (str): path to a file containing a list of taxonomic codes to filter from the tree - taxmask (str): path to a file containing a list of taxonomic codes to mask from the tree - verbose (bool): whether to print verbose output (default: False) + - limit_species (int): the minimum number of species in a subHOG that is included in the database (default: 10) + """ if h5_oma: @@ -77,6 +80,7 @@ def __init__(self,h5_oma=None,fileglob = None, taxa=None,masterTree=None, saving self.fileglob = fileglob self.idmapper = None self.date_string = "{:%B_%d_%Y_%H_%M}".format(datetime.now()) + self.limit_species = limit_species if saving_name: self.saving_name= saving_name if self.saving_name[-1]!= '/': @@ -190,10 +194,12 @@ def __init__(self,h5_oma=None,fileglob = None, taxa=None,masterTree=None, saving if self.h5OMA: self.HAM_PIPELINE = functools.partial( pyhamutils.get_ham_treemap_from_row, tree=self.tree_string , swap_ids=self.swap2taxcode , reformat_names = self.reformat_names , - orthoXML_as_string = True , use_phyloxml = self.use_phyloxml , orthomapper = self.idmapper , levels = None ) + orthoXML_as_string = True , use_phyloxml = self.use_phyloxml , orthomapper = self.idmapper , levels = None, + ) else: self.HAM_PIPELINE = functools.partial( pyhamutils.get_ham_treemap_from_row, tree=self.tree_string , swap_ids=self.swap2taxcode , - orthoXML_as_string = False , reformat_names = self.reformat_names , use_phyloxml = self.use_phyloxml , orthomapper = self.idmapper , levels = None ) + orthoXML_as_string = False , reformat_names = self.reformat_names , use_phyloxml = self.use_phyloxml , orthomapper = self.idmapper , levels = None, + ) self.HASH_PIPELINE = functools.partial( hashutils.row2hash , taxaIndex=self.taxaIndex, treeweights=self.treeweights, wmg=wmg , lossonly = lossonly, duplonly = duplonly) if self.h5OMA: @@ -469,7 +475,7 @@ def mp_with_timeout(functypes, data_generator): gc.collect() print('DONE!') - mp_with_timeout(functypes=functype_dict, data_generator=self.generates_dataframes(100)) + mp_with_timeout(functypes=functype_dict, data_generator=self.generates_dataframes(size=100, minhog_size=self.limit_species)) return self.hashes_path, self.lshforestpath , self.mat_path @@ -494,6 +500,9 @@ def main(): parser.add_argument('--taxcodes', help='use taxid info in HOGs' , type = bool) parser.add_argument('--verbose', help='print verbose output' , type = bool) parser.add_argument('--reformat_names', help='try to correct broken species trees by replacing all names with numbers.' , type = bool) + parser.add_argument('--specieslim', help='minimum number of species in a subhog' , type = int, default=10) + parser.add_argument('--eventslim', help='minimum number of events (loss/duplication) in a subhog' , type = int, default=0) + dbdict = { 'all': { 'taxfilter': None , 'taxmask': None }, 'plants': { 'taxfilter': None , 'taxmask': 33090 }, @@ -595,12 +604,14 @@ def main(): with open_file( omafile , mode="r") as h5_oma: lsh_builder = LSHBuilder(h5_oma = h5_oma, fileglob=orthoglob ,saving_name=dbname , numperm = nperm , treeweights= weights , taxfilter = taxfilter, taxmask=taxmask , masterTree =mastertree , - lossonly = lossonly , duplonly = duplonly , use_taxcodes = taxcodes , reformat_names=reformat_names, verbose=verbose ) + lossonly = lossonly , duplonly = duplonly , use_taxcodes = taxcodes , reformat_names=reformat_names, verbose=verbose, + limit_species=args['specieslim']) lsh_builder.run_pipeline(threads) else: lsh_builder = LSHBuilder(h5_oma = None, fileglob=orthoglob ,saving_name=dbname , numperm = nperm , treeweights= weights , taxfilter = taxfilter, taxmask=taxmask , - masterTree =mastertree , lossonly = lossonly , duplonly = duplonly , use_taxcodes = taxcodes , reformat_names=reformat_names, verbose=verbose) + masterTree =mastertree , lossonly = lossonly , duplonly = duplonly , use_taxcodes = taxcodes , reformat_names=reformat_names, verbose=verbose, + limit_species=args['specieslim']) lsh_builder.run_pipeline(threads) print(time.time() - start) #save corrected tree From aeec03a78d5ae26898f4fca5a10c4d5ad0f7dc0a Mon Sep 17 00:00:00 2001 From: AthinaGav Date: Wed, 23 Sep 2026 15:07:26 +0200 Subject: [PATCH 4/4] add ONLY specieslim --- src/HogProf/lshbuilder.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/src/HogProf/lshbuilder.py b/src/HogProf/lshbuilder.py index 2d90309..dd77d04 100755 --- a/src/HogProf/lshbuilder.py +++ b/src/HogProf/lshbuilder.py @@ -501,8 +501,7 @@ def main(): parser.add_argument('--verbose', help='print verbose output' , type = bool) parser.add_argument('--reformat_names', help='try to correct broken species trees by replacing all names with numbers.' , type = bool) parser.add_argument('--specieslim', help='minimum number of species in a subhog' , type = int, default=10) - parser.add_argument('--eventslim', help='minimum number of events (loss/duplication) in a subhog' , type = int, default=0) - + dbdict = { 'all': { 'taxfilter': None , 'taxmask': None }, 'plants': { 'taxfilter': None , 'taxmask': 33090 },