{
  "release": {
    "date": "2026-07-15",
    "id": "236",
    "name": "VOGDB release 236",
    "data_source": "NCBI Refseq release 236",
    "url": "https://fileshare.csb.univie.ac.at/vog/vog236",
    "license": "All data published are licensed under CC BY 4.0 (https://creativecommons.org/licenses/by/4.0/)",
    "authors": "Lovro Trgovec-Greif, Hans-Joerg Hellinger, Jean Mainguy, Alexander Pfundner, Dmitrij Frishman, Michael Kiening, Nicole Webster, Patrick Laffy, Thomas Rattei",
    "contact": "Thomas Rattei, Centre for Microbiology and Environmental Systems Science, University of Vienna, Austria, thomas.rattei@univie.ac.at",
    "number_proteins": 724147,
    "number_genomes": 15791
  },
  "groups": [
    {
      "name": "vfam",
      "description": "Virus protein families (built from vogs by HMM-HMM clustering)",
      "number": 40080,
      "summary": "Number of VFAM: 40080 (Virus protein families)"
    },
    {
      "name": "vog",
      "description": "Virus orthologous groups (built from bidirectional sequence similarities)",
      "number": 49116,
      "summary": "Number of VOG: 49116 (Virus orthologous groups)"
    },
    {
      "name": "vfold",
      "description": "Virus protein structural folds (built from vfams by clustering of predicted 3D structures of representative proteins)",
      "number": 33327,
      "summary": "Number of VFOLD: 33327 (Virus protein structural folds)"
    }
  ],
  "files": [
    {
      "name": "vog.raw_algs.alistat.txt",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vog.raw_algs.alistat.txt",
      "description": "Statistics of multiple alignments according to minimum reporting standard for multiple sequence alignments (https://doi.org/10.1093/nargab/lqaa024).",
      "md5sum": "8d661024de71d8b556c5de3168a7e6c2",
      "bytes": 3306620,
      "url_label": "vog.raw_algs.alistat.txt (Statistics of multiple alignments): 3,306,620 bytes, MD5 checksum 8d661024de71d8b556c5de3168a7e6c2"
    },
    {
      "name": "vog.annotations.tsv.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vog.annotations.tsv.gz",
      "description": "Tab separated file of groups and their consensus functional annotations (preferrably from Swissprot annotations, if not available then the annotations from RefSeq were used). Columns: GroupName|ProteinCount|SpeciesCount|FunctionalCategory|ConsensusFunctionalDescription",
      "md5sum": "30b42e52252552bbf138f171d39fb122",
      "bytes": 374556,
      "url_label": "vog.annotations.tsv.gz (Funcational annotations of groups): 374,556 bytes, MD5 checksum 30b42e52252552bbf138f171d39fb122"
    },
    {
      "name": "vfam.members.tsv.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vfam.members.tsv.gz",
      "description": "Tab separated file of VOGs and the comma separated lists of their member protein ids. Columns: GroupName|ProteinCount|SpeciesCount|FunctionalCategory|ProteinIDs",
      "md5sum": "935b5f768b615b8365aa70ecd9a5c86c",
      "bytes": 4580149,
      "url_label": "vfam.members.tsv.gz (Member protein ids of groups): 4,580,149 bytes, MD5 checksum 935b5f768b615b8365aa70ecd9a5c86c"
    },
    {
      "name": "vfam.virusonly.tsv.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vfam.virusonly.tsv.gz",
      "description": "Tab separated file of VOGs and their specificic occurrence in virus genomes. For this purpose the homology of all member proteins to cellular genomes from eggNOG 4.5 have been determined with three different stringencies: High stringency: blastp e-Value <=1e-04 and hits in maximal 2 cellular genomes; Medium stringency: blastp e-Value <=1e-10 and hits in maximal 3 cellular genomes; Low stringency: blastp e-Value <=1e-15 and hits in maximal 4 cellular genomes; The column Only_in_viruses has been set true if members matched not more than the maximal number of genomes at the e-Value threshold for each stringency level. Columns: GroupName|Only in viruses (high stringency)|Only in viruses (medium stringency)|Only in viruses (low stringency) 1=True; 0=False. This file is useful to extract virus-specific markers from all VOGs, based on your preferred level of stringency.",
      "md5sum": "8a777299696e8ba1d20d454e3611fbc7",
      "bytes": 105937,
      "url_label": "vfam.virusonly.tsv.gz (Specificity if groups to Viruses): 105,937 bytes, MD5 checksum 8a777299696e8ba1d20d454e3611fbc7"
    },
    {
      "name": "vfam.lca.tsv.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vfam.lca.tsv.gz",
      "description": "Tab separated file of VOGs and the taxonomic lineage of the last common aencestor (LCA) of member genomes. Genomes with unclassified taxonomic lineages have not been used for LCA determination, which can result in VOG without lca (if all proteins of a VOG are from unclassified lineages). The numbers of genomes per VOG and LCA, as well as the total numbers of genomes in the LCA are given. Columns: GroupName|GenomesInGroupAndLCA|GenomesTotalInLCA|LastCommonAncestor_TaxonName|LastCommonAncestor_TaxonID",
      "md5sum": "a0b54cc7220a353d18107017d82e15f3",
      "bytes": 505954,
      "url_label": "vfam.lca.tsv.gz (Last common aencestors of groups): 505,954 bytes, MD5 checksum a0b54cc7220a353d18107017d82e15f3"
    },
    {
      "name": "vog.hmm.tar.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vog.hmm.tar.gz",
      "description": "Compressed archive of the HMMER3 compatible Hidden Markov Models obtained from the multiple sequence alignments for each vogdb group.",
      "md5sum": "4e17111ca1dc3a7529ef5dae26a642f6",
      "bytes": 580452971,
      "url_label": "vog.hmm.tar.gz (Hidden Markov Models of groups): 580,452,971 bytes, MD5 checksum 4e17111ca1dc3a7529ef5dae26a642f6"
    },
    {
      "name": "vfam.annotations.tsv.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vfam.annotations.tsv.gz",
      "description": "Tab separated file of groups and their consensus functional annotations (preferrably from Swissprot annotations, if not available then the annotations from RefSeq were used). Columns: GroupName|ProteinCount|SpeciesCount|FunctionalCategory|ConsensusFunctionalDescription",
      "md5sum": "e8b95279ecf9c7cbe8632cd289bf89eb",
      "bytes": 297147,
      "url_label": "vfam.annotations.tsv.gz (Funcational annotations of groups): 297,147 bytes, MD5 checksum e8b95279ecf9c7cbe8632cd289bf89eb"
    },
    {
      "name": "vog.faa.tar.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vog.faa.tar.gz",
      "description": "Compressed archive of FASTA formatted files of the proteins per vogdb group.",
      "md5sum": "e9a4566afaafb8067274d67a0c05f519",
      "bytes": 66910600,
      "url_label": "vog.faa.tar.gz (Protein sequences of groups): 66,910,600 bytes, MD5 checksum e9a4566afaafb8067274d67a0c05f519"
    },
    {
      "name": "vogdb.host.txt",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vogdb.host.txt",
      "description": "Tab separated file of host information and classification for virus taxa. Columns: taxon id|phage/nonphage|host|superkingdom of host.",
      "md5sum": "d3abb871cbff4ef611d352faaa3a7892",
      "bytes": 487967,
      "url_label": "vogdb.host.txt (Host information and classification for genomes): 487,967 bytes, MD5 checksum d3abb871cbff4ef611d352faaa3a7892"
    },
    {
      "name": "vfold.faa.tar.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vfold.faa.tar.gz",
      "description": "Compressed archive of FASTA formatted files of the proteins per vogdb group.",
      "md5sum": "9b324507cf06502c74dfd6e1dd334f64",
      "bytes": 64791369,
      "url_label": "vfold.faa.tar.gz (Protein sequences of groups): 64,791,369 bytes, MD5 checksum 9b324507cf06502c74dfd6e1dd334f64"
    },
    {
      "name": "vfam.raw_algs.tar.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vfam.raw_algs.tar.gz",
      "description": "Compressed archive of multiple sequence alignments for each VOGDB group.",
      "md5sum": "e5e6056fd1e7cc76b50dd671dd5c44c8",
      "bytes": 63745537,
      "url_label": "vfam.raw_algs.tar.gz (Multiple sequence alignments of groups): 63,745,537 bytes, MD5 checksum e5e6056fd1e7cc76b50dd671dd5c44c8"
    },
    {
      "name": "vog.members.tsv.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vog.members.tsv.gz",
      "description": "Tab separated file of VOGs and the comma separated lists of their member protein ids. Columns: GroupName|ProteinCount|SpeciesCount|FunctionalCategory|ProteinIDs",
      "md5sum": "928ae101116c543a0fde6649c12e08aa",
      "bytes": 4666375,
      "url_label": "vog.members.tsv.gz (Member protein ids of groups): 4,666,375 bytes, MD5 checksum 928ae101116c543a0fde6649c12e08aa"
    },
    {
      "name": "vfold.members.tsv.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vfold.members.tsv.gz",
      "description": "Tab separated file of VOGs and the comma separated lists of their member protein ids. Columns: GroupName|ProteinCount|SpeciesCount|FunctionalCategory|ProteinIDs",
      "md5sum": "f84c3318c2e1a45fec2e912d7bdc5218",
      "bytes": 4541056,
      "url_label": "vfold.members.tsv.gz (Member protein ids of groups): 4,541,056 bytes, MD5 checksum f84c3318c2e1a45fec2e912d7bdc5218"
    },
    {
      "name": "vfam.hmm.tar.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vfam.hmm.tar.gz",
      "description": "Compressed archive of the HMMER3 compatible Hidden Markov Models obtained from the multiple sequence alignments for each vogdb group.",
      "md5sum": "6d5c43b3bdcd465ac98e1a54fb7d84af",
      "bytes": 468559551,
      "url_label": "vfam.hmm.tar.gz (Hidden Markov Models of groups): 468,559,551 bytes, MD5 checksum 6d5c43b3bdcd465ac98e1a54fb7d84af"
    },
    {
      "name": "vfold.annotations.tsv.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vfold.annotations.tsv.gz",
      "description": "Tab separated file of groups and their consensus functional annotations (preferrably from Swissprot annotations, if not available then the annotations from RefSeq were used). Columns: GroupName|ProteinCount|SpeciesCount|FunctionalCategory|ConsensusFunctionalDescription",
      "md5sum": "15a947d9fa09527f389a69237a9b8789",
      "bytes": 244554,
      "url_label": "vfold.annotations.tsv.gz (Funcational annotations of groups): 244,554 bytes, MD5 checksum 15a947d9fa09527f389a69237a9b8789"
    },
    {
      "name": "vfam.representatives.colabfold_predictions.tar.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vfam.representatives.colabfold_predictions.tar.gz",
      "description": "Colabfold predictions of protein structures, represented by best ranked PDB and score files.",
      "md5sum": "325ef2d0ebc52157cbef7dd826227494",
      "bytes": 7925261503,
      "url_label": "vfam.representatives.colabfold_predictions.tar.gz (Protein structure predictions): 7,925,261,503 bytes, MD5 checksum 325ef2d0ebc52157cbef7dd826227494"
    },
    {
      "name": "vogdb.proteins.all.fa.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vogdb.proteins.all.fa.gz",
      "description": "FASTA formatted file of all proteins from the genomes in vog.species.list. Protein IDs encode the taxonomy id of the genome and the RefSeq protein id. For peptides from polyproteins also the corresponding protein id of the polyprotein (CDS) is given.",
      "md5sum": "fd9b8461a358bece4ed7aea1c2a7be44",
      "bytes": 107411905,
      "url_label": "vogdb.proteins.all.fa.gz (Protein sequences of all genomes): 107,411,905 bytes, MD5 checksum fd9b8461a358bece4ed7aea1c2a7be44"
    },
    {
      "name": "vogdb.genes.all.fa.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vogdb.genes.all.fa.gz",
      "description": "FASTA formatted file of all gene sequences from the genomes in vog.species.list. Same IDs as in the protein file are used. For polyprotein genes the partial gene sequences of the peptides as well as the complete gene sequences of the polyprotein are contained.",
      "md5sum": "cf872ae19547faeb98296a39fec03b99",
      "bytes": 172872866,
      "url_label": "vogdb.genes.all.fa.gz (Gene sequences of all genomes): 172,872,866 bytes, MD5 checksum cf872ae19547faeb98296a39fec03b99"
    },
    {
      "name": "vfam.raw_algs.alistat.txt",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vfam.raw_algs.alistat.txt",
      "description": "Statistics of multiple alignments according to minimum reporting standard for multiple sequence alignments (https://doi.org/10.1093/nargab/lqaa024).",
      "md5sum": "661d7ae44c3a8c097afba2a86296b148",
      "bytes": 2742146,
      "url_label": "vfam.raw_algs.alistat.txt (Statistics of multiple alignments): 2,742,146 bytes, MD5 checksum 661d7ae44c3a8c097afba2a86296b148"
    },
    {
      "name": "vfam.representatives.colabfold_mean_plddt.txt",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vfam.representatives.colabfold_mean_plddt.txt",
      "description": "Mean pLDDT values for colabfold predictions of protein structures.",
      "md5sum": "09cfdb728ff85fbbdbc72d37648f0cae",
      "bytes": 640240,
      "url_label": "vfam.representatives.colabfold_mean_plddt.txt (Mean pLDDT values of protein structure predictions): 640,240 bytes, MD5 checksum 09cfdb728ff85fbbdbc72d37648f0cae"
    },
    {
      "name": "vog.virusonly.tsv.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vog.virusonly.tsv.gz",
      "description": "Tab separated file of VOGs and their specificic occurrence in virus genomes. For this purpose the homology of all member proteins to cellular genomes from eggNOG 4.5 have been determined with three different stringencies: High stringency: blastp e-Value <=1e-04 and hits in maximal 2 cellular genomes; Medium stringency: blastp e-Value <=1e-10 and hits in maximal 3 cellular genomes; Low stringency: blastp e-Value <=1e-15 and hits in maximal 4 cellular genomes; The column Only_in_viruses has been set true if members matched not more than the maximal number of genomes at the e-Value threshold for each stringency level. Columns: GroupName|Only in viruses (high stringency)|Only in viruses (medium stringency)|Only in viruses (low stringency) 1=True; 0=False. This file is useful to extract virus-specific markers from all VOGs, based on your preferred level of stringency.",
      "md5sum": "9e704afdd5c747d52b290241cf6a63a4",
      "bytes": 127389,
      "url_label": "vog.virusonly.tsv.gz (Specificity if groups to Viruses): 127,389 bytes, MD5 checksum 9e704afdd5c747d52b290241cf6a63a4"
    },
    {
      "name": "vog.raw_algs.tar.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vog.raw_algs.tar.gz",
      "description": "Compressed archive of multiple sequence alignments for each VOGDB group.",
      "md5sum": "96bad38890073f2413ab9acd7c90ac86",
      "bytes": 62748765,
      "url_label": "vog.raw_algs.tar.gz (Multiple sequence alignments of groups): 62,748,765 bytes, MD5 checksum 96bad38890073f2413ab9acd7c90ac86"
    },
    {
      "name": "vfold.lca.tsv.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vfold.lca.tsv.gz",
      "description": "Tab separated file of VOGs and the taxonomic lineage of the last common aencestor (LCA) of member genomes. Genomes with unclassified taxonomic lineages have not been used for LCA determination, which can result in VOG without lca (if all proteins of a VOG are from unclassified lineages). The numbers of genomes per VOG and LCA, as well as the total numbers of genomes in the LCA are given. Columns: GroupName|GenomesInGroupAndLCA|GenomesTotalInLCA|LastCommonAncestor_TaxonName|LastCommonAncestor_TaxonID",
      "md5sum": "5e964ff2dc1d9e50ceb6e8aa422bc69f",
      "bytes": 418338,
      "url_label": "vfold.lca.tsv.gz (Last common aencestors of groups): 418,338 bytes, MD5 checksum 5e964ff2dc1d9e50ceb6e8aa422bc69f"
    },
    {
      "name": "vogdb.functional_categories.txt",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vogdb.functional_categories.txt",
      "description": "Text file listing the lettercodes of functional categories. These consist of X (unused in NCBI COG functional categories), followed by a lower case character indicating the functional category.",
      "md5sum": "6b816cc49c17d0095da91bad4e7552fa",
      "bytes": 308,
      "url_label": "vogdb.functional_categories.txt (Lettercodes of functional categories): 308 bytes, MD5 checksum 6b816cc49c17d0095da91bad4e7552fa"
    },
    {
      "name": "vfold.virusonly.tsv.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vfold.virusonly.tsv.gz",
      "description": "Tab separated file of VOGs and their specificic occurrence in virus genomes. For this purpose the homology of all member proteins to cellular genomes from eggNOG 4.5 have been determined with three different stringencies: High stringency: blastp e-Value <=1e-04 and hits in maximal 2 cellular genomes; Medium stringency: blastp e-Value <=1e-10 and hits in maximal 3 cellular genomes; Low stringency: blastp e-Value <=1e-15 and hits in maximal 4 cellular genomes; The column Only_in_viruses has been set true if members matched not more than the maximal number of genomes at the e-Value threshold for each stringency level. Columns: GroupName|Only in viruses (high stringency)|Only in viruses (medium stringency)|Only in viruses (low stringency) 1=True; 0=False. This file is useful to extract virus-specific markers from all VOGs, based on your preferred level of stringency.",
      "md5sum": "d09430d162f930b9aea65d626a976a14",
      "bytes": 87287,
      "url_label": "vfold.virusonly.tsv.gz (Specificity if groups to Viruses): 87,287 bytes, MD5 checksum d09430d162f930b9aea65d626a976a14"
    },
    {
      "name": "vog.lca.tsv.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vog.lca.tsv.gz",
      "description": "Tab separated file of VOGs and the taxonomic lineage of the last common aencestor (LCA) of member genomes. Genomes with unclassified taxonomic lineages have not been used for LCA determination, which can result in VOG without lca (if all proteins of a VOG are from unclassified lineages). The numbers of genomes per VOG and LCA, as well as the total numbers of genomes in the LCA are given. Columns: GroupName|GenomesInGroupAndLCA|GenomesTotalInLCA|LastCommonAncestor_TaxonName|LastCommonAncestor_TaxonID",
      "md5sum": "c467f8b28cf36ec0315d1063b6f8cfc0",
      "bytes": 625089,
      "url_label": "vog.lca.tsv.gz (Last common aencestors of groups): 625,089 bytes, MD5 checksum c467f8b28cf36ec0315d1063b6f8cfc0"
    },
    {
      "name": "vogdb.species.txt",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vogdb.species.txt",
      "description": "Tab separated file of virus genomes used for VOG construction. Columns: species name|taxon id|source|source version",
      "md5sum": "404cbfc03498ec5aef6f614a0a7990eb",
      "bytes": 823396,
      "url_label": "vogdb.species.txt (Virus genomes used for VOG construction): 823,396 bytes, MD5 checksum 404cbfc03498ec5aef6f614a0a7990eb"
    },
    {
      "name": "vogdb.taxonomy.krona.html",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vogdb.taxonomy.krona.html",
      "description": "Interactive chart of virus genome taxonomies.",
      "md5sum": "0952976de71b5c0a7d6c0de482d16f74",
      "bytes": 7046195,
      "url_label": "vogdb.taxonomy.krona.html (Distribution of virus genome taxonomies): 7,046,195 bytes, MD5 checksum 0952976de71b5c0a7d6c0de482d16f74"
    },
    {
      "name": "vfam.faa.tar.gz",
      "url": "https://fileshare.csb.univie.ac.at/vog/vog236/vfam.faa.tar.gz",
      "description": "Compressed archive of FASTA formatted files of the proteins per vogdb group.",
      "md5sum": "d684fdf7ee19f837b96966b0b4f59cc4",
      "bytes": 66062485,
      "url_label": "vfam.faa.tar.gz (Protein sequences of groups): 66,062,485 bytes, MD5 checksum d684fdf7ee19f837b96966b0b4f59cc4"
    }
  ]
}
