[
  {
    "id": "bfm-enformer",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "Enformer",
    "description": "Convolution/transformer model for predicting gene expression and regulatory tracks from long DNA context.",
    "paper_links": [
      {
        "label": "Nature Methods",
        "url": "https://www.nature.com/articles/s41592-021-01252-x"
      }
    ],
    "code_links": [
      {
        "label": "DeepMind research code",
        "url": "https://github.com/google-deepmind/deepmind-research/tree/master/enformer"
      }
    ],
    "status": "Open code + weights",
    "io": "DNA sequence window -> tissue/cell-type regulatory predictions",
    "use_cases": "Variant effect prediction, enhancer/promoter analysis, expression impact",
    "year": "2021",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-021-01252-x",
      "https://github.com/google-deepmind/deepmind-research/tree/master/enformer"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Nature Methods paper and DeepMind Apache research code match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-dnabert",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "DNABERT",
    "description": "BERT-style k-mer DNA language model trained on the human genome.",
    "paper_links": [
      {
        "label": "Bioinformatics",
        "url": "https://academic.oup.com/bioinformatics/article/37/15/2112/6128680"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/jerryji1993/DNABERT"
      },
      {
        "label": "NVIDIA BioNeMo docs",
        "url": "https://docs.nvidia.com/bionemo-framework/1.10/models/dnabert.html"
      }
    ],
    "status": "Open code + weights / BioNeMo",
    "io": "DNA k-mer sequence -> embeddings or task logits",
    "use_cases": "Promoter/enhancer/TFBS classification, sequence embeddings",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://academic.oup.com/bioinformatics/article/37/15/2112/6128680",
      "https://github.com/jerryji1993/DNABERT",
      "https://docs.nvidia.com/bionemo-framework/1.10/models/dnabert.html"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-dnabert-2",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "DNABERT-2",
    "description": "More efficient multi-species genome transformer with BPE tokenization and GUE benchmark.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2306.15006"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/zhihan1996/DNABERT-2-117M"
      }
    ],
    "status": "Model hub",
    "io": "DNA sequence up to about 2 kb -> embeddings/log-likelihood/task heads",
    "use_cases": "Regulatory element classification, variant scoring, similarity",
    "year": "2023",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2306.15006",
      "https://huggingface.co/zhihan1996/DNABERT-2-117M"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "checkpoint exists, but model card lacks exact licence. Add official repository/licence evidence.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-nucleotide-transformer",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "Nucleotide Transformer",
    "description": "Family of DNA foundation models, 50M to 2.5B parameters, trained on human and multi-species genomes.",
    "paper_links": [
      {
        "label": "Nature Methods",
        "url": "https://www.nature.com/articles/s41592-024-02523-z"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/instadeepai/nucleotide-transformer"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/InstaDeepAI/nucleotide-transformer-2.5b-multi-species"
      }
    ],
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "io": "DNA sequence -> nucleotide/sequence embeddings, fine-tuned predictions",
    "use_cases": "Regulatory genomics, splice/variant effect prediction, transfer learning",
    "year": "2024",
    "canonical": true,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-024-02523-z",
      "https://github.com/instadeepai/nucleotide-transformer",
      "https://huggingface.co/InstaDeepAI/nucleotide-transformer-2.5b-multi-species"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact. Action: set noncommercial:true; official licence is CC BY-NC-SA 4.0.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-hyenadna",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "HyenaDNA",
    "description": "Long-context genomic model using Hyena operators, single-nucleotide resolution up to about 1M tokens.",
    "paper_links": [
      {
        "label": "NeurIPS paper",
        "url": "https://proceedings.neurips.cc/paper_files/paper/2023/hash/86ab6927ee4ae9bde4247793c46797c7-Abstract-Conference.html"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/HazyResearch/hyena-dna"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/LongSafari/hyenadna-large-1m-seqlen"
      }
    ],
    "status": "Open code + weights",
    "io": "Long DNA sequence -> embeddings or task logits",
    "use_cases": "Long-range genomics, regulatory classification, synthetic long-context benchmarks",
    "year": "",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://proceedings.neurips.cc/paper_files/paper/2023/hash/86ab6927ee4ae9bde4247793c46797c7-Abstract-Conference.html",
      "https://github.com/HazyResearch/hyena-dna",
      "https://huggingface.co/LongSafari/hyenadna-large-1m-seqlen"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-caduceus",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "Caduceus",
    "description": "Bidirectional long-range DNA language model with reverse-complement equivariance, based on Mamba-style blocks.",
    "paper_links": [
      {
        "label": "ICML",
        "url": "https://proceedings.mlr.press/v235/schiff24a.html"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/kuleshov-group/caduceus"
      },
      {
        "label": "project page",
        "url": "https://caduceus-dna.github.io/"
      }
    ],
    "status": "Open code + weights",
    "io": "DNA sequence -> embeddings/log-likelihood/task outputs",
    "use_cases": "Variant effect, long-range regulatory modeling, RC-aware genomics",
    "year": "2024",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://proceedings.mlr.press/v235/schiff24a.html",
      "https://github.com/kuleshov-group/caduceus",
      "https://caduceus-dna.github.io/"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-gena-lm",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "GENA-LM",
    "description": "Open-source long-sequence DNA language model family.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2023.06.12.544594"
      },
      {
        "label": "NAR",
        "url": "https://academic.oup.com/nar/article/53/2/gkae1310/7954523"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/AIRI-Institute/GENA_LM"
      }
    ],
    "status": "Open code + weights",
    "io": "DNA sequence -> embeddings/fine-tuned predictions",
    "use_cases": "Long DNA classification, regulatory genomics, model benchmarking",
    "year": "2023",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2023.06.12.544594",
      "https://academic.oup.com/nar/article/53/2/gkae1310/7954523",
      "https://github.com/AIRI-Institute/GENA_LM"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-gpn-msa",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "GPN-MSA",
    "description": "Alignment-based DNA language model for genome-wide variant effect prediction.",
    "paper_links": [
      {
        "label": "PMC article",
        "url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC10592768/"
      },
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2023.10.10.561776v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/songlab-cal/gpn"
      }
    ],
    "status": "Open code + weights",
    "io": "MSA/genomic context -> variant effect score",
    "use_cases": "Genome-wide variant scoring, evolutionary constraint",
    "year": "2023",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://pmc.ncbi.nlm.nih.gov/articles/PMC10592768/",
      "https://www.biorxiv.org/content/10.1101/2023.10.10.561776v1",
      "https://github.com/songlab-cal/gpn",
      "https://www.nature.com/articles/s41587-024-02511-w"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Final paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-evo",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "Evo",
    "description": "7B long-context biological sequence model trained on prokaryotic and phage genomes; spans DNA/RNA/protein coding signals.",
    "paper_links": [
      {
        "label": "Science",
        "url": "https://www.science.org/doi/10.1126/science.ado9336"
      },
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.02.27.582234v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/evo-design/evo"
      }
    ],
    "status": "Open code + weights",
    "io": "Nucleotide byte sequence -> next-token probabilities, generated sequence, embeddings",
    "use_cases": "Genome-scale generation, CRISPR/Cas system design, molecular function prediction",
    "year": "2024",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.science.org/doi/10.1126/science.ado9336",
      "https://www.biorxiv.org/content/10.1101/2024.02.27.582234v1",
      "https://github.com/evo-design/evo"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-evo-2",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "Evo 2",
    "description": "Genome model trained across all domains of life with long context and single-nucleotide resolution.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-026-10176-5"
      },
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.02.18.638918v1.full"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/arcinstitute/evo2"
      }
    ],
    "status": "Open code + weights",
    "io": "DNA sequence up to large genomic context -> likelihoods, variants, generated genomes",
    "use_cases": "Coding/noncoding variant effects, genome design, cross-domain genomics",
    "year": "2025",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41586-026-10176-5",
      "https://www.biorxiv.org/content/10.1101/2025.02.18.638918v1.full",
      "https://github.com/arcinstitute/evo2"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-lucaone",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "LucaOne",
    "description": "Unified nucleic-acid and protein biological foundation model.",
    "paper_links": [
      {
        "label": "Nature Machine Intelligence",
        "url": "https://www.nature.com/articles/s42256-025-01044-4"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/LucaOne/LucaOne"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/collections/LucaGroup/lucaone"
      }
    ],
    "status": "Open code + weights",
    "io": "DNA/RNA/protein sequence -> embeddings and downstream task predictions",
    "use_cases": "Cross-modality sequence embeddings, central-dogma tasks, functional prediction",
    "year": "2025",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s42256-025-01044-4",
      "https://github.com/LucaOne/LucaOne",
      "https://huggingface.co/collections/LucaGroup/lucaone"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-bioseq-blm",
    "date_added": "2026-06-16",
    "modality": "platform",
    "name": "BioSeq-BLM",
    "description": "Platform for applying biological language models to DNA, RNA, and protein sequences.",
    "paper_links": [
      {
        "label": "NAR",
        "url": "https://academic.oup.com/nar/article/49/22/e129/6377401"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Zimiao1025/BioSeq-BLM"
      }
    ],
    "status": "Open code + web",
    "io": "Biological sequence -> features/predictors/evaluation outputs",
    "use_cases": "Building sequence predictors without hand-engineered features",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://academic.oup.com/nar/article/49/22/e129/6377401",
      "https://github.com/Zimiao1025/BioSeq-BLM"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact. Action: reclassify from DNA to platform; it is a DNA/RNA/protein web and software platform.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-rna-fm",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "RNA-FM",
    "description": "RNA foundation model trained on millions of ncRNA sequences; extracts structure/function-aware RNA embeddings.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2022.08.06.503062v2.full-text"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ml4bio/RNA-FM"
      },
      {
        "label": "Hugging Face mirror",
        "url": "https://huggingface.co/multimolecule/rnafm"
      }
    ],
    "status": "Open code + weights",
    "io": "RNA sequence -> residue/sequence embeddings, task heads",
    "use_cases": "RNA secondary structure, function prediction, ncRNA annotation",
    "year": "2022",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2022.08.06.503062v2.full-text",
      "https://github.com/ml4bio/RNA-FM",
      "https://huggingface.co/multimolecule/rnafm"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-rinalmo",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "RiNALMo",
    "description": "650M-parameter RNA language model trained on noncoding RNA sequences.",
    "paper_links": [
      {
        "label": "Nature Communications",
        "url": "https://www.nature.com/articles/s41467-025-60872-5"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/lbcb-sci/RiNALMo"
      },
      {
        "label": "Zenodo weights",
        "url": "https://zenodo.org/records/15043668"
      }
    ],
    "status": "Open code + weights",
    "io": "RNA sequence -> embeddings/secondary-structure features",
    "use_cases": "RNA structure prediction, functional annotation, transfer learning",
    "year": "2025",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-025-60872-5",
      "https://github.com/lbcb-sci/RiNALMo",
      "https://zenodo.org/records/15043668"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "published paper, Apache code, and Zenodo weights match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-uni-rna",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "UNI-RNA",
    "description": "Universal RNA pretraining model for structure and function tasks.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2023.07.11.548588v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ComDec/unirna_tf"
      }
    ],
    "status": "Inference code public; weights via Google Drive, CC BY-NC (non-commercial)",
    "io": "RNA sequence -> embeddings/task predictions",
    "use_cases": "RNA structure/function prediction, RNA sequence representation",
    "year": "2023",
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2023.07.11.548588v1",
      "https://github.com/ComDec/unirna_tf"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-ernie-rna",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "ERNIE-RNA",
    "description": "Structure-enhanced RNA language model incorporating base-pairing restrictions.",
    "paper_links": [
      {
        "label": "Nature Communications",
        "url": "https://www.nature.com/articles/s41467-025-64972-0"
      },
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.03.17.585376v1.full-text"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Bruce-ywj/ERNIE-RNA"
      },
      {
        "label": "Hugging Face mirror",
        "url": "https://huggingface.co/multimolecule/ernierna"
      }
    ],
    "status": "Open code + weights",
    "io": "RNA sequence/structure-aware tokens -> embeddings, attention, structure outputs",
    "use_cases": "Secondary structure, RNA functional annotation",
    "year": "2024",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-025-64972-0",
      "https://www.biorxiv.org/content/10.1101/2024.03.17.585376v1.full-text",
      "https://github.com/Bruce-ywj/ERNIE-RNA",
      "https://huggingface.co/multimolecule/ernierna"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-rnabert",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "RNABERT",
    "description": "Early BERT-style RNA sequence model for structural alignment and clustering.",
    "paper_links": [
      {
        "label": "NAR Genomics and Bioinformatics",
        "url": "https://academic.oup.com/nargab/article/4/1/lqac012/6546209"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/mana438/RNABERT"
      }
    ],
    "status": "Open code + weights",
    "io": "RNA sequence -> base embeddings",
    "use_cases": "RNA clustering, structural alignment, representation learning",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://academic.oup.com/nargab/article/4/1/lqac012/6546209",
      "https://github.com/mana438/RNABERT"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "source and weights exist, but no reuse licence is declared. State “no licence found.”",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-splicebert",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "SpliceBERT",
    "description": "Pre-mRNA sequence model for splicing-related prediction.",
    "paper_links": [
      {
        "label": "Briefings in Bioinformatics",
        "url": "https://academic.oup.com/bib/article/25/3/bbae163/7644137"
      },
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2023.01.31.526427v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/chenkenbio/SpliceBERT"
      },
      {
        "label": "Hugging Face mirror",
        "url": "https://huggingface.co/multimolecule/splicebert"
      }
    ],
    "status": "Open code + weights",
    "io": "Pre-mRNA sequence -> embeddings/splicing predictions",
    "use_cases": "Splice site and splicing regulatory variant analysis",
    "year": "2023",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://academic.oup.com/bib/article/25/3/bbae163/7644137",
      "https://www.biorxiv.org/content/10.1101/2023.01.31.526427v1",
      "https://github.com/chenkenbio/SpliceBERT",
      "https://huggingface.co/multimolecule/splicebert"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "original project is BSD-3-Clause while the MultiMolecule mirror is AGPL-3.0. Record component-specific licences.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-rna-msm",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "RNA-MSM",
    "description": "MSA-based RNA language modeling for structure inference.",
    "paper_links": [
      {
        "label": "NAR",
        "url": "https://academic.oup.com/nar/article/52/1/e3/7369930"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/yikunpku/RNA-MSM"
      }
    ],
    "status": "Open code + weights",
    "io": "RNA MSA -> embeddings/structure-related predictions",
    "use_cases": "Homology-aware RNA structure inference",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://academic.oup.com/nar/article/52/1/e3/7369930",
      "https://github.com/yikunpku/RNA-MSM"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-generrna",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "GenerRNA",
    "description": "Generative pretrained model for de novo RNA design.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.02.01.578496v1"
      },
      {
        "label": "PLOS One",
        "url": "https://journals.plos.org/plosone/article?id=10.1371/journal.pone.0310814"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/pfnet-research/GenerRNA"
      }
    ],
    "status": "Open code + weights",
    "io": "Prompt/seed RNA context -> generated RNA sequences",
    "use_cases": "RNA design, synthetic sequence exploration",
    "year": "2024",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.02.01.578496v1",
      "https://journals.plos.org/plosone/article?id=10.1371/journal.pone.0310814",
      "https://github.com/pfnet-research/GenerRNA"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-aido-rna",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "AIDO.RNA",
    "description": "Large RNA function/structure foundation model in the AIDO/GenBio ecosystem.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.11.28.625345"
      }
    ],
    "code_links": [
      {
        "label": "ModelGenerator",
        "url": "https://github.com/genbio-ai/modelgenerator"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/genbio-ai/AIDO.RNA-1.6B"
      }
    ],
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "io": "RNA sequence -> embeddings, structure/function task heads",
    "use_cases": "RNA function, structure prediction, downstream adaptation",
    "year": "2024",
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.11.28.625345",
      "https://github.com/genbio-ai/modelgenerator",
      "https://huggingface.co/genbio-ai/AIDO.RNA-1.6B",
      "https://github.com/genbio-ai/ModelGenerator/blob/main/LICENSE"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Same GenBio non-commercial restriction as AIDO.Cell. Paper, model, licence.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-esm-1b-esm-1v",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "ESM-1b / ESM-1v",
    "description": "Evolutionary-scale protein masked language models; ESM-1v is widely used for zero-shot variant effect prediction.",
    "paper_links": [
      {
        "label": "ESM-1b (PNAS)",
        "url": "https://www.pnas.org/doi/10.1073/pnas.2016239118"
      },
      {
        "label": "ESM-1v (bioRxiv)",
        "url": "https://www.biorxiv.org/content/10.1101/2021.07.09.450648"
      }
    ],
    "code_links": [
      {
        "label": "ESM code",
        "url": "https://github.com/facebookresearch/esm"
      }
    ],
    "status": "Open code + weights",
    "io": "Amino acid sequence -> residue embeddings, log-likelihoods",
    "use_cases": "Protein embeddings, mutation effect scoring, function/structure transfer",
    "year": "2021",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.pnas.org/doi/10.1073/pnas.2016239118",
      "https://www.biorxiv.org/content/10.1101/2021.07.09.450648",
      "https://github.com/facebookresearch/esm"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "repository is archived. Mark historical/archived while retaining accessible MIT code and weights.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-esm-2",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "ESM-2",
    "description": "Larger ESM protein LM family; basis for ESMFold and many embeddings workflows.",
    "paper_links": [
      {
        "label": "Science",
        "url": "https://www.science.org/doi/10.1126/science.ade2574"
      }
    ],
    "code_links": [
      {
        "label": "ESM code",
        "url": "https://github.com/facebookresearch/esm"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/facebook/esm2_t33_650M_UR50D"
      }
    ],
    "status": "Open code + weights",
    "io": "Protein sequence -> embeddings, contacts, structure via ESMFold",
    "use_cases": "Structure prediction, embeddings, zero-shot scoring",
    "year": "",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.science.org/doi/10.1126/science.ade2574",
      "https://github.com/facebookresearch/esm",
      "https://huggingface.co/facebook/esm2_t33_650M_UR50D"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact; weights.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-esmfold",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "ESMFold",
    "description": "Fast protein structure prediction from a single sequence using ESM-2 representations.",
    "paper_links": [
      {
        "label": "Science",
        "url": "https://www.science.org/doi/10.1126/science.ade2574"
      }
    ],
    "code_links": [
      {
        "label": "ESM code",
        "url": "https://github.com/facebookresearch/esm"
      }
    ],
    "status": "Open code + weights",
    "io": "Protein sequence -> 3D protein structure",
    "use_cases": "MSA-free structure prediction, metagenomic structure annotation",
    "year": "",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.science.org/doi/10.1126/science.ade2574",
      "https://github.com/facebookresearch/esm"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-esm3",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "ESM3",
    "description": "Multimodal generative protein model over sequence, structure, and function tracks.",
    "paper_links": [
      {
        "label": "Science",
        "url": "https://www.science.org/doi/10.1126/science.ads0018"
      },
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.07.01.600583v1"
      }
    ],
    "code_links": [
      {
        "label": "EvolutionaryScale ESM",
        "url": "https://github.com/evolutionaryscale/esm"
      },
      {
        "label": "EvolutionaryScale",
        "url": "https://www.evolutionaryscale.ai/"
      }
    ],
    "status": "Open smaller models + API/gated larger models",
    "io": "Sequence/structure/function prompts -> completed/generated protein",
    "use_cases": "Controllable protein generation, function-guided design",
    "year": "2024",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.science.org/doi/10.1126/science.ads0018",
      "https://www.biorxiv.org/content/10.1101/2024.07.01.600583v1",
      "https://github.com/evolutionaryscale/esm",
      "https://www.evolutionaryscale.ai/"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "old URL redirects to Biohub’s ESMC/ESMFold2 repository. Add the specific ESM3 README/model-card locator and current MIT terms.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-esm-c-esm-cambrian",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "ESM-C / ESM Cambrian",
    "description": "Biohub-maintained protein sequence encoder family for efficient representation learning; current releases include openly available ESMC checkpoints and provide the language-model backbone used by ESMFold2.",
    "paper_links": [
      {
        "label": "bioRxiv (ESMC / ESMFold2)",
        "url": "https://www.biorxiv.org/content/10.64898/2026.06.03.729735v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Biohub/esm"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/biohub"
      }
    ],
    "status": "Open code + weights (MIT); local and Biohub Platform inference",
    "io": "Protein sequence -> embeddings/logits",
    "use_cases": "Protein search, annotation, downstream feature extraction",
    "year": "",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.06.03.729735v1",
      "https://github.com/Biohub/esm",
      "https://huggingface.co/biohub"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-prottrans-prott5-protbert",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "ProtTrans / ProtT5 / ProtBERT",
    "description": "Large transformer protein language models trained on UniProt/BFD-scale protein corpora.",
    "paper_links": [
      {
        "label": "PubMed",
        "url": "https://pubmed.ncbi.nlm.nih.gov/34232869/"
      },
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2007.06225"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/agemagician/ProtTrans"
      },
      {
        "label": "Hugging Face Rostlab",
        "url": "https://huggingface.co/Rostlab"
      }
    ],
    "status": "Open code + weights",
    "io": "Protein sequence -> embeddings/task predictions",
    "use_cases": "General protein embeddings, fine-tuning, annotation",
    "year": "2020",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://pubmed.ncbi.nlm.nih.gov/34232869/",
      "https://arxiv.org/abs/2007.06225",
      "https://github.com/agemagician/ProtTrans",
      "https://huggingface.co/Rostlab"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "MIT repository and Rostlab models match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-proteinbert",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "ProteinBERT",
    "description": "Universal protein sequence/function model combining local and global tokens.",
    "paper_links": [
      {
        "label": "Bioinformatics",
        "url": "https://academic.oup.com/bioinformatics/article/38/8/2102/6502274"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/nadavbra/protein_bert"
      }
    ],
    "status": "Open code + weights",
    "io": "Protein sequence +/- annotations -> embeddings/task outputs",
    "use_cases": "Function prediction, protein classification",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://academic.oup.com/bioinformatics/article/38/8/2102/6502274",
      "https://github.com/nadavbra/protein_bert"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-progen-progen2",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "ProGen / ProGen2",
    "description": "Autoregressive protein language models for sequence generation and fitness scoring.",
    "paper_links": [
      {
        "label": "ProGen (Nat Biotech)",
        "url": "https://www.nature.com/articles/s41587-022-01618-2"
      },
      {
        "label": "ProGen2 (OpenReview)",
        "url": "https://openreview.net/forum?id=ZOn4HXehSJ6"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/salesforce/progen"
      },
      {
        "label": "Hugging Face mirror",
        "url": "https://huggingface.co/jinyuan22/ProGen2-base"
      }
    ],
    "status": "Open code + weights/mirrors",
    "io": "Prompt/control tokens + sequence context -> generated protein sequence",
    "use_cases": "De novo protein generation, family-conditioned design, zero-shot fitness",
    "year": "2022",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41587-022-01618-2",
      "https://openreview.net/forum?id=ZOn4HXehSJ6",
      "https://github.com/salesforce/progen",
      "https://huggingface.co/jinyuan22/ProGen2-base"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "ProGen · ProGen2 · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-progen3",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "ProGen3",
    "description": "Scaled sparse autoregressive protein LM for generation and protein understanding.",
    "paper_links": [
      {
        "label": "bioRxiv v2",
        "url": "https://www.biorxiv.org/content/10.1101/2025.04.15.649055v2"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Profluent-AI/progen3"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/Profluent-Bio/progen3-3b"
      }
    ],
    "status": "Open code + weights through 3B (non-commercial); 46B remains API-only",
    "io": "Protein prompt/context -> generated sequence/log-likelihood",
    "use_cases": "Large-scale protein generation and scoring",
    "year": "",
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.04.15.649055v2",
      "https://github.com/Profluent-AI/progen3",
      "https://huggingface.co/Profluent-Bio/progen3-3b"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-ankh",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "Ankh",
    "description": "Efficient general-purpose protein language model.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2023.01.16.524265v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/agemagician/Ankh"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/ElnaggarLab/ankh-large"
      }
    ],
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "io": "Protein sequence -> embeddings/task heads",
    "use_cases": "Lightweight transfer learning, protein classification",
    "year": "2023",
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2023.01.16.524265v1",
      "https://github.com/agemagician/Ankh",
      "https://huggingface.co/ElnaggarLab/ankh-large"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P1",
    "audit_note": "set NC true; weights are CC BY-NC-SA 4.0.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-protgpt2",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "ProtGPT2",
    "description": "GPT-style protein generator.",
    "paper_links": [
      {
        "label": "Nature Communications",
        "url": "https://www.nature.com/articles/s41467-022-32007-7"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/nferruz/ProtGPT2"
      }
    ],
    "status": "Model hub",
    "io": "Prompt/seed sequence -> generated protein sequence",
    "use_cases": "De novo sequence generation, novelty exploration",
    "year": "2022",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-022-32007-7",
      "https://huggingface.co/nferruz/ProtGPT2"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "published paper and Apache model card match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-xtrimopglm",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "xTrimoPGLM",
    "description": "Very large protein language model family from BioMap.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2401.06199"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/biomap-research/xTrimoPGLM"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/biomap-research/xtrimopglm-1b-mlm"
      }
    ],
    "status": "Open code + public weights (non-commercial)",
    "io": "Protein sequence -> embeddings/generation",
    "use_cases": "Protein understanding, large-scale protein representation",
    "year": "2023",
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2401.06199",
      "https://github.com/biomap-research/xTrimoPGLM",
      "https://huggingface.co/biomap-research/xtrimopglm-1b-mlm"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-iglm",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "IgLM",
    "description": "Autoregressive antibody language model.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2021.12.13.472419v2"
      },
      {
        "label": "Cell Systems DOI",
        "url": "https://doi.org/10.1016/j.cels.2023.10.001"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Graylab/IgLM"
      }
    ],
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "io": "Antibody chain context -> generated antibody sequence",
    "use_cases": "Antibody design, repertoire modeling",
    "year": "2021",
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2021.12.13.472419v2",
      "https://doi.org/10.1016/j.cels.2023.10.001",
      "https://github.com/Graylab/IgLM"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P1",
    "audit_note": "set NC true and cite JHU Academic Software Licence for code and pretrained models.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-antiberta",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "AntiBERTa",
    "description": "Antibody-specific BERT-style representation model.",
    "paper_links": [
      {
        "label": "PubMed",
        "url": "https://pubmed.ncbi.nlm.nih.gov/35845836/"
      },
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2021.11.10.468064.full"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/alchemab/antiberta"
      }
    ],
    "status": "Open code + weights",
    "io": "Antibody sequence -> embeddings",
    "use_cases": "Antibody clustering, developability, paratope-related tasks",
    "year": "2021",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://pubmed.ncbi.nlm.nih.gov/35845836/",
      "https://www.biorxiv.org/content/10.1101/2021.11.10.468064.full",
      "https://github.com/alchemab/antiberta"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-ablang",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "AbLang",
    "description": "Antibody language model for completing and restoring antibody sequences.",
    "paper_links": [
      {
        "label": "Bioinformatics Advances",
        "url": "https://academic.oup.com/bioinformaticsadvances/article/2/1/vbac046/6627685"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/oxpig/AbLang"
      }
    ],
    "status": "Open code + weights",
    "io": "Antibody sequence with masks/gaps -> completed sequence/embeddings",
    "use_cases": "Antibody sequence completion, repertoire cleanup",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://academic.oup.com/bioinformaticsadvances/article/2/1/vbac046/6627685",
      "https://github.com/oxpig/AbLang"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "BSD code and released weights match. Persist; no substantive change.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-eve",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "EVE",
    "description": "VAE over protein family MSAs for disease variant prediction.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-021-04043-8"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/OATML-Markslab/EVE"
      }
    ],
    "status": "Open code + weights/models",
    "io": "Protein MSA -> variant effect score",
    "use_cases": "Human disease variant interpretation, protein family fitness",
    "year": "2021",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41586-021-04043-8",
      "https://github.com/OATML-Markslab/EVE"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Nature paper, MIT code, and released models match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-msa-transformer",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "MSA Transformer",
    "description": "Transformer over multiple sequence alignments.",
    "paper_links": [
      {
        "label": "ICML / bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2021.02.12.430858v1"
      }
    ],
    "code_links": [
      {
        "label": "ESM code",
        "url": "https://github.com/facebookresearch/esm"
      }
    ],
    "status": "Open code + weights",
    "io": "MSA -> embeddings, contacts, mutation scores",
    "use_cases": "Contact prediction, variant scoring with homologs",
    "year": "2021",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2021.02.12.430858v1",
      "https://github.com/facebookresearch/esm"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-tranception",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "Tranception",
    "description": "Autoregressive protein model with inference-time retrieval over homologs.",
    "paper_links": [
      {
        "label": "ICML / arXiv",
        "url": "https://arxiv.org/abs/2205.13760"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/OATML-Markslab/Tranception"
      }
    ],
    "status": "Open code + weights",
    "io": "Protein sequence + optional MSA/retrieval -> fitness score/generated sequence",
    "use_cases": "Protein fitness prediction, variant prioritization",
    "year": "2022",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2205.13760",
      "https://github.com/OATML-Markslab/Tranception"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-poet",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "PoET",
    "description": "Generative model of protein families as sequence sets.",
    "paper_links": [
      {
        "label": "NeurIPS paper",
        "url": "https://papers.nips.cc/paper_files/paper/2023/hash/f4366126eba252699b280e8f93c0ab2f-Abstract-Conference.html"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/OpenProteinAI/PoET"
      }
    ],
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "io": "Family sequence set -> generated/scored sequences",
    "use_cases": "Family-specific protein design, few-shot fitness",
    "year": "2023",
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://papers.nips.cc/paper_files/paper/2023/hash/f4366126eba252699b280e8f93c0ab2f-Abstract-Conference.html",
      "https://github.com/OpenProteinAI/PoET",
      "https://github.com/OpenProteinAI/PoET#license",
      "https://zenodo.org/records/10061322"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "PoET weights are CC BY-NC-SA for academic use; noncommercial:false. Replace arXiv with NeurIPS paper and resolve overlap with bfm-poet-2 and platform record bfm-openprotein-ai-poet-poet-2. NeurIPS paper, repo terms, weights.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-proteinmpnn",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "ProteinMPNN",
    "description": "Neural inverse folding model for designing sequences for a given backbone.",
    "paper_links": [
      {
        "label": "Science",
        "url": "https://www.science.org/doi/10.1126/science.add2187"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/dauparas/ProteinMPNN"
      }
    ],
    "status": "Open code + weights",
    "io": "Protein backbone/PDB -> designed FASTA sequences",
    "use_cases": "Stabilizing backbones, binder scaffolds, redesign",
    "year": "",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.science.org/doi/10.1126/science.add2187",
      "https://github.com/dauparas/ProteinMPNN"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-esm-if1",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "ESM-IF1",
    "description": "Inverse folding model trained from large predicted structure sets.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2022.04.10.487779v1"
      }
    ],
    "code_links": [
      {
        "label": "ESM code",
        "url": "https://github.com/facebookresearch/esm"
      }
    ],
    "status": "Open code + weights",
    "io": "Backbone coordinates -> sequence likelihood/design",
    "use_cases": "Inverse folding, structure-conditioned mutation scoring",
    "year": "2022",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2022.04.10.487779v1",
      "https://github.com/facebookresearch/esm"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-rfdiffusion",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "RFdiffusion / RFdiffusion3 (RFD3)",
    "description": "Protein-backbone and all-atom diffusion family for functional design; RFD3 extends the family to proteins in the context of ligands, nucleic acids, and other non-protein atoms.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-023-06415-8"
      },
      {
        "label": "bioRxiv (RFD3 v2)",
        "url": "https://www.biorxiv.org/content/10.1101/2025.09.18.676967v2"
      }
    ],
    "code_links": [
      {
        "label": "RFdiffusion GitHub",
        "url": "https://github.com/RosettaCommons/RFdiffusion"
      },
      {
        "label": "RFD3 Foundry",
        "url": "https://github.com/RosettaCommons/foundry/tree/production/models/rfd3"
      }
    ],
    "status": "Open code + weights",
    "io": "Motif/target constraints -> protein backbone design",
    "use_cases": "Binder design, enzyme scaffolding, symmetric assemblies",
    "year": "2023",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "aliases": [
      "RFdiffusion",
      "RFdiffusion3",
      "RFD3"
    ],
    "date_modified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41586-023-06415-8",
      "https://www.biorxiv.org/content/10.1101/2025.09.18.676967v2",
      "https://github.com/RosettaCommons/RFdiffusion",
      "https://github.com/RosettaCommons/foundry/tree/production/models/rfd3"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md"
  },
  {
    "id": "bfm-chroma",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "Chroma",
    "description": "Programmable generative protein model.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-023-06728-8"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/generatebio/chroma"
      }
    ],
    "status": "Open code; gated academic/non-profit weights under a restrictive parameters licence; non-commercial terms apply to at least one artifact or access route",
    "io": "Constraints/prompts -> protein backbone/sequence designs",
    "use_cases": "Controllable protein generation",
    "year": "2023",
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41586-023-06728-8",
      "https://github.com/generatebio/chroma"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-dplm",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "DPLM",
    "description": "Diffusion language model for protein sequence learning and generation.",
    "paper_links": [
      {
        "label": "ICML / arXiv",
        "url": "https://arxiv.org/abs/2402.18567"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/bytedance/dplm"
      }
    ],
    "status": "Open code + weights",
    "io": "Protein sequence masks/noise -> denoised/generated sequence",
    "use_cases": "Sequence generation, protein representation learning",
    "year": "2024",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2402.18567",
      "https://github.com/bytedance/dplm"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Apache code and DPLM checkpoints match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-bioemu-1",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "BioEmu-1",
    "description": "Generative emulator of protein equilibrium ensembles.",
    "paper_links": [
      {
        "label": "Science",
        "url": "https://www.science.org/doi/10.1126/science.adq5170"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/microsoft/bioemu"
      }
    ],
    "status": "Open code + weights",
    "io": "Protein sequence -> conformational ensemble samples",
    "use_cases": "Protein dynamics, conformational heterogeneity",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.science.org/doi/10.1126/science.adq5170",
      "https://github.com/microsoft/bioemu"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-alphafold-3",
    "date_added": "2026-06-16",
    "modality": "complex",
    "name": "AlphaFold 3",
    "description": "Diffusion-based model for structures of proteins, nucleic acids, ligands, ions, and modifications.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-024-07487-w"
      }
    ],
    "code_links": [
      {
        "label": "AlphaFold 3 GitHub",
        "url": "https://github.com/google-deepmind/alphafold3"
      },
      {
        "label": "AlphaFold Server",
        "url": "https://alphafoldserver.com/"
      }
    ],
    "status": "Open code, gated weights / server; non-commercial terms apply to at least one artifact or access route",
    "io": "Protein/DNA/RNA/ligand specification -> complex 3D structure",
    "use_cases": "Biomolecular interaction modeling, drug discovery, complex hypotheses",
    "year": "2024",
    "canonical": true,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41586-024-07487-w",
      "https://github.com/google-deepmind/alphafold3",
      "https://alphafoldserver.com/"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P1",
    "audit_note": "set NC true and distinguish Apache code, gated NC parameters, and NC server.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-chai-1",
    "date_added": "2026-06-16",
    "modality": "complex",
    "name": "Chai-1",
    "description": "Multimodal structure model for proteins, small molecules, DNA, RNA, glycosylations, and more.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.10.10.615955v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/chaidiscovery/chai-lab"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/chaidiscovery/chai-1"
      }
    ],
    "status": "Open code + weights",
    "io": "FASTA/SMILES/CCD-like inputs -> complex 3D structure",
    "use_cases": "Open AlphaFold3-like complex prediction, ligand/protein/nucleic-acid modeling",
    "year": "2024",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.10.10.615955v1",
      "https://github.com/chaidiscovery/chai-lab",
      "https://huggingface.co/chaidiscovery/chai-1"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Apache code and weights are openly available. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-boltz-1-boltz-2",
    "date_added": "2026-06-16",
    "modality": "complex",
    "name": "Boltz-1 / Boltz-2",
    "description": "Open biomolecular interaction model family; Boltz-2 adds affinity modeling.",
    "paper_links": [
      {
        "label": "Boltz-1 (PMC)",
        "url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC11601547/"
      },
      {
        "label": "Boltz-2 (bioRxiv)",
        "url": "https://www.biorxiv.org/content/10.1101/2025.06.14.659707v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/jwohlwend/boltz"
      }
    ],
    "status": "Open code + weights",
    "io": "Biomolecular complex specification -> structure +/- affinity",
    "use_cases": "Complex prediction, virtual screening, affinity-aware molecular design",
    "year": "2025",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://pmc.ncbi.nlm.nih.gov/articles/PMC11601547/",
      "https://www.biorxiv.org/content/10.1101/2025.06.14.659707v1",
      "https://github.com/jwohlwend/boltz"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Boltz-1 paper; Boltz-2 paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-rosettafold-all-atom",
    "date_added": "2026-06-16",
    "modality": "complex",
    "name": "RoseTTAFold All-Atom / RoseTTAFold3 (RF3)",
    "description": "RoseTTAFold all-atom family for predicting arbitrary biomolecular assemblies; RF3 extends the family across proteins, RNA, DNA, ligands, ions, and covalent modifications.",
    "paper_links": [
      {
        "label": "Science",
        "url": "https://www.science.org/doi/10.1126/science.adl2528"
      },
      {
        "label": "bioRxiv (RF3 v2)",
        "url": "https://www.biorxiv.org/content/10.1101/2025.08.14.670328v2"
      }
    ],
    "code_links": [
      {
        "label": "Foundry",
        "url": "https://github.com/RosettaCommons/foundry"
      },
      {
        "label": "RF3 model",
        "url": "https://github.com/RosettaCommons/foundry/tree/production/models/rf3"
      }
    ],
    "status": "Open code + checkpoints (BSD-3-Clause); RF3 inference API still stabilizing",
    "io": "Protein/nucleic acid/ligand assembly inputs -> all-atom structure",
    "use_cases": "General biomolecular assembly modeling",
    "year": "",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "aliases": [
      "RoseTTAFold All-Atom",
      "RoseTTAFold3",
      "RF3"
    ],
    "date_modified": "2026-07-13",
    "sources": [
      "https://www.science.org/doi/10.1126/science.adl2528",
      "https://www.biorxiv.org/content/10.1101/2025.08.14.670328v2",
      "https://github.com/RosettaCommons/foundry",
      "https://github.com/RosettaCommons/foundry/tree/production/models/rf3"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md"
  },
  {
    "id": "bfm-openfold",
    "date_added": "2026-06-16",
    "modality": "complex",
    "name": "OpenFold",
    "description": "Open reimplementation/training framework for AlphaFold2-style protein structure prediction.",
    "paper_links": [
      {
        "label": "Nature Methods",
        "url": "https://www.nature.com/articles/s41592-024-02272-z"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/aqlaboratory/openfold"
      }
    ],
    "status": "Open code + weights",
    "io": "Protein sequence/MSA/templates -> 3D protein structure",
    "use_cases": "Open AF2 reproduction, training, structure prediction",
    "year": "2024",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-024-02272-z",
      "https://github.com/aqlaboratory/openfold"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-diffdock",
    "date_added": "2026-06-16",
    "modality": "complex",
    "name": "DiffDock",
    "description": "Diffusion model for protein-ligand docking.",
    "paper_links": [
      {
        "label": "ICLR",
        "url": "https://openreview.net/forum?id=kKF8_K-mBbS"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/gcorso/DiffDock"
      }
    ],
    "status": "Open code + weights",
    "io": "Protein structure + ligand -> ligand pose(s)",
    "use_cases": "Docking, pose generation, blind docking workflows",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://openreview.net/forum?id=kKF8_K-mBbS",
      "https://github.com/gcorso/DiffDock"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "MIT code and weights remain available. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-chemberta-chemberta-2",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "ChemBERTa / ChemBERTa-2",
    "description": "BERT/RoBERTa-style chemical language models over SMILES.",
    "paper_links": [
      {
        "label": "ChemBERTa-2",
        "url": "https://huggingface.co/papers/2209.01712"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/seyonechithrananda/bert-loves-chemistry"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/seyonec/ChemBERTa-zinc-base-v1"
      }
    ],
    "status": "Open code + weights",
    "io": "SMILES -> embeddings/masked-token predictions/task heads",
    "use_cases": "Molecular property prediction, QSAR, transfer learning",
    "year": "",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://huggingface.co/papers/2209.01712",
      "https://github.com/seyonechithrananda/bert-loves-chemistry",
      "https://huggingface.co/seyonec/ChemBERTa-zinc-base-v1",
      "https://arxiv.org/abs/2209.01712",
      "https://huggingface.co/DeepChem/ChemBERTa-77M-MTR",
      "https://huggingface.co/DeepChem/ChemBERTa-77M-MLM"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Family record cites ChemBERTa-2 but links only the older ChemBERTa v1 repo/model. Add the official ChemBERTa-2 77M MLM/MTR artifacts or split versions. ChemBERTa-2 paper, v1 repo, ChemBERTa-2 MTR, MLM.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-molformer",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "MoLFormer",
    "description": "Large-scale chemical language model over SMILES with linear attention.",
    "paper_links": [
      {
        "label": "Nature Machine Intelligence",
        "url": "https://www.nature.com/articles/s42256-022-00580-7"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/IBM/molformer"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/ibm-research/MoLFormer-XL-both-10pct"
      }
    ],
    "status": "Open code + weights",
    "io": "SMILES -> molecular embeddings",
    "use_cases": "Molecular property prediction, similarity, screening",
    "year": "2022",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s42256-022-00580-7",
      "https://github.com/IBM/molformer",
      "https://huggingface.co/ibm-research/MoLFormer-XL-both-10pct"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact; weights.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-gp-molformer",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "GP-MoLFormer",
    "description": "Autoregressive MoLFormer-style generator trained on large SMILES corpora.",
    "paper_links": [
      {
        "label": "Digital Discovery",
        "url": "https://pubs.rsc.org/en/content/articlehtml/2025/dd/d5dd00122f"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/IBM/gp-molformer"
      }
    ],
    "status": "Open code + weights",
    "io": "SMILES prompt/context -> generated SMILES",
    "use_cases": "De novo molecular generation, analog generation",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://pubs.rsc.org/en/content/articlehtml/2025/dd/d5dd00122f",
      "https://github.com/IBM/gp-molformer"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-grover",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "GROVER",
    "description": "Self-supervised graph transformer for molecular representation learning.",
    "paper_links": [
      {
        "label": "NeurIPS / arXiv",
        "url": "https://arxiv.org/abs/2007.02835"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/tencent-ailab/grover"
      }
    ],
    "status": "Open code + weights",
    "io": "Molecular graph/SMILES-derived graph -> embeddings/task heads",
    "use_cases": "Property prediction, graph-based QSAR",
    "year": "2020",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2007.02835",
      "https://github.com/tencent-ailab/grover"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-graphmvp",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "GraphMVP",
    "description": "Multi-view pretraining linking 2D molecular graphs and 3D geometry.",
    "paper_links": [
      {
        "label": "OpenReview",
        "url": "https://openreview.net/forum?id=xQUe1pOKPam"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/chao1224/GraphMVP"
      }
    ],
    "status": "Open code + weights",
    "io": "2D graph + 3D conformer -> embeddings",
    "use_cases": "Property prediction, 3D-aware molecular representation",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://openreview.net/forum?id=xQUe1pOKPam",
      "https://github.com/chao1224/GraphMVP"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "MIT code and public pretrained weights match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-uni-mol",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "Uni-Mol",
    "description": "Universal 3D molecular representation framework with molecular and pocket pretraining;",
    "paper_links": [
      {
        "label": "ICLR",
        "url": "https://openreview.net/forum?id=6K2RM6wVqKu"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/deepmodeling/Uni-Mol"
      }
    ],
    "status": "Open code + weights",
    "io": "3D conformers/pockets -> embeddings, docking/property outputs",
    "use_cases": "Molecular property, protein-ligand binding, docking, virtual screening",
    "year": "",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://openreview.net/forum?id=6K2RM6wVqKu",
      "https://github.com/deepmodeling/Uni-Mol",
      "https://arxiv.org/abs/2508.00920"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Base paper; official artifact; Uni-Mol3 paper. Action: remove the unsupported Uni-Mol3 merge from this verified base record, or create a separate held record; the official artifact does not substantiate that family merge.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": [
      "Uni-Mol base"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "bfm-megamolbart",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "MegaMolBART",
    "description": "BART-style generative model for SMILES in NVIDIA BioNeMo.",
    "paper_links": [],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/NVIDIA/MegaMolBART"
      },
      {
        "label": "BioNeMo Framework",
        "url": "https://github.com/NVIDIA/bionemo-framework"
      }
    ],
    "status": "Open code / BioNeMo",
    "io": "SMILES -> latent embeddings or generated SMILES",
    "use_cases": "Molecular generation, analog search, property optimization",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://github.com/NVIDIA/MegaMolBART",
      "https://github.com/NVIDIA/bionemo-framework",
      "https://docs.nvidia.com/bionemo-framework/1.10/notebooks/MMB_GenerativeAI_Inference_with_examples.html"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "Official repository; technical guide. Action: stop labelling the duplicated GitHub URL as a paper; identify it as official software/model documentation.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-molmim",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "MolMIM",
    "description": "Mutual-information-based molecular generator used in BioNeMo workflows.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/pdf/2208.09016"
      }
    ],
    "code_links": [
      {
        "label": "BioNeMo Framework",
        "url": "https://github.com/NVIDIA/bionemo-framework"
      }
    ],
    "status": "BioNeMo / platform",
    "io": "Molecular seed/objective -> optimized generated molecules",
    "use_cases": "Goal-directed generation, molecular optimization",
    "year": "2022",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/pdf/2208.09016",
      "https://github.com/NVIDIA/bionemo-framework"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; platform.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-molt5",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "MolT5",
    "description": "Molecule-language sequence-to-sequence model for molecule captioning and text-to-molecule generation.",
    "paper_links": [
      {
        "label": "EMNLP",
        "url": "https://aclanthology.org/2022.emnlp-main.26/"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/blender-nlp/MolT5"
      }
    ],
    "status": "Open code + weights",
    "io": "SMILES <-> natural-language description",
    "use_cases": "Molecule captioning, text-guided molecule generation",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://aclanthology.org/2022.emnlp-main.26/",
      "https://github.com/blender-nlp/MolT5"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-3d-molt5",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "3D-MolT5",
    "description": "Multimodal molecule-text model integrating SMILES, 3D structural tokens, and text.",
    "paper_links": [
      {
        "label": "OpenReview",
        "url": "https://openreview.net/forum?id=eGqQyTAbXC"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/qizhipei/3d-molt5"
      },
      {
        "label": "Weights",
        "url": "https://huggingface.co/QizhiPei/3d-molt5-base"
      }
    ],
    "status": "Open MIT code + public pretrained and fine-tuned weights",
    "io": "SMILES/3D/text -> generated text or molecule",
    "use_cases": "3D-aware molecule-text tasks, molecular design with descriptions",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://openreview.net/forum?id=eGqQyTAbXC",
      "https://github.com/qizhipei/3d-molt5",
      "https://huggingface.co/QizhiPei/3d-molt5-base"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Status is stale: official pretrained and fine-tuned weights now exist, including MIT-licensed 3d-molt5-base. Paper, repo, weights.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-mole",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "MolE",
    "description": "Recursion chemistry foundation model combining geometric deep learning and transformers.",
    "paper_links": [
      {
        "label": "Paper",
        "url": "https://www.nature.com/articles/s41467-024-53751-y"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/recursionpharma/mole_public"
      }
    ],
    "status": "Open code, weights not released (CC-BY-NC); non-commercial terms apply to at least one artifact or access route",
    "io": "Molecule structure -> embeddings/property predictions",
    "use_cases": "Low-data property prediction, drug discovery representation learning",
    "year": "",
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-024-53751-y",
      "https://github.com/recursionpharma/mole_public"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "Primary paper; artifact. Action: replace the duplicated repository entry in paper_links with the actual paper.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-geneformer",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "Geneformer",
    "description": "Transformer pretrained on large human single-cell transcriptome corpora for network biology.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-023-06139-9"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/ctheodoris/Geneformer"
      },
      {
        "label": "NVIDIA BioNeMo docs",
        "url": "https://docs.nvidia.com/bionemo-framework/latest/models/geneformer/"
      }
    ],
    "status": "Model hub",
    "io": "Ranked/selected expressed genes per cell -> cell/gene embeddings, predictions",
    "use_cases": "In silico perturbation, dosage sensitivity, target discovery, classification",
    "year": "2023",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41586-023-06139-9",
      "https://huggingface.co/ctheodoris/Geneformer",
      "https://docs.nvidia.com/bionemo-framework/latest/models/geneformer/"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · model",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-geneformerv2",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "GeneformerV2",
    "description": "Scaled and quantized Geneformer family for resource-efficient network biology.",
    "paper_links": [
      {
        "label": "Nature Computational Science",
        "url": "https://www.nature.com/articles/s43588-026-00972-4"
      }
    ],
    "code_links": [
      {
        "label": "Geneformer hub",
        "url": "https://huggingface.co/ctheodoris/Geneformer"
      }
    ],
    "status": "Model hub",
    "io": "scRNA cell profile -> embeddings/perturbation outputs",
    "use_cases": "Efficient Geneformer workflows, larger corpora, network predictions",
    "year": "2026",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s43588-026-00972-4",
      "https://huggingface.co/ctheodoris/Geneformer"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "add a version-specific GeneformerV2 artifact locator and document boundary from the Geneformer family record.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-scgpt",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "scGPT",
    "description": "Generative pre-trained transformer for single-cell and multi-omics analysis.",
    "paper_links": [
      {
        "label": "Nature Methods",
        "url": "https://www.nature.com/articles/s41592-024-02201-0"
      },
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2023.04.30.538439v2"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/bowang-lab/scGPT"
      },
      {
        "label": "docs",
        "url": "https://scgpt.readthedocs.io/en/latest/introduction.html"
      },
      {
        "label": "CZI VCP",
        "url": "https://virtualcellmodels.cziscience.com/model/scgpt"
      }
    ],
    "status": "Open code + weights",
    "io": "AnnData/count matrix -> embeddings, reconstructed genes, task predictions",
    "use_cases": "Annotation, integration, perturbation prediction, gene network inference",
    "year": "2023",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-024-02201-0",
      "https://www.biorxiv.org/content/10.1101/2023.04.30.538439v2",
      "https://github.com/bowang-lab/scGPT",
      "https://scgpt.readthedocs.io/en/latest/introduction.html",
      "https://virtualcellmodels.cziscience.com/model/scgpt"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-scfoundation",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "scFoundation",
    "description": "Large single-cell transcriptomics model trained on more than 50M cells.",
    "paper_links": [
      {
        "label": "Nature Methods",
        "url": "https://www.nature.com/articles/s41592-024-02305-7"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/biomap-research/scFoundation"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/genbio-ai/scFoundation"
      }
    ],
    "status": "Open code + weights",
    "io": "Gene expression vector -> embeddings/reconstruction/task heads",
    "use_cases": "Cell annotation, drug response, gene module inference",
    "year": "2024",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-024-02305-7",
      "https://github.com/biomap-research/scFoundation",
      "https://huggingface.co/genbio-ai/scFoundation"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-uce",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "UCE",
    "description": "Universal Cell Embeddings model trained across human and multiple species.",
    "paper_links": [
      {
        "label": "Final paper",
        "url": "https://www.nature.com/articles/s41586-026-10689-z"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/snap-stanford/UCE"
      },
      {
        "label": "CZI VCP",
        "url": "https://virtualcellmodels.cziscience.com/model/uce"
      }
    ],
    "status": "Open code + weights",
    "io": "AnnData expression profile -> universal cell embedding",
    "use_cases": "Cross-species/cross-tissue cell search, annotation, atlas alignment",
    "year": "2023",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41586-026-10689-z",
      "https://github.com/snap-stanford/UCE",
      "https://virtualcellmodels.cziscience.com/model/uce"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "UCE now has a final Nature article published July 2026; replace the preprint as primary while preserving it as history. Nature paper, repo.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-cellplm",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "CellPLM",
    "description": "Cell language model that encodes cell-cell relationships beyond isolated cells.",
    "paper_links": [
      {
        "label": "ICLR",
        "url": "https://openreview.net/forum?id=BKXvPDekud"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/OmicsML/CellPLM"
      }
    ],
    "status": "Open code + weights",
    "io": "Tissue/cell collection expression data -> cell embeddings/task outputs",
    "use_cases": "Annotation, spatial-informed tasks, fast embedding",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://openreview.net/forum?id=BKXvPDekud",
      "https://github.com/OmicsML/CellPLM"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-scprint",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "scPRINT",
    "description": "Large cell model trained on more than 50M cells for gene network inference.",
    "paper_links": [
      {
        "label": "Nature Communications",
        "url": "https://www.nature.com/articles/s41467-025-58699-1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/cantinilab/scPRINT"
      },
      {
        "label": "CZI VCP",
        "url": "https://virtualcellmodels.cziscience.com/model/scprint"
      }
    ],
    "status": "Open code + weights",
    "io": "scRNA count data -> embeddings, gene networks, denoised outputs",
    "use_cases": "Gene regulatory network inference, denoising, label prediction",
    "year": "2025",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-025-58699-1",
      "https://github.com/cantinilab/scPRINT",
      "https://virtualcellmodels.cziscience.com/model/scprint"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-cellfm",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "CellFM",
    "description": "800M-parameter foundation model trained on about 100M human cells.",
    "paper_links": [
      {
        "label": "Nature Communications",
        "url": "https://www.nature.com/articles/s41467-025-59926-5"
      },
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.06.04.597369v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/biomed-AI/CellFM"
      }
    ],
    "status": "Open code + weights/data; non-commercial terms apply to at least one artifact or access route",
    "io": "Single-cell expression -> embeddings/task predictions",
    "use_cases": "Annotation, disease-state analysis, gene function prediction",
    "year": "2024",
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-025-59926-5",
      "https://www.biorxiv.org/content/10.1101/2024.06.04.597369v1",
      "https://github.com/biomed-AI/CellFM"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact. Action: set noncommercial:true; official terms are CC BY-NC-ND 4.0.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-transcriptformer",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "TranscriptFormer",
    "description": "Cross-species generative cell atlas trained on up to 112M cells across 12 species.",
    "paper_links": [
      {
        "label": "Science",
        "url": "https://www.science.org/doi/10.1126/science.aec8514"
      },
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.04.25.650731v2.full-text"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/czi-ai/transcriptformer"
      },
      {
        "label": "CZI VCP",
        "url": "https://virtualcellmodels.cziscience.com/model/transcriptformer"
      }
    ],
    "status": "Open code + weights",
    "io": "Single-cell transcriptome -> embeddings/generative outputs",
    "use_cases": "Cross-species atlas modeling, cell-state representation, virtual-cell workflows",
    "year": "2025",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.science.org/doi/10.1126/science.aec8514",
      "https://www.biorxiv.org/content/10.1101/2025.04.25.650731v2.full-text",
      "https://github.com/czi-ai/transcriptformer",
      "https://virtualcellmodels.cziscience.com/model/transcriptformer"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-aido-cell",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "AIDO.Cell",
    "description": "Transcriptome-scale single-cell foundation model capable of full human transcriptome context.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.11.28.625303v1"
      },
      {
        "label": "AIDO.Cell model card",
        "url": "https://virtualcellmodels.cziscience.com/model/aido-cell"
      }
    ],
    "code_links": [
      {
        "label": "ModelGenerator",
        "url": "https://github.com/genbio-ai/modelgenerator"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/genbio-ai/AIDO.Cell-10M"
      }
    ],
    "status": "Model hub + toolkit; non-commercial terms apply to at least one artifact or access route",
    "io": "Full transcriptome expression vector -> dense representation/task outputs",
    "use_cases": "Target ID, in silico knockout, transcriptome-scale representations",
    "year": "2024",
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.11.28.625303v1",
      "https://virtualcellmodels.cziscience.com/model/aido-cell",
      "https://github.com/genbio-ai/modelgenerator",
      "https://huggingface.co/genbio-ai/AIDO.Cell-10M",
      "https://github.com/genbio-ai/ModelGenerator/blob/main/LICENSE"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "noncommercial:false and generic hub wording omit the GenBio AI Community License, which limits code, weights, derivatives, and outputs to non-commercial purposes. Paper, model, licence.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-scimilarity",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "SCimilarity",
    "description": "Cell atlas foundation model for scalable search of similar human cells.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-024-08411-y"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Genentech/scimilarity"
      }
    ],
    "status": "Open code + weights",
    "io": "scRNA profile -> embedding / nearest atlas cells",
    "use_cases": "Cell-state search, annotation, reference mapping",
    "year": "2024",
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41586-024-08411-y",
      "https://github.com/Genentech/scimilarity"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-scbert",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "scBERT",
    "description": "Early BERT-style scRNA model for cell type annotation.",
    "paper_links": [
      {
        "label": "Nature Machine Intelligence",
        "url": "https://www.nature.com/articles/s42256-022-00534-z"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/TencentAILabHealthcare/scBERT"
      }
    ],
    "status": "Open code + weights",
    "io": "scRNA expression ranks -> cell type prediction",
    "use_cases": "Cell annotation, transfer learning baseline",
    "year": "2022",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s42256-022-00534-z",
      "https://github.com/TencentAILabHealthcare/scBERT"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-scmulan",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "scMulan",
    "description": "Multitask generative pre-trained model for single-cell analysis.",
    "paper_links": [
      {
        "label": "RECOMB / Springer",
        "url": "https://link.springer.com/chapter/10.1007/978-1-0716-3989-4_57"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/SuperBianC/scMulan"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/deeplife/scmulan_model"
      },
      {
        "label": "Current checkpoint",
        "url": "https://cloud.tsinghua.edu.cn/f/2250c5df51034b2e9a85/?dl=1"
      }
    ],
    "status": "Public code + official externally hosted checkpoint; previous model-hub link is unavailable",
    "io": "scRNA/multi-task inputs -> annotation/integration/generation outputs",
    "use_cases": "Multi-task single-cell analysis",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://link.springer.com/chapter/10.1007/978-1-0716-3989-4_57",
      "https://github.com/SuperBianC/scMulan",
      "https://huggingface.co/deeplife/scmulan_model",
      "https://cloud.tsinghua.edu.cn/f/2250c5df51034b2e9a85/?dl=1"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Listed HF artifact is gone, but official repo still provides a checkpoint through Tsinghua Cloud. Replace the dead link; status remains code + weights. Paper, repo, current checkpoint, dead listed HF.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-sccello",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "scCello",
    "description": "Cell-ontology guided transcriptome foundation model.",
    "paper_links": [
      {
        "label": "NeurIPS",
        "url": "https://neurips.cc/virtual/2024/poster/94537"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/DeepGraphLearning/scCello"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/katarinayuan/scCello-zeroshot"
      }
    ],
    "status": "Open code + weights",
    "io": "scRNA expression + ontology context -> cell representations/labels",
    "use_cases": "Ontology-aware annotation",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://neurips.cc/virtual/2024/poster/94537",
      "https://github.com/DeepGraphLearning/scCello",
      "https://huggingface.co/katarinayuan/scCello-zeroshot"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "MIT project and official public checkpoint match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-atacformer",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "Atacformer",
    "description": "Transformer foundation model for ATAC-seq analysis and interpretation.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.11.03.685753v1"
      }
    ],
    "code_links": [
      {
        "label": "Documentation",
        "url": "https://docs.bedbase.org/atacformer/"
      },
      {
        "label": "Weights",
        "url": "https://huggingface.co/databio/atacformer-base-hg38"
      }
    ],
    "status": "Public code/documentation and model weights; exact reuse licence requires confirmation",
    "io": "ATAC-seq peaks/accessibility profiles -> embeddings/predictions",
    "use_cases": "Chromatin accessibility modeling, regulatory interpretation",
    "year": "2025",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.11.03.685753v1",
      "https://docs.bedbase.org/atacformer/",
      "https://huggingface.co/databio/atacformer-base-hg38",
      "https://pmc.ncbi.nlm.nih.gov/articles/PMC12637716/"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Current paper; official documentation; weights. Action: replace the stale paper-only description, add the official artifacts, and record public code/weights while holding any stronger reuse claim until an exact licence is captured.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-cell2sentence",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "Cell2Sentence",
    "description": "Representation strategy treating cells as sentences for LLM-compatible biology workflows.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2023.09.11.557287v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/vandijklab/cell2sentence"
      }
    ],
    "status": "Open code + weights/workflows",
    "io": "scRNA profile -> text-like sequence or LLM-compatible representation",
    "use_cases": "Cell annotation, LLM-assisted single-cell reasoning",
    "year": "2023",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2023.09.11.557287v1",
      "https://github.com/vandijklab/cell2sentence"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "document family boundary against C2S-Scale or merge under a family record.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-chatcell",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "ChatCell",
    "description": "Natural-language interface for single-cell analysis.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2402.08303"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/zjunlp/ChatCell"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/zjunlp/chatcell-large"
      }
    ],
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "io": "User question + single-cell data/context -> analysis narrative/outputs",
    "use_cases": "Interactive single-cell analysis, analyst assistance",
    "year": "2024",
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2402.08303",
      "https://github.com/zjunlp/ChatCell",
      "https://huggingface.co/zjunlp/chatcell-large",
      "https://github.com/zjunlp/ChatCell/blob/main/LICENSE"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Repository is CC BY-NC-SA 4.0, but noncommercial:false. Paper, repo, licence.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-biot5",
    "date_added": "2026-06-16",
    "modality": "multimodal",
    "name": "BioT5",
    "description": "Text-molecule-protein style sequence-to-sequence model integrating biology, chemistry, and language.",
    "paper_links": [
      {
        "label": "EMNLP 2023",
        "url": "https://aclanthology.org/2023.emnlp-main.70/"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/QizhiPei/BioT5"
      }
    ],
    "status": "Open code + weights",
    "io": "Text/SMILES/protein tokens -> generated text/sequence/task output",
    "use_cases": "Cross-modal bio/chem text tasks, molecule/protein captioning",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://aclanthology.org/2023.emnlp-main.70/",
      "https://github.com/QizhiPei/BioT5"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-biot5-plus",
    "date_added": "2026-06-16",
    "modality": "multimodal",
    "name": "BioT5+",
    "description": "Expanded BioT5 model with IUPAC integration and multi-task tuning.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2402.17810"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/QizhiPei/BioT5"
      }
    ],
    "status": "Open code + weights",
    "io": "Text + biological/chemical sequence tokens -> generated outputs",
    "use_cases": "Generalized bio/chem language tasks",
    "year": "2024",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2402.17810",
      "https://github.com/QizhiPei/BioT5"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "official code and pretrained/fine-tuned models are released. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-protst",
    "date_added": "2026-06-16",
    "modality": "multimodal",
    "name": "ProtST",
    "description": "Protein sequence and biomedical text contrastive model.",
    "paper_links": [
      {
        "label": "ICML 2023 (PMLR)",
        "url": "https://proceedings.mlr.press/v202/xu23t.html"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/DeepGraphLearning/ProtST"
      }
    ],
    "status": "Open code + weights",
    "io": "Protein sequence + text descriptions -> aligned embeddings",
    "use_cases": "Text-guided protein retrieval/function annotation",
    "year": "2023",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://proceedings.mlr.press/v202/xu23t.html",
      "https://github.com/DeepGraphLearning/ProtST"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-proteindt",
    "date_added": "2026-06-16",
    "modality": "multimodal",
    "name": "ProteinDT",
    "description": "Text-guided protein design framework.",
    "paper_links": [
      {
        "label": "Nat Mach Intell",
        "url": "https://www.nature.com/articles/s42256-025-01011-z"
      },
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2302.04611"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/chao1224/ProteinDT"
      }
    ],
    "status": "Open code + weights",
    "io": "Natural-language prompt -> protein representation/design candidates",
    "use_cases": "Text-conditioned protein design",
    "year": "2023",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s42256-025-01011-z",
      "https://arxiv.org/abs/2302.04611",
      "https://github.com/chao1224/ProteinDT"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "published paper, MIT code, and official checkpoints match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-protrek",
    "date_added": "2026-06-16",
    "modality": "multimodal",
    "name": "ProTrek",
    "description": "Tri-modal contrastive learning across protein sequence, structure, and text.",
    "paper_links": [
      {
        "label": "Final paper",
        "url": "https://www.nature.com/articles/s41587-025-02836-0"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/westlake-repl/ProTrek"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/westlake-repl/ProTrek_650M"
      }
    ],
    "status": "Open code + weights",
    "io": "Sequence/structure/text -> shared embeddings",
    "use_cases": "Protein retrieval, annotation, multimodal search",
    "year": "2024",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41587-025-02836-0",
      "https://github.com/westlake-repl/ProTrek",
      "https://huggingface.co/westlake-repl/ProTrek_650M"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Replace bioRxiv with final Nature Biotechnology article. Final paper, repo, model.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-galactica",
    "date_added": "2026-06-16",
    "modality": "multimodal",
    "name": "Galactica",
    "description": "General scientific LLM sometimes used for scientific/bio text, but withdrawn from original public demo.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2211.09085"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/paperswithcode/galai"
      }
    ],
    "status": "Open code + weights, caution; non-commercial terms apply to at least one artifact or access route",
    "io": "Scientific text -> generated text",
    "use_cases": "Literature/text assistance; not a biology-sequence model",
    "year": "2022",
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2211.09085",
      "https://github.com/paperswithcode/galai"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P1",
    "audit_note": "set NC true for model weights; retain Apache code and withdrawn-demo caution separately.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-czi-virtual-cells-platform",
    "date_added": "2026-06-16",
    "modality": "platform",
    "name": "Biohub AI-Supported Virtual Cells Platform",
    "description": "Biohub-operated platform for cell foundation models, datasets, benchmarking, model cards, and AI Workspace workflows.",
    "paper_links": [],
    "code_links": [
      {
        "label": "Model catalog",
        "url": "https://virtualcellmodels.cziscience.com/models"
      },
      {
        "label": "AI Workspace",
        "url": "https://virtualcellmodels.cziscience.com/ai-workspace"
      }
    ],
    "status": "Web/CLI/model hub",
    "io": "",
    "use_cases": "Useful for Geneformer, scGPT, UCE, scPRINT, TranscriptFormer, AIDO.Cell model cards and quickstarts.",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "aliases": [
      "CZI Virtual Cells Platform",
      "CZI Virtual Cell Models"
    ],
    "date_modified": "2026-07-13",
    "sources": [
      "https://virtualcellmodels.cziscience.com/models",
      "https://virtualcellmodels.cziscience.com/ai-workspace"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md"
  },
  {
    "id": "bfm-helical",
    "date_added": "2026-06-16",
    "modality": "platform",
    "name": "Helical",
    "description": "Python framework wrapping genomics, transcriptomics, and single-cell foundation models.",
    "paper_links": [],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/helicalAI/helical"
      },
      {
        "label": "docs",
        "url": "https://helical.readthedocs.io/"
      }
    ],
    "status": "Platform/framework; no standalone primary model paper identified",
    "io": "",
    "use_cases": "Helpful when comparing models with a unified API; commercial offers also exist.",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://github.com/helicalAI/helical",
      "https://helical.readthedocs.io/"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "No standalone primary paper identified; framework; documentation. Action: explicitly label it as a platform/framework without a standalone paper.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-nvidia-bionemo",
    "date_added": "2026-06-16",
    "modality": "platform",
    "name": "NVIDIA BioNeMo",
    "description": "Framework, recipes, NIM microservices, and model collection for DNA/RNA/protein/small-molecule workflows.",
    "paper_links": [],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/NVIDIA/BioNeMo"
      },
      {
        "label": "framework",
        "url": "https://github.com/NVIDIA/bionemo-framework"
      },
      {
        "label": "docs",
        "url": "https://docs.nvidia.com/bionemo-framework/1.10/"
      }
    ],
    "status": "Open framework + NGC/commercial services",
    "io": "",
    "use_cases": "Includes or wraps DNABERT, Geneformer, ESM, MegaMolBART/MolMIM-like workflows, and newer BioNeMo models.",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://github.com/NVIDIA/BioNeMo",
      "https://github.com/NVIDIA/bionemo-framework",
      "https://docs.nvidia.com/bionemo-framework/1.10/",
      "https://docs.nvidia.com/bionemo-framework/latest/"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "framework · docs",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-aido-aido-modelgenerator",
    "date_added": "2026-06-16",
    "modality": "platform",
    "name": "AIDO / AIDO.ModelGenerator",
    "description": "GenBio AI software stack for adapting multiscale biological foundation models.",
    "paper_links": [],
    "code_links": [
      {
        "label": "AIDO",
        "url": "https://github.com/genbio-ai/AIDO"
      },
      {
        "label": "ModelGenerator",
        "url": "https://github.com/genbio-ai/modelgenerator"
      },
      {
        "label": "docs",
        "url": "https://genbio-ai.github.io/ModelGenerator/"
      }
    ],
    "status": "Open toolkit + model hub; non-commercial terms apply to at least one artifact or access route",
    "io": "",
    "use_cases": "Covers AIDO DNA/RNA/protein/cell model families and downstream adapters.",
    "year": "",
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://github.com/genbio-ai/AIDO",
      "https://github.com/genbio-ai/modelgenerator",
      "https://genbio-ai.github.io/ModelGenerator/",
      "https://raw.githubusercontent.com/genbio-ai/ModelGenerator/main/LICENSE"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "No standalone primary paper identified; toolkit; licence. Action: set noncommercial:true; identify the GenBio AI Community License as non-commercial and state that this is a platform/toolkit.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-alphafold-server",
    "date_added": "2026-06-16",
    "modality": "platform",
    "name": "AlphaFold Server",
    "description": "Hosted structure prediction for proteins and biomolecular complexes.",
    "paper_links": [],
    "code_links": [
      {
        "label": "AlphaFold Server",
        "url": "https://alphafoldserver.com/"
      }
    ],
    "status": "Hosted server; outputs restricted to non-commercial use with additional prohibited uses",
    "io": "",
    "use_cases": "Practical entry point for users who do not need local AF3 execution.",
    "year": "",
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://alphafoldserver.com/",
      "https://alphafoldserver.com/faq",
      "https://alphafoldserver.com/output-terms"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Status says only “terms apply,” while use and outputs are explicitly non-commercial and additionally prohibit docking/screening and structure-model training. Set explicit NC wording/flag. FAQ, output terms.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-hugging-face-biology-orgs",
    "date_added": "2026-06-16",
    "modality": "platform",
    "name": "Hugging Face biology orgs",
    "description": "Model-card access to many pretrained models.",
    "paper_links": [],
    "code_links": [
      {
        "label": "EvolutionaryScale",
        "url": "https://huggingface.co/EvolutionaryScale"
      },
      {
        "label": "InstaDeepAI",
        "url": "https://huggingface.co/InstaDeepAI"
      },
      {
        "label": "Rostlab",
        "url": "https://huggingface.co/Rostlab"
      },
      {
        "label": "GenBio AI",
        "url": "https://huggingface.co/genbio-ai"
      },
      {
        "label": "MultiMolecule",
        "url": "https://huggingface.co/multimolecule"
      }
    ],
    "status": "Model hub",
    "io": "",
    "use_cases": "Always check whether a model card is official or an independently validated mirror.",
    "year": "",
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://huggingface.co/EvolutionaryScale",
      "https://huggingface.co/InstaDeepAI",
      "https://huggingface.co/Rostlab",
      "https://huggingface.co/genbio-ai",
      "https://huggingface.co/multimolecule"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "platform classification is acceptable, but it is a curated org list rather than one official biological hub. Add explicit selection/provenance rationale.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-alphagenome",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "AlphaGenome",
    "description": "DeepMind's unified DNA sequence model that takes up to 1 Mb of DNA and predicts thousands of regulatory tracks (gene expression, splicing, chromatin accessibility, histone marks, TF binding, contact maps) at single-base resolution, plus variant-effect scores.",
    "io": "DNA sequence up to 1 Mb -> multimodal genomic tracks + variant effect scores at base-pair resolution",
    "status": "Open research code + gated public weights under non-commercial terms; free non-commercial API",
    "use_cases": "Regulatory variant effect prediction, splicing/expression impact, enhancer-promoter analysis, in silico mutagenesis",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-025-10014-0"
      }
    ],
    "code_links": [
      {
        "label": "Research code",
        "url": "https://github.com/google-deepmind/alphagenome_research"
      },
      {
        "label": "Kaggle weights",
        "url": "https://www.kaggle.com/models/google/alphagenome"
      },
      {
        "label": "Model terms",
        "url": "https://deepmind.google.com/science/alphagenome/model-terms"
      }
    ],
    "canonical": true,
    "noncommercial": true,
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41586-025-10014-0",
      "https://github.com/google-deepmind/alphagenome_research",
      "https://www.kaggle.com/models/google/alphagenome",
      "https://deepmind.google.com/science/alphagenome/model-terms"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-borzoi",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "Borzoi",
    "description": "Calico convolutional model that predicts base-resolution RNA-seq coverage (plus ChIP/DNase/ATAC/CAGE) from ~524 kb of DNA as a unifying model of gene regulation; built on the Enformer architecture.",
    "io": "~524 kb DNA window -> RNA-seq coverage at 32 bp resolution across many assays/tissues; derived variant scores",
    "status": "Open code + weights",
    "use_cases": "eQTL/sQTL/paQTL/ipaQTL variant scoring, transcription/splicing/polyadenylation modeling, regulatory variant interpretation",
    "year": "2024",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41588-024-02053-6"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/calico/borzoi"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41588-024-02053-6",
      "https://github.com/calico/borzoi"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-sei",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "Sei",
    "description": "Deep convolutional model from the Troyanskaya lab predicting 21,907 chromatin profiles (>1,300 cell lines/tissues) and organizing them into 40 interpretable regulatory 'sequence classes' for variant interpretation.",
    "io": "DNA sequence (~4 kb context) -> 21,907 chromatin-profile predictions + sequence-class scores; variant effect on regulatory activity",
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Noncoding variant interpretation, regulatory activity classification, disease/trait genetics, sequence-class annotation",
    "year": "2022",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41588-022-01102-2"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/FunctionLab/sei-framework"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41588-022-01102-2",
      "https://github.com/FunctionLab/sei-framework",
      "https://raw.githubusercontent.com/FunctionLab/sei-framework/main/LICENSE.txt"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact; licence. Action: set noncommercial:true; redistribution/use is limited to academic and research purposes.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-gpn-genomic-pre-trained-network",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "GPN (Genomic Pre-trained Network)",
    "description": "Self-supervised convolutional DNA language model that learns gene structure and motifs unsupervised and gives zero-shot genome-wide variant effect scores (original plant/Arabidopsis + Brassicales version).",
    "io": "DNA sequence -> per-nucleotide log-likelihoods/embeddings; zero-shot variant effect scores",
    "status": "Open code + weights",
    "use_cases": "Unsupervised variant effect prediction, evolutionary constraint, motif/gene-structure discovery, plant and other genomes",
    "year": "2023",
    "paper_links": [
      {
        "label": "PNAS",
        "url": "https://www.pnas.org/doi/abs/10.1073/pnas.2311219120"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/songlab-cal/gpn"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.pnas.org/doi/abs/10.1073/pnas.2311219120",
      "https://github.com/songlab-cal/gpn"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-generator",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "GENERator",
    "description": "Long-context generative (decoder/transformer) genomic foundation model with 98 kb context, ~1.2B params, pretrained on 386 Gbp of eukaryotic DNA (RefSeq) with 6-mer tokenization.",
    "io": "DNA prompt/context -> generated DNA sequences (coding regions, cis-regulatory elements), embeddings, likelihoods",
    "status": "Open code + weights",
    "use_cases": "Synthetic enhancer/CRE design, protein-coding sequence generation, long-context genomic representation, variant analysis",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2502.07272"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/GenerTeam/GENERator"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2502.07272",
      "https://github.com/GenerTeam/GENERator"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "MIT repository and current v1/v2 model cards match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-genomeocean",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "GenomeOcean",
    "description": "Efficient 4-billion-parameter generative genome foundation model trained on 600+ Gbp of contigs from 220 TB of diverse metagenomic co-assemblies, using BPE tokenization.",
    "io": "DNA sequence -> embeddings, next-token likelihoods, generated genomic/metagenomic sequences",
    "status": "Model hub",
    "use_cases": "Metagenomic sequence modeling, biosynthetic gene cluster generation, genome representation, sequence generation",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.01.30.635558v1"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/pGenomeOcean/GenomeOcean-4B"
      },
      {
        "label": "Code",
        "url": "https://github.com/jgi-genomeocean/genomeocean"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.01.30.635558v1",
      "https://huggingface.co/pGenomeOcean/GenomeOcean-4B",
      "https://github.com/jgi-genomeocean/genomeocean"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "Paper; code; weights. Action: add the official source repository and upgrade the model-hub-only description.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-megadna",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "megaDNA",
    "description": "Multiscale hierarchical transformer pretrained on ~100k unannotated bacteriophage genomes with nucleotide-level tokenization; can generate de novo phage genomes up to 96 kb.",
    "io": "DNA context -> generated genome-scale sequences (up to 96 kb), zero-shot predictions (essential genes, variant effects, regulatory activity, taxonomy)",
    "status": "Open code + weights",
    "use_cases": "Whole-phage-genome generation/design, essential-gene and variant-effect prediction, regulatory element and taxonomy prediction",
    "year": "2024",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-024-53759-4"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/lingxusb/megaDNA"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-024-53759-4",
      "https://github.com/lingxusb/megaDNA"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-dnagpt",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "DNAGPT",
    "description": "Generalized GPT-style DNA model pretrained on 200B+ bp from all mammals, augmented with numerical regression (GC content) and binary-classification (sequence order) objectives and a multi-modal token language.",
    "io": "DNA sequence (+numbers) -> embeddings, classifications, regressions, generated sequences",
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Genomic signal/region recognition, mRNA abundance regression, artificial genome generation, multi-task DNA analysis",
    "year": "2023",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2307.05628"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/TencentAILabHealthcare/DNAGPT"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2307.05628",
      "https://github.com/TencentAILabHealthcare/DNAGPT"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact. Action: set noncommercial:true; official licence is PolyForm Noncommercial 1.0.0.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-grover-dna-language-model",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "GROVER (DNA language model)",
    "description": "BERT-style human-genome language model (Poetsch lab, TU Dresden) trained with byte-pair encoding whose vocabulary is selected via a next-k-mer prediction task to capture genomic 'grammar'. Name = Genome Rules Obtained Via Extracted Representations.",
    "io": "DNA sequence -> embeddings/masked-token predictions; fine-tuned task outputs",
    "status": "Public model hub artifact; exact model reuse licence not declared",
    "use_cases": "Genome element identification, protein-DNA binding prediction, sequence embeddings, genome-stability analysis",
    "year": "2024",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s42256-024-00872-0"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/PoetschLab/GROVER"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s42256-024-00872-0",
      "https://huggingface.co/PoetschLab/GROVER"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "Model hub and paper match, but the HF card exposes no explicit model licence. Preserve as accessible-with-unclear-reuse, not implicitly permissive. Paper, model.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-agront-agronomic-nucleotide-transformer",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "AgroNT (Agronomic Nucleotide Transformer)",
    "description": "1B-parameter DNA foundation model from InstaDeep trained on 48 edible/crop plant genomes for plant regulatory and trait genomics; a Nucleotide-Transformer sibling.",
    "io": "Plant DNA sequence (6-mer tokens, ~6 kb) -> embeddings; fine-tuned regulatory/expression/TFBS/variant predictions",
    "status": "Model hub; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Crop regulatory annotation, tissue-specific expression prediction, TF binding site and promoter/terminator strength, trait/crop-protection genomics",
    "year": "2024",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s42003-024-06465-2"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/InstaDeepAI/agro-nucleotide-transformer-1b"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s42003-024-06465-2",
      "https://huggingface.co/InstaDeepAI/agro-nucleotide-transformer-1b"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P1",
    "audit_note": "set NC true and record CC BY-NC-SA 4.0 model terms.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-plantcaduceus-plantcad-plantcad2",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "PlantCaduceus (PlantCAD / PlantCAD2)",
    "description": "Plant genomic foundation-model family; PlantCAD2 expands the successor model to 65 angiosperm species and 8,192-base context for transferable plant-regulatory representations.",
    "io": "Plant DNA window (512 bp) -> embeddings/masked-token predictions; zero-shot variant scores, fine-tuned annotations",
    "status": "Open code + public weights; checkpoint reuse terms unresolved",
    "use_cases": "Cross-species splice/TIS/TTS annotation, deleterious-variant identification, transfer learning across diverged plant species (e.g. Arabidopsis->maize)",
    "year": "2024",
    "paper_links": [
      {
        "label": "bioRxiv (PlantCAD2 v3)",
        "url": "https://www.biorxiv.org/content/10.1101/2025.08.27.672609v3.full"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/plantcad/plantcad"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/collections/kuleshov-group/plantcad2-67e437e241a382671371a572"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.08.27.672609v3.full",
      "https://github.com/plantcad/plantcad",
      "https://huggingface.co/collections/kuleshov-group/plantcad2-67e437e241a382671371a572"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-specieslm-species-aware-dna-language-model",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "SpeciesLM (species-aware DNA language model)",
    "description": "Species-aware masked DNA language model from the Gagneur lab trained on 800+ species spanning ~500M years to capture regulatory elements and their evolution alignment-free.",
    "io": "DNA sequence + species token -> masked-nucleotide reconstruction, embeddings/representations",
    "status": "Open code + weights",
    "use_cases": "Regulatory element/motif discovery, TF/RBP motif detection, evolutionary conservation modeling, representation learning",
    "year": "2024",
    "paper_links": [
      {
        "label": "Springer",
        "url": "https://link.springer.com/article/10.1186/s13059-024-03221-x"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/gagneurlab/SpeciesLM"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://link.springer.com/article/10.1186/s13059-024-03221-x",
      "https://github.com/gagneurlab/SpeciesLM"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-glm-genomic-language-model",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "gLM (genomic language model)",
    "description": "Transformer (19-layer) trained on ~7M metagenomic scaffolds over ESM-2 protein embeddings to learn genomic context, operon/co-regulation structure and contextualized gene function.",
    "io": "Contig of 15-30 genes (as ESM-2 protein embeddings) -> contextualized gene embeddings; masked-gene predictions",
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Protein function/co-regulation prediction, operon detection, MAG binning/assembly correction, genomic-context transfer learning",
    "year": "2024",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-024-46947-9"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/y-hwang/gLM"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-024-46947-9",
      "https://github.com/y-hwang/gLM"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact. Action: set noncommercial:true; official licence limits use to academic and non-commercial research.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-omnina",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "OmniNA",
    "description": "Generative (decoder, Llama-based) foundation model jointly trained on 91.7M NCBI nucleotide sequences and their natural-language annotations to unify sequence understanding with text; 66M-1.7B params.",
    "io": "Nucleotide sequence (+/- text prompt) -> generated annotations/text, task predictions, embeddings",
    "status": "Open code + weights",
    "use_cases": "Sequence annotation, functional element recognition, species classification, sequence-to-text tasks, multi-task nucleotide analysis",
    "year": "2024",
    "paper_links": [
      {
        "label": "Oxford Acad.",
        "url": "https://academic.oup.com/nar/article/54/6/gkag083/8528802"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/xilinshen/OmniNA"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://academic.oup.com/nar/article/54/6/gkag083/8528802",
      "https://github.com/xilinshen/OmniNA"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-orca",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "Orca",
    "description": "Sequence-based deep-learning model (Zhou lab) predicting multiscale 3D genome organization (TADs, A/B compartments, polycomb interactions, enhancer-promoter contacts) from kilobase to whole-chromosome scale.",
    "io": "DNA sequence (up to whole-chromosome) -> predicted Hi-C/Micro-C 3D contact maps; structural-variant effects",
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "3D genome structure prediction, structural-variant effect modeling, in silico virtual genetic screens, TAD/compartment analysis",
    "year": "2022",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41588-022-01065-4"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/jzhoulab/orca"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41588-022-01065-4",
      "https://github.com/jzhoulab/orca"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact. Action: set noncommercial:true; official licence permits academic-research use only.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-get-general-expression-transformer",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "GET (General Expression Transformer)",
    "description": "Interpretable foundation model of transcriptional regulation across 213 human fetal/adult cell types that predicts gene expression from chromatin accessibility (peak x TF-motif) and sequence.",
    "io": "Cell-type accessibility (peak x TF-motif matrix) + sequence -> predicted gene expression, regulatory syntax, TF interactions",
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Cross-cell-type expression prediction (incl. unseen cell types), regulatory grammar discovery, TF interaction inference, in silico perturbation",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-024-08391-z"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/GET-Foundation/get_model"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41586-024-08391-z",
      "https://github.com/GET-Foundation/get_model"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P1",
    "audit_note": "set NC true; repository is CC BY-NC 4.0.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-decima",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "Decima",
    "description": "Genentech sequence model predicting cell-type- and condition-specific gene expression from genomic sequence, trained on pseudobulked single-cell RNA-seq from 22M+ cells; built on the gReLU framework.",
    "io": "Genomic sequence around a gene -> predicted cell-type/condition-specific expression; regulatory element attributions",
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Single-cell-resolution expression prediction (including unseen genes), regulatory-element discovery, disease/tissue-state expression change analysis",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41592-026-03102-0"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Genentech/decima"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-026-03102-0",
      "https://github.com/Genentech/decima"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact. Action: set noncommercial:true; official licence is the Genentech Non-Commercial Software License.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-scooby",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "scooby",
    "description": "Single-cell-resolution sequence model (Gagneur lab) that fine-tunes Borzoi with a cell-specific decoder to predict scRNA-seq and scATAC-seq genomic profiles directly from DNA.",
    "io": "DNA sequence + cell representation -> predicted single-cell multi-omic coverage (scRNA-seq + scATAC-seq) profiles",
    "status": "Open code + weights",
    "use_cases": "Single-cell multi-omic profile prediction from sequence, cell-state-specific regulatory variant effects, accessibility/expression modeling",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41592-025-02854-5"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/gagneurlab/scooby"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-025-02854-5",
      "https://github.com/gagneurlab/scooby"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "published paper, MIT code, and official model cards match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-segmentnt",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "SegmentNT",
    "description": "InstaDeep segmentation model built on the Nucleotide Transformer encoder + 1D U-Net that predicts the location of 14 classes of genomic elements at single-nucleotide resolution over up to 30 kb (generalizes to 50 kb).",
    "io": "DNA sequence up to 30 kb -> per-nucleotide segmentation across 14 element classes (genes, UTRs, exons/introns, splice sites, promoters, enhancers, CTCF, polyA)",
    "status": "Model hub; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Single-nucleotide genome annotation, regulatory/gene-structure element localization, downstream regulatory genomics",
    "year": "2024",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.03.14.584712v2"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/InstaDeepAI/segment_nt"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.03.14.584712v2",
      "https://huggingface.co/InstaDeepAI/segment_nt"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P1",
    "audit_note": "set NC true and cite CC BY-NC-SA 4.0 model terms.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-janusdna",
    "date_added": "2026-06-16",
    "modality": "dna",
    "name": "JanusDNA",
    "description": "Bidirectional hybrid DNA foundation model using a Mamba-Attention Mixture-of-Experts design that combines autoregressive training efficiency with masked-model bidirectional understanding; processes up to 1 Mb on a single 80GB GPU.",
    "io": "DNA sequence up to ~1 Mb (single 80GB GPU) -> embeddings/representations; fine-tuned task outputs",
    "status": "Open code + weights",
    "use_cases": "Long-range genomic representation learning, regulatory benchmarks (Genomic Benchmarks/NT/DNALongBench tasks), bidirectional promoter modeling",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.17257"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Qihao-Duan/JanusDNA"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2505.17257",
      "https://github.com/Qihao-Duan/JanusDNA"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-orthrus",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "Orthrus",
    "description": "Mamba-based mature-RNA foundation model pretrained by self-supervised contrastive learning over splice isoforms (10 organisms) and orthologous transcripts (400+ mammals, Zoonomia) to learn function/evolution-aware RNA representations. Two sizes: ~1M and ~10M params.",
    "io": "Mature RNA/mRNA transcript sequence -> dense embeddings (and fine-tuned property predictions)",
    "status": "Open code + weights (MIT; 4-track base and 6-track large checkpoints, HF mirrors exist)",
    "use_cases": "mRNA property prediction (half-life, translation, expression), isoform function discrimination, sample-efficient transfer learning",
    "year": "2024",
    "paper_links": [
      {
        "label": "Nature Methods",
        "url": "https://www.nature.com/articles/s41592-026-03064-3"
      },
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.10.10.617658v2"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/bowang-lab/Orthrus"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-026-03064-3",
      "https://www.biorxiv.org/content/10.1101/2024.10.10.617658v2",
      "https://github.com/bowang-lab/Orthrus"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-utr-lm",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "UTR-LM",
    "description": "Semi-supervised 5' UTR language model pretrained on endogenous and random 5' UTRs across species, augmented with secondary-structure and minimum-free-energy supervision to decode untranslated-region function.",
    "io": "5' UTR RNA sequence -> embeddings and predictions of mean ribosome loading (MRL), translation efficiency (TE), mRNA expression level (EL), IRES",
    "status": "Open code + weights (GPL-3.0; checkpoints on Google Drive/CodeOcean/Zenodo)",
    "use_cases": "mRNA/UTR therapeutic design, MRL/TE/expression prediction, IRES detection, wet-lab-validated 5' UTR optimization (211-UTR library)",
    "year": "2024",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s42256-024-00823-9"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/a96123155/UTR-LM"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s42256-024-00823-9",
      "https://github.com/a96123155/UTR-LM"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "published paper, GPL-3.0 code and public checkpoints match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-codonbert",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "CodonBERT",
    "description": "BERT-style mRNA language model that tokenizes coding sequences by codons, pretrained on >10M mRNA sequences from diverse organisms for mRNA design and optimization (Sanofi).",
    "io": "mRNA coding sequence (codon tokens) -> contextual codon embeddings (and downstream regression/prediction heads)",
    "status": "Open code + weights (dual license: separate code and model-artifact licenses; PyTorch weights via download)",
    "use_cases": "mRNA vaccine/therapeutic codon optimization, protein expression prediction, mRNA stability/degradation prediction",
    "year": "2024",
    "paper_links": [
      {
        "label": "Genome Research",
        "url": "https://genome.cshlp.org/content/34/7/1027"
      },
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2023.09.09.556981"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Sanofi-Public/CodonBERT"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "sources": [
      "https://genome.cshlp.org/content/34/7/1027",
      "https://www.biorxiv.org/content/10.1101/2023.09.09.556981",
      "https://github.com/Sanofi-Public/CodonBERT"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-rnaernie",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "RNAErnie",
    "description": "Motif-aware RNA language model (ERNIE-based) that adds RNA-motif-level random masking and RNA-type stop-word tokens during pretraining, plus type-guided fine-tuning for multi-purpose RNA modeling.",
    "io": "RNA sequence -> embeddings/attention maps and task predictions (classification, interaction, structure)",
    "status": "Open code + weights (Google Drive + HF mirrors: multimolecule/rnaernie, WANGNingroci/RNAErnie)",
    "use_cases": "ncRNA classification, RNA-RNA/RNA-protein interaction prediction, secondary structure prediction, general RNA representation",
    "year": "2024",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s42256-024-00836-4"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/CatIIIIIIII/RNAErnie"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s42256-024-00836-4",
      "https://github.com/CatIIIIIIII/RNAErnie"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-birna-bert",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "BiRNA-BERT",
    "description": "117M-parameter ncRNA transformer encoder trained on ~36M ncRNA sequences with adaptive dual tokenization (nucleotide-level NUC plus byte-pair encoding BPE) selected by input length, using ALiBi for context extension.",
    "io": "RNA sequence (NUC or BPE tokens, length-adaptive) -> embeddings and task predictions",
    "status": "Open code + weights (HF collection buetnlpbio/birna-bert)",
    "use_cases": "Short-sequence RNA classification, long-context RNA modeling, fine-grained nucleotide-level structural prediction",
    "year": "2025",
    "paper_links": [
      {
        "label": "Communications Biology",
        "url": "https://www.nature.com/articles/s42003-025-08982-0"
      },
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.07.02.601703"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/buetnlpbio/BiRNA-BERT"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s42003-025-08982-0",
      "https://www.biorxiv.org/content/10.1101/2024.07.02.601703",
      "https://github.com/buetnlpbio/BiRNA-BERT"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-helix-mrna",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "Helix-mRNA",
    "description": "Hybrid Mamba2 state-space + transformer-attention foundation model for full-length mRNA therapeutics, single-nucleotide tokenization with codon separators; effective context ~12,288 tokens at ~10% of comparable transformer params.",
    "io": "Full-length mRNA sequence (5'UTR + CDS + 3'UTR) -> embeddings and fine-tuned property predictions",
    "status": "Open code, gated weights (HF helical-ai/helix-mRNA, CC-BY-NC-SA, login + non-commercial terms)",
    "use_cases": "mRNA therapeutic/vaccine optimization, translation efficiency/stability/degradation analysis across coding and untranslated regions",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2502.13785"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/helicalAI/helical"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2502.13785",
      "https://github.com/helicalAI/helical"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-rhofold-plus",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "RhoFold+",
    "description": "RNA language-model-based deep learning method for single-chain RNA 3D structure prediction, integrating the pretrained RNA-FM language model (trained on ~23.7M RNAs) with MSA and structure modules; end-to-end, fast (~0.14s without MSA search).",
    "io": "RNA sequence (FASTA, optional A3M MSA) -> 3D tertiary structure (PDB + pLDDT), secondary structure, distograms",
    "status": "Open code + weights (Apache-2.0; checkpoint auto-download / cuhkaih/rhofold on HF; public server)",
    "use_cases": "RNA 3D structure prediction, secondary structure inference, structure-based RNA analysis; competitive with top CASP15/RNA-Puzzles methods",
    "year": "2024",
    "paper_links": [
      {
        "label": "Nature Methods",
        "url": "https://www.nature.com/articles/s41592-024-02487-0"
      },
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2207.01586"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ml4bio/RhoFold"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-024-02487-0",
      "https://arxiv.org/abs/2207.01586",
      "https://github.com/ml4bio/RhoFold"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-rnagenesis",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "RNAGenesis",
    "description": "~1B-parameter generalist RNA foundation model unifying sequence representation (BERT-style encoder), de novo functional design (query-based latent compression + diffusion-guided decoder), and 3D structure prediction in one framework.",
    "io": "RNA sequence and/or structural/functional constraints -> embeddings, generated/optimized RNA sequences, predicted 3D structures",
    "status": "Open code + weights (repo zaixizhang/RNAGenesis)",
    "use_cases": "Functional RNA therapeutic design (ASO/siRNA/shRNA/circRNA/aptamer/UTR), inverse folding, RNA 3D structure and de novo structure design",
    "year": "2024",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.12.30.630826v2"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/zaixizhang/RNAGenesis"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.12.30.630826v2",
      "https://github.com/zaixizhang/RNAGenesis"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-lamar",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "LAMAR",
    "description": "12-layer transformer foundation language model for multilayer RNA regulation, pretrained on ~15M genome/transcriptome sequences from 225 mammals and 1569 viruses via masked language modeling.",
    "io": "RNA/pre-mRNA sequence -> embeddings and task predictions (translation efficiency, half-life, splice sites, IRES)",
    "status": "Open code + weights",
    "use_cases": "mRNA translation efficiency and half-life prediction, pre-mRNA splice-site prediction, IRES prediction, RNA regulatory modeling",
    "year": "2025",
    "paper_links": [
      {
        "label": "Genome Biology",
        "url": "https://genomebiology.biomedcentral.com/articles/10.1186/s13059-025-03752-x"
      },
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.10.12.617732v2"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/zhw-e8/LAMAR"
      },
      {
        "label": "GitHub mirror",
        "url": "https://github.com/rnasys/LAMAR"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "sources": [
      "https://genomebiology.biomedcentral.com/articles/10.1186/s13059-025-03752-x",
      "https://www.biorxiv.org/content/10.1101/2024.10.12.617732v2",
      "https://github.com/zhw-e8/LAMAR",
      "https://github.com/rnasys/LAMAR"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-omnigenome",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "OmniGenome",
    "description": "RNA foundation model introducing sequence-structure alignment, enabling bidirectional mapping between RNA sequences and secondary structures (52M and 186M variants).",
    "io": "RNA sequence and/or secondary structure -> embeddings, predicted structure, or designed sequences for a target structure",
    "status": "Open code + weights (HF: yangheng/OmniGenome-186M and yangheng/OmniGenome-52M; ships with OmniGenBench)",
    "use_cases": "RNA secondary structure prediction, RNA design for target structures (74% EternaV2 solved vs up to 3% prior), TF binding prediction, genomic FM benchmarking",
    "year": "2024",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2407.11242"
      },
      {
        "label": "AAAI 2025",
        "url": "https://ojs.aaai.org/index.php/AAAI/article/view/35500"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/yangheng95/OmniGenBench"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2407.11242",
      "https://ojs.aaai.org/index.php/AAAI/article/view/35500",
      "https://github.com/yangheng95/OmniGenBench"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-mrnabert",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "mRNABERT",
    "description": "Universal 86M-parameter mRNA language model pretrained on >18M non-redundant mRNAs with dual tokenization (single-nucleotide for UTRs, codons for CDS) and cross-modality contrastive learning against frozen ProtT5-XL protein embeddings.",
    "io": "mRNA sequence -> embeddings and predictions/designs for 5' UTR, CDS, and RBP sites",
    "status": "Open code + weights (HF: YYLY66/mRNABERT)",
    "use_cases": "End-to-end mRNA sequence design (5' UTR and CDS), RNA-binding-protein site prediction, full-length mRNA property/optimization (R2=0.66 on TE)",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-025-65340-8"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/yyly6/mRNABERT"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-025-65340-8",
      "https://github.com/yyly6/mRNABERT"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-trrosettarna",
    "date_added": "2026-06-16",
    "modality": "rna",
    "name": "trRosettaRNA",
    "description": "Automated RNA 3D structure prediction that feeds MSA and predicted secondary structure into a transformer (RNAformer) to predict inter-nucleotide 1D/2D geometries, then folds 3D models by energy minimization.",
    "io": "RNA sequence (+ auto-generated MSA and secondary structure) -> predicted 2D geometries and 3D tertiary structure",
    "status": "Open code + weights (Apache-2.0; params downloadable from Yang Lab server; public web server)",
    "use_cases": "De novo RNA 3D structure prediction; CASP15/RNA-Puzzles-competitive automated predictions",
    "year": "2023",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-023-42528-4"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/YangLab-SDU/trRosettaRNA2"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-023-42528-4",
      "https://github.com/YangLab-SDU/trRosettaRNA2"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "published paper, Apache code, downloadable parameters, and server match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-rosettafoldna",
    "date_added": "2026-06-16",
    "modality": "complex",
    "name": "RoseTTAFoldNA",
    "description": "Single three-track network extending RoseTTAFold2 to predict 3D structures of nucleic acids and protein-DNA/protein-RNA complexes with per-prediction confidence estimates.",
    "io": "Protein and/or nucleic-acid (RNA/DNA) sequences -> 3D structure of the complex with confidence",
    "status": "Open code + weights (weights downloadable via repo)",
    "use_cases": "Protein-RNA and protein-DNA complex modeling, RNA monomer/dimer structure, design of sequence-specific nucleic-acid-binding proteins",
    "year": "2023",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41592-023-02086-5"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/uw-ipd/RoseTTAFold2NA"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-023-02086-5",
      "https://github.com/uw-ipd/RoseTTAFold2NA"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-saprot",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "SaProt",
    "description": "Structure-aware protein language model that augments amino acids with Foldseek 3Di structure tokens to form a combined sequence-structure (SA) vocabulary; trained on ~40M sequence-structure pairs.",
    "io": "Amino acid sequence + Foldseek 3Di structure tokens -> residue/sequence embeddings, masked-token logits, zero-shot variant scores",
    "status": "Open code + weights",
    "use_cases": "Zero-shot variant effect / fitness prediction (top ProteinGym performer), structure-aware embeddings, function/localization prediction, cross-modal search",
    "year": "2024",
    "paper_links": [
      {
        "label": "OpenReview",
        "url": "https://openreview.net/forum?id=6MRm3G4NiU"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/westlake-repl/SaProt"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://openreview.net/forum?id=6MRm3G4NiU",
      "https://github.com/westlake-repl/SaProt"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-prostt5",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "ProstT5",
    "description": "Bilingual T5 protein language model (fine-tuned from ProtT5-XL-U50 on 17M AlphaFoldDB structures) that translates bidirectionally between amino acid sequence and Foldseek 3Di structure tokens (folding and inverse folding).",
    "io": "Amino acid sequence <-> 3Di structure token string; encoder also yields embeddings",
    "status": "Open code + weights",
    "use_cases": "Fast structure-aware remote homology detection, sequence-to-3Di prediction for structureless Foldseek search, structure-conditioned sequence design, embeddings",
    "year": "2024",
    "paper_links": [
      {
        "label": "Oxford Acad.",
        "url": "https://academic.oup.com/nargab/article/6/4/lqae150/7901286"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/mheinzinger/ProstT5"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://academic.oup.com/nargab/article/6/4/lqae150/7901286",
      "https://github.com/mheinzinger/ProstT5"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-prosst",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "ProSST",
    "description": "Structure-aware protein language model combining GVP-based quantized local-structure tokens (18.8M structures) with a sequence-structure disentangled-attention transformer.",
    "io": "Protein sequence + quantized local-structure tokens -> embeddings, masked-token logits, zero-shot mutation scores",
    "status": "Open code + weights",
    "use_cases": "State-of-the-art zero-shot mutation effect prediction on ProteinGym (~0.504 Spearman), supervised tasks (thermostability, metal-ion binding, localization), structure-aware representation learning",
    "year": "2024",
    "paper_links": [
      {
        "label": "OpenReview",
        "url": "https://openreview.net/forum?id=4Z7RZixpJQ"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/openmedlab/ProSST"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://openreview.net/forum?id=4Z7RZixpJQ",
      "https://github.com/openmedlab/ProSST"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "GPL-3.0 code and official checkpoints match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-amplify",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "AMPLIFY",
    "description": "Efficient ESM-style masked protein language model (120M/350M) from Amgen/Mila/Chandar-lab built to match much larger PLMs at far lower cost by prioritizing data quality over scale.",
    "io": "Amino acid sequence -> residue/sequence embeddings, masked-token logits",
    "status": "Open code + weights",
    "use_cases": "General protein embeddings, fine-tuning for property/function prediction, zero-shot variant scoring, low-cost replacement for large PLMs",
    "year": "2024",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.09.23.614603v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/chandar-lab/AMPLIFY"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.09.23.614603v1",
      "https://github.com/chandar-lab/AMPLIFY"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-carp",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "CARP",
    "description": "Convolutional (ByteNet/CNN) masked protein language model family from Microsoft Research (CARP-640M) showing convolutions are competitive with transformers for protein pretraining; scales linearly with sequence length.",
    "io": "Amino acid sequence -> residue/sequence embeddings, masked-token logits",
    "status": "Open code + weights",
    "use_cases": "Sequence embeddings, structure/function transfer, zero-shot mutation effect prediction, long-sequence modeling without quadratic attention cost; basis for EvoDiff",
    "year": "2024",
    "paper_links": [
      {
        "label": "Paper",
        "url": "https://www.cell.com/cell-systems/fulltext/S2405-4712(24)00029-2"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/microsoft/protein-sequence-models"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.cell.com/cell-systems/fulltext/S2405-4712(24)00029-2",
      "https://github.com/microsoft/protein-sequence-models"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-ligandmpnn",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "LigandMPNN",
    "description": "Inverse-folding sequence design model extending ProteinMPNN to explicitly condition on non-protein atomic context (small molecules, nucleotides, metals); also outputs side-chain conformations.",
    "io": "Protein backbone + ligand/nucleotide/metal atomic context (PDB) -> designed sequences and side-chain conformations",
    "status": "Open code + weights",
    "use_cases": "Sequence design for ligand/DNA/metal-binding sites, enzyme and binder design, redesign of interfaces involving non-protein atoms",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41592-025-02626-1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/dauparas/LigandMPNN"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-025-02626-1",
      "https://github.com/dauparas/LigandMPNN"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Nature Methods paper, MIT code, and weights match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-antifold",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "AntiFold",
    "description": "Antibody-specific inverse-folding model fine-tuned from ESM-IF1 on solved and predicted antibody structures (Deane lab / OPIG).",
    "io": "Antibody backbone structure -> CDR-aware designed sequences and per-residue probabilities",
    "status": "Open code + weights",
    "use_cases": "Antibody CDR sequence design, zero-shot antibody-antigen binding affinity ranking, structure-based antibody optimization",
    "year": "2025",
    "paper_links": [
      {
        "label": "Oxford Acad.",
        "url": "https://academic.oup.com/bioinformaticsadvances/article/5/1/vbae202/8090019"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/oxpig/AntiFold"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://academic.oup.com/bioinformaticsadvances/article/5/1/vbae202/8090019",
      "https://github.com/oxpig/AntiFold"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "paper, BSD repository, and included weights match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-rfdiffusion2",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "RFdiffusion2",
    "description": "Next-generation RoseTTAFold-All-Atom-based diffusion/flow-matching model for de novo enzyme/protein design directly from atom-level functional-group constraints without specifying residue order.",
    "io": "Atomic motif / functional-group geometry constraints -> protein backbone scaffolds",
    "status": "Open code + weights",
    "use_cases": "Enzyme active-site scaffolding, atom-level (unindexed) motif scaffolding, small-molecule-aware de novo design",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41592-025-02975-x"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/RosettaCommons/RFdiffusion2"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-025-02975-x",
      "https://github.com/RosettaCommons/RFdiffusion2"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-genie-2",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "Genie 2",
    "description": "SE(3)-equivariant DDPM protein-structure diffusion model for unconditional C-alpha backbone generation and single- and multi-motif scaffolding, trained with large AlphaFold-DB augmentation.",
    "io": "Length / motif constraints -> generated protein backbones (C-alpha frames)",
    "status": "Open code + weights",
    "use_cases": "Unconditional protein backbone generation, single- and multi-motif scaffolding for functional-site design, diversity/novelty-focused design",
    "year": "2024",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2405.15489"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/aqlaboratory/genie2"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2405.15489",
      "https://github.com/aqlaboratory/genie2"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-proteus",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "Proteus",
    "description": "Graph-triangle-based protein backbone diffusion model achieving high designability and efficiency without RoseTTAFold/AlphaFold structure-prediction pretraining.",
    "io": "Length / noise -> generated protein backbones",
    "status": "Open code + weights",
    "use_cases": "Efficient de novo backbone generation, designable monomer scaffold generation",
    "year": "2024",
    "paper_links": [
      {
        "label": "PMLR",
        "url": "https://proceedings.mlr.press/v235/wang24bi.html"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Wangchentong/Proteus"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://proceedings.mlr.press/v235/wang24bi.html",
      "https://github.com/Wangchentong/Proteus"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-evodiff",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "EvoDiff",
    "description": "Microsoft discrete-diffusion framework for controllable protein generation in sequence space, with sequence (EvoDiff-Seq, 38M/640M) and MSA (EvoDiff-MSA) models; built on CARP/ByteNet.",
    "io": "Noise / partial sequence / MSA / motif constraints -> generated or inpainted protein sequences",
    "status": "Open code + weights",
    "use_cases": "Unconditional and conditional sequence generation, motif scaffolding/inpainting in sequence space, generation of disordered regions inaccessible to structure-based models",
    "year": "2024",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2023.09.11.556673v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/microsoft/evodiff"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2023.09.11.556673v1",
      "https://github.com/microsoft/evodiff"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "official MIT code and pretrained models match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-dplm-2",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "DPLM-2",
    "description": "Multimodal extension of the diffusion protein language model (DPLM) that jointly models sequence and structure via lookup-free-quantized structure tokens, enabling single-stage co-generation.",
    "io": "Sequence and/or structure-token prompts/masks -> jointly generated sequence + 3D structure (folding, inverse folding, co-design)",
    "status": "Open code + weights",
    "use_cases": "Simultaneous sequence-structure co-generation, folding, inverse folding, multimodal motif scaffolding, structure-aware representations",
    "year": "2024",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2410.13782"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/bytedance/dplm"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2410.13782",
      "https://github.com/bytedance/dplm"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-pinal-denovo-pinal",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "Pinal (Denovo-Pinal)",
    "description": "Text-to-protein design framework that generates protein sequences from natural-language descriptions via a structure-then-sequence two-stage 16B-parameter pipeline trained on ~1.7B protein-text pairs.",
    "io": "Natural-language prompt (keywords/phrases/paragraph) -> designed protein sequence(s)",
    "status": "Open code + weights",
    "use_cases": "Language-guided de novo protein design, function-conditioned generation, accessible 'prompt a protein' workflows",
    "year": "2024",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.08.01.606258v7"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/westlake-repl/Denovo-Pinal"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.08.01.606258v7",
      "https://github.com/westlake-repl/Denovo-Pinal"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-aido-protein-16b",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "AIDO.Protein-16B",
    "description": "First mixture-of-experts protein language model (16B params, 8 experts/2 active, ~4.5B active) in GenBio's AIDO multiscale foundation-model system, pretrained on 1.2T amino acids (UniRef90 + ColabFoldDB).",
    "io": "Amino acid sequence -> embeddings, masked-token logits, zero-shot variant scores, generated sequence",
    "status": "Model hub; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Protein understanding (xTrimoPGLM benchmark SOTA), zero-shot DMS/ProteinGym fitness prediction, structure-conditioned sequence generation, downstream adaptation via ModelGenerator",
    "year": "2024",
    "paper_links": [
      {
        "label": "OpenReview",
        "url": "https://openreview.net/forum?id=6VldeCDKpH"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/genbio-ai/AIDO.Protein-16B"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://openreview.net/forum?id=6VldeCDKpH",
      "https://huggingface.co/genbio-ai/AIDO.Protein-16B",
      "https://raw.githubusercontent.com/genbio-ai/ModelGenerator/main/LICENSE"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; model; licence. Action: set noncommercial:true and record the GenBio AI Community License.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-ptm-mamba",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "PTM-Mamba",
    "description": "PTM-aware protein language model using bidirectional gated Mamba blocks fused with ESM-2 embeddings to represent post-translationally modified residues.",
    "io": "Amino acid sequence + PTM tokens -> PTM-aware residue embeddings, task predictions",
    "status": "Open code + weights",
    "use_cases": "PTM effect prediction on protein-protein interactions, disease-association and druggability prediction, zero-shot PTM discovery",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41592-025-02656-9"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/programmablebio/ptm-mamba"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-025-02656-9",
      "https://github.com/programmablebio/ptm-mamba"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-prothyena",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "ProtHyena",
    "description": "Hyena-operator protein language model (~1.6M params) giving sub-quadratic single-amino-acid-resolution modeling of long sequences; protein analog of HyenaDNA, trained on Pfam.",
    "io": "Amino acid sequence -> embeddings, next-token predictions, task heads",
    "status": "Open code + weights",
    "use_cases": "Long-sequence protein modeling, fluorescence/stability/secondary-structure prediction, efficient embeddings",
    "year": "2024",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.01.18.576206v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ZHymLumine/ProtHyena"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.01.18.576206v1",
      "https://github.com/ZHymLumine/ProtHyena"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-lc-plm",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "LC-PLM",
    "description": "Long-context protein language model based on a bidirectional Mamba (BiMamba-S, shared projection layers) architecture from Amazon Science for efficient very-long-sequence modeling; graph-contextual variant LC-PLM-G.",
    "io": "Amino acid sequence (long context) -> residue/sequence embeddings, masked-token logits",
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Long protein / multi-domain modeling, length extrapolation, embeddings for structure and function tasks, PPI-graph-contextual protein modeling",
    "year": "2024",
    "paper_links": [
      {
        "label": "TMLR paper",
        "url": "https://openreview.net/forum?id=dWvztQzfy4"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/amazon-science/LC-PLM"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://openreview.net/forum?id=dWvztQzfy4",
      "https://github.com/amazon-science/LC-PLM",
      "https://github.com/amazon-science/LC-PLM/blob/main/LICENSE"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Repository explicitly uses CC BY-NC 4.0, but noncommercial:false; paper is now accepted/published in TMLR. TMLR record, repo, licence.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-antiberty",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "AntiBERTy",
    "description": "Antibody-specific BERT masked language model (26M params) pretrained on 558M natural antibody heavy and light chains; the representation model underlying IgFold (Gray lab, Johns Hopkins).",
    "io": "Antibody sequence -> residue embeddings, attention, pseudo-log-likelihoods",
    "status": "Open code + weights",
    "use_cases": "Antibody embeddings, affinity-maturation trajectory analysis, clustering, input features for structure prediction (IgFold)",
    "year": "2021",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2112.07782"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/jeffreyruffolo/AntiBERTy"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2112.07782",
      "https://github.com/jeffreyruffolo/AntiBERTy"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-ablang2-tcrlang",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "AbLang2 / TCRLang",
    "description": "Second-generation antibody language model trained on paired and unpaired OAS data, focused on suggesting diverse non-germline residues and correcting germline bias; includes a paired TCR-weights variant (TCRLang).",
    "io": "Paired/unpaired antibody (or TCR) sequence with masks/gaps -> restored residues, embeddings, mutation suggestions",
    "status": "Open code + weights",
    "use_cases": "Antibody sequence restoration, germline-bias-corrected mutation suggestion, paired-chain antibody and TCR representation, antibody design",
    "year": "2024",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.02.02.578678"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/oxpig/AbLang2"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.02.02.578678",
      "https://github.com/oxpig/AbLang2"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-igbert-igt5",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "IgBert / IgT5",
    "description": "Large-scale antibody-specific BERT and T5 language models (Exscientia/Oxford) trained on 2B+ unpaired and 2M+ paired OAS sequences, handling paired and unpaired variable regions.",
    "io": "Paired or unpaired antibody heavy/light sequences -> embeddings, masked-token logits (IgT5 is a T5/encoder-decoder; IgBert is encoder)",
    "status": "Model hub",
    "use_cases": "Antibody engineering regression/design tasks, developability and binding-related prediction, paired-chain antibody embeddings",
    "year": "2024",
    "paper_links": [
      {
        "label": "PLOS",
        "url": "https://journals.plos.org/ploscompbiol/article?id=10.1371/journal.pcbi.1012646"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/Exscientia/IgT5"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://journals.plos.org/ploscompbiol/article?id=10.1371/journal.pcbi.1012646",
      "https://huggingface.co/Exscientia/IgT5"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "paper and MIT model card match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-p-iggen",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "p-IgGen",
    "description": "Generative paired antibody language model (GPT-2-like decoder, OPIG/AstraZeneca) that produces full paired heavy-light antibody sequences, with a developability-biased fine-tuned variant.",
    "io": "Optional seed/one chain -> generated paired antibody sequences (or complementary chain); also scores sequences",
    "status": "Open code + weights",
    "use_cases": "De novo paired antibody generation, chain pairing/completion, developability-aware therapeutic antibody design",
    "year": "2024",
    "paper_links": [
      {
        "label": "Oxford Acad.",
        "url": "https://academic.oup.com/bioinformatics/article/40/11/btae659/7888884"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/oxpig/p-IgGen"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://academic.oup.com/bioinformatics/article/40/11/btae659/7888884",
      "https://github.com/oxpig/p-IgGen"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-balm-bio-inspired-antibody-language-model",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "BALM (Bio-inspired Antibody Language Model)",
    "description": "Antibody language model (BEAM-Labs) trained on ~336M nonredundant antibody sequences for antibody function/structure prediction; basis for the BALMFold structure predictor.",
    "io": "Antibody sequence -> embeddings, task predictions; BALMFold -> antibody structure",
    "status": "Open code + weights",
    "use_cases": "Antigen-binding prediction, paratope prediction, binding-affinity prediction, maturation-trajectory modeling, antibody structure prediction (BALMFold)",
    "year": "2024",
    "paper_links": [
      {
        "label": "Oxford Acad.",
        "url": "https://academic.oup.com/bib/article/25/4/bbae245/7682462"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/BEAM-Labs/BALM"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://academic.oup.com/bib/article/25/4/bbae245/7682462",
      "https://github.com/BEAM-Labs/BALM"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-sapiens",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "Sapiens",
    "description": "Human antibody BERT masked language models (separate heavy/light) from Merck's BioPhi platform, trained on OAS for in-silico antibody humanization.",
    "io": "Antibody variable-region sequence -> per-position human residue probabilities / humanized sequence, embeddings",
    "status": "Open code + weights",
    "use_cases": "Automated antibody humanization, humanness evaluation (within BioPhi/OASis), antibody sequence repair",
    "year": "2022",
    "paper_links": [
      {
        "label": "Paper",
        "url": "https://www.tandfonline.com/doi/full/10.1080/19420862.2021.2020203"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Merck/Sapiens"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.tandfonline.com/doi/full/10.1080/19420862.2021.2020203",
      "https://github.com/Merck/Sapiens"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-proteinnpt",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "ProteinNPT",
    "description": "Non-parametric transformer (Marks/OATML lab) that jointly attends over batches of protein sequences and their property labels (plus PLM/MSA embeddings) for semi-supervised fitness prediction and design.",
    "io": "Sequence batch + (partial) property labels + PLM/MSA embeddings -> predicted properties, conditionally sampled sequences",
    "status": "Open code + weights",
    "use_cases": "Single/multi-property fitness prediction, conditional sequence generation, iterative protein redesign via Bayesian optimization",
    "year": "2023",
    "paper_links": [
      {
        "label": "NeurIPS",
        "url": "https://proceedings.neurips.cc/paper_files/paper/2023/hash/6a4d5d85f7a52f062d23d98d544a5578-Abstract-Conference.html"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/OATML-Markslab/ProteinNPT"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://proceedings.neurips.cc/paper_files/paper/2023/hash/6a4d5d85f7a52f062d23d98d544a5578-Abstract-Conference.html",
      "https://github.com/OATML-Markslab/ProteinNPT"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-poet-2",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "PoET-2",
    "description": "Retrieval-augmented multimodal protein foundation model (OpenProtein.AI, ~182M params) accepting sequence and structural context with dual causal + masked decoders; in-context family-evolutionary conditioning.",
    "io": "Conditioning set of homologs/structures + sequence context -> zero/few-shot fitness scores, generated sequences, embeddings",
    "status": "Open code + downloadable weights under a non-commercial model licence",
    "use_cases": "Zero- and few-shot fitness/stability/binding prediction, conditional protein generation, in-context protein engineering with minimal data",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2508.04724"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/OpenProteinAI/PoET-2"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2508.04724",
      "https://github.com/OpenProteinAI/PoET-2",
      "https://www.openprotein.ai/",
      "https://docs.openprotein.ai/rest-api/prompt.html"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Model identity is verified; keep distinct from the OpenProtein platform record and expose exact model terms.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-zymctrl",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "ZymCTRL",
    "description": "Conditional autoregressive protein language model (~738M params, Ferruz lab / Basecamp Research) that generates artificial enzyme sequences for a user-specified Enzyme Commission (EC) number.",
    "io": "EC number (+ optional sequence context) -> generated enzyme sequences",
    "status": "Model hub",
    "use_cases": "Function-conditioned enzyme design, de novo enzyme generation for a target reaction, controllable family-specific generation",
    "year": "2024",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.05.03.592223v1"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/AI4PD/ZymCTRL"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.05.03.592223v1",
      "https://huggingface.co/AI4PD/ZymCTRL"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; model.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-alphafold2",
    "date_added": "2026-06-16",
    "modality": "complex",
    "name": "AlphaFold2",
    "description": "MSA- and template-based deep learning model that predicts highly accurate single-chain protein 3D structures; the breakthrough CASP14 system and foundation for most downstream structure workflows and the AlphaFold DB.",
    "io": "Protein sequence + MSA/templates -> 3D protein structure + per-residue confidence (pLDDT/PAE)",
    "status": "Open code + weights",
    "use_cases": "Monomer structure prediction, structural annotation, input to docking/design pipelines, source of structural embeddings",
    "year": "2021",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-021-03819-2"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/google-deepmind/alphafold"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41586-021-03819-2",
      "https://github.com/google-deepmind/alphafold"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Apache code and CC BY 4.0 parameters match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-alphafold-multimer",
    "date_added": "2026-06-16",
    "modality": "complex",
    "name": "AlphaFold-Multimer",
    "description": "Extension of AlphaFold2 trained specifically on multimeric inputs to predict multi-chain protein complex (protein-protein assembly) structures; the standard pre-AlphaFold3 complex-prediction method.",
    "io": "Multi-chain protein FASTA + paired/unpaired MSAs -> protein complex 3D structure with interface confidence (ipTM/pTM)",
    "status": "Open code + weights",
    "use_cases": "Protein-protein complex prediction, interface/epitope modeling, PPI screening, assembly hypotheses",
    "year": "2021",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2021.10.04.463034"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/google-deepmind/alphafold"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2021.10.04.463034",
      "https://github.com/google-deepmind/alphafold"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-protenix",
    "date_added": "2026-06-16",
    "modality": "complex",
    "name": "Protenix",
    "description": "ByteDance's trainable, fully open-source PyTorch reproduction of AlphaFold3 for all-atom biomolecular complex structure prediction; the v1 release (2026) added protein-template integration and RNA MSA support to match or exceed AF3 on several benchmarks, and v2 (2026) adds large antibody-antigen gains, GPCR hit discovery, and zero-shot antibody design.",
    "io": "JSON spec of protein/RNA/DNA/ligand (+optional MSA, templates, atom-level constraints) -> all-atom complex 3D structure",
    "status": "Open code + weights",
    "use_cases": "Open AF3-style complex prediction, antibody-antigen modeling, drug-target structures, retraining/fine-tuning research",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.01.08.631967v1"
      },
      {
        "label": "bioRxiv (v1)",
        "url": "https://www.biorxiv.org/content/10.64898/2026.02.05.703733v1"
      },
      {
        "label": "bioRxiv (v2)",
        "url": "https://www.biorxiv.org/content/10.64898/2026.04.10.717613v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/bytedance/Protenix"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.01.08.631967v1",
      "https://www.biorxiv.org/content/10.64898/2026.02.05.703733v1",
      "https://www.biorxiv.org/content/10.64898/2026.04.10.717613v1",
      "https://github.com/bytedance/Protenix"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Apache code and model parameters, including current family releases, match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-helixfold3",
    "date_added": "2026-06-16",
    "modality": "complex",
    "name": "HelixFold3",
    "description": "Baidu PaddleHelix's AlphaFold3 reproduction for all-atom biomolecular complex structure prediction across proteins, nucleic acids, ligands, ions and modifications; one of the first public AF3 reproductions (Aug 2024).",
    "io": "Protein/nucleic-acid/ligand specification -> all-atom complex 3D structure",
    "status": "Open code, gated weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "AF3-style complex prediction, ligand and nucleic-acid complexes, antibody-antigen (HelixFold-Multimer variant), drug discovery",
    "year": "2024",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2408.16975"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/PaddlePaddle/PaddleHelix/tree/dev/apps/protein_folding/helixfold3"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2408.16975",
      "https://github.com/PaddlePaddle/PaddleHelix/tree/dev/apps/protein_folding/helixfold3"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P1",
    "audit_note": "set NC true; code/parameters are non-commercial and server terms differ by free versus paid route.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-openfold3",
    "date_added": "2026-06-16",
    "modality": "complex",
    "name": "OpenFold3",
    "description": "OpenFold Consortium / AlQuraishi Lab research-preview reproduction of AlphaFold3 for protein/RNA/DNA/small-molecule complex prediction, with Apache-2.0 code and weights plus public training data.",
    "io": "Protein/RNA/DNA/small-molecule inputs -> all-atom biomolecular complex 3D structure",
    "status": "Open code, gated weights (Apache-2.0; Hugging Face login/contact-sharing gate); public training data",
    "use_cases": "Reproducible AF3-class complex prediction, benchmarking, retraining/fine-tuning, commercial structure prediction (Apache 2.0)",
    "year": "2026",
    "paper_links": [
      {
        "label": "Technical report",
        "url": "https://portal.openfold.omsf.io/reports/of3p2_technical_report.pdf"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/aqlaboratory/openfold-3"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/OpenFold/OpenFold3"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "sources": [
      "https://portal.openfold.omsf.io/reports/of3p2_technical_report.pdf",
      "https://github.com/aqlaboratory/openfold-3",
      "https://huggingface.co/OpenFold/OpenFold3"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-chai-2",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "Chai-2",
    "description": "Chai Discovery's all-atom multimodal generative foundation model for zero-shot de novo antibody/nanobody and miniprotein binder design at atomic precision; successor to Chai-1.",
    "io": "Target epitope/structure -> de novo antibody/nanobody/binder sequences and designed complex structures",
    "status": "Commercial access plus limited non-commercial academic access; terms apply",
    "use_cases": "De novo antibody/nanobody design, binder generation against novel epitopes, therapeutic discovery without high-throughput screening",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.07.05.663018v1"
      }
    ],
    "code_links": [
      {
        "label": "Official access",
        "url": "https://www.chaidiscovery.com/product"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.07.05.663018v1",
      "https://www.chaidiscovery.com/product"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; official access page. Action: add the official access link and describe commercial access plus limited non-commercial academic access, rather than generic “Web/API”.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-umol",
    "date_added": "2026-06-16",
    "modality": "complex",
    "name": "Umol",
    "description": "Sequence-based deep learning model that predicts fully flexible all-atom protein-ligand complex structures without requiring a known protein structure (protein as MSA, ligand as SMILES).",
    "io": "Protein sequence (MSA) + ligand SMILES + optional binding-site info -> protein-ligand complex 3D structure + confidence (plDDT)",
    "status": "Open code + weights",
    "use_cases": "Protein-ligand co-folding from sequence, binder vs non-binder discrimination via confidence, structure-based virtual screening",
    "year": "2024",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-024-48837-6"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/patrickbryant1/Umol"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-024-48837-6",
      "https://github.com/patrickbryant1/Umol"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Apache code and CC BY 4.0 parameters match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-diffdock-l",
    "date_added": "2026-06-16",
    "modality": "complex",
    "name": "DiffDock-L",
    "description": "Generalizing version of the DiffDock diffusion docking model using confidence-bootstrapping self-training for much better cross-pocket generalization (DockGen success 10%->24%).",
    "io": "Protein structure + ligand -> ranked ligand binding poses",
    "status": "Open code + weights",
    "use_cases": "Blind molecular docking, pose generation, docking generalization to novel protein domains (DockGen benchmark)",
    "year": "2024",
    "paper_links": [
      {
        "label": "OpenReview",
        "url": "https://openreview.net/forum?id=UfBIxpTK10"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/gcorso/DiffDock"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://openreview.net/forum?id=UfBIxpTK10",
      "https://github.com/gcorso/DiffDock"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-flowdock",
    "date_added": "2026-06-16",
    "modality": "complex",
    "name": "FlowDock",
    "description": "Geometric flow-matching generative model for simultaneous protein-ligand docking (apo-to-holo) and binding-affinity prediction.",
    "io": "Protein sequence(s) + ligand SMILES -> protein-ligand complex structure + predicted binding affinity",
    "status": "Open code + weights",
    "use_cases": "Apo-state blind docking, generative co-folding, affinity estimation for virtual screening",
    "year": "2025",
    "paper_links": [
      {
        "label": "Oxford Acad.",
        "url": "https://academic.oup.com/bioinformatics/article/41/Supplement_1/i198/8199366"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/BioinfoMachineLearning/FlowDock"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://academic.oup.com/bioinformatics/article/41/Supplement_1/i198/8199366",
      "https://github.com/BioinfoMachineLearning/FlowDock"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "published paper, MIT code, and Zenodo checkpoints match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-igfold",
    "date_added": "2026-06-16",
    "modality": "protein",
    "name": "IgFold",
    "description": "Fast, accurate antibody/nanobody structure prediction model that leverages embeddings from the AntiBERTy antibody language model to directly predict backbone coordinates in seconds.",
    "io": "Antibody/nanobody sequence(s) -> 3D antibody structure (+ refinement)",
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "High-throughput antibody structure prediction, repertoire-scale modeling (OAS-scale databases), antibody engineering pipelines",
    "year": "2023",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-023-38063-x"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Graylab/IgFold"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-023-38063-x",
      "https://github.com/Graylab/IgFold"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact. Action: set noncommercial:true; official JHU Academic Software License restricts use.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-rosettafold2",
    "date_added": "2026-06-16",
    "modality": "complex",
    "name": "RoseTTAFold2",
    "description": "Updated full three-track RoseTTAFold network for protein monomer and complex structure prediction, matching AF2 (monomer) and AF2-Multimer (complex) accuracy with better scaling on large proteins/complexes.",
    "io": "Protein sequence(s) + MSA -> monomer or complex 3D structure",
    "status": "Open code + weights",
    "use_cases": "Protein and complex structure prediction, backbone scaffold for downstream design tools (e.g. RFdiffusion), large-protein modeling",
    "year": "2023",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2023.05.24.542179v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/uw-ipd/RoseTTAFold2"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2023.05.24.542179v1",
      "https://github.com/uw-ipd/RoseTTAFold2"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-omegafold",
    "date_added": "2026-06-16",
    "modality": "complex",
    "name": "OmegaFold",
    "description": "MSA-free protein structure predictor that folds from a single primary sequence using the OmegaPLM protein language model plus a geometry-aware Geoformer.",
    "io": "Single protein sequence -> 3D protein structure",
    "status": "Open code + weights",
    "use_cases": "Structure prediction for orphan proteins and antibodies (noisy/absent MSAs), fast single-sequence / metagenomic structure annotation",
    "year": "2022",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2022.07.21.500999v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/HeliXonProtein/OmegaFold"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2022.07.21.500999v1",
      "https://github.com/HeliXonProtein/OmegaFold"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "accessible Apache code/weights are historical and repository activity is dormant. Mark maintenance state.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-chemformer-molbart",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "Chemformer / MolBART",
    "description": "BART-style encoder-decoder transformer pretrained on SMILES with a denoising/heteroencoding objective, fine-tunable for both sequence-to-sequence and discriminative cheminformatics tasks.",
    "io": "SMILES (optionally corrupted) -> embeddings or generated/transformed SMILES",
    "status": "Open code + weights",
    "use_cases": "Reaction prediction, retrosynthesis, molecular optimization, property prediction, SMILES generation",
    "year": "2022",
    "paper_links": [
      {
        "label": "Paper",
        "url": "https://iopscience.iop.org/article/10.1088/2632-2153/ac3ffb"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/MolecularAI/Chemformer"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://iopscience.iop.org/article/10.1088/2632-2153/ac3ffb",
      "https://github.com/MolecularAI/Chemformer"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-molgpt",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "MolGPT",
    "description": "GPT/transformer-decoder language model trained with next-token prediction over SMILES for conditional and unconditional drug-like molecule generation.",
    "io": "Optional property/scaffold conditioning + SMILES context -> generated SMILES",
    "status": "Open code + weights",
    "use_cases": "De novo molecule generation, property/scaffold-conditioned generation, generative-model benchmarking",
    "year": "2021",
    "paper_links": [
      {
        "label": "Paper",
        "url": "https://pubs.acs.org/doi/10.1021/acs.jcim.1c00600"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/devalab/molgpt"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://pubs.acs.org/doi/10.1021/acs.jcim.1c00600",
      "https://github.com/devalab/molgpt"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "MIT code is verified, but weights are externally hosted on Kaggle. Add the exact official artifact provenance/licence.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-safe-gpt",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "SAFE-GPT",
    "description": "87M-parameter GPT-2-style model trained on ~1.1B molecules encoded in the SAFE (Sequential Attachment-based Fragment Embedding) line notation, enabling fragment-constrained generation in one autoregressive framework.",
    "io": "SAFE/SMILES fragment or scaffold context -> generated molecules (SAFE/SMILES)",
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Scaffold decoration, fragment linking, scaffold hopping, motif extension, de novo generation",
    "year": "2024",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2310.10773"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/datamol-io/safe"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2310.10773",
      "https://github.com/datamol-io/safe"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P1",
    "audit_note": "set NC true; code is Apache, data CC BY, model weights CC BY-NC.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-genmol",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "GenMol",
    "description": "Generalist masked discrete-diffusion model over SAFE molecular sequences using non-autoregressive parallel decoding, fragment remasking, and molecular context guidance for unified drug-discovery generation.",
    "io": "Molecular/fragment context (SAFE) -> generated valid molecules",
    "status": "Open code + weights",
    "use_cases": "De novo generation, fragment-constrained design, goal-directed hit generation, lead optimization",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2501.06158"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/NVIDIA-Digital-Bio/genmol"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2501.06158",
      "https://github.com/NVIDIA-Digital-Bio/genmol"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "explicitly separate Apache-2.0 code from NVIDIA Open Model Licence weights.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-megalodon",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "Megalodon",
    "description": "Scalable transformer with lightweight equivariant layers for de novo 3D molecule generation, trained with a joint continuous-and-discrete denoising (diffusion / flow-matching) co-design objective.",
    "io": "Optional conditioning -> 3D molecular structures (atoms + coordinates)",
    "status": "Open code + weights",
    "use_cases": "Unconditional/conditional 3D molecule generation, low-energy structure generation, structure-energy benchmarks",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.18392"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/NVIDIA-Digital-Bio/megalodon"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2505.18392",
      "https://github.com/NVIDIA-Digital-Bio/megalodon"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "add exact code licence and NVIDIA model-weight licence instead of generic “open.”",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-uni-mol2",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "Uni-Mol2",
    "description": "Two-track (atom/graph/geometry) 3D molecular pretraining model scaled from 84M to 1.1B parameters on 800M conformations, the largest molecular pretraining model in the Uni-Mol series and a scaling-law study.",
    "io": "Molecular graph + 3D conformer -> embeddings / property predictions",
    "status": "Open code + weights",
    "use_cases": "Molecular property prediction (e.g., QM9, COMPAS), 3D-aware representation learning, transfer learning",
    "year": "2024",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2406.14969"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/deepmodeling/Uni-Mol"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2406.14969",
      "https://github.com/deepmodeling/Uni-Mol"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "MIT family repository and Uni-Mol2 code/weights match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-molclr",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "MolCLR",
    "description": "Self-supervised graph-contrastive learning framework that pretrains GNN (GCN/GIN) encoders on ~10M molecules via atom-masking, bond-deletion, and subgraph-removal augmentations.",
    "io": "Molecular graph -> embeddings / fine-tuned property predictions",
    "status": "Open code + weights",
    "use_cases": "Molecular property prediction, transferable graph representations, contrastive-pretraining baseline",
    "year": "2022",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s42256-022-00447-x"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/yuyangw/MolCLR"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s42256-022-00447-x",
      "https://github.com/yuyangw/MolCLR"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "MIT code and bundled pretrained checkpoints match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-mole-bert",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "Mole-BERT",
    "description": "Graph pretraining framework with a VQ-VAE context-aware atom tokenizer plus Masked Atoms Modeling and Triplet Masked Contrastive Learning to fix mismatches in masked-atom GNN pretraining.",
    "io": "Molecular graph -> embeddings / fine-tuned predictions",
    "status": "Open code + weights",
    "use_cases": "Molecular property prediction, self-supervised graph representation learning",
    "year": "2023",
    "paper_links": [
      {
        "label": "OpenReview",
        "url": "https://openreview.net/forum?id=jevY-DtiZTR"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/junxia97/Mole-BERT"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://openreview.net/forum?id=jevY-DtiZTR",
      "https://github.com/junxia97/Mole-BERT"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-kpgt-knowledge-guided-pre-training-of-graph-transformer",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "KPGT (Knowledge-Guided Pre-training of Graph Transformer)",
    "description": "Line Graph Transformer (LiGhT) emphasizing chemical bonds, pretrained on ~2M molecules with a knowledge-guided strategy using molecular descriptors/fingerprints as additional supervision.",
    "io": "Molecular graph -> embeddings / property predictions",
    "status": "Open code + weights",
    "use_cases": "Molecular property prediction (evaluated on 63 datasets), knowledge-guided representation learning",
    "year": "2022",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-023-43214-1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/lihan97/KPGT"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-023-43214-1",
      "https://github.com/lihan97/KPGT"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-molgen",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "MolGen",
    "description": "Pretrained BART-style molecular language model over 100M+ SELFIES with domain-agnostic prefix tuning and a chemical-feedback paradigm to avoid invalid 'molecular hallucinations'.",
    "io": "SELFIES context / property targets -> generated valid molecules",
    "status": "Open code + weights",
    "use_cases": "De novo and property-targeted molecular generation, multi-task molecular design",
    "year": "2024",
    "paper_links": [
      {
        "label": "Final paper",
        "url": "https://proceedings.iclr.cc/paper_files/paper/2024/hash/ed7dd1e32cf9b0abf664bf0e891527e5-Abstract-Conference.html"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/zjunlp/MolGen"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://proceedings.iclr.cc/paper_files/paper/2024/hash/ed7dd1e32cf9b0abf664bf0e891527e5-Abstract-Conference.html",
      "https://github.com/zjunlp/MolGen"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Replace arXiv with final ICLR 2024 paper. ICLR paper, repo.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-molbert",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "MolBERT",
    "description": "BERT-style SMILES language model from BenevolentAI pretrained with masked language modeling plus chemistry-relevant auxiliary tasks (physicochemical property regression on 200 RDKit descriptors, SMILES equivalence).",
    "io": "SMILES -> molecular embeddings / property predictions",
    "status": "Open code + weights",
    "use_cases": "Virtual screening, QSAR/property prediction, molecular similarity, transfer learning",
    "year": "2020",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2011.13230"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/BenevolentAI/MolBERT"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2011.13230",
      "https://github.com/BenevolentAI/MolBERT"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-smi-ted-materials-smi-ted",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "SMI-TED (materials.smi-ted)",
    "description": "Large encoder-decoder ('SMILES Transformer Encoder-Decoder') chemical foundation model pretrained on 91M PubChem SMILES (~4B tokens), released in 289M and 8x289M (mixture-of-experts) variants.",
    "io": "SMILES -> embeddings or reconstructed/generated SMILES",
    "status": "Open code + weights",
    "use_cases": "Quantum-property prediction, reaction-yield prediction, classification/regression, few-shot molecular tasks",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2407.20267"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/IBM/materials"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2407.20267",
      "https://github.com/IBM/materials"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Apache FM4M repository and official SMI-TED model artifact match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-smi-ssed-materials-smi-ssed",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "SMI-SSED (materials.smi_ssed)",
    "description": "Mamba-based (state-space) encoder-decoder chemical foundation model pretrained on 91M PubChem SMILES (~4B tokens), released in 336M and 8x336M variants.",
    "io": "SMILES -> embeddings or reconstructed/generated SMILES",
    "status": "Open code + weights",
    "use_cases": "Quantum-property prediction, molecular reconstruction, classification/regression, synthesis-yield prediction",
    "year": "2024",
    "paper_links": [
      {
        "label": "Paper",
        "url": "https://research.ibm.com/publications/a-mamba-based-foundation-model-for-chemistry"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/IBM/materials"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://research.ibm.com/publications/a-mamba-based-foundation-model-for-chemistry",
      "https://github.com/IBM/materials"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-chemfm",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "ChemFM",
    "description": "Scaling-law-guided causal decoder-only (TinyLlama-style) chemical language model (1B and 3B parameters) pretrained via self-supervised causal LM on 178M UniChem molecules (~1.78B augmented SMILES).",
    "io": "SMILES -> embeddings / property predictions / generated SMILES",
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Molecular property prediction (34+ benchmarks), molecular design/generation, conditional generation",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s42004-025-01793-8"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/TheLuoFengLab/ChemFM"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s42004-025-01793-8",
      "https://github.com/TheLuoFengLab/ChemFM"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact. Action: set noncommercial:true; repository licence is CC BY-NC 4.0.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-mist",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "MIST",
    "description": "Family of encoder-only SMILES transformer foundation models (MIST-28M to MIST-1.8B) pretrained with masked language modeling on up to ~2B Enamine REAL Space molecules using the Smirk tokenizer.",
    "io": "SMILES -> embeddings / fine-tuned property predictions",
    "status": "Open code + weights",
    "use_cases": "Molecular and formulation property prediction across 400+ tasks, large-scale chemical-space exploration",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2510.18900"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/BattModels/mist-demo"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2510.18900",
      "https://github.com/BattModels/mist-demo"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-smiles-bert",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "SMILES-BERT",
    "description": "Early transformer encoder pretrained on large unlabeled SMILES via a masked SMILES-recovery objective, then fine-tuned for molecular property prediction.",
    "io": "SMILES tokens -> embeddings / property predictions",
    "status": "Public source code; pretrained weights not released",
    "use_cases": "Molecular property prediction, transfer learning from unlabeled chemical data",
    "year": "2019",
    "paper_links": [
      {
        "label": "Paper",
        "url": "https://dl.acm.org/doi/10.1145/3307339.3342186"
      }
    ],
    "code_links": [
      {
        "label": "Code",
        "url": "https://github.com/uta-smile/SMILES-BERT"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://dl.acm.org/doi/10.1145/3307339.3342186",
      "https://github.com/uta-smile/SMILES-BERT"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; official code. Action: upgrade from paper-only to public code, explicitly noting that pretrained weights remain unreleased.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-cddd-continuous-and-data-driven-molecular-descriptors",
    "date_added": "2026-06-16",
    "modality": "molecule",
    "name": "CDDD (Continuous and Data-Driven molecular Descriptors)",
    "description": "Translation (seq2seq autoencoder) model mapping between equivalent SMILES to learn a fixed-size continuous latent representation usable as molecular descriptors and for generation.",
    "io": "SMILES -> continuous latent descriptor vector (and back to SMILES via decoder)",
    "status": "Open code + weights",
    "use_cases": "Descriptor extraction for QSAR/property prediction, latent-space molecular optimization/generation",
    "year": "2019",
    "paper_links": [
      {
        "label": "RSC",
        "url": "https://pubs.rsc.org/en/content/articlehtml/2019/sc/c8sc04175j"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/jrwnter/cddd"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://pubs.rsc.org/en/content/articlehtml/2019/sc/c8sc04175j",
      "https://github.com/jrwnter/cddd"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "MIT implementation and pretrained model link match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-state-arc-institute",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "STATE (Arc Institute)",
    "description": "Arc Institute's first virtual-cell model: a perturbation-trained pair of modules (State Embedding SE + State Transition ST) that predicts single-cell transcriptional responses to genetic, chemical, and cytokine perturbations.",
    "io": "Cell expression context + perturbation specification -> predicted post-perturbation expression / cell embeddings",
    "status": "Open code + weights (non-commercial; Arc Research Institute State Model license + acceptable-use policy; SE-600M weights downloadable)",
    "use_cases": "In silico perturbation screening, drug/cytokine response prediction, perturbation effect ranking, virtual-cell modeling",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.06.26.661135v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ArcInstitute/state"
      }
    ],
    "canonical": true,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.06.26.661135v1",
      "https://github.com/ArcInstitute/state"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-tahoe-x1",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "Tahoe-x1",
    "description": "Family of perturbation-trained single-cell foundation models (Tx1-70M, Tx1-1B, Tx1-3B) from Tahoe/Vevo Therapeutics, pretrained on large single-cell corpora including the Tahoe-100M drug-perturbation compendium.",
    "io": "scRNA-seq expression profile -> cell-state embeddings / perturbation-aware predictions",
    "status": "Open code + weights (Hugging Face: tahoebio/Tahoe-x1)",
    "use_cases": "Cancer-relevant cell-state representation, drug perturbation modeling, target/biomarker discovery, gigascale single-cell analysis",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.10.23.683759v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/tahoebio/tahoe-x1"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.10.23.683759v1",
      "https://github.com/tahoebio/tahoe-x1"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-sclong",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "scLong",
    "description": "Billion-parameter single-cell transcriptomics foundation model that runs self-attention across the full ~28,000-gene set (including lowly expressed genes) and injects Gene Ontology knowledge via a graph convolutional network.",
    "io": "scRNA-seq expression vector (full gene set) -> gene/cell embeddings, task predictions",
    "status": "Open code + weights",
    "use_cases": "Genetic/chemical perturbation response prediction, cancer drug response, gene regulatory network inference, long-range gene-context modeling",
    "year": "2026",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-026-69102-y"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/BaiDing1234/scLong"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-026-69102-y",
      "https://github.com/BaiDing1234/scLong"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "checkpoint is public via SharePoint, but no repository or weight licence was found. State this explicitly.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-sctab",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "scTab",
    "description": "Feature-attention (TabNet-based) foundation-scale cross-tissue cell-type annotation model trained with a novel augmentation scheme on 22.2M human cells spanning 164 cell types and 56 tissues.",
    "io": "scRNA-seq cell expression -> cell-type label (with deep-ensemble uncertainty)",
    "status": "Open code + weights",
    "use_cases": "Cross-tissue automated cell type annotation, large-scale reference annotation, uncertainty-aware labeling",
    "year": "2024",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-024-51059-5"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/theislab/scTab"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-024-51059-5",
      "https://github.com/theislab/scTab"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-genept",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "GenePT",
    "description": "Embedding model for genes and cells built from GPT-3.5/ChatGPT embeddings of NCBI gene text descriptions, requiring no gene-expression pretraining.",
    "io": "Gene names / expression-ordered gene lists -> LLM-derived gene and cell embeddings",
    "status": "Open code + precomputed gene embeddings; no conventional trained model weights",
    "use_cases": "Cell type classification, gene-property prediction, perturbation embedding (GenePert), lightweight single-cell representation",
    "year": "2024",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41551-024-01284-6"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/yiqunchen/GenePT"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41551-024-01284-6",
      "https://github.com/yiqunchen/GenePT",
      "https://doi.org/10.5281/zenodo.10833191"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "GenePT releases code and precomputed OpenAI gene embeddings, not conventional trained model weights. Status should say code + precomputed embeddings. Paper, repo, embeddings DOI.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-scelmo",
    "date_added": "2026-06-16",
    "modality": "platform",
    "name": "scELMo",
    "description": "Framework turning LLM text embeddings of gene/metadata descriptions into reusable single-cell representations usable zero-shot or fine-tuned, without large-scale expression pretraining.",
    "io": "Gene/metadata descriptions + expression data -> LLM-based gene/cell embeddings",
    "status": "Open code + weights",
    "use_cases": "Cell clustering, batch-effect correction, cell type annotation, in-silico treatment / perturbation modeling",
    "year": "2025",
    "paper_links": [
      {
        "label": "Paper",
        "url": "https://www.cell.com/patterns/fulltext/S2666-3899(25)00279-X"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/HelloWorldLTY/scELMo"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.cell.com/patterns/fulltext/S2666-3899(25)00279-X",
      "https://github.com/HelloWorldLTY/scELMo"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact. Action: reclassify as a platform/method or HOLD/remove; it depends on external LLM APIs and does not release dedicated biological-model weights.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-langcell",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "LangCell",
    "description": "Language-cell pre-training framework aligning scRNA-seq profiles with cell-identity text (scLibrary, 27.5M cell-text pairs) for zero/few-shot cell-identity understanding.",
    "io": "scRNA-seq cell profile (+ optional text) -> cell embeddings / zero-shot cell-type and identity predictions",
    "status": "Open code + weights",
    "use_cases": "Zero-shot and few-shot cell type annotation, novel cell-type identification, cell-text retrieval, batch integration",
    "year": "2024",
    "paper_links": [
      {
        "label": "Final paper",
        "url": "https://proceedings.mlr.press/v235/zhao24u.html"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/PharMolix/LangCell"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://proceedings.mlr.press/v235/zhao24u.html",
      "https://github.com/PharMolix/LangCell"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Replace arXiv-only source with final ICML publication. ICML paper, repo.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-cellwhisperer",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "CellWhisperer",
    "description": "Multimodal CLIP-style model aligning transcriptomes (via Geneformer) and text (via BioBERT) with contrastive learning, enabling natural-language chat over single-cell data.",
    "io": "Transcriptome and/or natural-language query -> shared embedding, text answers, cell/gene retrieval",
    "status": "Open code + weights",
    "use_cases": "Conversational scRNA-seq exploration, automated cell annotation/description, transcriptome-text retrieval",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41587-025-02857-9"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/epigen/cellwhisperer"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41587-025-02857-9",
      "https://github.com/epigen/cellwhisperer"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-c2s-scale-cell2sentence-scale",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "C2S-Scale (Cell2Sentence-Scale)",
    "description": "Scaled Cell2Sentence family (up to 27B params, Gemma-2 based) from Google Research/DeepMind and Yale representing cells as 'cell sentences' of ranked gene names for LLM-native single-cell reasoning.",
    "io": "scRNA-seq profile as cell sentence (+ text prompt) -> generated text, predictions, hypotheses, cell sentences",
    "status": "Model hub (Hugging Face: C2S-Scale-Gemma-2 2B and 27B; Apache-2.0 code)",
    "use_cases": "Cell annotation, multicellular/multimodal reasoning, biological hypothesis generation, drug-response prediction",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.04.14.648850v2"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/vandijklab/cell2sentence"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.04.14.648850v2",
      "https://github.com/vandijklab/cell2sentence"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "add direct 2B/27B model-card and model-term URLs; document why this remains separate from Cell2Sentence.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-teddy",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "TEDDY",
    "description": "Family of single-cell foundation models (70M/160M/400M params) trained on a large annotated corpus via corrupted-expression reconstruction, with two variants (Teddy-G Geneformer-style, Teddy-X scGPT-style), introduced alongside a disease-focused benchmark.",
    "io": "scRNA-seq expression (corrupted) -> reconstructed expression / cell embeddings",
    "status": "Public Apache-2.0 code/configuration; model weight files not confirmed",
    "use_cases": "Cell representation learning, disease-state classification, scaling-law studies for cell FMs",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2503.03485"
      }
    ],
    "code_links": [
      {
        "label": "Official repository",
        "url": "https://huggingface.co/Merck/TEDDY"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2503.03485",
      "https://huggingface.co/Merck/TEDDY"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; official repository. Action: upgrade from paper-only to public Apache-2.0 code/configuration; state that model weight files were not confirmed.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-cellvq",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "CellVQ",
    "description": "Interpretable 500M-parameter single-cell foundation model with a Single-Cell Discretization (vector-quantization) module that turns sparse expression into discrete 'cell codes', trained on 68M cells; ships CellVQ-Graph for multimodal knowledge graphs.",
    "io": "scRNA-seq expression -> discrete cell code / embeddings; CellVQ-Graph builds multimodal knowledge graphs",
    "status": "Public code + pretrained checkpoint access",
    "use_cases": "Cell-state identification, interpretable representation learning, multimodal knowledge-graph construction for discovery",
    "year": "2026",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-026-70071-5"
      }
    ],
    "code_links": [
      {
        "label": "Code",
        "url": "https://github.com/A4Bio/CellVQ"
      },
      {
        "label": "Weights",
        "url": "https://modelscope.cn/models/wj1006/CellVQ/files"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-026-70071-5",
      "https://github.com/A4Bio/CellVQ",
      "https://modelscope.cn/models/wj1006/CellVQ/files"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; code; weights. Action: replace paper-only status with public code and pretrained checkpoint access.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-regformer",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "RegFormer",
    "description": "Single-cell foundation model that embeds gene regulatory network hierarchies into a Mamba state-space architecture for biologically grounded, efficient long-sequence modeling.",
    "io": "scRNA-seq expression (+ GRN priors) -> cell/gene embeddings, GRNs, task predictions",
    "status": "Open MIT code + public CC BY 4.0 checkpoints",
    "use_cases": "Cell annotation, gene regulatory network construction, genetic perturbation and drug-response prediction",
    "year": "2026",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-026-72198-x"
      }
    ],
    "code_links": [
      {
        "label": "Code",
        "url": "https://github.com/BGIResearch/RegFormer"
      },
      {
        "label": "Checkpoints",
        "url": "https://figshare.com/articles/online_resource/pretraining_models/28645493"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-026-72198-x",
      "https://github.com/BGIResearch/RegFormer",
      "https://figshare.com/articles/online_resource/pretraining_models/28645493"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; MIT code; checkpoints. Action: replace paper-only status with open MIT code and public CC BY 4.0 checkpoints.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-scprint-2",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "scPRINT-2",
    "description": "Next-generation cell foundation model from the Cantini lab pretrained on 350M cells across 16 organisms, adding a generative cell-level architecture for imputation and counterfactual reasoning, with a new benchmark suite.",
    "io": "scRNA-seq count data -> denoised expression, cell embeddings, cell-type predictions, generated/counterfactual cells",
    "status": "Open code + weights",
    "use_cases": "Expression denoising, cell embedding/annotation, imputation, counterfactual/generative single-cell modeling, benchmarking",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2025.12.11.693702v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/cantinilab/scPRINT-2"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2025.12.11.693702v1",
      "https://github.com/cantinilab/scPRINT-2"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-sclinguist",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "scLinguist",
    "description": "Pre-trained Hyena-based foundation model for cross-modality translation in single-cell multi-omics (scRNA-seq, single-cell proteomics, etc.), trained with self-supervised unimodal pretraining + paired post-pretraining.",
    "io": "Single-cell data in one modality -> translated/predicted profile in another modality + shared embeddings",
    "status": "Open MIT code, documentation and released checkpoints",
    "use_cases": "Cross-modality translation, multi-omics integration, missing-modality (e.g., protein-from-RNA) imputation",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.09.30.679123v1"
      }
    ],
    "code_links": [
      {
        "label": "Code",
        "url": "https://github.com/OmicsML/scLinguist"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.09.30.679123v1",
      "https://github.com/OmicsML/scLinguist",
      "https://doi.org/10.1101/2025.09.30.679123",
      "https://sclinguist.readthedocs.io/en/stable/install.html"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "“Paper/preprint only” is stale: official MIT repository, documentation, and three released checkpoints exist. Preprint, repo, documentation.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-captain",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "CAPTAIN",
    "description": "Multimodal single-cell foundation model jointly pretrained on co-assayed RNA and surface protein (CITE-seq-style) data over ~4M+ cells with 382 standardized surface proteins; initialized from scGPT.",
    "io": "Single-cell transcriptome (+ optional protein) -> joint RNA-protein embeddings, imputed proteins, annotations",
    "status": "Public MIT code + public pretrained checkpoints/data; checkpoint terms not separately stated",
    "use_cases": "Surface-protein imputation/expansion, cell type annotation, batch harmonization, RNA-protein cross-modal analysis",
    "year": "2026",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-026-72882-y"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/iamjiboya/CAPTAIN"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-026-72882-y",
      "https://github.com/iamjiboya/CAPTAIN"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-scconcept",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "scConcept",
    "description": "Contrastive-pretraining single-cell foundation model that learns count- and gene-panel-invariant cell embeddings by contrasting multiple cell views instead of gene-level reconstruction; pretrained on >30M scRNA-seq profiles.",
    "io": "scRNA-seq profile / gene-panel subset -> panel-invariant cell embedding",
    "status": "Open MIT code, package and public pretrained checkpoints",
    "use_cases": "Technology-agnostic cell representation, cross-platform integration, dissociated-to-spatial transfer, gene-panel optimization",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.10.14.682419v1"
      }
    ],
    "code_links": [
      {
        "label": "Code",
        "url": "https://github.com/theislab/scConcept"
      },
      {
        "label": "Weights",
        "url": "https://huggingface.co/theislab/scConcept"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.10.14.682419v1",
      "https://github.com/theislab/scConcept",
      "https://huggingface.co/theislab/scConcept",
      "https://doi.org/10.1101/2025.10.14.682419"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "“Paper/preprint only” is stale: official MIT code, PyPI package, and pretrained HF checkpoints now exist. Preprint, repo, model hub.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-pulsar",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "PULSAR",
    "description": "Multi-scale, multicellular foundation model from the Leskovec lab that aggregates gene->cell->donor representations, trained self-supervised on 36.2M immune cells from 6,807 donors to link single-cell profiles to clinical phenotypes.",
    "io": "Donor-level collection of single-cell profiles -> unified donor representation, disease/phenotype predictions, perturbation simulations",
    "status": "Open MIT code + public checkpoints",
    "use_cases": "Disease classification, biomarker prediction, clinical-event forecasting (e.g., RA onset), cytokine perturbation simulation across resolutions",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.11.24.685470v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/snap-stanford/PULSAR"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.11.24.685470v1",
      "https://github.com/snap-stanford/PULSAR"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "public MIT checkpoints now exist. Change “unclear weights” to open code + weights and add model-card URLs.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-stack-arc-institute",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "Stack (Arc Institute)",
    "description": "Arc Institute single-cell foundation model enabling in-context learning at inference time via a tabular-attention architecture: cells act as 'prompts' to instruct predictions on other cells without fine-tuning.",
    "io": "Context cells / prompt + query -> in-context predictions of cell state without parameter updates",
    "status": "Open code + weights (Hugging Face: arcinstitute/Stack-Large); non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Few-shot/zero-shot cell-state prediction, prompt-based condition simulation, rapid adaptation without fine-tuning",
    "year": "2026",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.01.09.698608v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ArcInstitute/stack"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.01.09.698608v1",
      "https://github.com/ArcInstitute/stack"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact. Action: set noncommercial:true; official licence is CC BY-NC-SA 4.0.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-sc-mamba2",
    "date_added": "2026-06-16",
    "modality": "singlecell",
    "name": "SC-MAMBA2",
    "description": "State-space (Mamba-2) single-cell foundation model (~625M params, pretrained on ~57M cells) for efficient ultra-long transcriptome modeling, capturing the full gene set with linear-time complexity and bidirectional processing.",
    "io": "Full-length scRNA-seq expression sequence -> cell/gene embeddings, task predictions",
    "status": "Open code, unclear weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Ultra-long transcriptome modeling, cell annotation, efficient large-context single-cell representation",
    "year": "2024",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.09.30.615775v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/xtalpi-xic/SC-MAMBA2"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.09.30.615775v1",
      "https://github.com/xtalpi-xic/SC-MAMBA2"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact. Action: set noncommercial:true; official licence is CC BY-NC 4.0.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-uni-uni2-h",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "UNI / UNI2-h",
    "description": "Self-supervised (DINOv2) ViT tile-level vision encoder for computational pathology. UNI is ViT-L/16 pretrained on 100M+ tiles from 100k+ H&E WSIs; UNI2-h is a ViT-H/14 (681M params) trained on 200M+ tiles from 350k H&E/IHC slides from Mass General Brigham.",
    "io": "H&E/IHC histology image tile (patch) -> dense tile embedding (feature vector) for downstream MIL/classification",
    "status": "Open code, gated weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Disease detection/diagnosis, cancer subtyping, biomarker/mutation prediction, survival, rare disease analysis, WSI feature-extraction backbone",
    "year": "2024",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41591-024-02857-3"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/mahmoodlab/UNI"
      }
    ],
    "canonical": true,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41591-024-02857-3",
      "https://github.com/mahmoodlab/UNI",
      "https://github.com/mahmoodlab/UNI#license-and-terms-of-tuse"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Gated weights and associated code/models are CC BY-NC-ND 4.0 for non-commercial academic use; noncommercial:false is wrong. Paper, repo terms.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-prov-gigapath",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "Prov-GigaPath",
    "description": "First whole-slide pathology foundation model combining a DINOv2 ViT-Giant tile encoder with a LongNet slide-level aggregator, pretrained on 1.3B tiles from 171,189 real-world WSIs (Providence health network).",
    "io": "WSI tiles -> tile embeddings -> LongNet slide-level embedding; supports tile-level and slide-level representation",
    "status": "Open code, gated weights",
    "use_cases": "Cancer subtyping, pathomics tasks, mutation/biomarker prediction, slide-level classification on gigapixel WSIs",
    "year": "2024",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-024-07441-w"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/prov-gigapath/prov-gigapath"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41586-024-07441-w",
      "https://github.com/prov-gigapath/prov-gigapath"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-conch",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "CONCH",
    "description": "Contrastive vision-language pathology foundation model (CoCa architecture, ViT-B/16 backbone) pretrained on 1.17M histopathology image-caption pairs, with aligned image and text encoders.",
    "io": "Histology image and/or pathology text -> aligned image/text embeddings; enables zero-shot classification, retrieval, captioning",
    "status": "Open code, gated weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Zero-shot tissue classification, image-to-text/text-to-image retrieval, captioning, segmentation, multimodal pathology search",
    "year": "2024",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41591-024-02856-4"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/mahmoodlab/CONCH"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41591-024-02856-4",
      "https://github.com/mahmoodlab/CONCH",
      "https://github.com/mahmoodlab/CONCH#license-and-terms-of-use"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Gated weights and code/model are CC BY-NC-ND 4.0 for non-commercial academic research; noncommercial:false is wrong. Paper, repo terms.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-titan",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "TITAN",
    "description": "Multimodal whole-slide foundation model learning slide-level representations via visual self-supervision plus vision-language alignment to 182k+ pathology reports and 423k synthetic captions, over 335,645 WSIs (Mass General Brigham).",
    "io": "Set of CONCH tile features for a WSI -> general-purpose slide-level embedding; can also generate pathology report text",
    "status": "Open code, gated weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Slide-level classification without fine-tuning (linear-probe/few-shot/zero-shot), rare disease retrieval, automated report generation, cross-modal slide retrieval",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41591-025-03982-3"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/mahmoodlab/TITAN"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41591-025-03982-3",
      "https://github.com/mahmoodlab/TITAN"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P1",
    "audit_note": "set NC true and state manual gating, CC BY-NC-ND terms, and decoder-weight omission.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-phikon-phikon-v2",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "Phikon / Phikon-v2",
    "description": "Owkin histology tile encoders. Phikon is an iBOT ViT-B pretrained on 40M TCGA tiles (~6k WSIs); Phikon-v2 is a DINOv2 ViT-L pretrained on 460M tiles from 55k+ public WSIs (PANCAN-XL: TCGA, CPTAC, GTEx).",
    "io": "H&E histology tile -> tile-level embedding for weakly-supervised WSI tasks",
    "status": "Model hub; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Cancer subtype classification, biomarker/molecular feature prediction, weakly-supervised WSI classification, feature extraction",
    "year": "2023",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2409.09173"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/owkin/HistoSSLscaling"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2409.09173",
      "https://github.com/owkin/HistoSSLscaling"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P1",
    "audit_note": "set NC true and cite separate Phikon and Phikon-v2 non-commercial model terms.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-musk",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "MUSK",
    "description": "Vision-language pathology foundation model using unified masked modeling on 50M pathology images and 1B text tokens (large-scale unpaired image-text), for precision/clinical oncology.",
    "io": "Pathology image and/or text -> aligned multimodal embeddings; patch- and slide-level outputs",
    "status": "Open code, gated weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Cross-modal retrieval, visual question answering, image classification, cancer prognosis and immunotherapy/relapse outcome prediction",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-024-08378-w"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/lilab-stanford/MUSK"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41586-024-08378-w",
      "https://github.com/lilab-stanford/MUSK"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact. Action: set noncommercial:true; code/model terms are CC BY-NC-ND 4.0 and academic-research-only.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-chief",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "CHIEF",
    "description": "Clinical Histopathology Imaging Evaluation Foundation model combining a self-supervised tile encoder with weakly-supervised whole-slide pretraining (15M tiles + 60k WSIs).",
    "io": "WSI tiles -> tile features -> slide-level representation/prediction (with optional anatomical site prompt)",
    "status": "Open code, gated weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Cancer detection, tumor origin prediction, genomic/mutation profile prediction, survival/prognosis prediction across 19 cancer types",
    "year": "2024",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-024-07894-z"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/hms-dbmi/CHIEF"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41586-024-07894-z",
      "https://github.com/hms-dbmi/CHIEF"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact. Action: set noncommercial:true and state academic/non-commercial restrictions alongside the GPLv3 reference.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-hibou-hibou-b-hibou-l",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "Hibou (Hibou-B / Hibou-L)",
    "description": "HistAI family of DINOv2 ViT pathology tile encoders pretrained on a proprietary dataset of 1M+ WSIs spanning diverse tissues and stains. Hibou-B (ViT-B) and Hibou-L (ViT-L).",
    "io": "Histology tile -> tile-level embedding for patch- and slide-level tasks",
    "status": "Open code + weights",
    "use_cases": "Patch and slide-level classification, biomarker prediction, WSI feature-extraction backbone",
    "year": "2024",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2406.05074"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/HistAI/hibou"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2406.05074",
      "https://github.com/HistAI/hibou"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-prism",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "PRISM",
    "description": "Paige multimodal generative slide-level pathology foundation model built on Virchow tile features (587k WSIs, 195k clinical reports); PRISM2 (2025) scales to 2.3M H&E WSIs with clinical reports and a 4B-parameter LLM backbone for clinical-grade cancer detection.",
    "io": "Set of Virchow tile features for a WSI -> slide-level embedding and/or generated diagnostic text",
    "status": "Model hub; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Slide-level cancer classification, zero-shot tasks, automated diagnostic report/summary generation",
    "year": "2024",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2405.10254"
      },
      {
        "label": "arXiv (PRISM2)",
        "url": "https://arxiv.org/abs/2506.13063"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/paige-ai/Prism"
      },
      {
        "label": "Hugging Face (Prism2)",
        "url": "https://huggingface.co/paige-ai/Prism2"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2405.10254",
      "https://arxiv.org/abs/2506.13063",
      "https://huggingface.co/paige-ai/Prism",
      "https://huggingface.co/paige-ai/Prism2"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P1",
    "audit_note": "set NC true; PRISM and PRISM2 are gated CC BY-NC-ND models.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-midnight-midnight-12k-midnight-92k",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "Midnight (Midnight-12k / Midnight-92k)",
    "description": "kaiko.ai DINOv2-based pathology tile foundation models showing SOTA-competitive performance with orders of magnitude fewer WSIs. Midnight-12k uses only public TCGA (12k WSIs); Midnight-92k adds proprietary NKI-80k.",
    "io": "H&E histology tile -> tile-level embedding",
    "status": "Open code + weights",
    "use_cases": "Patch/slide classification, biomarker prediction, data-efficient pathology feature-extraction backbone",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.05186"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/kaiko-ai/midnight"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2504.05186",
      "https://github.com/kaiko-ai/midnight"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-gpfm-generalizable-pathology-foundation-model",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "GPFM (Generalizable Pathology Foundation Model)",
    "description": "Pathology tile foundation model trained with a unified expert+self knowledge-distillation framework on 190M images from ~95k public WSIs across 34 tissue types.",
    "io": "Histology tile -> tile-level embedding",
    "status": "Open code, gated weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Broad pan-task pathology evaluation: classification, retrieval, survival, biomarker prediction, VQA, report generation across 72 tasks",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41551-025-01488-4"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/birkhoffkiki/GPFM"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41551-025-01488-4",
      "https://github.com/birkhoffkiki/GPFM"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P1",
    "audit_note": "set NC true and change “gated weights” to directly downloadable NC weights.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-openphenom-s-16-phenom-family",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "OpenPhenom-S/16 (Phenom family)",
    "description": "Recursion channel-agnostic masked-autoencoder (CA-MAE) ViT-S/16 microscopy foundation model for image-based phenotypic profiling, trained on millions of Cell Painting images (RxRx3, JUMP-CP). Open variant of proprietary Phenom-1/Phenom-2.",
    "io": "Multi-channel microscopy image (1/4/6/11 channels) -> phenotypic embedding (channel-agnostic)",
    "status": "Model hub; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Morphological/phenotypic profiling, mechanism-of-action classification, perturbation/drug-response analysis, batch-corrected representation learning",
    "year": "2024",
    "paper_links": [
      {
        "label": "Paper",
        "url": "https://openaccess.thecvf.com/content/CVPR2024/html/Kraus_Masked_Autoencoders_for_Microscopy_are_Scalable_Learners_of_Cellular_Biology_CVPR_2024_paper.html"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/recursionpharma/maes_microscopy"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "sources": [
      "https://openaccess.thecvf.com/content/CVPR2024/html/Kraus_Masked_Autoencoders_for_Microscopy_are_Scalable_Learners_of_Cellular_Biology_CVPR_2024_paper.html",
      "https://github.com/recursionpharma/maes_microscopy",
      "https://www.rxrx.ai/phenom"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Official project terms restrict model use to non-commercial purposes; the official project hub is cited.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-subcell",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "SubCell",
    "description": "Suite of proteome-aware self-supervised ViT vision foundation models for fluorescence microscopy, trained on Human Protein Atlas single-cell images (13k+ genes, 37 cell lines) with multitask learning over organelles/protein localization.",
    "io": "Multi-channel immunofluorescence single-cell image -> cell morphology / protein-localization embedding",
    "status": "Model hub",
    "use_cases": "Subcellular localization classification, cell-cycle modeling, drug-response prediction, mechanism-of-action identification, cross-dataset single-cell profiling",
    "year": "2024",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.12.06.627299v2"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/CellProfiling/subcell-embed"
      },
      {
        "label": "Model card",
        "url": "https://virtualcellmodels.cziscience.com/model/0193323e-ebd5-727c-bb32-87ed8f737213"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.12.06.627299v2",
      "https://github.com/CellProfiling/subcell-embed",
      "https://virtualcellmodels.cziscience.com/model/0193323e-ebd5-727c-bb32-87ed8f737213"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "“Model hub” is supportable, but the record links only training code. Add the CZI Virtual Cells model card, which identifies MIT licensing and the usable model. Preprint, repo, model card.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-cell-dino",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "Cell-DINO",
    "description": "Self-supervised DINOv2-adapted vision-transformer foundation model for cell fluorescence microscopy, learning single-cell morphology representations without manual annotation (Meta/Camille Couprie et al.).",
    "io": "Fluorescence microscopy single-cell image -> morphology embedding",
    "status": "Open code + weights",
    "use_cases": "Protein localization classification in low-annotation regimes, mechanism-of-action classification on Cell Painting, image-based profiling",
    "year": "2025",
    "paper_links": [
      {
        "label": "PLOS",
        "url": "https://journals.plos.org/ploscompbiol/article?id=10.1371/journal.pcbi.1013828"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/facebookresearch/dinov2/blob/main/docs/README_CELL_DINO.md"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://journals.plos.org/ploscompbiol/article?id=10.1371/journal.pcbi.1013828",
      "https://github.com/facebookresearch/dinov2/blob/main/docs/README_CELL_DINO.md"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-cell-painting-cnn-deepprofiler-model",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "Cell Painting CNN (DeepProfiler model)",
    "description": "Weakly-supervised EfficientNet CNN trained on Cell Painting Gallery datasets (488 treatments from 5 source studies) to produce generalizable single-cell morphological feature embeddings, used via the DeepProfiler tool.",
    "io": "Cell Painting microscopy single-cell crop -> morphological feature embedding (replacement for CellProfiler features)",
    "status": "Open code + weights",
    "use_cases": "Image-based morphological profiling, compound mechanism-of-action, genetic perturbation phenotyping, replicate reproducibility analysis",
    "year": "2024",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-024-45999-1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/cytomining/DeepProfiler"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-024-45999-1",
      "https://github.com/cytomining/DeepProfiler"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-nicheformer",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "Nicheformer",
    "description": "Transformer foundation model jointly pretrained on dissociated single-cell and spatially-resolved transcriptomics (SpatialCorpus-110M: ~57M dissociated + ~53M spatial cells across 73 tissues, human and mouse).",
    "io": "Single-cell or spatial transcriptome (tokenized expression) -> cell embedding; predicts spatial context for dissociated cells",
    "status": "Open code + weights",
    "use_cases": "Spatial composition/label prediction, niche/neighborhood analysis, transferring spatial context to scRNA-seq, spatial domain tasks",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41592-025-02814-z"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/theislab/nicheformer"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-025-02814-z",
      "https://github.com/theislab/nicheformer"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-scgpt-spatial",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "scGPT-spatial",
    "description": "Spatial-omics foundation model continually pretrained from scGPT with a Mixture-of-Experts decoder and coordinate-aware training on SpatialHuman30M (30M spatial cells/spots across Visium, Visium HD, MERFISH, Xenium).",
    "io": "Spatial transcriptomics expression (cells/spots with coordinates) -> spatially-aware cell/spot embeddings, imputed expression, deconvolution",
    "status": "Open code + weights",
    "use_cases": "Spatial domain clustering, multimodal integration, spot deconvolution, gene-expression imputation across spatial protocols",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.02.05.636714v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/bowang-lab/scGPT-spatial"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.02.05.636714v1",
      "https://github.com/bowang-lab/scGPT-spatial"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "MIT code and Figshare weights match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-mstar",
    "date_added": "2026-06-16",
    "modality": "spatial",
    "name": "mSTAR",
    "description": "Multimodal knowledge-enhanced whole-slide pathology foundation model (mSTAR = Multimodal Self-TAught pRetraining) that injects report text and gene-expression context into patch/slide representations; trained on 26,169 slide-level modality pairs across 32 cancer types (116M+ patches).",
    "io": "WSI tiles (+ optional report/gene-expression context at pretraining) -> multimodal slide- and patch-level representations",
    "status": "Open code, gated weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Molecular/biomarker prediction, multimodal slide-level classification, survival prediction across a 97-task oncology benchmark",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-025-66220-x"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Innse/mSTAR"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-025-66220-x",
      "https://github.com/Innse/mSTAR"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P1",
    "audit_note": "set NC true and state manually gated CC BY-NC-ND weights.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-biomedgpt",
    "date_added": "2026-06-16",
    "modality": "multimodal",
    "name": "BioMedGPT",
    "description": "Open multimodal generative biomedical LLM (BioMedGPT-10B) that aligns the feature spaces of molecules (SMILES), proteins, and natural language so users can query biological modalities in free text; also ships BioMedGPT-LM-7B, a Llama2-based biomedical text model.",
    "io": "Free-text question + molecule (SMILES)/protein sequence -> natural-language answer or property/QA output",
    "status": "Open code + weights",
    "use_cases": "Biomedical QA, molecule QA, protein QA, drug/target discovery assistance",
    "year": "2023",
    "paper_links": [
      {
        "label": "Final paper",
        "url": "https://ieeexplore.ieee.org/document/10767279/"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/PharMolix/OpenBioMed"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://ieeexplore.ieee.org/document/10767279/",
      "https://github.com/PharMolix/OpenBioMed",
      "https://github.com/PharMolix/OpenBioMed/blob/main/USE_POLICY.md"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Replace arXiv with the final IEEE article; expose direct model artifacts and the BioMedGPT Acceptable Use Policy rather than only generic “open code + weights.” Final paper, repo, AUP.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-chatnt",
    "date_added": "2026-06-16",
    "modality": "multimodal",
    "name": "ChatNT",
    "description": "Multimodal conversational agent built on the Nucleotide Transformer (NT-v2 500M DNA encoder + Perceiver resampler + Vicuna-7B decoder) that casts supervised DNA/RNA/protein genomics tasks as English text-to-text problems.",
    "io": "One or more DNA/RNA/protein sequences + English instruction -> natural-language answer/prediction",
    "status": "Model hub; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Conversational genomics, multi-task prediction in plain English, generalization to unseen questions on sequence tasks",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s42256-025-01047-1"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/InstaDeepAI/ChatNT"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s42256-025-01047-1",
      "https://huggingface.co/InstaDeepAI/ChatNT"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P1",
    "audit_note": "set NC true and cite the model-card non-commercial licence.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-instructprotein",
    "date_added": "2026-06-16",
    "modality": "multimodal",
    "name": "InstructProtein",
    "description": "Generative LLM with bidirectional protein<->text capability, instruction-tuned via a knowledge-graph-derived instruction dataset to align human and protein language.",
    "io": "Protein sequence -> text function description, OR natural-language prompt -> generated protein sequence",
    "status": "Open code + weights",
    "use_cases": "Protein function description, text-conditioned protein generation, protein-text alignment",
    "year": "2024",
    "paper_links": [
      {
        "label": "ACL",
        "url": "https://aclanthology.org/2024.acl-long.62/"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/HICAI-ZJU/InstructProtein"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://aclanthology.org/2024.acl-long.62/",
      "https://github.com/HICAI-ZJU/InstructProtein"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-prott3",
    "date_added": "2026-06-16",
    "modality": "multimodal",
    "name": "ProtT3",
    "description": "Protein-to-text framework coupling a protein language model encoder to a language model via a BLIP-2-style Q-Former cross-modal projector for text-based protein understanding.",
    "io": "Protein sequence -> generated text (captions, QA, descriptions)",
    "status": "Open code + weights",
    "use_cases": "Protein captioning, protein QA, text-based protein retrieval/understanding",
    "year": "2024",
    "paper_links": [
      {
        "label": "ACL",
        "url": "https://aclanthology.org/2024.acl-long.324/"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/acharkq/ProtT3"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://aclanthology.org/2024.acl-long.324/",
      "https://github.com/acharkq/ProtT3"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "Public code/checkpoints verify, but no explicit repository licence was found; mark reuse licence unclear. Paper, repo.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-mol-instructions",
    "date_added": "2026-06-16",
    "modality": "benchmark",
    "name": "Mol-Instructions",
    "description": "Large-scale biomolecular instruction dataset (~2M instructions: molecule, protein, biotext tasks) with released LLaMA/LLaMA2/LLaMA3-based instruction-tuned models for biomolecular LLMs.",
    "io": "Instruction + molecule (SELFIES/SMILES)/protein sequence -> task output (property, description, designed molecule/protein)",
    "status": "Open code + weights",
    "use_cases": "Instruction-tuning biomolecular LLMs, molecule/protein design and description, benchmark for bio-LLM instruction following",
    "year": "2024",
    "paper_links": [
      {
        "label": "OpenReview",
        "url": "https://openreview.net/forum?id=Tlsdsb6l9n"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/zjunlp/Mol-Instructions"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://openreview.net/forum?id=Tlsdsb6l9n",
      "https://github.com/zjunlp/Mol-Instructions"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-prot2text",
    "date_added": "2026-06-16",
    "modality": "multimodal",
    "name": "Prot2Text",
    "description": "Multimodal encoder-decoder that fuses a protein structure GNN (RGCN) and an ESM sequence encoder with a transformer/GPT-style decoder to generate free-text protein function descriptions.",
    "io": "Protein sequence + structure graph -> free-text function description",
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Free-text protein function prediction, annotation of novel proteins, structure-aware function generation",
    "year": "2024",
    "paper_links": [
      {
        "label": "Paper",
        "url": "https://ojs.aaai.org/index.php/AAAI/article/view/28948"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/hadi-abdine/Prot2Text"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://ojs.aaai.org/index.php/AAAI/article/view/28948",
      "https://github.com/hadi-abdine/Prot2Text"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Repository/model release is CC BY-NC-SA 4.0, but noncommercial:false. Paper, repo.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-txgemma",
    "date_added": "2026-06-16",
    "modality": "multimodal",
    "name": "TxGemma",
    "description": "Open Gemma-2-based LLM family (2B/9B/27B, predict + chat variants) from Google fine-tuned on a comprehensive therapeutics dataset (built largely on Therapeutics Data Commons) for prediction and agentic/conversational therapeutic tasks across 66 tasks.",
    "io": "Text prompt describing molecule/target/therapeutic task -> classification/regression/generation answer or conversational explanation",
    "status": "Gated/authenticated model access under Gemma terms",
    "use_cases": "ADMET/toxicity/BBB prediction, binding-affinity regression, retrosynthesis suggestions, agentic drug-discovery workflows (Agentic-Tx)",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.06196"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/google/txgemma-9b-chat"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2504.06196",
      "https://huggingface.co/google/txgemma-9b-chat"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "Paper; model. Action: specify gated/authenticated model access and Gemma terms rather than generic “model hub”.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-scchat",
    "date_added": "2026-06-16",
    "modality": "platform",
    "name": "scChat",
    "description": "LLM-powered multi-agent co-pilot that adds research context (RAG over CellMarker + GO/KEGG/Reactome/GSEA knowledge graphs, plus planner/executor agents) to scRNA-seq analysis through conversation.",
    "io": "AnnData scRNA-seq dataset + user question -> annotated cell types, enrichment, visualizations, hypotheses, next-step suggestions",
    "status": "Open code, unclear weights",
    "use_cases": "Conversational single-cell analysis, hypothesis validation, mechanistic interpretation, experiment planning",
    "year": "2024",
    "paper_links": [
      {
        "label": "Paper",
        "url": "https://onlinelibrary.wiley.com/doi/10.1002/aic.70285"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/li-group/scChat"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://onlinelibrary.wiley.com/doi/10.1002/aic.70285",
      "https://github.com/li-group/scChat"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact. Action: reclassify as a multi-agent analysis platform, not a reusable single-cell foundation model.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-tamarind-bio",
    "date_added": "2026-06-16",
    "modality": "platform",
    "name": "Tamarind Bio",
    "description": "No-code web platform/serving layer that hosts 250+ AI and simulation models (folding, design, docking, ADMET, MD) for running biological foundation models at scale without scripting.",
    "io": "Sequence/structure/ligand inputs via GUI or API -> structure predictions, designs, docking/affinity, property outputs",
    "status": "Web/API/commercial",
    "use_cases": "Running AlphaFold/RFdiffusion/Boltz/Chai/BoltzGen/GROMACS and many FMs at scale, protein/antibody/peptide design, binding prediction",
    "year": "2024",
    "paper_links": [],
    "code_links": [
      {
        "label": "Code",
        "url": "https://www.tamarind.bio/"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.tamarind.bio/"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "official platform",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-neurosnap",
    "date_added": "2026-06-16",
    "modality": "platform",
    "name": "Neurosnap",
    "description": "Zero-code browser platform exposing 100+ AI-driven bioinformatics tools (folding, binder/protein design, docking, ADMET, MD) through a guided interface.",
    "io": "Protein/ligand/sequence inputs -> folded structures, designs, docking poses, annotations, property predictions",
    "status": "Web/API/commercial",
    "use_cases": "Antibody/enzyme/peptide engineering, structure prediction, docking and screening without code",
    "year": "2024",
    "paper_links": [],
    "code_links": [
      {
        "label": "Code",
        "url": "https://neurosnap.ai/"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://neurosnap.ai/"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "platform identity is current, but add official product terms/pricing/API evidence; website availability alone is insufficient licence evidence.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-latch-bio",
    "date_added": "2026-06-16",
    "modality": "platform",
    "name": "Latch Bio",
    "description": "Cloud platform and open-source Python SDK (built on Flyte) for building, deploying, and running bioinformatics workflows (including FM-based pipelines) with auto-generated GUIs.",
    "io": "Nextflow/Snakemake/Python workflow + data -> managed cloud execution with GUI and storage integration",
    "status": "Web/API/commercial",
    "use_cases": "Deploying and scaling FM/bioinformatics pipelines, no-code workflow execution for wet-lab scientists, data+compute management",
    "year": "2022",
    "paper_links": [],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/latchbio/latch"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://github.com/latchbio/latch"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "commercial platform plus MIT SDK is correctly scoped. Persist project/licence sources.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-ginkgo-model-api-aa-0",
    "date_added": "2026-06-16",
    "modality": "platform",
    "name": "Ginkgo Model API (AA-0)",
    "description": "Ginkgo's 2024 hosted-model initiative, launched with AA-0 (AminoAcid-0), a 650M-parameter ESM-2-architecture protein LLM trained on public data plus 2B+ proprietary metagenomic protein sequences (UMDB).",
    "io": "Protein sequence -> embeddings/features, with masked/iterative design outputs (AA-0)",
    "status": "Historical API announcement; original developer portal unavailable as of 2026-07-13",
    "use_cases": "Historical reference for hosted AA-0 embeddings and masked-sequence generation; current access not confirmed",
    "year": "2024",
    "paper_links": [
      {
        "label": "Ginkgo Datapoints announcement",
        "url": "https://datapoints.ginkgo.bio/updates/introducing-ginkgo-s-model-api-a-programmable-interface-for-ginkgo-s-ai-research"
      }
    ],
    "code_links": [],
    "canonical": false,
    "noncommercial": false,
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "sources": [
      "https://datapoints.ginkgo.bio/updates/introducing-ginkgo-s-model-api-a-programmable-interface-for-ginkgo-s-ai-research"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-proteingym",
    "date_added": "2026-06-16",
    "modality": "benchmark",
    "name": "ProteinGym",
    "description": "Large-scale benchmark of ~2.7M missense variants across 217 deep-mutational-scanning assays (substitutions) plus ~300K indel mutants across 74 assays, with clinical variants, for evaluating protein fitness/variant-effect predictors and design models.",
    "io": "Protein sequence/MSA/structure + variants -> standardized Spearman/AUC/NDCG/MCC leaderboard scores",
    "status": "Open benchmark code + data (MIT); public leaderboard",
    "use_cases": "Benchmarking ESM/Tranception/EVE/PoET-style variant-effect models, substitution and indel fitness evaluation",
    "year": "2023",
    "paper_links": [
      {
        "label": "NeurIPS",
        "url": "https://proceedings.neurips.cc/paper_files/paper/2023/hash/cac723e5ff29f65e3fcbb0739ae91bee-Abstract-Datasets_and_Benchmarks.html"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/OATML-Markslab/ProteinGym"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "sources": [
      "https://proceedings.neurips.cc/paper_files/paper/2023/hash/cac723e5ff29f65e3fcbb0739ae91bee-Abstract-Datasets_and_Benchmarks.html",
      "https://github.com/OATML-Markslab/ProteinGym"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-bend",
    "date_added": "2026-06-16",
    "modality": "benchmark",
    "name": "BEND",
    "description": "Benchmark of biologically meaningful downstream tasks on the human genome (gene finding, enhancer/chromatin, variant effects, histone marks, CpG methylation) for evaluating DNA language models.",
    "io": "DNA LM embeddings + genome annotation tasks -> standardized task metrics",
    "status": "Open benchmark code, data and wrappers around third-party checkpoints; no own model weights",
    "use_cases": "Comparing DNABERT/Nucleotide Transformer/HyenaDNA/Caduceus on gene finding, enhancer, variant, and chromatin tasks",
    "year": "2024",
    "paper_links": [
      {
        "label": "Final paper",
        "url": "https://proceedings.iclr.cc/paper_files/paper/2024/hash/429e7b31625a8b7839f9e4d6e2aa9bb9-Abstract-Conference.html"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/frederikkemarin/BEND"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://proceedings.iclr.cc/paper_files/paper/2024/hash/429e7b31625a8b7839f9e4d6e2aa9bb9-Abstract-Conference.html",
      "https://github.com/frederikkemarin/BEND"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Benchmark role is correct, but “Open code + weights” is not: BEND provides benchmark data/code and wrappers around third-party checkpoints. Link the final ICLR paper. ICLR paper, repo.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-genomic-benchmarks",
    "date_added": "2026-06-16",
    "modality": "benchmark",
    "name": "Genomic Benchmarks",
    "description": "Collection of ready-to-use datasets for genomic sequence classification (promoters, enhancers, open chromatin across human/mouse/worm), distributed as a pip package with train/test splits and baselines.",
    "io": "Genomic sequences -> classification labels with train/test splits and baseline notebooks",
    "status": "Open benchmark datasets, package and baselines; no own model weights",
    "use_cases": "Prototyping and benchmarking DNA models on regulatory-element classification tasks",
    "year": "2023",
    "paper_links": [
      {
        "label": "Springer",
        "url": "https://link.springer.com/article/10.1186/s12863-023-01123-8"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ML-Bioinfo-CEITEC/genomic_benchmarks"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://link.springer.com/article/10.1186/s12863-023-01123-8",
      "https://github.com/ML-Bioinfo-CEITEC/genomic_benchmarks"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Benchmark distributes datasets/package/baselines, not model weights. Paper, repo.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-sceval",
    "date_added": "2026-06-16",
    "modality": "benchmark",
    "name": "scEval",
    "description": "Systematic task-level evaluation framework for single-cell foundation models across eight downstream tasks (annotation, batch integration, imputation, perturbation, GRN, multi-omics, etc.), with guidelines on pretraining/fine-tuning stability.",
    "io": "Single-cell FM + datasets/tasks -> standardized metrics and hyperparameter/stability analyses",
    "status": "Open benchmark code using external models/checkpoints; repository licence not declared",
    "use_cases": "Benchmarking scGPT/Geneformer/scBERT-style models on annotation, integration, imputation, perturbation, etc.",
    "year": "2024",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2023.09.08.555192v4"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/HelloWorldLTY/scEval"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2023.09.08.555192v4",
      "https://github.com/HelloWorldLTY/scEval"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "benchmark code uses external models/checkpoints. Change status accordingly; also record that repository has no explicit licence.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-open-problems-in-single-cell-analysis",
    "date_added": "2026-06-16",
    "modality": "benchmark",
    "name": "Open Problems in Single-Cell Analysis",
    "description": "Community-driven, containerized benchmarking platform that formalizes single-cell genomics tasks (12 tasks: denoising, label projection, batch integration, spatial decomposition, etc.) with standardized datasets, methods, and metrics, plus competitions.",
    "io": "Single-cell datasets + method/metric components -> reproducible leaderboards across tasks",
    "status": "Open benchmark code, tasks, datasets and containers; evaluated checkpoints are external",
    "use_cases": "Benchmarking and comparing single-cell methods/FMs, hosting open challenges, reproducible evaluation at scale",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41587-025-02694-w"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/openproblems-bio/openproblems"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41587-025-02694-w",
      "https://github.com/openproblems-bio/openproblems"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "change to “open benchmark code, tasks, datasets and containers”; it has no proprietary model weights.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-proteinbench",
    "date_added": "2026-06-16",
    "modality": "benchmark",
    "name": "ProteinBench",
    "description": "Holistic evaluation framework for protein foundation models spanning structure prediction, protein design, and conformational dynamics, scoring quality/novelty/diversity/robustness with a public leaderboard.",
    "io": "Protein FM outputs + task suite -> multi-metric scores and leaderboard rankings",
    "status": "Web leaderboard; code announced but repository unavailable",
    "use_cases": "Comparing protein generative/structure/design FMs (RFdiffusion, ESM3, Chroma, AlphaFold-style) across diverse tasks",
    "year": "2024",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2409.06744"
      }
    ],
    "code_links": [
      {
        "label": "Project / leaderboard",
        "url": "https://proteinbench.github.io/"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2409.06744",
      "https://proteinbench.github.io/"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-therapeutics-data-commons-tdc",
    "date_added": "2026-06-16",
    "modality": "benchmark",
    "name": "Therapeutics Data Commons (TDC)",
    "description": "Open platform curating 66+ ML-ready datasets across 22 drug-discovery tasks (ADMET, DTI, generation oracles, etc.) with leaderboards and a Python library (PyTDC).",
    "io": "Therapeutic ML task name -> standardized datasets, splits, oracles, and evaluation metrics",
    "status": "Open platform/benchmark datasets, library, oracles and leaderboards; no own model weights",
    "use_cases": "Benchmarking molecular/protein FMs on ADMET, binding, DTI, retrosynthesis; molecule-generation oracles; basis for TxGemma training",
    "year": "2021",
    "paper_links": [
      {
        "label": "Final paper",
        "url": "https://datasets-benchmarks-proceedings.neurips.cc/paper_files/paper/2021/hash/4c56ff4ce4aaf9573aa5dff913df997a-Abstract-round1.html"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/mims-harvard/TDC"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://datasets-benchmarks-proceedings.neurips.cc/paper_files/paper/2021/hash/4c56ff4ce4aaf9573aa5dff913df997a-Abstract-round1.html",
      "https://github.com/mims-harvard/TDC"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Benchmark/platform supplies datasets, library, oracles, and leaderboards—not weights. Replace arXiv with NeurIPS Datasets and Benchmarks paper. Final paper, repo.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-genslms",
    "date_added": "2026-06-20",
    "modality": "dna",
    "name": "GenSLMs",
    "description": "Genome-scale language models (2.5B-25B parameters) pretrained on 110M prokaryotic gene sequences and fine-tuned on 1.5M SARS-CoV-2 genomes to model evolutionary dynamics and identify variants of concern.",
    "io": "prokaryotic gene sequences / SARS-CoV-2 genomes -> evolutionary dynamics predictions, variant-of-concern classification",
    "status": "Open code + weights",
    "use_cases": "variant-of-concern identification, viral evolutionary dynamics, pandemic surveillance, genome-scale sequence modeling",
    "year": "2022",
    "paper_links": [
      {
        "label": "Final paper",
        "url": "https://journals.sagepub.com/doi/10.1177/10943420231201154"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ramanathanlab/genslm"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://journals.sagepub.com/doi/10.1177/10943420231201154",
      "https://github.com/ramanathanlab/genslm"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Replace preprint-only primary source with the published article. Final paper, repo.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-metagene-1",
    "date_added": "2026-06-20",
    "modality": "dna",
    "name": "METAGENE-1",
    "description": "A 7B-parameter autoregressive transformer pretrained on 1.5 trillion base pairs of metagenomic DNA and RNA from human wastewater, designed for pandemic monitoring, pathogen detection, and biosurveillance.",
    "io": "metagenomic DNA/RNA sequence -> sequence likelihoods, embeddings, and pathogen classification outputs",
    "status": "Open code + weights",
    "use_cases": "pandemic monitoring, pathogen detection, biosurveillance, metagenomic sequence analysis",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2501.02045"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/metagene-ai/METAGENE-1"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2501.02045",
      "https://huggingface.co/metagene-ai/METAGENE-1"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · model",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-cpgpt",
    "date_added": "2026-06-20",
    "modality": "dna",
    "name": "CpGPT",
    "description": "Transformer model pretrained on 150,000+ DNA methylation datasets for zero-shot methylation imputation, array conversion, aging biomarker prediction, and cross-tissue and cross-species transfer.",
    "io": "DNA methylation array profiles -> imputed methylation values, age predictions, cross-platform conversions",
    "status": "Open code + weights",
    "use_cases": "methylation imputation, array platform conversion, epigenetic aging clocks, cross-tissue transfer, cross-species analysis",
    "year": "2024",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.10.24.619766v3"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/lucascamillomd/CpGPT"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.10.24.619766v3",
      "https://github.com/lucascamillomd/CpGPT"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "MIT code and pretrained models match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-methylgpt",
    "date_added": "2026-06-20",
    "modality": "dna",
    "name": "MethylGPT",
    "description": "Transformer foundation model pretrained on 226,555 human DNA methylation profiles across 5,281 datasets, enabling methylome-based imputation, biological age prediction, and disease risk assessment across diverse tissue types.",
    "io": "DNA methylation profile (beta values) -> imputed methylation, predicted biological age, disease risk scores",
    "status": "Open code + weights",
    "use_cases": "methylation imputation, biological age prediction, disease risk assessment, tissue-type analysis",
    "year": "2024",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.10.30.621013v2"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/albert-ying/MethylGPT"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.10.30.621013v2",
      "https://github.com/albert-ying/MethylGPT"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Apache code and pretrained models match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-hicfoundation",
    "date_added": "2026-06-20",
    "modality": "dna",
    "name": "HiCFoundation",
    "description": "Self-supervised foundation model trained on 118 million Hi-C contact matrix submatrices for chromatin 3D architecture analysis across 316 species, supporting resolution enhancement, loop detection, single-cell Hi-C, and multi-omics integration.",
    "io": "Hi-C contact matrix submatrices -> chromatin 3D architecture predictions, enhanced resolution maps, loop calls, single-cell Hi-C profiles",
    "status": "Open code + weights",
    "use_cases": "chromatin 3D architecture analysis, Hi-C resolution enhancement, loop detection, single-cell Hi-C analysis, multi-omics integration",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41592-026-03097-8"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Noble-Lab/HiCFoundation"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-026-03097-8",
      "https://github.com/Noble-Lab/HiCFoundation"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-phylogpn",
    "date_added": "2026-06-20",
    "modality": "dna",
    "name": "PhyloGPN",
    "description": "Convolutional genomic language model (83M parameters) trained with a phylogenetic-tree loss on the Zoonomia multi-species alignment, producing per-nucleotide embeddings and variant-effect scores without requiring alignment at inference.",
    "io": "multi-species DNA alignment (training) / single-species nucleotide sequence (inference) -> per-nucleotide embeddings, variant-effect scores",
    "status": "Open code + weights",
    "use_cases": "variant effect prediction, regulatory element analysis, cross-species genomic comparison, non-coding variant interpretation",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2503.03773"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/songlab/PhyloGPN"
      },
      {
        "label": "Shared GPN code",
        "url": "https://github.com/songlab-cal/gpn"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2503.03773",
      "https://huggingface.co/songlab/PhyloGPN",
      "https://github.com/songlab-cal/gpn"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "Paper; weights; official code. Action: add the official shared GPN source repository.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-aido-dna",
    "date_added": "2026-06-20",
    "modality": "dna",
    "name": "AIDO.DNA",
    "description": "7B-parameter encoder-only transformer trained on 10.6 billion nucleotides from 796 species, achieving state-of-the-art performance on supervised, generative, and zero-shot functional genomics benchmarks.",
    "io": "DNA nucleotide sequence -> functional genomic annotations, sequence embeddings, zero-shot variant effect predictions",
    "status": "Open weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "variant effect prediction, regulatory element annotation, gene expression modeling, zero-shot functional genomics",
    "year": "2024",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.12.01.625444v2"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/genbio-ai/AIDO.DNA-7B"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.12.01.625444v2",
      "https://huggingface.co/genbio-ai/AIDO.DNA-7B"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "GenBio community terms are already conservatively represented. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-genos",
    "date_added": "2026-06-20",
    "modality": "dna",
    "name": "Genos",
    "description": "Human-centric genomic foundation model (1.2B and 10B parameters, MoE architecture) trained on 636 high-quality human assemblies with 1M-token context at single-nucleotide resolution for pathogenic mutation interpretation.",
    "io": "DNA sequence (human genome, single-nucleotide resolution) -> genomic variant pathogenicity predictions and mutation interpretations",
    "status": "Open code + weights",
    "use_cases": "pathogenic mutation interpretation, genomic variant analysis, human genome modeling",
    "year": "2025",
    "paper_links": [
      {
        "label": "Oxford",
        "url": "https://academic.oup.com/gigascience/article/doi/10.1093/gigascience/giaf132/8296738"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/BGI-HangzhouAI/Genos"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://academic.oup.com/gigascience/article/doi/10.1093/gigascience/giaf132/8296738",
      "https://github.com/BGI-HangzhouAI/Genos"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "published paper, Apache code, and model artifacts match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-genos-m",
    "date_added": "2026-06-20",
    "modality": "dna",
    "name": "Genos-m",
    "description": "Sparse mixture-of-experts genomic foundation model (4.7B total / ~330M active parameters) pretrained on 1.2 trillion nucleotides from human-associated microbial genomes, MAGs, and bacteriophages at single-nucleotide resolution with 1M-token context.",
    "io": "microbial nucleotide sequence -> genomic embeddings, sequence generation, functional annotations",
    "status": "Open code + weights",
    "use_cases": "microbial genome annotation, bacteriophage analysis, metagenome-assembled genome analysis, microbial sequence generation",
    "year": "2026",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.05.21.726868v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/BGI-HangzhouAI/Genos-m"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.05.21.726868v1",
      "https://github.com/BGI-HangzhouAI/Genos-m"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-epiagent",
    "date_added": "2026-06-20",
    "modality": "singlecell",
    "name": "EpiAgent",
    "description": "Transformer foundation model pretrained on ~5 million single-cell ATAC-seq profiles for chromatin accessibility analysis, supporting cell type annotation, data imputation, and in silico cis-regulatory perturbation.",
    "io": "single-cell ATAC-seq chromatin accessibility profiles -> cell type annotations, imputed accessibility, cis-regulatory perturbation predictions",
    "status": "Open code + weights",
    "use_cases": "cell type annotation, chromatin accessibility imputation, cis-regulatory element perturbation, single-cell epigenomics analysis",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41592-025-02822-z"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/xy-chen16/EpiAgent"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-025-02822-z",
      "https://github.com/xy-chen16/EpiAgent"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-eva",
    "date_added": "2026-06-20",
    "modality": "rna",
    "name": "EVA",
    "description": "Generative RNA foundation model (1.4B MoE, 114M sequences, 8k context) enabling controllable sequence generation across 11 RNA classes including tRNAs and therapeutic mRNAs.",
    "io": "RNA class label + optional conditioning -> RNA sequence",
    "status": "Open code + weights",
    "use_cases": "tRNA design, mRNA therapeutic design, controllable RNA sequence generation, multi-class RNA engineering",
    "year": "2026",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.03.17.712398v2"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/GENTEL-lab/EVA"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.03.17.712398v2",
      "https://github.com/GENTEL-lab/EVA"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-exai-1",
    "date_added": "2026-06-20",
    "modality": "rna",
    "name": "Exai-1",
    "description": "Multimodal cell-free RNA foundation model integrating sequence embeddings with cfRNA abundance data, pretrained on 306B tokens from blood samples for liquid biopsy and cancer detection.",
    "io": "cfRNA sequence + abundance data -> liquid biopsy embeddings for cancer detection",
    "status": "Open code, gated weights",
    "use_cases": "liquid biopsy, cancer detection, cfRNA analysis",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.03.26.645557v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/exai-oss/exai-1"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.03.26.645557v1",
      "https://github.com/exai-oss/exai-1"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-proteina",
    "date_added": "2026-06-20",
    "modality": "protein",
    "name": "Proteina",
    "description": "Flow-based protein backbone generator using hierarchical fold-class conditioning and a scalable non-equivariant transformer for de novo backbone design up to 800 residues.",
    "io": "fold-class conditioning + noise -> protein backbone coordinates (up to 800 residues)",
    "status": "Open weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "de novo protein backbone design, fold-conditioned structure generation, large protein design",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2503.00710"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/NVIDIA-BioNeMo/proteina"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2503.00710",
      "https://github.com/NVIDIA-BioNeMo/proteina"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "NC flag is already correct; official repository provides research-only weights. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-la-proteina",
    "date_added": "2026-06-20",
    "modality": "protein",
    "name": "La-Proteina",
    "description": "Partially-latent flow matching model from NVIDIA for joint de novo generation of amino acid sequence and full atomistic structure (backbone + side chains) without requiring a sequence as input.",
    "io": "noise -> amino acid sequence + all-atom 3D protein structure",
    "status": "Open code + weights",
    "use_cases": "de novo protein design, co-design of sequence and structure, drug discovery, synthetic biology",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2507.09466"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/NVIDIA-Digital-Bio/la-proteina"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2507.09466",
      "https://github.com/NVIDIA-Digital-Bio/la-proteina"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "Paper; artifact. Action: encode the separate terms: Apache-2.0 source, NVIDIA Open Model License weights, and CC BY materials.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-proteina-complexa",
    "date_added": "2026-06-20",
    "modality": "complex",
    "name": "Proteina-Complexa",
    "description": "Fully-atomistic protein binder design model from NVIDIA using flow-based generative pretraining with test-time compute scaling to design binders against protein and small-molecule targets.",
    "io": "protein or small-molecule target structure -> all-atom binder structure",
    "status": "Open code + weights",
    "use_cases": "protein binder design, drug target engagement, small-molecule co-binder generation, therapeutic protein engineering",
    "year": "2026",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.27950"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/NVIDIA-BioNeMo/Proteina-Complexa"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2603.27950",
      "https://github.com/NVIDIA-BioNeMo/Proteina-Complexa"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "Paper; artifact. Action: encode Apache-2.0 source, NVIDIA Open Model License weights, and CC BY 4.0 datasets separately.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-foldflow-2",
    "date_added": "2026-06-20",
    "modality": "protein",
    "name": "FoldFlow-2",
    "year": "2024",
    "description": "SE(3)-equivariant flow matching model for joint protein backbone and sequence co-design, trained on PDB and large synthetic structure datasets using sequence conditioning.",
    "io": "protein sequence (optional) -> protein backbone 3D structure + sequence",
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "de novo protein design, backbone generation, sequence-structure co-design, protein engineering",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2405.20313"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/DreamFold/FoldFlow"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2405.20313",
      "https://github.com/DreamFold/FoldFlow"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact. Action: set noncommercial:true; official licence is CC BY-NC 4.0.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-iggm",
    "date_added": "2026-06-20",
    "modality": "protein",
    "name": "IgGM",
    "description": "Generative foundation model for antibody and nanobody design that co-generates CDR sequences and full antibody-antigen complex structures conditioned on a target antigen.",
    "io": "antigen structure -> CDR sequences + antibody-antigen complex 3D structure",
    "status": "Open code + weights",
    "use_cases": "antibody design, nanobody design, CDR sequence generation, antibody-antigen complex modeling",
    "year": "2024",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.09.19.613838v2"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/TencentAI4S/IgGM"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.09.19.613838v2",
      "https://github.com/TencentAI4S/IgGM"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · repo",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-ablingua",
    "date_added": "2026-06-20",
    "modality": "protein",
    "name": "AbLingua",
    "year": "2026",
    "description": "Family of encoder-based antibody language models scaled to 1.7B parameters trained on 1.4B sequences, featuring improved tokenization that captures structural motifs; the largest antibody-specific encoder LM to date.",
    "io": "antibody sequence -> sequence embeddings / representations",
    "status": "Open public Apache-2.0 checkpoint; larger family models not released",
    "use_cases": "antibody design, CDR representation, binding affinity prediction, developability screening",
    "canonical": false,
    "noncommercial": false,
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s42003-026-10283-z"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/IDEA-AI4S/AbLingua"
      }
    ],
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s42003-026-10283-z",
      "https://huggingface.co/IDEA-AI4S/AbLingua"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-sceptr",
    "date_added": "2026-06-20",
    "modality": "protein",
    "name": "SCEPTR",
    "description": "TCR-specific protein language model pre-trained with autocontrastive learning and masked language modelling for data-efficient transfer learning on TCR-pMHC specificity prediction tasks.",
    "io": "TCR amino acid sequence -> TCR-pMHC binding specificity prediction",
    "status": "Open code + weights",
    "use_cases": "TCR-pMHC specificity prediction, T cell receptor binding analysis, antigen-specific T cell identification",
    "year": "2025",
    "paper_links": [
      {
        "label": "Final paper",
        "url": "https://www.sciencedirect.com/science/article/pii/S2405471224003697"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/yutanagano/sceptr"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.sciencedirect.com/science/article/pii/S2405471224003697",
      "https://github.com/yutanagano/sceptr",
      "https://doi.org/10.5281/zenodo.14286003"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Replace arXiv with final Cell Systems paper; weights are bundled/released and MIT. Final paper, repo, software DOI.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-mammal",
    "date_added": "2026-06-20",
    "modality": "multimodal",
    "name": "MAMMAL",
    "description": "IBM multi-modal multi-task foundation model trained on 2 billion biological samples spanning proteins, small molecules, and single-cell gene expression, achieving state-of-the-art on 9 of 11 drug-discovery benchmarks.",
    "io": "protein sequences, small molecules, single-cell gene expression -> task-specific predictions (property, interaction, generation)",
    "status": "Open code + weights",
    "use_cases": "drug discovery, protein property prediction, molecule-protein interaction, single-cell analysis",
    "year": "2024",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2410.22367"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/BiomedSciAI/biomed-multi-alignment"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2410.22367",
      "https://github.com/BiomedSciAI/biomed-multi-alignment"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "IBM MAMMAL repository, Apache code, and official model artifact match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-intellifold-2",
    "date_added": "2026-06-20",
    "modality": "complex",
    "name": "IntelliFold-2",
    "description": "Architectural refinement of IntelliFold surpassing AlphaFold3 on FoldBench via latent-space scaling in Pairformer, stochastic atomization, and policy-guided diffusion sampling for biomolecular structure prediction.",
    "io": "protein/nucleic acid/ligand sequences and features -> all-atom 3D complex structures",
    "status": "Open code + weights",
    "use_cases": "protein structure prediction, biomolecular complex modeling, drug target analysis, protein-ligand interaction studies",
    "year": "2026",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.02.09.704787v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/IntelliGen-AI/IntelliFold"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.02.09.704787v1",
      "https://github.com/IntelliGen-AI/IntelliFold"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "official Apache repository and IntelliFold weights match; upstream AF3 weights are explicitly excluded. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-placer",
    "date_added": "2026-06-20",
    "modality": "complex",
    "name": "PLACER",
    "description": "Graph neural network for rapidly generating conformational ensembles of protein-small molecule complexes by denoising partially corrupted atomic coordinates, enabling fast sampling of binding-relevant structural diversity.",
    "io": "protein-ligand complex structure (partially corrupted) -> conformational ensemble of protein-small molecule complex",
    "status": "Open code + weights",
    "use_cases": "conformational ensemble generation, protein-ligand docking, drug discovery, binding pose sampling",
    "year": "2025",
    "paper_links": [
      {
        "label": "PNAS",
        "url": "https://www.pnas.org/doi/10.1073/pnas.2427161122"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/baker-laboratory/PLACER"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.pnas.org/doi/10.1073/pnas.2427161122",
      "https://github.com/baker-laboratory/PLACER"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "BSD licence explicitly covers source and weights. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-atomica",
    "date_added": "2026-06-20",
    "modality": "complex",
    "name": "ATOMICA",
    "description": "SE(3)-equivariant geometric foundation model pretrained on 2 million molecular interaction interfaces from PDB and CSD to learn universal representations of intermolecular interactions across all biomolecular modality pairs.",
    "io": "3D molecular interaction interface (protein, nucleic acid, ligand, ion) -> interaction embeddings and interface representations",
    "status": "Open code + weights",
    "use_cases": "protein-ligand binding prediction, drug discovery, molecular docking, biomolecular interaction modeling, interface representation learning",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.04.02.646906v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/mims-harvard/ATOMICA"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.04.02.646906v1",
      "https://github.com/mims-harvard/ATOMICA"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "MIT code and official released checkpoints match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-pharmolixfm",
    "date_added": "2026-06-20",
    "modality": "complex",
    "name": "PharMolixFM",
    "description": "All-atom multimodal foundation model supporting diffusion, flow matching, and Bayesian flow network paradigms for protein-small molecule docking, structure-based drug design, and peptide design.",
    "io": "protein structures + small molecules -> docked complexes, designed ligands, peptide sequences",
    "status": "Open code + weights",
    "use_cases": "protein-ligand docking, structure-based drug design, peptide design",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2503.21788"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/PharMolix/OpenBioMed"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2503.21788",
      "https://github.com/PharMolix/OpenBioMed",
      "https://cloud.tsinghua.edu.cn/f/8f337ed5b58f45138659/"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "Identity and preview weights verify, but add the model download directly instead of relying on the umbrella OpenBioMed repo. Paper, repo, model.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-coati",
    "date_added": "2026-06-20",
    "modality": "molecule",
    "name": "COATI",
    "description": "Multimodal contrastive encoder-decoder aligning 3D conformer (E(3)-equivariant GNN) and SMILES (RoFormer) representations via CLIP-style learning, with autoregressive SMILES reconstruction for druglike chemical space.",
    "io": "3D molecular conformer or SMILES string -> aligned joint embedding; embedding -> SMILES sequence",
    "status": "Open code + weights",
    "use_cases": "molecular representation learning, drug discovery, virtual screening, molecule generation",
    "year": "2024",
    "paper_links": [
      {
        "label": "ACS",
        "url": "https://pubs.acs.org/doi/10.1021/acs.jcim.3c01753"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/terraytherapeutics/COATI"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://pubs.acs.org/doi/10.1021/acs.jcim.3c01753",
      "https://github.com/terraytherapeutics/COATI"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "distinguish original COATI release from COATI2, whose training code is not released.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-gem-chemrl-gem",
    "date_added": "2026-06-20",
    "modality": "molecule",
    "name": "GEM (ChemRL-GEM)",
    "description": "Geometry-Enhanced Molecular representation learning model that jointly encodes atoms, bonds, and bond angles via dual-graph GNN, pretrained with geometry-level self-supervised tasks including bond length, angle, and inter-atom distance prediction.",
    "io": "2D molecular graph + 3D geometry -> molecular property predictions",
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "molecular property prediction, drug discovery, ADMET modeling, quantum chemistry property estimation",
    "year": "2022",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2106.06130"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/PaddlePaddle/PaddleHelix/tree/dev/apps/pretrained_compound/ChemRL/GEM"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2106.06130",
      "https://github.com/PaddlePaddle/PaddleHelix/tree/dev/apps/pretrained_compound/ChemRL/GEM"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-chemgpt",
    "date_added": "2026-06-20",
    "modality": "molecule",
    "name": "ChemGPT",
    "description": "Autoregressive GPT-Neo model pretrained on SELFIES strings from PubChem at scales up to 1.3B parameters, used to establish neural scaling laws for generative molecular chemistry.",
    "io": "SELFIES molecular string -> generated SELFIES molecular string",
    "status": "Open weights",
    "use_cases": "molecule generation, drug discovery, molecular language modeling, scaling law benchmarking",
    "year": "2023",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s42256-023-00740-3"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/ncfrey/ChemGPT-1.2B"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s42256-023-00740-3",
      "https://huggingface.co/ncfrey/ChemGPT-1.2B"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "model exists but its current model card has no explicit licence. State “licence not declared,” not implied open reuse.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-chemdfm",
    "date_added": "2026-06-20",
    "modality": "molecule",
    "name": "ChemDFM",
    "description": "Chemistry dialogue foundation model built on LLaMA-13B, domain-pretrained on 34B tokens from chemical literature and fine-tuned on 2.7M chemistry instructions across property prediction, reaction prediction, molecule captioning, and retrosynthesis.",
    "io": "chemical text or molecular SMILES -> property predictions, reaction products, retrosynthesis routes, molecule captions",
    "status": "Open code + weights",
    "use_cases": "property prediction, reaction prediction, retrosynthesis, molecule captioning",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2401.14818"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/OpenDFM/ChemDFM"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2401.14818",
      "https://github.com/OpenDFM/ChemDFM"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "current repository links released model variants. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-nafm",
    "date_added": "2026-06-20",
    "modality": "molecule",
    "name": "NaFM",
    "description": "Foundation model for small-molecule natural products pretrained with scaffold-aware contrastive learning and masked graph objectives to capture evolutionary scaffold and side-chain variation.",
    "io": "natural product molecular graph -> scaffold-aware embeddings for classification, screening, and property prediction",
    "status": "Open code + weights",
    "use_cases": "taxonomic classification, virtual screening, drug discovery, natural product property prediction",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2503.17656"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/TomAIDD/NaFM-Official"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2503.17656",
      "https://github.com/TomAIDD/NaFM-Official"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "MIT code and Zenodo checkpoint match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-molfm",
    "date_added": "2026-06-20",
    "modality": "molecule",
    "name": "MolFM",
    "description": "Multimodal molecular foundation model that jointly learns from molecular graphs, biomedical text, and knowledge graphs via cross-modal attention, enabling unified representation across chemical and textual modalities.",
    "io": "molecular graph + biomedical text + knowledge graph -> joint molecular-text embeddings",
    "status": "Open code + weights",
    "use_cases": "molecule-text retrieval, drug property prediction, molecular captioning, cross-modal molecular search",
    "year": "2023",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2307.09484"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/PharMolix/OpenBioMed"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2307.09484",
      "https://github.com/PharMolix/OpenBioMed"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "generic OpenBioMed repository links a Baidu-hosted model. Add the direct official checkpoint and component-specific licence.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-novae",
    "date_added": "2026-06-20",
    "modality": "spatial",
    "name": "Novae",
    "description": "Self-supervised graph attention network trained on 29 million spatially resolved cells across 18 tissues for zero-shot spatial domain inference, batch correction, and spatially variable gene analysis across spatial transcriptomics technologies.",
    "io": "spatially resolved transcriptomics data -> spatial domain labels, batch-corrected embeddings, spatially variable genes",
    "status": "Open code + weights",
    "use_cases": "spatial domain inference, batch correction, spatially variable gene detection, cross-technology integration",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41592-025-02899-6"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/prism-oncology/novae"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-025-02899-6",
      "https://github.com/prism-oncology/novae"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "published paper, BSD code, and model card match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-maxtoki",
    "date_added": "2026-06-20",
    "modality": "singlecell",
    "name": "MaxToki",
    "description": "Autoregressive single-cell language model built on LLaMA (NVIDIA/Gladstone), pretrained on 175M single-cell transcriptomes; uses in-context learning to predict cell-state trajectories and age-modulating perturbations validated in vivo.",
    "io": "single-cell transcriptome profiles -> cell-state trajectory predictions, perturbation effect scores",
    "status": "Open code + weights",
    "use_cases": "cell-state trajectory inference, aging perturbation prediction, in-context cell reasoning, single-cell gene expression modeling",
    "year": "2026",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.03.30.715396v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/NVIDIA-Digital-Bio/maxToki"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.03.30.715396v1",
      "https://github.com/NVIDIA-Digital-Bio/maxToki"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-procyon",
    "date_added": "2026-06-20",
    "modality": "multimodal",
    "name": "ProCyon",
    "description": "11B parameter multimodal protein foundation model from Harvard MIMS that jointly encodes protein sequence, structure, small-molecule structure, and natural language to generate and predict protein phenotypes.",
    "io": "protein sequence, structure, small-molecule structure, natural language -> protein phenotype predictions, cross-modal retrieval, text-conditioned protein understanding",
    "status": "Open code + weights",
    "use_cases": "protein phenotype prediction, cross-modal protein retrieval, protein-drug interaction analysis, natural language querying of protein function",
    "year": "2024",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.12.10.627665v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/mims-harvard/ProCyon"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.12.10.627665v1",
      "https://github.com/mims-harvard/ProCyon"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-instructbiomol",
    "date_added": "2026-06-20",
    "modality": "multimodal",
    "name": "InstructBioMol",
    "description": "Any-to-any multimodal LLM that aligns natural language, small molecules, and proteins within a unified framework for biomolecular instruction following and cross-modal design.",
    "io": "text / molecule / protein sequence -> text / molecule / protein sequence (any-to-any)",
    "status": "Open code + weights",
    "use_cases": "molecule captioning, protein function description, molecule generation from text, protein design, cross-modal biomolecular retrieval",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2410.07919"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/HICAI-ZJU/InstructBioMol"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2410.07919",
      "https://github.com/HICAI-ZJU/InstructBioMol"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-naturelm",
    "date_added": "2026-06-20",
    "modality": "multimodal",
    "name": "NatureLM",
    "description": "Microsoft Research foundation model (1B-46.7B parameters) unifying small molecules, materials, proteins, DNA, RNA, and cells under a single sequence-based architecture with text-driven cross-domain generation.",
    "io": "text + molecular/biological sequences (small molecules, proteins, DNA, RNA, materials, cells) -> generated sequences, cross-domain predictions",
    "status": "Open weights",
    "use_cases": "protein design, molecule generation, materials discovery, cross-domain biological sequence modeling, text-guided molecular generation",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2502.07527"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/microsoft/NatureLM-8x7B"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2502.07527",
      "https://huggingface.co/microsoft/NatureLM-8x7B"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; model.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-biobridge",
    "date_added": "2026-06-20",
    "modality": "multimodal",
    "name": "BioBridge",
    "description": "Knowledge-graph framework that bridges independently trained unimodal biomedical foundation models (protein, small molecule, clinical text) enabling cross-modal retrieval and prediction without fine-tuning the base encoders.",
    "io": "protein sequence / small molecule / clinical text -> cross-modal embeddings and predictions",
    "status": "Open code + weights",
    "use_cases": "cross-modal biomedical retrieval, drug-target interaction, protein-phenotype linking, multimodal biomedical reasoning",
    "year": "2023",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2310.03320"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/RyanWangZf/BioBridge"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2310.03320",
      "https://github.com/RyanWangZf/BioBridge"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "paper/repository/checkpoints and MIT licence match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-openmedlab",
    "date_added": "2026-06-20",
    "modality": "platform",
    "name": "OpenMEDLab",
    "description": "Open-source platform and model hub for sharing and deploying foundation models across medical imaging, clinical NLP, bioinformatics, and protein engineering modalities.",
    "io": "multi-modal medical and biological data -> pretrained foundation models and benchmarks",
    "status": "Model hub",
    "use_cases": "medical image analysis, clinical NLP, bioinformatics, protein engineering, model sharing and deployment",
    "year": "2024",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2402.18028"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/openmedlab"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2402.18028",
      "https://github.com/openmedlab"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper · official organization",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "aliases": []
  },
  {
    "id": "bfm-amazon-bio-discovery",
    "date_added": "2026-06-20",
    "modality": "platform",
    "name": "Amazon Bio Discovery",
    "description": "AWS cloud platform providing access to 40+ wet-lab-validated biological foundation models via an AI agent interface, integrated with a laboratory partner network for closed-loop drug discovery workflows.",
    "io": "biological targets, sequences, or molecules -> drug candidates, binder designs, and experimental predictions",
    "status": "Web/API/commercial",
    "use_cases": "drug discovery, protein design, nanobody engineering, closed-loop wet-lab experimentation",
    "year": "2026",
    "paper_links": [
      {
        "label": "Amazon Science",
        "url": "https://www.amazon.science/publications/agent-guided-de-novo-design-of-nanobody-binders-against-a-novel-cancer-target"
      }
    ],
    "code_links": [
      {
        "label": "AWS",
        "url": "https://aws.amazon.com/biodiscovery/"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.amazon.science/publications/agent-guided-de-novo-design-of-nanobody-binders-against-a-novel-cancer-target",
      "https://aws.amazon.com/biodiscovery/"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; platform.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-latent-labs-latent-x-latent-x2",
    "date_added": "2026-06-20",
    "modality": "protein",
    "name": "Latent Labs (Latent-X / Latent-X2)",
    "description": "Generative all-atom protein design platform offering no-code and API access to models for binder and drug-like antibody design, with reported experimental hit rates up to 90%.",
    "io": "target protein structure -> designed binder or antibody sequences and all-atom structures",
    "status": "Web/API/commercial",
    "use_cases": "protein binder design, antibody engineering, drug-like antibody generation, therapeutic lead discovery",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2507.19375"
      }
    ],
    "code_links": [
      {
        "label": "Latent Labs",
        "url": "https://www.latentlabs.com/latent-x/"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2507.19375",
      "https://www.latentlabs.com/latent-x/"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; official product.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-omiclip-loki",
    "date_added": "2026-06-20",
    "modality": "spatial",
    "name": "OmiCLIP / Loki",
    "description": "Dual-encoder foundation model trained on 2.2 million paired histology tiles and spatial transcriptomics profiles from 32 organs, enabling spatially resolved gene expression prediction, cell-type decomposition, and cross-modal retrieval from H&E images.",
    "io": "H&E histology tile -> spatial transcriptomics expression profile, cell-type composition",
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "spatial transcriptomics prediction from histology, cell-type deconvolution, cross-modal retrieval, multi-organ spatial analysis",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41592-025-02707-1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/GuangyuWangLab2021/Loki"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-025-02707-1",
      "https://github.com/GuangyuWangLab2021/Loki"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; artifact. Action: set noncommercial:true and identify the custom academic/non-commercial restrictions rather than relying on its “BSD-3” label.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-kronos",
    "date_added": "2026-06-20",
    "modality": "spatial",
    "name": "KRONOS",
    "description": "Self-supervised foundation model for spatial proteomics trained on 47 million multiplex imaging patches across 175 protein markers, 16 tissue types, and 8 fluorescence platforms including CODEX, MxIF, and IBEX.",
    "io": "multiplex protein imaging patches -> spatial proteomics embeddings",
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "spatial proteomics analysis, tissue phenotyping, protein marker embedding, multiplex imaging interpretation",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2506.03373"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/mahmoodlab/KRONOS"
      }
    ],
    "canonical": true,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2506.03373",
      "https://github.com/mahmoodlab/KRONOS"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P1",
    "audit_note": "set NC true and state gated CC BY-NC-ND weights.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-virtues",
    "date_added": "2026-06-20",
    "modality": "spatial",
    "name": "VirTues",
    "description": "Foundation model for multiplex spatial proteomics that learns marker-aware, multi-scale representations from protein imaging data, enabling zero-shot cell typing, niche annotation, and patient stratification.",
    "io": "multiplex spatial protein images -> multi-scale cell and tissue representations",
    "status": "Open code + weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "zero-shot cell typing, tissue niche annotation, patient stratification, spatial proteomics analysis",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2501.06039"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/bunnelab/virtues"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2501.06039",
      "https://github.com/bunnelab/virtues",
      "https://github.com/bunnelab/virtues#models"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Mixed access is flattened incorrectly: code is MIT, virtues-sp32 and virtues-imc14 weights are CC BY-NC 4.0, while virtues-sp31 is MIT. Status must expose per-checkpoint terms instead of generic open + noncommercial:false. Paper, repo model table.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-stofm",
    "date_added": "2026-06-20",
    "modality": "spatial",
    "name": "SToFM",
    "description": "Multi-scale spatial transcriptomics foundation model pretrained on SToCorpus-88M using an SE(2) Transformer to jointly model macro tissue morphology, cellular niche context, and gene expression across resolution scales.",
    "io": "spatial transcriptomics tissue image + gene expression -> multi-scale spatial embeddings",
    "status": "Open code + weights",
    "use_cases": "spatial gene expression prediction, tissue morphology analysis, cellular niche characterization, spatial omics representation learning",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2507.11588"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/PharMolix/SToFM"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2507.11588",
      "https://github.com/PharMolix/SToFM"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "MIT code, checkpoint, and released corpus match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-storm",
    "date_added": "2026-06-20",
    "modality": "spatial",
    "name": "STORM",
    "description": "Multimodal foundation model jointly pretrained on 1.2 million spatially resolved transcriptomic profiles and paired H&E images across 18 organs, enabling virtual ST prediction, spatial domain discovery, and clinical outcome prediction.",
    "io": "H&E histology image + spatial transcriptomics profiles -> predicted spatial gene expression, spatial domain annotations, clinical outcome predictions",
    "status": "Paper + official project page; no public code or weights verified",
    "use_cases": "virtual spatial transcriptomics from H&E, spatial domain discovery, clinical outcome prediction, multi-organ spatial analysis",
    "year": "2026",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.03630"
      }
    ],
    "code_links": [
      {
        "label": "Official project",
        "url": "https://storm-web-demo.vercel.app/"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2604.03630",
      "https://storm-web-demo.vercel.app/"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Paper; official project. Action: add the project page, state that no public code/weights were found, and set noncommercial:false unless an actual non-commercial artifact licence is sourced.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-pathorchestra",
    "date_added": "2026-06-20",
    "modality": "spatial",
    "name": "PathOrchestra",
    "description": "Pathology foundation model trained on 300,000 whole-slide images across 20 tissue types, evaluated on 112 clinical-grade tasks spanning pan-cancer classification, biomarker assessment, and structured report generation.",
    "io": "whole-slide pathology image -> cancer classification, biomarker predictions, structured pathology reports",
    "status": "Open code, gated weights; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "pan-cancer classification, biomarker assessment, pathology report generation, tissue type analysis",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2503.24345"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/yanfang-research/PathOrchestra"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2503.24345",
      "https://github.com/yanfang-research/PathOrchestra"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-cellsam",
    "date_added": "2026-06-20",
    "modality": "spatial",
    "name": "CellSAM",
    "description": "Universal cell segmentation foundation model combining a learned cell detector (CellFinder) with SAM to generalize across mammalian cells, yeast, and bacteria in diverse imaging modalities.",
    "io": "microscopy images -> cell instance segmentation masks",
    "status": "Open code + weights",
    "use_cases": "cell segmentation, microscopy image analysis, multi-organism cell detection, high-throughput biological imaging",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41592-025-02879-w"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/vanvalenlab/cellSAM"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41592-025-02879-w",
      "https://github.com/vanvalenlab/cellSAM"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "current Nature Methods paper, Apache inference code, and weights match. Persist.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "aliases": []
  },
  {
    "id": "bfm-tape-tasks-assessing-protein-embeddings",
    "date_added": "2026-06-20",
    "modality": "benchmark",
    "name": "TAPE (Tasks Assessing Protein Embeddings)",
    "description": "NeurIPS 2019 benchmark of five semi-supervised protein tasks spanning secondary structure prediction, contact prediction, remote homology detection, fluorescence, and stability, used to evaluate protein language models.",
    "io": "protein sequence -> task-specific predictions (structure, homology, fitness)",
    "status": "Open code + weights",
    "use_cases": "protein language model evaluation, secondary structure prediction, remote homology detection, protein engineering fitness prediction",
    "year": "2019",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/1906.08230"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/songlab-cal/tape"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/1906.08230",
      "https://github.com/songlab-cal/tape"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "benchmark, BSD code and baseline weights remain available, but README says training code is no longer maintained. Mark historical maintenance state.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-peer-protein-sequence-understanding",
    "date_added": "2026-06-20",
    "modality": "benchmark",
    "name": "PEER (Protein sEquence undERstanding)",
    "description": "Multi-task benchmark covering 17 protein tasks across five groups (function, localization, structure, protein-protein interaction, and protein-ligand interaction) for systematic evaluation of protein language models.",
    "io": "protein sequence -> performance scores across 17 protein understanding tasks",
    "status": "Open benchmark code, datasets and configurations; no own model weights",
    "use_cases": "protein FM evaluation, multi-task benchmarking, protein function prediction, protein structure assessment",
    "year": "2022",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2206.02096"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/DeepGraphLearning/PEER_Benchmark"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2206.02096",
      "https://github.com/DeepGraphLearning/PEER_Benchmark"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "change to open benchmark code/datasets/configurations, not own weights.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-genbench-genomic-foundation-models",
    "date_added": "2026-06-20",
    "modality": "benchmark",
    "name": "GenBench (Genomic Foundation Models)",
    "description": "NeurIPS 2024 benchmarking suite for systematic evaluation of genomic foundation models across short- and long-range DNA tasks spanning coding regions, non-coding regions, and genome structure.",
    "io": "DNA sequences -> benchmark scores across genomic tasks",
    "status": "Open code + weights",
    "use_cases": "genomic model evaluation, DNA sequence modeling, regulatory element analysis, genome structure prediction",
    "year": "2024",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2406.01627"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/jimmylihui/GenBench"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2406.01627",
      "https://github.com/jimmylihui/GenBench"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "say “open benchmark code/data/baseline assets,” not generically “weights.”",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-dnalongbench",
    "date_added": "2026-06-20",
    "modality": "benchmark",
    "name": "DNALONGBENCH",
    "description": "Benchmark suite of five genomic tasks with long-range dependencies up to 1 million base pairs, designed to evaluate and expose limitations of DNA foundation models on long-range prediction.",
    "io": "DNA sequences (up to 1 Mbp) -> genomic task predictions (classification/regression across five long-range tasks)",
    "status": "Open benchmark datasets, evaluation code and baselines; no own model weights",
    "use_cases": "benchmarking DNA foundation models, long-range genomic prediction, model evaluation, genome biology research",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-025-65077-4"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/wenduocheng/DNALongBench"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41467-025-65077-4",
      "https://github.com/wenduocheng/DNALongBench"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Benchmark provides datasets, evaluation code, and baselines, not released benchmark “weights.” Change status to code + benchmark data/baselines. Paper, repo.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-nabench",
    "date_added": "2026-06-20",
    "modality": "benchmark",
    "name": "NABench",
    "description": "Large-scale benchmark aggregating 162 high-throughput assays and 2.6 million mutated sequences across diverse DNA and RNA families to evaluate nucleotide foundation models on fitness prediction tasks.",
    "io": "nucleotide sequences (DNA/RNA mutants) -> fitness prediction scores",
    "status": "Open code + weights",
    "use_cases": "nucleotide model evaluation, fitness prediction benchmarking, DNA/RNA model comparison",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2511.02888"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/mrzzmrzz/NABench"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2511.02888",
      "https://github.com/mrzzmrzz/NABench"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Benchmark provides data, scores, and evaluation code; contributors are instructed to supply their own model checkpoints. Change status from “weights” to benchmark data/code/results. Paper, repo.",
    "audit_source": "evidence/source-audit/foundation_residual_mod0.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-rnagym",
    "date_added": "2026-06-20",
    "modality": "benchmark",
    "name": "RNAGym",
    "description": "Benchmark suite for RNA fitness and structure prediction covering 31 fitness assays with over 1.1 million variants and secondary/tertiary structure tasks, serving as the ProteinGym analog for RNA.",
    "io": "RNA sequence variants -> fitness scores and secondary/tertiary structure predictions",
    "status": "Open code + weights",
    "use_cases": "RNA fitness prediction, RNA secondary structure evaluation, RNA tertiary structure benchmarking, foundation model assessment",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.06.16.660049v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/MarksLab-DasLab/RNAGym"
      }
    ],
    "canonical": true,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.06.16.660049v1",
      "https://github.com/MarksLab-DasLab/RNAGym"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P3",
    "audit_note": "describe open benchmark code/data/baseline checkpoints rather than generic own weights.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-perteval-scfm",
    "date_added": "2026-06-20",
    "modality": "benchmark",
    "name": "PertEval-scFM",
    "description": "ICML 2025 benchmarking framework evaluating single-cell foundation models on perturbation effect prediction, finding that scFM zero-shot embeddings provide minimal improvement over simple baselines under distribution shift.",
    "io": "scRNA-seq profiles + perturbation conditions -> perturbation effect prediction benchmarks",
    "status": "Open code",
    "use_cases": "benchmarking scFMs, perturbation effect prediction, model evaluation under distribution shift, single-cell transcriptomics",
    "year": "2025",
    "paper_links": [
      {
        "label": "PMLR",
        "url": "https://proceedings.mlr.press/v267/wenteler25a.html"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/aaronwtr/PertEval"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://proceedings.mlr.press/v267/wenteler25a.html",
      "https://github.com/aaronwtr/PertEval"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; artifact.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-pathbench",
    "date_added": "2026-06-20",
    "modality": "benchmark",
    "name": "PathBench",
    "description": "Benchmark evaluating 19 pathology foundation models across 64 clinical diagnosis and prognosis tasks using 15,888 whole-slide images from 8,549 patients across 10 hospitals, with rigorous data leakage prevention.",
    "io": "whole-slide images -> diagnostic and prognostic performance scores across 64 tasks",
    "status": "Open benchmark application and data; evaluated model weights are external",
    "use_cases": "pathology model evaluation, cancer diagnosis benchmarking, clinical prognosis assessment, foundation model comparison",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.20202"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/birkhoffkiki/PathBench"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2505.20202",
      "https://github.com/birkhoffkiki/PathBench"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "change from “code + weights” to open benchmark application/data; evaluated models are external.",
    "audit_source": "evidence/source-audit/foundation_residual_mod1.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-comet-multi-omics-evaluation",
    "date_added": "2026-06-20",
    "modality": "benchmark",
    "name": "COMET (Multi-Omics Evaluation)",
    "description": "Comprehensive benchmark evaluating biological language models across single-omics, cross-omics, and multi-omics tasks spanning DNA, RNA, and protein modalities.",
    "io": "DNA/RNA/protein sequences -> standardized benchmark scores across omics tasks",
    "status": "Paper/preprint only",
    "use_cases": "model evaluation, multi-omics integration assessment, biological language model benchmarking, cross-omics task comparison",
    "year": "2024",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2412.10347"
      }
    ],
    "code_links": [],
    "canonical": false,
    "noncommercial": false,
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2412.10347"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Paper; no official reusable artifact found. Action: retain conservative paper-only status.",
    "audit_source": "evidence/source-audit/foundation_residual_mod2.md",
    "aliases": []
  },
  {
    "id": "bfm-space",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "dna",
    "name": "SPACE",
    "description": "Multi-species genomic-profile foundation model for transferable DNA representations and cross-species regulatory prediction.",
    "io": "DNA sequence -> genomic profiles and transferable embeddings",
    "status": "Public code + checkpoint; checkpoint reuse terms not separately stated",
    "use_cases": "Regulatory prediction, genomic representation learning, cross-species transfer",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2506.01833"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ZhuJiwei111/SPACE"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/yangyz1230/space"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://arxiv.org/abs/2506.01833",
      "https://github.com/ZhuJiwei111/SPACE",
      "https://huggingface.co/yangyz1230/space"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-bmfm-dna",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "dna",
    "name": "BMFM-DNA",
    "description": "SNP-aware and reference 113M ModernBERT genomic model family for learning DNA representations with or without explicit variants.",
    "io": "DNA sequence +/- variants -> genomic representations",
    "status": "Public code + reference and SNP-aware checkpoints",
    "use_cases": "Variant-effect modeling, regulatory prediction, transferable genomic embeddings",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2507.05265"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/BiomedSciAI/biomed-multi-omic"
      },
      {
        "label": "Reference model",
        "url": "https://huggingface.co/ibm-research/biomed.dna.ref.modernbert.113m.v1"
      },
      {
        "label": "SNP model",
        "url": "https://huggingface.co/ibm-research/biomed.dna.snp.modernbert.113m.v1"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://arxiv.org/abs/2507.05265",
      "https://github.com/BiomedSciAI/biomed-multi-omic",
      "https://huggingface.co/ibm-research/biomed.dna.ref.modernbert.113m.v1",
      "https://huggingface.co/ibm-research/biomed.dna.snp.modernbert.113m.v1"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-omni-dna",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "dna",
    "name": "Omni-DNA",
    "description": "Autoregressive 20M-to-1B genomic model family trained at multi-species scale for long-context DNA understanding, annotation, and generation.",
    "io": "DNA sequence and optional textual context -> sequence likelihoods, annotations, or generated DNA",
    "status": "Public code + weights",
    "use_cases": "Genomic sequence generation, annotation, long-context representation learning",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2502.03499"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Zehui127/Omni-DNA"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/zehui127/Omni-DNA-1B"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://arxiv.org/abs/2502.03499",
      "https://github.com/Zehui127/Omni-DNA",
      "https://huggingface.co/zehui127/Omni-DNA-1B"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-nucel",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "dna",
    "name": "NucEL",
    "description": "Single-nucleotide ELECTRA-style genomic representation model designed to learn nucleotide- and sequence-level features.",
    "io": "DNA sequence -> nucleotide and sequence representations",
    "status": "Public source repository + checkpoint",
    "use_cases": "Genomic representation learning, sequence classification, regulatory transfer",
    "year": "2026",
    "paper_links": [
      {
        "label": "AAAI",
        "url": "https://ojs.aaai.org/index.php/AAAI/article/view/36982"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/FreakingPotato/NucEL"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/FreakingPotato/NucEL"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://ojs.aaai.org/index.php/AAAI/article/view/36982",
      "https://github.com/FreakingPotato/NucEL",
      "https://huggingface.co/FreakingPotato/NucEL"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-d3lm",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "dna",
    "name": "D3LM",
    "description": "Discrete masked-diffusion DNA language model for genomic representations and mammalian-sequence generation.",
    "io": "DNA sequence or masked sequence -> representations or generated/completed DNA",
    "status": "Public checkpoints with custom model code; standalone training repository not confirmed",
    "use_cases": "DNA generation, sequence completion, genomic representation learning",
    "year": "2026",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.01780"
      }
    ],
    "code_links": [
      {
        "label": "NT-initialized model",
        "url": "https://huggingface.co/Hengchang-Liu/D3LM-from-nt"
      },
      {
        "label": "Scratch model",
        "url": "https://huggingface.co/Hengchang-Liu/D3LM-scratch"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://arxiv.org/abs/2603.01780",
      "https://huggingface.co/Hengchang-Liu/D3LM-from-nt",
      "https://huggingface.co/Hengchang-Liu/D3LM-scratch"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-onegenome-rice",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "dna",
    "name": "OneGenome-Rice",
    "description": "Rice genomic mixture-of-experts generative model trained across cultivated and wild genomes with context up to one megabase.",
    "io": "Rice genomic sequence -> long-context representations or generated sequence",
    "status": "Public code + weights",
    "use_cases": "Rice regulatory genomics, introgression analysis, breeding and sequence generation",
    "year": "2026",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://doi.org/10.64898/2026.04.21.719822"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/zhejianglab/OneGenome-Rice"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/ZhejiangLab/OneGenome-Rice"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "aliases": [
      "OGR"
    ],
    "sources": [
      "https://doi.org/10.64898/2026.04.21.719822",
      "https://github.com/zhejianglab/OneGenome-Rice",
      "https://huggingface.co/ZhejiangLab/OneGenome-Rice"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md"
  },
  {
    "id": "bfm-hydrarna",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "rna",
    "name": "HydraRNA",
    "description": "Hybrid state-space and attention RNA language model trained on full-length coding and non-coding RNAs with long sequence support.",
    "io": "RNA sequence -> embeddings, likelihoods, or downstream predictions",
    "status": "Public code + downloadable checkpoints",
    "use_cases": "RNA representation learning, long-RNA analysis, coding and non-coding RNA transfer",
    "year": "2025",
    "paper_links": [
      {
        "label": "Genome Biology",
        "url": "https://doi.org/10.1186/s13059-025-03853-7"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/GuipengLi/HydraRNA"
      },
      {
        "label": "Weights",
        "url": "https://drive.google.com/drive/folders/14ZXi_aANEEdPa_Sc2cQZtUa4dENPDTkz"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://doi.org/10.1186/s13059-025-03853-7",
      "https://github.com/GuipengLi/HydraRNA",
      "https://drive.google.com/drive/folders/14ZXi_aANEEdPa_Sc2cQZtUa4dENPDTkz"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-structrfm",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "rna",
    "name": "structRFM",
    "description": "Structure-guided RNA foundation model learning joint sequence, secondary-structure, and functional representations.",
    "io": "RNA sequence and structure context -> transferable embeddings and predictions",
    "status": "Public code + dataset + checkpoints",
    "use_cases": "RNA structure/function prediction, embeddings, downstream transfer",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.08.06.668731v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/heqin-zhu/structRFM"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/heqin-zhu/structRFM"
      },
      {
        "label": "Zenodo",
        "url": "https://doi.org/10.5281/zenodo.16754363"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.08.06.668731v1",
      "https://github.com/heqin-zhu/structRFM",
      "https://huggingface.co/heqin-zhu/structRFM",
      "https://doi.org/10.5281/zenodo.16754363"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-codonfm-encodon",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "rna",
    "name": "CodonFM / EnCodon",
    "description": "Codon-level foundation-model family trained across species for representations, sequence-property prediction, and codon-aware generation.",
    "io": "Coding sequence at codon resolution -> embeddings, predictions, or optimized sequence",
    "status": "Public code + checkpoints under NVIDIA model terms",
    "use_cases": "Expression and stability prediction, codon optimization, coding-sequence design",
    "year": "2025",
    "paper_links": [
      {
        "label": "NVIDIA preprint",
        "url": "https://research.nvidia.com/labs/dbr/assets/data/manuscripts/nv-codonfm-preprint.pdf"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/NVIDIA-BioNeMo/CodonFM"
      },
      {
        "label": "EnCodon weights",
        "url": "https://huggingface.co/nvidia/NV-CodonFM-Encodon-80M-v1"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "aliases": [
      "CodonFM",
      "EnCodon"
    ],
    "sources": [
      "https://research.nvidia.com/labs/dbr/assets/data/manuscripts/nv-codonfm-preprint.pdf",
      "https://github.com/NVIDIA-BioNeMo/CodonFM",
      "https://huggingface.co/nvidia/NV-CodonFM-Encodon-80M-v1"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md"
  },
  {
    "id": "bfm-ankh3",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "protein",
    "name": "Ankh3",
    "description": "Multi-task denoising and sequence-completion protein language model for transferable protein representations.",
    "io": "Protein sequence -> embeddings, recovered sequence, or downstream predictions",
    "status": "Public code + non-commercial weights",
    "use_cases": "Protein embeddings, sequence completion, function and property transfer",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.20052"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/agemagician/Ankh"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/ElnaggarLab/ankh3-large"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "sources": [
      "https://arxiv.org/abs/2505.20052",
      "https://github.com/agemagician/Ankh",
      "https://huggingface.co/ElnaggarLab/ankh3-large"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-esm-s",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "protein",
    "name": "ESM-S",
    "description": "Structure-supervised protein language model producing sequence representations informed by three-dimensional structure.",
    "io": "Protein sequence during inference, structure supervision during training -> structure-aware embeddings",
    "status": "Public code + weights; exact repository licence unresolved",
    "use_cases": "Structure-aware protein embeddings, function prediction, downstream transfer",
    "year": "2024",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2402.05856"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/DeepGraphLearning/esm-s"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/Oxer11/ESM-S"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://arxiv.org/abs/2402.05856",
      "https://github.com/DeepGraphLearning/esm-s",
      "https://huggingface.co/Oxer11/ESM-S"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-proust",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "protein",
    "name": "Proust",
    "description": "Causal protein language model combining sequence generation, embeddings, and zero-shot fitness estimation.",
    "io": "Protein sequence or prompt -> generated sequence, embeddings, or fitness score",
    "status": "Public code + non-commercial checkpoint",
    "use_cases": "Protein generation, representation learning, zero-shot fitness estimation",
    "year": "2026",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2602.01845"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Furkan9015/proust-inference"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/nappenstance/proust_v0"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "sources": [
      "https://arxiv.org/abs/2602.01845",
      "https://github.com/Furkan9015/proust-inference",
      "https://huggingface.co/nappenstance/proust_v0"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-ab-roberta",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "protein",
    "name": "Ab-RoBERTa",
    "description": "Antibody-specific masked language model trained on antibody sequence repertoires for reusable antibody embeddings and fine-tuning.",
    "io": "Antibody sequence -> embeddings or masked-token predictions",
    "status": "Public checkpoint; separate source repository not confirmed",
    "use_cases": "Antibody representation learning, sequence classification, downstream fine-tuning",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2506.13006"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/mogam-ai/Ab-RoBERTa"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://arxiv.org/abs/2506.13006",
      "https://huggingface.co/mogam-ai/Ab-RoBERTa"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-oneprot",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "multimodal",
    "name": "OneProt",
    "description": "Cross-modal protein representation model aligning sequence, structure, binding-site, and textual information in a shared space.",
    "io": "Protein sequence/structure/binding site/text -> aligned embeddings",
    "status": "Public code + weights",
    "use_cases": "Cross-modal protein retrieval, representation learning, annotation and transfer",
    "year": "2025",
    "paper_links": [
      {
        "label": "PLOS Computational Biology",
        "url": "https://journals.plos.org/ploscompbiol/article?id=10.1371/journal.pcbi.1013679"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/klemens-floege/oneprot/"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/HelmholtzAI-FZJ/oneprot-4"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://journals.plos.org/ploscompbiol/article?id=10.1371/journal.pcbi.1013679",
      "https://github.com/klemens-floege/oneprot/",
      "https://huggingface.co/HelmholtzAI-FZJ/oneprot-4"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-protdat",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "multimodal",
    "name": "ProtDAT",
    "description": "Text-conditioned protein generation model aligning natural-language descriptions with protein sequences.",
    "io": "Functional text prompt -> generated protein sequence",
    "status": "Public code + non-commercial weights",
    "use_cases": "Text-guided protein design, controllable generation, protein-language alignment",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature Communications",
        "url": "https://doi.org/10.1038/s41467-025-65562-w"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/GXY0116/ProtDAT/tree/v1.0.0"
      },
      {
        "label": "Zenodo weights",
        "url": "https://zenodo.org/records/14264096"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "sources": [
      "https://doi.org/10.1038/s41467-025-65562-w",
      "https://github.com/GXY0116/ProtDAT/tree/v1.0.0",
      "https://zenodo.org/records/14264096"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-esmfold2",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "complex",
    "name": "ESMFold2",
    "description": "ESMC-6B-conditioned diffusion model for all-atom prediction of proteins, nucleic acids, ligands, complexes, and binder candidates.",
    "io": "Biomolecular sequence/specification with optional MSA -> all-atom structure",
    "status": "Open code + weights (MIT); local and Biohub Platform inference",
    "use_cases": "Fast structure prediction, complex modeling, interaction and binder assessment",
    "year": "2026",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.06.03.729735v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Biohub/esm"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/biohub/ESMFold2"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.06.03.729735v1",
      "https://github.com/Biohub/esm",
      "https://huggingface.co/biohub/ESMFold2"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-opendde-preview",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "complex",
    "name": "OpenDDE Preview",
    "description": "Preview all-atom co-folding and design model for proteins, nucleic acids, ligands, ions, covalent bonds, and mixed biomolecular assemblies.",
    "io": "Biomolecular JSON specification -> predicted or designed all-atom structure",
    "status": "Open preview code + checkpoints + Docker (Apache-2.0); unstable and not production-ready",
    "use_cases": "All-atom structure prediction, co-folding, interaction modeling and early design research",
    "year": "2026",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2607.03787"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/aurekaresearch/OpenDDE"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/aurekaresearch/OpenDDE"
      },
      {
        "label": "Project",
        "url": "https://aurekaresearch.github.io/OpenDDE-Website/"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://arxiv.org/abs/2607.03787",
      "https://github.com/aurekaresearch/OpenDDE",
      "https://huggingface.co/aurekaresearch/OpenDDE",
      "https://aurekaresearch.github.io/OpenDDE-Website/"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-pocketxmol",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "molecule",
    "name": "PocketXMol",
    "description": "Pocket-interacting molecular generative model supporting docking, conformer generation, structure-based design, linking, growing, PROTACs, and peptides.",
    "io": "Pocket or partial-complex constraints -> molecules, peptide structures, or poses",
    "status": "Open code + weights + train/test data",
    "use_cases": "Structure-based drug design, docking, conformer generation, molecular and peptide design",
    "year": "2026",
    "paper_links": [
      {
        "label": "Cell",
        "url": "https://doi.org/10.1016/j.cell.2026.01.003"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/pengxingang/PocketXMol"
      },
      {
        "label": "Zenodo",
        "url": "https://doi.org/10.5281/zenodo.17801271"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://doi.org/10.1016/j.cell.2026.01.003",
      "https://github.com/pengxingang/PocketXMol",
      "https://doi.org/10.5281/zenodo.17801271"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-suiren-1",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "molecule",
    "name": "Suiren-1.0",
    "description": "Molecular foundation-model family for quantum properties, energies, forces, embeddings, conformer aggregation, and intermolecular interactions.",
    "io": "Molecular geometry or representation -> energies, forces, properties, or embeddings",
    "status": "Public code + Base/Dimer/ConfAvg checkpoints; modified MIT-style terms",
    "use_cases": "Molecular property prediction, force modeling, conformer and interaction analysis",
    "year": "2026",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.21942"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/golab-ai/Suiren-Foundation-Model"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/ajy112/Suiren-Base"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "aliases": [
      "Suiren-Base",
      "Suiren-Dimer",
      "Suiren-ConfAvg"
    ],
    "sources": [
      "https://arxiv.org/abs/2603.21942",
      "https://github.com/golab-ai/Suiren-Foundation-Model",
      "https://huggingface.co/ajy112/Suiren-Base"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md"
  },
  {
    "id": "bfm-ubio-molfm",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "molecule",
    "name": "UBio-MolFM",
    "description": "Universal biomolecular machine-learning potential based on an equivariant all-atom representation model.",
    "io": "All-atom biomolecular geometry -> energies, forces, embeddings, or molecular-dynamics potential",
    "status": "Open code + public checkpoint (MIT); training-data release is partial/gated",
    "use_cases": "Biomolecular energy and force prediction, simulation, transferable atomistic embeddings",
    "year": "2026",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2602.17709"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/IQuestLab/UBio-MolFM"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/IQuestLab/IQuest-UBio-MolFM-V1"
      },
      {
        "label": "Data subset",
        "url": "https://huggingface.co/datasets/IQuestLab/UBio-Protein26"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://arxiv.org/abs/2602.17709",
      "https://github.com/IQuestLab/UBio-MolFM",
      "https://huggingface.co/IQuestLab/IQuest-UBio-MolFM-V1",
      "https://huggingface.co/datasets/IQuestLab/UBio-Protein26"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-uma",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "molecule",
    "name": "UMA",
    "description": "Universal atomistic potential spanning molecules, materials, and catalysts; its OMol task covers drug-like molecules and biomolecular use cases.",
    "io": "Atomic structure with task, charge, and spin context -> energies and forces",
    "status": "Open code; gated weights under custom FAIR Chemistry Licence/AUP with geographic restrictions",
    "use_cases": "Molecular and biomolecular simulation, energy/force prediction, materials and catalyst modeling",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2506.23971"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/facebookresearch/fairchem"
      },
      {
        "label": "Documentation",
        "url": "https://fair-chem.github.io/uma/"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/facebook/UMA"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://arxiv.org/abs/2506.23971",
      "https://github.com/facebookresearch/fairchem",
      "https://fair-chem.github.io/uma/",
      "https://huggingface.co/facebook/UMA"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-seedfold",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "complex",
    "name": "SeedFold / SeedFold-Linear",
    "description": "All-atom folding family for protein-protein, antibody-antigen, protein-ligand, RNA, and DNA complexes.",
    "io": "Biomolecular sequence/complex specification -> predicted all-atom structure",
    "status": "Registration-gated web server; no public code, weights, or explicit model licence located",
    "use_cases": "Complex structure prediction, antibody-antigen and protein-ligand modeling",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2512.24354"
      }
    ],
    "code_links": [
      {
        "label": "Project",
        "url": "https://seedfold.github.io/"
      },
      {
        "label": "Web server",
        "url": "https://seedfold.io/"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "aliases": [
      "SeedFold",
      "SeedFold-Linear"
    ],
    "sources": [
      "https://arxiv.org/abs/2512.24354",
      "https://seedfold.github.io/",
      "https://seedfold.io/"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md"
  },
  {
    "id": "bfm-boltzgen",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "protein",
    "name": "BoltzGen",
    "description": "All-atom generative binder-design model and workflow for proteins, peptides, nanobodies, small molecules, and other biomolecular targets.",
    "io": "Target structure and design constraints -> ranked binder sequences and structures",
    "status": "Open code + weights + training data (MIT)",
    "use_cases": "Protein and peptide binder design, nanobody design, target-conditioned generation",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://doi.org/10.1101/2025.11.20.689494"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/HannesStark/boltzgen"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/boltzgen/boltzgen-1"
      },
      {
        "label": "Project",
        "url": "https://boltz.bio/boltzgen"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://doi.org/10.1101/2025.11.20.689494",
      "https://github.com/HannesStark/boltzgen",
      "https://huggingface.co/boltzgen/boltzgen-1",
      "https://boltz.bio/boltzgen"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-pxdesign",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "protein",
    "name": "PXDesign",
    "description": "Diffusion-based protein-binder design suite with structure-prediction and filtering stages.",
    "io": "Target structure/MSA, hotspots, and binder length -> binder sequence/structure candidates and scores",
    "status": "Open code + checkpoint + free web server (Apache-2.0); dependencies have separate terms",
    "use_cases": "De novo protein binder design, target-conditioned generation, candidate filtering",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://doi.org/10.1101/2025.08.15.670450"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/bytedance/PXDesign"
      },
      {
        "label": "Project",
        "url": "https://protenix.github.io/pxdesign/"
      },
      {
        "label": "Server",
        "url": "https://protenix-server.com/"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://doi.org/10.1101/2025.08.15.670450",
      "https://github.com/bytedance/PXDesign",
      "https://protenix.github.io/pxdesign/",
      "https://protenix-server.com/"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-boltzmol-1",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "molecule",
    "name": "BoltzMol-1",
    "description": "Commercial prospective hit-discovery service combining Boltz-2-style screening, make-on-demand generation, and ADME filtering.",
    "io": "Target and screening/design request -> ranked or generated molecular candidates",
    "status": "Commercial web/API access only; no public code or weights",
    "use_cases": "Virtual screening, hit discovery, molecular generation and prioritization",
    "year": "2026",
    "paper_links": [
      {
        "label": "Technical report",
        "url": "https://boltz.bio/boltzmol1-technical-report.pdf"
      }
    ],
    "code_links": [
      {
        "label": "Announcement",
        "url": "https://boltz.bio/boltzmol-boltzprot-api"
      },
      {
        "label": "API",
        "url": "https://api.boltz.bio/"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://boltz.bio/boltzmol1-technical-report.pdf",
      "https://boltz.bio/boltzmol-boltzprot-api",
      "https://api.boltz.bio/"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-boltzprot-1",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "protein",
    "name": "BoltzProt-1",
    "description": "Commercial protein-binder and nanobody-design service combining generative design with interaction ranking.",
    "io": "Target structure and binder-design request -> ranked protein or nanobody candidates",
    "status": "Commercial web/API access only; no public code or weights",
    "use_cases": "Protein binder design, nanobody generation, candidate ranking",
    "year": "2026",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.06.23.733997v1"
      },
      {
        "label": "Technical report",
        "url": "https://boltz.bio/boltzprot1-technical-report.pdf"
      }
    ],
    "code_links": [
      {
        "label": "Announcement",
        "url": "https://boltz.bio/boltzmol-boltzprot-api"
      },
      {
        "label": "API",
        "url": "https://api.boltz.bio/"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.06.23.733997v1",
      "https://boltz.bio/boltzprot1-technical-report.pdf",
      "https://boltz.bio/boltzmol-boltzprot-api",
      "https://api.boltz.bio/"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-seedproteo",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "protein",
    "name": "SeedProteo",
    "description": "All-atom binder and unconditional protein-design system accessed through the Seed web service.",
    "io": "Target/design constraints -> protein binder or unconditional design candidates",
    "status": "Registration-gated web server; no public code, weights, or explicit model licence located",
    "use_cases": "Protein binder design, unconditional protein generation, structure-conditioned design",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2512.24192"
      }
    ],
    "code_links": [
      {
        "label": "Project",
        "url": "https://seedfold.github.io/"
      },
      {
        "label": "Web server",
        "url": "https://seedfold.io/proteinDesign"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://arxiv.org/abs/2512.24192",
      "https://seedfold.github.io/",
      "https://seedfold.io/proteinDesign"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-alphaproteo",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "protein",
    "name": "AlphaProteo",
    "description": "DeepMind protein-binder design system for generating candidate binders against specified target proteins.",
    "io": "Target protein structure -> designed binder sequences and structures",
    "status": "Paper/project only; no public code, weights, server/API, or model licence",
    "use_cases": "De novo protein binder design and experimental candidate generation",
    "year": "2024",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2409.08022"
      }
    ],
    "code_links": [
      {
        "label": "Official project",
        "url": "https://deepmind.google/blog/alphaproteo-generates-novel-proteins-for-biology-and-health-research/"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://arxiv.org/abs/2409.08022",
      "https://deepmind.google/blog/alphaproteo-generates-novel-proteins-for-biology-and-health-research/"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-rosetta-foundry",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "platform",
    "name": "Rosetta Foundry",
    "description": "Shared training, inference, packaging, and model-registry platform for RF3, RFD3/RFD3NA, ProteinMPNN, LigandMPNN, and related RosettaCommons models.",
    "io": "Model specification and biomolecular inputs -> installed models, training or inference workflows",
    "status": "Open platform code + model checkpoints (BSD-3-Clause); some model APIs still stabilizing",
    "use_cases": "Reproducible model installation, training, inference, and Rosetta model discovery",
    "year": "2025",
    "paper_links": [],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/RosettaCommons/foundry"
      },
      {
        "label": "Documentation",
        "url": "https://rosettacommons.github.io/foundry/"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://github.com/RosettaCommons/foundry",
      "https://rosettacommons.github.io/foundry/"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-foldbench",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "benchmark",
    "name": "FoldBench",
    "description": "Low-homology benchmark of biological assemblies across all-atom structure-prediction tasks involving proteins, nucleic acids, ligands, and interactions.",
    "io": "Predicted assemblies + references -> quality and generalization metrics",
    "status": "Open benchmark targets + evaluation code + samples (MIT); public results",
    "use_cases": "All-atom structure-prediction evaluation and generalization testing",
    "year": "2025",
    "paper_links": [
      {
        "label": "Nature Communications",
        "url": "https://doi.org/10.1038/s41467-025-67127-3"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/BEAM-Labs/FoldBench"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://doi.org/10.1038/s41467-025-67127-3",
      "https://github.com/BEAM-Labs/FoldBench"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-pxmeter",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "benchmark",
    "name": "PXMeter",
    "description": "Structural-quality evaluation toolkit and curated benchmark pipeline for all-atom structure prediction.",
    "io": "Predicted structures + benchmark references -> structural quality metrics",
    "status": "Open evaluation toolkit + benchmark datasets/pipelines (Apache-2.0)",
    "use_cases": "All-atom model evaluation, structural-quality diagnostics, reproducible benchmarking",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://doi.org/10.1101/2025.07.17.664878"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/bytedance/PXMeter"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://doi.org/10.1101/2025.07.17.664878",
      "https://github.com/bytedance/PXMeter"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-pfmbench",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "benchmark",
    "name": "PFMBench",
    "description": "Protein-foundation-model evaluation suite spanning dozens of tasks, multiple protein-science areas, and fine-tuning and zero-shot settings.",
    "io": "Protein foundation model + task datasets -> standardized evaluation results",
    "status": "Open benchmark code + task-data links (Apache-2.0); no public leaderboard confirmed",
    "use_cases": "Comparative evaluation and model selection across protein-science tasks",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2506.14796"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/biomap-research/PFMBench"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://arxiv.org/abs/2506.14796",
      "https://github.com/biomap-research/PFMBench"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-protein-se3",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "benchmark",
    "name": "Protein-SE(3)",
    "description": "Unified retraining and evaluation framework for SE(3)-equivariant protein structure-generation models.",
    "io": "Structure-generation model + standardized data -> retrained models and comparable metrics",
    "status": "Open benchmark/training framework + preprocessed data (MIT)",
    "use_cases": "Evaluation of RFdiffusion, Genie, FrameDiff, FoldFlow/FrameFlow and related generators",
    "year": "2025",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2507.20243"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/BruthYU/protein-se3"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://arxiv.org/abs/2507.20243",
      "https://github.com/BruthYU/protein-se3"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-esm-atlas",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "platform",
    "name": "ESM Atlas",
    "description": "Public protein-sequence and structure atlas with large-scale ESM-derived embeddings, predicted structures, search, and an alpha API.",
    "io": "Protein sequence/query -> atlas search results, embeddings, structures, and metadata",
    "status": "Public web atlas + alpha API; explicit current dataset licence not located",
    "use_cases": "Protein-space exploration, similarity search, predicted-structure access and programmatic retrieval",
    "year": "2026",
    "paper_links": [],
    "code_links": [
      {
        "label": "Biohub resources",
        "url": "https://biohub.ai/resources"
      },
      {
        "label": "ESM repository",
        "url": "https://github.com/Biohub/esm"
      },
      {
        "label": "Atlas API",
        "url": "https://biohub.ai/esm/protein/atlas/api-docs/"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://biohub.ai/resources",
      "https://github.com/Biohub/esm",
      "https://biohub.ai/esm/protein/atlas/api-docs/"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-biomatrix",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "multimodal",
    "name": "BioMatrix",
    "description": "Native multimodal biomolecular generalist spanning molecule and protein sequences, three-dimensional structures, and natural language.",
    "io": "Biomolecular sequence/structure/text -> cross-modal embeddings, understanding, or generation",
    "status": "Open code + weights (Apache-2.0)",
    "use_cases": "Folding, inverse folding, design, captioning, affinity and interaction prediction",
    "year": "2026",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2606.22138"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/QizhiPei/BioMatrix"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/QizhiPei/BioMatrix-4B-SFT"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://arxiv.org/abs/2606.22138",
      "https://github.com/QizhiPei/BioMatrix",
      "https://huggingface.co/QizhiPei/BioMatrix-4B-SFT"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-genejepa",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "singlecell",
    "name": "GeneJEPA",
    "description": "Joint-embedding predictive transcriptomic model trained on large-scale single-cell data for transferable cell and gene representations.",
    "io": "Sparse single-cell counts -> cell/gene embeddings and latent predictions",
    "status": "Public source + checkpoint; checkpoint MIT, source-code licence not stated",
    "use_cases": "Cell clustering, representation transfer, perturbation and drug-response prediction",
    "year": "2025",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.10.14.682378v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/BiostateAI/GeneJEPA"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/elonlit/GeneJEPA"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.10.14.682378v1",
      "https://github.com/BiostateAI/GeneJEPA",
      "https://huggingface.co/elonlit/GeneJEPA"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-eva-scienta",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "multimodal",
    "name": "EVA (Scienta)",
    "description": "Cross-species immunology and inflammation model combining transcriptomic and histology information for patient-level and translational representations; only EVA-RNA is publicly released.",
    "io": "Bulk, microarray, or pseudobulk RNA with optional histology -> sample/gene embeddings and translational predictions",
    "status": "Gated EVA-RNA weights under custom Scienta licence; full multimodal system closed; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Target efficacy, perturbation transfer, stratification and response prediction",
    "year": "2026",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2602.10168"
      }
    ],
    "code_links": [
      {
        "label": "EVA-RNA",
        "url": "https://huggingface.co/ScientaLab/eva-rna"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "sources": [
      "https://arxiv.org/abs/2602.10168",
      "https://huggingface.co/ScientaLab/eva-rna"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-genbio-pathfm",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "spatial",
    "name": "GenBio-PathFM",
    "description": "Large histopathology representation model trained from public pathology data for transferable H&E patch embeddings.",
    "io": "H&E image tiles -> transferable patch embeddings",
    "status": "Public code + gated weights under a restrictive non-commercial licence",
    "use_cases": "Pathology representation learning, biomarker inference, classification and robustness analysis",
    "year": "2026",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.03.17.712534v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/genbio-ai/genbio-pathfm"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/genbio-ai/genbio-pathfm"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.03.17.712534v1",
      "https://github.com/genbio-ai/genbio-pathfm",
      "https://huggingface.co/genbio-ai/genbio-pathfm"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-squall",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "spatial",
    "name": "SQUALL",
    "description": "Joint histology and spatial-transcriptomics model pretrained on paired image and spatial molecular observations.",
    "io": "H&E + spatial expression -> joint embeddings, reconstructed expression, and biomarker profiles",
    "status": "Public code + weights + pretraining data under CC BY-NC-ND-4.0; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Virtual biomarkers, cross-platform spatial profiling, clustering and representation learning",
    "year": "2026",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.06.01.729028v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/OswaldZhang/SQUALL-release"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/zongxu/SQUALL"
      },
      {
        "label": "Zenodo data",
        "url": "https://zenodo.org/records/17318279"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.06.01.729028v1",
      "https://github.com/OswaldZhang/SQUALL-release",
      "https://huggingface.co/zongxu/SQUALL",
      "https://zenodo.org/records/17318279"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-spatialwhisperer",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "spatial",
    "name": "SpatialWhisperer",
    "description": "Trimodal histology, expression, and language model placing image, molecular, and text signals in a shared embedding space.",
    "io": "Image, expression, or text -> shared embeddings and natural-language cell queries",
    "status": "Public code + non-commercial checkpoint; depends on gated UNI2",
    "use_cases": "Zero-shot cell typing, spatial annotation, cross-modal retrieval and natural-language querying",
    "year": "2026",
    "paper_links": [
      {
        "label": "OpenReview",
        "url": "https://openreview.net/forum?id=Ze7U293Zw4"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/zinagoodlab/spatialwhisperer"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/Good-Lab/spatialwhisperer"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "sources": [
      "https://openreview.net/forum?id=Ze7U293Zw4",
      "https://github.com/zinagoodlab/spatialwhisperer",
      "https://huggingface.co/Good-Lab/spatialwhisperer"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-deepspot-m",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "spatial",
    "name": "DeepSpot-M",
    "description": "Gene-query model for transcriptome-wide virtual spatial transcriptomics from routine histology.",
    "io": "H&E tile -> spatial expression predictions across a transcriptome-scale gene set",
    "status": "Public non-commercial code + weights",
    "use_cases": "Virtual spatial transcriptomics, biomarker mapping, adaptation and large-cohort profiling",
    "year": "2026",
    "paper_links": [
      {
        "label": "medRxiv",
        "url": "https://www.medrxiv.org/content/10.64898/2026.06.19.26356060v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ratschlab/DeepSpotM"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/ratschlab/DeepSpotM"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "sources": [
      "https://www.medrxiv.org/content/10.64898/2026.06.19.26356060v1",
      "https://github.com/ratschlab/DeepSpotM",
      "https://huggingface.co/ratschlab/DeepSpotM"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-h2o-spatial",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "spatial",
    "name": "H2O",
    "description": "Histopathology-to-spatial-transcriptomics and proteomics framework for predicting molecular landscapes from routine H&E images.",
    "io": "H&E patches + spatial context -> spatial transcriptomic or proteomic expression",
    "status": "Public source; referenced checkpoint is absent; all rights reserved and no reuse licence; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Virtual spatial multi-omics, clustering, communication and trajectory analysis",
    "year": "2026",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.04.21.717342v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/TencentAILabHealthcare/H2O"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.04.21.717342v1",
      "https://github.com/TencentAILabHealthcare/H2O"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-lemon",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "spatial",
    "name": "LEMON",
    "description": "Self-supervised single-nucleus morphology foundation model trained on millions of cell images.",
    "io": "H&E nucleus crop -> compact cell-morphology embedding",
    "status": "Public inference code + weights (MIT)",
    "use_cases": "Cell typing, gene-expression prediction and quantitative morphology studies",
    "year": "2026",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.25802"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/aliceblondel/LEMON"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://arxiv.org/abs/2603.25802",
      "https://huggingface.co/aliceblondel/LEMON"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-mupd",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "spatial",
    "name": "MuPD (Multimodal Pathology Diffusion)",
    "description": "Generative multimodal pathology diffusion model sharing histology, transcriptomic, and clinical-text context.",
    "io": "Text, RNA, and image combinations -> generated H&E, IHC, or multiplex-immunofluorescence images",
    "status": "Public model artifact under CC BY-NC-ND terms; non-commercial terms apply to at least one artifact or access route",
    "use_cases": "Virtual staining, missing-modality generation and pathology-image augmentation",
    "year": "2026",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.03635"
      }
    ],
    "code_links": [
      {
        "label": "Hugging Face code + weights",
        "url": "https://huggingface.co/xiangjx/MuPaD-256"
      }
    ],
    "canonical": false,
    "noncommercial": true,
    "aliases": [
      "MuPaD"
    ],
    "sources": [
      "https://arxiv.org/abs/2604.03635",
      "https://huggingface.co/xiangjx/MuPaD-256"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "The official artifact is public rather than gated and uses CC BY-NC-ND terms.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "date_modified": "2026-07-13"
  },
  {
    "id": "bfm-bioairepo",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "platform",
    "name": "BioAIrepo",
    "description": "EMBL-EBI BioStudies pilot repository for depositing and discovering life-science AI models, metadata, weights, and associated data.",
    "io": "Model submission with metadata/assets -> searchable reusable model record",
    "status": "Open institutional model repository / pilot; individual deposits retain their own terms",
    "use_cases": "FAIR-oriented model deposit, discovery, metadata and reuse",
    "year": "2026",
    "paper_links": [
      {
        "label": "EMBL-EBI announcement",
        "url": "https://www.ebi.ac.uk/about/news/technology-and-innovation/bioairepo-embl-ebis-hub-for-life-science-ai-models/"
      }
    ],
    "code_links": [
      {
        "label": "Repository",
        "url": "https://www.ebi.ac.uk/biostudies/BioAIrepo"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://www.ebi.ac.uk/about/news/technology-and-innovation/bioairepo-embl-ebis-hub-for-life-science-ai-models/",
      "https://www.ebi.ac.uk/biostudies/BioAIrepo"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-vcbench",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "benchmark",
    "name": "VCBench",
    "description": "Operational benchmark for single-cell foundation models with capability evaluation and a contamination-reporting schema.",
    "io": "Single-cell model checkpoints + datasets -> capability, baseline, and contamination results",
    "status": "Open code + evaluation artifacts",
    "use_cases": "Perturbation, cross-species, GRN, RNA-protein and temporal single-cell evaluation",
    "year": "2026",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.06.18.733146v1"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/AppliedScientific/VCBench"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/collections/appliedscientific/vcbench-v100-single-cell-foundation-model-benchmark"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.06.18.733146v1",
      "https://github.com/AppliedScientific/VCBench",
      "https://huggingface.co/collections/appliedscientific/vcbench-v100-single-cell-foundation-model-benchmark"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-scmbench",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "benchmark",
    "name": "SCMBench",
    "description": "Benchmark of domain-specific and foundation models for paired and unpaired single-cell multi-omics integration.",
    "io": "Multi-omics datasets + integration methods -> integration, biological-conservation, and batch metrics",
    "status": "Open code + public data (MIT)",
    "use_cases": "RNA/ATAC/protein integration benchmarking and method selection",
    "year": "2026",
    "paper_links": [
      {
        "label": "Nature Communications",
        "url": "https://www.nature.com/articles/s41467-026-72570-x"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ml4bio/SCMBench"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://www.nature.com/articles/s41467-026-72570-x",
      "https://github.com/ml4bio/SCMBench"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-spapath-bench",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "benchmark",
    "name": "SpaPath-Bench",
    "description": "Representation-level evaluation of pathology foundation models using paired whole-slide images and spatial transcriptomics.",
    "io": "Pathology-model embeddings + paired WSI/ST slides -> spatial-domain scores",
    "status": "Public paper and results dashboard; benchmark execution pipeline not released",
    "use_cases": "Pathology-encoder selection and spatial representation diagnostics",
    "year": "2026",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2605.25764"
      }
    ],
    "code_links": [
      {
        "label": "Dashboard",
        "url": "https://bokai-zhao.github.io/SpaPath-benchboard/"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://arxiv.org/abs/2605.25764",
      "https://bokai-zhao.github.io/SpaPath-benchboard/"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-gpt-rosalind",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "multimodal",
    "name": "GPT-Rosalind",
    "description": "Purpose-built life-science reasoning model for chemistry, proteins, genomics, evidence synthesis, tools, and experiment planning.",
    "io": "Scientific questions + literature/data/tools -> synthesis, analysis, or experimental plans",
    "status": "Hosted Enterprise access for qualified customers; initial availability is United States only",
    "use_cases": "Life-science reasoning, evidence synthesis, data analysis and experiment planning",
    "year": "2026",
    "paper_links": [
      {
        "label": "Official release",
        "url": "https://openai.com/index/introducing-gpt-rosalind/"
      }
    ],
    "code_links": [
      {
        "label": "Access request",
        "url": "https://openai.com/form/life-sciences-access/"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://openai.com/index/introducing-gpt-rosalind/",
      "https://openai.com/form/life-sciences-access/"
    ],
    "audit_state": "verified_with_limits",
    "audit_priority": "P2",
    "audit_note": "Access is limited to qualified Enterprise customers and initially scoped to the United States.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "date_modified": "2026-07-13",
    "aliases": []
  },
  {
    "id": "bfm-mimic-polymathic",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "multimodal",
    "name": "MIMIC",
    "description": "Any-to-any biomolecular model spanning DNA, RNA, proteins, structures, and cellular and semantic context.",
    "io": "Biological sequence/structure/context modalities -> aligned or generated multimodal outputs",
    "status": "Paper and informational repository; model code, weights, and LORE assets announced but not released",
    "use_cases": "Cross-modal biomolecular representation, retrieval, understanding and generation",
    "year": "2026",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.24506"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/PolymathicAI/MIMIC"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://arxiv.org/abs/2604.24506",
      "https://github.com/PolymathicAI/MIMIC"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-holocell",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "singlecell",
    "name": "HoloCell",
    "description": "Generative single-cell model integrating epigenomic, transcriptomic, and proteomic modalities.",
    "io": "Partial multi-omic cell profile -> cell embeddings and missing-modality generation",
    "status": "Paper only; project repository says artifacts are in preparation and licence remains unresolved",
    "use_cases": "Multi-omic cell representation, missing-modality prediction and generative integration",
    "year": "2026",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.06.07.730684v2"
      }
    ],
    "code_links": [
      {
        "label": "Placeholder repository",
        "url": "https://github.com/bjzgcai/HoloCell"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.06.07.730684v2",
      "https://github.com/bjzgcai/HoloCell"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-cellos",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "singlecell",
    "name": "CellOS",
    "description": "Large multi-view single-cell joint-embedding predictive model for cell-state representation and perturbation prediction.",
    "io": "Single-cell transcriptome -> cell-state embeddings and perturbation predictions",
    "status": "Paper only; no public code or weights found",
    "use_cases": "Cell-state representation, transfer and perturbation prediction",
    "year": "2026",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.06.18.733163v2"
      }
    ],
    "code_links": [],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.06.18.733163v2"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-omnicell-bgi",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "spatial",
    "name": "OmniCell",
    "description": "Tissue-contextual model pretrained across dissociated and spatial expression profiles for unified cellular and molecular representations.",
    "io": "Single-cell or spatial expression + tissue context -> unified cell and molecular representations",
    "status": "Public source + checkpoint; source MIT and checkpoint labelled Apache-2.0",
    "use_cases": "Single-cell and spatial representation, tissue-context transfer and downstream prediction",
    "year": "2026",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2025.12.29.696804v3"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/BGIResearch/omnicell"
      },
      {
        "label": "ModelScope",
        "url": "https://modelscope.cn/models/PJSucas/OmniCell-v1"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2025.12.29.696804v3",
      "https://github.com/BGIResearch/omnicell",
      "https://modelscope.cn/models/PJSucas/OmniCell-v1"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  },
  {
    "id": "bfm-bioreason-pro",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "modality": "multimodal",
    "name": "BioReason-Pro",
    "description": "Protein-function reasoning model combining protein sequence understanding with language-model reasoning and tool/evidence workflows; distinct from the DNA-focused BioReason model.",
    "io": "Protein sequence + scientific question/evidence -> multi-step protein-function reasoning",
    "status": "Public source + web interface + Apache-2.0 weights; source-code licence not separately stated",
    "use_cases": "Protein-function reasoning, annotation, evidence synthesis and mechanistic explanation",
    "year": "2026",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/early/2026/03/20/2026.03.19.712954"
      }
    ],
    "code_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/bowang-lab/BioReason-Pro"
      },
      {
        "label": "Hugging Face collection",
        "url": "https://huggingface.co/collections/wanglab/bioreason-pro"
      },
      {
        "label": "SFT weights",
        "url": "https://huggingface.co/wanglab/bioreason-pro-sft"
      },
      {
        "label": "RL weights",
        "url": "https://huggingface.co/wanglab/bioreason-pro-rl"
      }
    ],
    "canonical": false,
    "noncommercial": false,
    "sources": [
      "https://www.biorxiv.org/content/early/2026/03/20/2026.03.19.712954",
      "https://github.com/bowang-lab/BioReason-Pro",
      "https://huggingface.co/collections/wanglab/bioreason-pro",
      "https://huggingface.co/wanglab/bioreason-pro-sft",
      "https://huggingface.co/wanglab/bioreason-pro-rl"
    ],
    "audit_state": "verified",
    "audit_priority": "P4",
    "audit_note": "Identity, scope, primary/official sources and current catalog claims were checked in the changed-record audit; no unresolved material error was recorded.",
    "audit_source": "evidence/source-audit/foundation_initial.md",
    "aliases": []
  }
]
