[
  {
    "id": "asa-the-ai-scientist",
    "date_added": "2026-06-16",
    "name": "The AI Scientist",
    "category": "crossdomain",
    "domain": "ML research; general template for in-silico science",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2408.06292"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/SakanaAI/AI-Scientist"
      }
    ],
    "access": "open-source",
    "inputs": "Research template, baseline code, topic area, model/API keys",
    "outputs": "Ideas, code, experiments, plots, paper draft, automated review",
    "autonomy": "A4",
    "notes": "Fully automated software-only ML research cycles (Sakana AI).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2408.06292",
      "https://github.com/SakanaAI/AI-Scientist"
    ]
  },
  {
    "id": "asa-the-ai-scientist-v2",
    "date_added": "2026-06-16",
    "name": "The AI Scientist-v2",
    "category": "crossdomain",
    "domain": "ML research",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.08066"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/SakanaAI/AI-Scientist-v2"
      }
    ],
    "access": "open-source",
    "inputs": "Research topic/template and compute environment",
    "outputs": "Workshop-style paper, experiments, analysis, review traces",
    "autonomy": "A4",
    "notes": "Agentic tree-search successor; produced first AI-generated peer-review-accepted workshop paper.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2504.08066",
      "https://github.com/SakanaAI/AI-Scientist-v2"
    ]
  },
  {
    "id": "asa-ai-researcher",
    "date_added": "2026-06-16",
    "name": "AI-Researcher",
    "category": "crossdomain",
    "domain": "AI/scientific research automation",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.18705"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/HKUDS/AI-Researcher"
      }
    ],
    "access": "open-source",
    "inputs": "Research objective or task; reference papers; benchmark settings",
    "outputs": "Literature review, hypothesis, implementation, manuscript",
    "autonomy": "A4",
    "notes": "End-to-end research automation with Scientist-Bench (NeurIPS 2025 spotlight).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2505.18705",
      "https://github.com/HKUDS/AI-Researcher"
    ]
  },
  {
    "id": "asa-agent-laboratory",
    "date_added": "2026-06-16",
    "name": "Agent Laboratory",
    "category": "crossdomain",
    "domain": "General research assistant, strongest examples in ML",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2501.04227"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/SamuelSchmidgall/AgentLaboratory"
      }
    ],
    "access": "open-source",
    "inputs": "User idea and feedback checkpoints",
    "outputs": "Literature review, experiment code, report",
    "autonomy": "A3",
    "notes": "Steerable LLM-agent research workflow with human feedback at each stage.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2501.04227",
      "https://github.com/SamuelSchmidgall/AgentLaboratory"
    ]
  },
  {
    "id": "asa-curie",
    "date_added": "2026-06-16",
    "name": "Curie",
    "category": "crossdomain",
    "domain": "Automated experimentation",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2502.16069"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Just-Curieous/Curie"
      }
    ],
    "access": "open-source",
    "inputs": "Scientific question or paper to reproduce/extend",
    "outputs": "Experiment plan, code, results, interpretation",
    "autonomy": "A3-A4",
    "notes": "Rigor-focused automated experimentation framework (pip curie-ai).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2502.16069",
      "https://github.com/Just-Curieous/Curie"
    ]
  },
  {
    "id": "asa-sciagents",
    "date_added": "2026-06-16",
    "name": "SciAgents",
    "category": "crossdomain",
    "domain": "Cross-domain discovery via graph reasoning",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2409.05556"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/lamm-mit/SciAgentsDiscovery"
      }
    ],
    "access": "open-source",
    "inputs": "Knowledge graph, scientific prompt, literature-derived context",
    "outputs": "Cross-disciplinary hypotheses, candidate materials/ideas",
    "autonomy": "A3",
    "notes": "Multi-agent graph reasoning for discovery, esp. bio-inspired materials (Buehler group).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2409.05556",
      "https://github.com/lamm-mit/SciAgentsDiscovery"
    ]
  },
  {
    "id": "asa-denario",
    "date_added": "2026-06-16",
    "name": "Denario",
    "category": "crossdomain",
    "domain": "Scientific research assistant, astrophysics/cosmology roots",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2510.26887"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/AstroPilot-AI/Denario"
      }
    ],
    "access": "open-source",
    "inputs": "Research question, analysis task, or paper-writing task",
    "outputs": "Ideas, literature checks, plans, code, plots, draft paper",
    "autonomy": "A3",
    "notes": "Modular multi-agent research assistant; astrophysics roots, multi-domain demos.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2510.26887",
      "https://github.com/AstroPilot-AI/Denario"
    ]
  },
  {
    "id": "asa-researchagent",
    "date_added": "2026-06-16",
    "name": "ResearchAgent",
    "category": "crossdomain",
    "domain": "Literature-grounded idea generation",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2404.07738"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/JinheonBaek/ResearchAgent"
      }
    ],
    "access": "open-source",
    "inputs": "Core scientific paper plus literature corpus",
    "outputs": "Iteratively refined research problems, methods, experiment designs",
    "autonomy": "A2-A3",
    "notes": "Literature-grounded idea generation with LLM reviewing agents (NAACL 2025).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2404.07738",
      "https://github.com/JinheonBaek/ResearchAgent"
    ]
  },
  {
    "id": "asa-paperqa2",
    "date_added": "2026-06-16",
    "name": "PaperQA2",
    "category": "crossdomain",
    "domain": "Scientific literature search and synthesis",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2409.13740"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Future-House/paper-qa"
      }
    ],
    "access": "open-source",
    "inputs": "Question plus paper corpus/web literature access",
    "outputs": "Cited answers, summaries, contradiction checks",
    "autonomy": "A2",
    "notes": "High-accuracy cited literature QA/review; superhuman synthesis claims (FutureHouse).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2409.13740",
      "https://github.com/Future-House/paper-qa"
    ]
  },
  {
    "id": "asa-google-ai-co-scientist",
    "date_added": "2026-06-16",
    "name": "Google AI Co-Scientist",
    "category": "crossdomain",
    "domain": "Biomedical hypothesis generation",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2502.18864"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Research objective and scientist guidance",
    "outputs": "Ranked/evolved hypotheses and proposals",
    "autonomy": "A3",
    "notes": "Google Research multi-agent hypothesis-generation system evaluated in biomedical research. No official public implementation was confirmed; community reimplementations are not treated as evidence for the Google system.",
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2502.18864"
    ]
  },
  {
    "id": "asa-robin",
    "date_added": "2026-06-16",
    "name": "Robin",
    "category": "crossdomain",
    "domain": "Experimental biology / therapeutic discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.13400"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Future-House/robin"
      }
    ],
    "access": "open-source",
    "inputs": "Disease or biological objective",
    "outputs": "Therapeutic candidates, ranked evidence, validation workflow",
    "autonomy": "A4-A5",
    "notes": "Lab-in-the-loop therapeutic discovery; identified ripasudil for dry AMD (Nature 2026).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2505.13400",
      "https://github.com/Future-House/robin"
    ]
  },
  {
    "id": "asa-tooluniverse",
    "date_added": "2026-06-16",
    "name": "ToolUniverse",
    "category": "crossdomain",
    "domain": "Scientific-agent tool ecosystem",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2509.23426"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/mims-harvard/ToolUniverse"
      }
    ],
    "access": "open-source",
    "inputs": "Agent request requiring tools, APIs, models, databases",
    "outputs": "Tool calls, retrieved evidence, analysis artifacts",
    "autonomy": "A2",
    "notes": "Open ecosystem of 600-1000+ standardized tools for building biomedical/scientific agents (pip tooluniverse).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2509.23426",
      "https://github.com/mims-harvard/ToolUniverse"
    ]
  },
  {
    "id": "asa-agentrxiv",
    "date_added": "2026-06-16",
    "name": "AgentRxiv",
    "category": "crossdomain",
    "domain": "Preprint/server concept for agent-generated work",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2503.18102"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/SamuelSchmidgall/AgentLaboratory"
      }
    ],
    "access": "open-source",
    "inputs": "Agent-produced preprints and metadata",
    "outputs": "Shared/citable agent outputs retrievable by other agent labs",
    "autonomy": "A3",
    "notes": "Shared preprint-server framework for collaborative autonomous-agent research, built on Agent Laboratory.",
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2503.18102",
      "https://github.com/SamuelSchmidgall/AgentLaboratory"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "asa-biomni",
    "date_added": "2026-06-16",
    "name": "Biomni",
    "category": "biology",
    "domain": "General biomedical research",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.05.30.656746v1"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/snap-stanford/Biomni"
      }
    ],
    "access": "open-source",
    "inputs": "Biomedical task in natural language, files/data, access to tools/databases",
    "outputs": "Literature work, executed code, protocols, hypotheses, multi-domain analyses",
    "autonomy": "A3-A4",
    "notes": "General-purpose biomedical AI agent (150 tools, 105 packages, 59 DBs) with hosted UI at biomni.stanford.edu.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/snap-stanford/Biomni",
      "https://www.biorxiv.org/content/10.1101/2025.05.30.656746v1"
    ]
  },
  {
    "id": "asa-stella",
    "date_added": "2026-06-16",
    "name": "STELLA",
    "category": "biology",
    "domain": "Biomedical self-evolving agent",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2507.02004"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/zaixizhang/STELLA"
      }
    ],
    "access": "open-source",
    "inputs": "Biomedical benchmark/task plus tool needs",
    "outputs": "Evolving reasoning templates, auto-created tools, benchmark answers",
    "autonomy": "A3",
    "notes": "Self-evolving biomedical LLM agent with Template Library and dynamic Tool Ocean; multi-agent (manager/dev/critic/tool-creation).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2507.02004",
      "https://github.com/zaixizhang/STELLA"
    ]
  },
  {
    "id": "asa-bioinformatics-agent-bia",
    "date_added": "2026-06-16",
    "name": "BioInformatics Agent (BIA)",
    "category": "biology",
    "domain": "Bioinformatics workflows",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.05.22.595240v1"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Natural-language bioinformatics task and datasets (esp. scRNA-seq)",
    "outputs": "Workflow design, executable code, comprehensive analytical reports",
    "autonomy": "A3",
    "notes": "Early LLM agent for autonomous bioinformatics workflows via chat interface.",
    "verified": "2026-06-16",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2024.05.22.595240v1"
    ]
  },
  {
    "id": "asa-crispr-gpt",
    "date_added": "2026-06-16",
    "name": "CRISPR-GPT",
    "category": "biology",
    "domain": "Gene-editing experiment design",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.04.25.591003v4"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/cong-lab/crispr-gpt-pub"
      }
    ],
    "access": "open-source",
    "inputs": "Gene-editing goal, target/gene context, constraints",
    "outputs": "CRISPR system choice, gRNA designs, delivery method, protocol, assay/analysis plans",
    "autonomy": "A2-A3",
    "notes": "LLM agent automating CRISPR experiment design; only a safety-limited 'light' version of code is public.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/cong-lab/crispr-gpt-pub",
      "https://www.biorxiv.org/content/10.1101/2024.04.25.591003v4"
    ]
  },
  {
    "id": "asa-cellvoyager",
    "date_added": "2026-06-16",
    "name": "CellVoyager",
    "category": "biology",
    "domain": "Single-cell RNA-seq exploratory analysis",
    "paper_links": [
      {
        "label": "Nature Methods",
        "url": "https://doi.org/10.1038/s41592-026-03029-6"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/zou-group/CellVoyager"
      }
    ],
    "access": "open-source",
    "inputs": "AnnData .h5ad, dataset summary, prior analyses, focus directions",
    "outputs": "Jupyter-style scRNA-seq analyses, figures, biological hypotheses",
    "autonomy": "A4",
    "notes": "Autonomous scRNA-seq exploratory analysis agent (Zou group); also in Nature Methods.",
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://doi.org/10.1038/s41592-026-03029-6",
      "https://github.com/zou-group/CellVoyager"
    ]
  },
  {
    "id": "asa-spatialagent",
    "date_added": "2026-06-16",
    "name": "SpatialAgent",
    "category": "biology",
    "domain": "Spatial biology and spatial transcriptomics",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.04.03.646459v1"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Genentech/SpatialAgent"
      }
    ],
    "access": "open-source",
    "inputs": "Spatial transcriptomics / single-cell data and research question",
    "outputs": "Experimental design, multimodal analysis, hypotheses",
    "autonomy": "A3-A4",
    "notes": "Autonomous agent for spatial biology (72 tools, 17 skill templates); Genentech.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/Genentech/SpatialAgent",
      "https://www.biorxiv.org/content/10.1101/2025.04.03.646459v1"
    ]
  },
  {
    "id": "asa-omnicellagent",
    "date_added": "2026-06-16",
    "name": "OmniCellAgent",
    "category": "biology",
    "domain": "Omics-driven precision medicine",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.07.31.667797v2"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/FuhaiLiAiLab/OmniCellAgent"
      }
    ],
    "access": "open-source",
    "inputs": "scRNA-seq/omics data and a disease/therapy question",
    "outputs": "Mechanism hypotheses, therapy candidates, analysis reports",
    "autonomy": "A3",
    "notes": "AI co-scientist for omics-driven precision medicine aimed at non-computational users.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/FuhaiLiAiLab/OmniCellAgent",
      "https://www.biorxiv.org/content/10.1101/2025.07.31.667797v2"
    ]
  },
  {
    "id": "asa-kbase-research-agent",
    "date_added": "2026-06-16",
    "name": "KBase Research Agent",
    "category": "biology",
    "domain": "Systems biology workflows",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.06.01.729336v1"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/kbaseincubator/narrative_llm_agent"
      }
    ],
    "access": "open-source",
    "inputs": "KBase narrative / systems-biology analysis task",
    "outputs": "Multi-agent KBase workflow, app execution, result summaries/reports",
    "autonomy": "A3",
    "notes": "LLM multi-agent (LangChain/LangGraph/CrewAI) workflow builder for DOE KBase.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/kbaseincubator/narrative_llm_agent",
      "https://www.biorxiv.org/content/10.64898/2026.06.01.729336v1"
    ]
  },
  {
    "id": "asa-biogaip",
    "date_added": "2026-06-16",
    "name": "BioGAIP",
    "category": "biology",
    "domain": "Bioinformatics analysis platform",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.05.16.720484v1"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Natural-language bioinformatics task and data",
    "outputs": "Automated, user-friendly bioinformatics analysis workflow",
    "autonomy": "A3",
    "notes": "LLM-powered multi-agent system for automated bioinformatics tasks (Fudan University).",
    "verified": "2026-06-16",
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.05.16.720484v1"
    ]
  },
  {
    "id": "asa-toolsgenie-2-0",
    "date_added": "2026-06-16",
    "name": "ToolsGenie 2.0",
    "category": "biology",
    "domain": "Biomedical tool agent / tool ecosystem",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.01.06.697527v1"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Natural-language bioinformatics task and file inputs",
    "outputs": "Tool-augmented, sandboxed bioinformatics analysis workflows",
    "autonomy": "A2-A3",
    "notes": "Scalable multi-agent (ReAct) system for bioinformatics automation with Docker sandboxing.",
    "verified": "2026-06-16",
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.01.06.697527v1"
    ]
  },
  {
    "id": "asa-medea",
    "date_added": "2026-06-16",
    "name": "Medea",
    "category": "biology",
    "domain": "Omics agent for therapeutic discovery",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.01.16.696667v1"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/mims-harvard/Medea"
      }
    ],
    "access": "open-source",
    "inputs": "Omics / therapeutic-discovery question",
    "outputs": "Transparent multi-step analyses, code execution, literature evidence, consensus therapeutic findings",
    "autonomy": "A3",
    "notes": "Verification-aware omics AI agent for therapeutic discovery (Zitnik lab, Harvard); 20 tools across single-cell/bulk transcriptomics, vulnerability maps, pathways.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/mims-harvard/Medea",
      "https://www.biorxiv.org/content/10.64898/2026.01.16.696667v1"
    ]
  },
  {
    "id": "asa-txagent",
    "date_added": "2026-06-16",
    "name": "TxAgent",
    "category": "biology",
    "domain": "Therapeutic reasoning",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2503.10970"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/mims-harvard/TxAgent"
      }
    ],
    "access": "open-source",
    "inputs": "Drug, disease, patient-context scenario, therapeutic question",
    "outputs": "Drug-interaction/contraindication analysis and treatment reasoning over 211 tools",
    "autonomy": "A2-A3",
    "notes": "Therapeutic-reasoning agent (Zitnik lab) using ToolUniverse; research/decision-support, not a standalone clinical tool.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2503.10970",
      "https://github.com/mims-harvard/TxAgent"
    ]
  },
  {
    "id": "asa-pharmagents",
    "date_added": "2026-06-16",
    "name": "PharmAgents",
    "category": "biology",
    "domain": "Small-molecule drug discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2503.22164"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Target/disease and small-molecule drug-discovery objective",
    "outputs": "Targets, lead compounds, affinity/property/toxicity/synthesizability analysis",
    "autonomy": "A3-A4",
    "notes": "Virtual-pharma multi-agent concept spanning target discovery to preclinical evaluation.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2503.22164"
    ]
  },
  {
    "id": "asa-drugagent",
    "date_added": "2026-06-16",
    "name": "DrugAgent",
    "category": "biology",
    "domain": "Drug-discovery ML programming",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2411.15692"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Drug-discovery ML programming task and data",
    "outputs": "ML pipeline code (Planner + Instructor agents), model selection, performance report",
    "autonomy": "A3",
    "notes": "Multi-agent drug-discovery workflow described in arXiv:2411.15692. The previously cited repository was a third-party reconstruction, not an author-owned release; no official runnable implementation was confirmed.",
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2411.15692"
    ]
  },
  {
    "id": "asa-bioagents",
    "date_added": "2026-06-16",
    "name": "BioAgents",
    "category": "biology",
    "domain": "Bioinformatics analysis with multi-agent system",
    "paper_links": [
      {
        "label": "Scientific Reports",
        "url": "https://doi.org/10.1038/s41598-025-25919-z"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/microsoft/bioinformagus"
      }
    ],
    "access": "open-source",
    "inputs": "Bioinformatics task/data (e.g., genomics pipeline questions)",
    "outputs": "Automated pipeline analysis, troubleshooting, interpretation",
    "autonomy": "A2",
    "notes": "Microsoft multi-agent bioinformatics workflow-design and troubleshooting assistant with limited code generation; it does not autonomously execute complete scientific workflows. Official implementation is MIT-licensed.",
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://doi.org/10.1038/s41598-025-25919-z",
      "https://github.com/microsoft/bioinformagus"
    ]
  },
  {
    "id": "asa-mmedagent",
    "date_added": "2026-06-16",
    "name": "MMedAgent",
    "category": "biology",
    "domain": "Multimodal medical tool-use agent",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2407.02483"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Wangyixinxin/MMedAgent"
      }
    ],
    "access": "open-source",
    "inputs": "Multimodal medical task (image + text)",
    "outputs": "Tool-mediated medical reasoning across modalities (7 tasks, 6 tools)",
    "autonomy": "A2",
    "notes": "First multimodal medical tool-use agent (built on LLaVA-Plus/LLaVA-Med); EMNLP 2024 Findings.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2407.02483",
      "https://github.com/Wangyixinxin/MMedAgent"
    ]
  },
  {
    "id": "asa-bioplanner",
    "date_added": "2026-06-16",
    "name": "BioPlanner",
    "category": "benchmark",
    "domain": "Biology protocol planning benchmark/agent evaluation",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2310.10632"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/bioplanner/bioplanner"
      }
    ],
    "access": "open-data",
    "inputs": "High-level biology protocol description plus admissible pseudocode functions",
    "outputs": "Reconstructed protocol pseudocode; automatic accuracy evaluation",
    "autonomy": "B",
    "notes": "Benchmark (BioProt dataset) auto-evaluating LLM protocol planning in biology.",
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2310.10632",
      "https://github.com/bioplanner/bioplanner"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "asa-talk2biomodels",
    "date_added": "2026-06-16",
    "name": "Talk2BioModels",
    "category": "biology",
    "domain": "Kinetic biological modeling assistant",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.03.11.642548v1"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/VirtualPatientEngine/AIAgents4Pharma"
      }
    ],
    "access": "open-source",
    "inputs": "Natural-language queries over SBML/BioModels kinetic models; uploaded papers for RAG",
    "outputs": "Model search, simulation/plotting, stability analysis, annotation retrieval",
    "autonomy": "A2",
    "notes": "Open-source agentic LLM platform for exploring kinetic biological (SBML) models.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/VirtualPatientEngine/AIAgents4Pharma",
      "https://www.biorxiv.org/content/10.1101/2025.03.11.642548v1"
    ]
  },
  {
    "id": "asa-biotrouble",
    "date_added": "2026-06-16",
    "name": "BioTrouble",
    "category": "biology",
    "domain": "Molecular-biology troubleshooting workflow",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2025.12.30.697016v1.full"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Molecular-biology troubleshooting description and protocol context",
    "outputs": "Diagnoses, suggested fixes, RAG-grounded reasoning with user feedback loop",
    "autonomy": "A2-A3",
    "notes": "Multi-agent RAG workflow for troubleshooting wet-lab molecular biology techniques.",
    "verified": "2026-06-16",
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2025.12.30.697016v1.full"
    ]
  },
  {
    "id": "asa-pantheonos-pantheon-evolve",
    "date_added": "2026-06-16",
    "name": "PantheonOS / Pantheon-Evolve",
    "category": "biology",
    "domain": "Bioinformatics algorithm evolution",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.02.26.707870v1"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/aristoteleo/PantheonOS"
      }
    ],
    "access": "open-source",
    "inputs": "Single-cell/multi-omics analysis task and fitness/evaluation loop",
    "outputs": "Composed agent workflows; autonomously evolved bioinformatics algorithms (batch correction, gene-panel design)",
    "autonomy": "A3-A4",
    "notes": "Evolvable multi-agent genomics framework; Pantheon-Evolve does agentic code evolution.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/aristoteleo/PantheonOS",
      "https://www.biorxiv.org/content/10.64898/2026.02.26.707870v1"
    ]
  },
  {
    "id": "asa-biolab",
    "date_added": "2026-06-16",
    "name": "BioLab",
    "category": "biology",
    "domain": "End-to-end life-sciences research with multi-agent system",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.09.03.674085v1"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Life-science research objective (e.g., target identification, antibody engineering)",
    "outputs": "Hypotheses, database queries, foundation-model predictions, experimental protocols; closed-loop wet-lab integration",
    "autonomy": "A4-A5",
    "notes": "Eight-agent system over 219 xBio-Tools/xTrimo foundation models for end-to-end life-science research.",
    "verified": "2026-06-16",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.09.03.674085v1"
    ]
  },
  {
    "id": "asa-coscientist",
    "date_added": "2026-06-16",
    "name": "Coscientist",
    "category": "chemistry",
    "domain": "Autonomous chemical experimentation",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-023-06792-0"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/gomesgroup/coscientist"
      }
    ],
    "access": "lab-gated",
    "inputs": "Chemistry objective, lab hardware/docs, web/literature context",
    "outputs": "Synthesis plans, robot/cloud-lab commands, experimental results, optimizations",
    "autonomy": "A5",
    "notes": "GPT-4 driven autonomous chemistry system that plans and runs experiments via a cloud robotic lab.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/gomesgroup/coscientist",
      "https://www.nature.com/articles/s41586-023-06792-0"
    ]
  },
  {
    "id": "asa-chemcrow",
    "date_added": "2026-06-16",
    "name": "ChemCrow",
    "category": "chemistry",
    "domain": "Chemistry tool-use agent",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2304.05376"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ur-whitelab/chemcrow-public"
      }
    ],
    "access": "open-source",
    "inputs": "Natural-language chemistry task",
    "outputs": "Tool calls, synthesis plans, property/drug/materials answers",
    "autonomy": "A2-A3",
    "notes": "LLM chemistry agent integrating 18 expert tools across synthesis, drug and materials design.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2304.05376",
      "https://github.com/ur-whitelab/chemcrow-public"
    ]
  },
  {
    "id": "asa-cactus",
    "date_added": "2026-06-16",
    "name": "CACTUS",
    "category": "chemistry",
    "domain": "Cheminformatics tool-use agent",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2405.00972"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/pnnl/cactus"
      }
    ],
    "access": "open-source",
    "inputs": "Chemistry or molecular-discovery question",
    "outputs": "Property prediction, similarity, drug-likeness, reasoning",
    "autonomy": "A2",
    "notes": "PNNL cheminformatics tool-use agent (Chemistry Agent Connecting Tool-Usage to Science).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2405.00972",
      "https://github.com/pnnl/cactus"
    ]
  },
  {
    "id": "asa-chemtoolagent",
    "date_added": "2026-06-16",
    "name": "ChemToolAgent",
    "category": "chemistry",
    "domain": "Chemistry problem-solving agent",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2411.07228"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/OSU-NLP-Group/ChemToolAgent"
      }
    ],
    "access": "open-source",
    "inputs": "Chemistry task needing general/molecule/reaction tools",
    "outputs": "Tool-mediated answer and analysis",
    "autonomy": "A2",
    "notes": "OSU-NLP chemistry agent studying when tools help LLMs; NAACL 2025 Findings.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2411.07228",
      "https://github.com/OSU-NLP-Group/ChemToolAgent"
    ]
  },
  {
    "id": "asa-chemmcp",
    "date_added": "2026-06-16",
    "name": "ChemMCP",
    "category": "chemistry",
    "domain": "Chemistry MCP toolkit for agents",
    "paper_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/OSU-NLP-Group/ChemMCP"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/OSU-NLP-Group/ChemMCP"
      }
    ],
    "access": "open-source",
    "inputs": "Agent requests for chemistry tools/data",
    "outputs": "Standardized chemistry tool calls (MCP)",
    "autonomy": "A2",
    "notes": "MCP-compatible chemistry toolkit; extends ChemToolAgent tools for building chemistry co-scientists.",
    "verified": "2026-07-13",
    "sources": [
      "https://github.com/OSU-NLP-Group/ChemMCP"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "asa-chemagent-ai4chem",
    "date_added": "2026-06-16",
    "name": "ChemAgent (AI4Chem)",
    "category": "chemistry",
    "domain": "Chemistry/materials tool learning",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2501.06590"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/gersteinlab/chemagent"
      }
    ],
    "access": "open-source",
    "inputs": "Chemistry reasoning task",
    "outputs": "Structured subtasks, memory-augmented answer",
    "autonomy": "A2",
    "notes": "Gerstein Lab ChemAgent: self-updating library/memory improves chemical reasoning; ICLR 2025.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2501.06590",
      "https://github.com/gersteinlab/chemagent"
    ]
  },
  {
    "id": "asa-chemagent-self-updating-memory",
    "date_added": "2026-06-16",
    "name": "ChemAgent (self-updating memory)",
    "category": "chemistry",
    "domain": "Chemical reasoning",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2501.06590"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/gersteinlab/chemagent"
      }
    ],
    "access": "open-source",
    "inputs": "Chemistry reasoning task",
    "outputs": "Structured subtasks, memory-augmented answer",
    "autonomy": "A2",
    "notes": "Gerstein Lab ChemAgent: self-updating library/memory improves chemical reasoning; ICLR 2025.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2501.06590",
      "https://github.com/gersteinlab/chemagent"
    ]
  },
  {
    "id": "asa-moose-chem",
    "date_added": "2026-06-16",
    "name": "MOOSE-Chem",
    "category": "chemistry",
    "domain": "Chemistry hypothesis discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2410.07076"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ZonglinY/MOOSE-Chem"
      }
    ],
    "access": "open-source",
    "inputs": "Research question plus background survey",
    "outputs": "Rediscovered/generated chemistry hypotheses",
    "autonomy": "A3",
    "notes": "LLM framework that decomposes and rediscovers unseen chemistry hypotheses; ICLR 2025.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2410.07076",
      "https://github.com/ZonglinY/MOOSE-Chem"
    ]
  },
  {
    "id": "asa-moose-chem2",
    "date_added": "2026-06-16",
    "name": "MOOSE-Chem2",
    "category": "chemistry",
    "domain": "Fine-grained chemistry hypothesis discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.19209"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ZonglinY/MOOSE-Chem2"
      }
    ],
    "access": "open-source",
    "inputs": "Research background and target question",
    "outputs": "More detailed, lab-testable hypotheses",
    "autonomy": "A3",
    "notes": "Fine-grained scientific hypothesis discovery via hierarchical search; NeurIPS 2025.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2505.19209",
      "https://github.com/ZonglinY/MOOSE-Chem2"
    ]
  },
  {
    "id": "asa-chemagents-robotic-ai-chemist",
    "date_added": "2026-06-16",
    "name": "ChemAgents robotic AI chemist",
    "category": "chemistry",
    "domain": "Robotic chemistry",
    "paper_links": [
      {
        "label": "ChemRxiv",
        "url": "https://chemrxiv.org/doi/10.26434/chemrxiv-2024-w953h"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/pic-ai-robotic-chemistry/ChemAgents"
      }
    ],
    "access": "lab-gated",
    "inputs": "On-demand chemistry task and robot lab setup",
    "outputs": "Protocol execution, experimental data, task reports",
    "autonomy": "A5",
    "notes": "Hierarchical multi-agent (Llama-3-70B) robotic chemist; ChemRxiv preprint, published JACS 2025 (10.1021/jacs.4c17738).",
    "verified": "2026-06-16",
    "sources": [
      "https://chemrxiv.org/doi/10.26434/chemrxiv-2024-w953h",
      "https://github.com/pic-ai-robotic-chemistry/ChemAgents"
    ]
  },
  {
    "id": "asa-llm-rdf",
    "date_added": "2026-06-16",
    "name": "LLM-RDF",
    "category": "chemistry",
    "domain": "End-to-end chemical synthesis development",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-024-54457-x"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Ruan-Yixiang/LLM-RDF"
      }
    ],
    "access": "open-source",
    "inputs": "Reaction target and synthesis-development objective",
    "outputs": "Literature extraction, condition screening, kinetics, optimization, scale-up",
    "autonomy": "A4-A5",
    "notes": "Six-agent GPT-4 reaction development framework with web app for end-to-end synthesis development.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/Ruan-Yixiang/LLM-RDF",
      "https://www.nature.com/articles/s41467-024-54457-x"
    ]
  },
  {
    "id": "asa-madd",
    "date_added": "2026-06-16",
    "name": "MADD",
    "category": "chemistry",
    "domain": "Multi-agent drug discovery orchestra",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2511.08217"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Drug design objective/properties (natural-language query)",
    "outputs": "De novo compound generation/screening, hit molecules, evaluation",
    "autonomy": "A3-A4",
    "notes": "Multi-agent drug discovery orchestra (four agents) for hit identification; EMNLP 2025 Findings.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2511.08217"
    ]
  },
  {
    "id": "asa-liddia",
    "date_added": "2026-06-16",
    "name": "LIDDIA",
    "category": "chemistry",
    "domain": "Language-based intelligent drug discovery agent",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2502.13959"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ninglab/LIDDiA"
      }
    ],
    "access": "open-source",
    "inputs": "Drug-discovery objective with target/property constraints",
    "outputs": "Generated and evaluated drug-candidate molecules",
    "autonomy": "A2-A3",
    "notes": "Language-based intelligent drug discovery agent navigating in-silico discovery; EMNLP 2025.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2502.13959",
      "https://github.com/ninglab/LIDDiA"
    ]
  },
  {
    "id": "asa-dziner",
    "date_added": "2026-06-16",
    "name": "dZiner",
    "category": "chemistry",
    "domain": "Inverse molecular/materials design",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2410.03963"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/mehradans92/dziner"
      }
    ],
    "access": "open-source",
    "inputs": "Target material/molecule requirements (property-to-structure)",
    "outputs": "Candidate designs and rationale with property inference",
    "autonomy": "A3",
    "notes": "Chemist AI agent for rational inverse design of materials (MOFs, surfactants, ligands).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2410.03963",
      "https://github.com/mehradans92/dziner"
    ]
  },
  {
    "id": "asa-llamp",
    "date_added": "2026-06-16",
    "name": "LLaMP",
    "category": "chemistry",
    "domain": "Materials informatics RAG/agent platform",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2401.17244"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/chiang-yuan/llamp"
      }
    ],
    "access": "open-source",
    "inputs": "Materials-science query/data",
    "outputs": "Materials knowledge retrieval, atomistic simulation orchestration, analysis",
    "autonomy": "A2",
    "notes": "Multimodal RAG framework of hierarchical ReAct agents grounded on Materials Project; EMNLP 2025.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2401.17244",
      "https://github.com/chiang-yuan/llamp"
    ]
  },
  {
    "id": "asa-physmaster",
    "date_added": "2026-06-16",
    "name": "PhysMaster",
    "category": "physics",
    "domain": "Theoretical and computational physics",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2512.19799"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Physics research problem, literature, computation needs",
    "outputs": "Reasoning traces, simulations, hypothesis loops, candidate discoveries",
    "autonomy": "A4",
    "notes": "Autonomous AI physicist for theoretical/computational physics; uses LANDAU knowledge repository and adaptive exploration.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2512.19799"
    ]
  },
  {
    "id": "asa-ai-mandel",
    "date_added": "2026-06-16",
    "name": "AI-Mandel",
    "category": "physics",
    "domain": "Quantum physics idea generation and implementation",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2511.11752"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/artificial-scientist-lab/ai-mandel"
      }
    ],
    "access": "open-source",
    "inputs": "Quantum-physics literature/context and target area",
    "outputs": "Implementable quantum-optics experiment designs and follow-up ideas",
    "autonomy": "A3-A4",
    "notes": "LLM multi-agent system that generates and executes novel quantum-optics ideas via PyTheus simulation.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2511.11752",
      "https://github.com/artificial-scientist-lab/ai-mandel"
    ]
  },
  {
    "id": "asa-physvec",
    "date_added": "2026-06-16",
    "name": "PhysVEC",
    "category": "physics",
    "domain": "Quantum many-body simulation verification",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.00149"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Paper-derived quantum many-body simulation task",
    "outputs": "Code, programming verification, scientific verification, corrections",
    "autonomy": "A3",
    "notes": "Verifiable, self-correcting AI-physicist framework for quantum many-body simulations; introduces QMP-Bench.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2604.00149"
    ]
  },
  {
    "id": "asa-dr-sai",
    "date_added": "2026-06-16",
    "name": "Dr.Sai",
    "category": "physics",
    "domain": "High-energy physics analysis at BESIII",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.22541"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Natural-language HEP analysis request",
    "outputs": "Simulation/reconstruction/statistical workflow and results",
    "autonomy": "A4",
    "notes": "Computational agent for BESIII data analysis, simulation, reconstruction, and statistics. It does not control the collider or detector, so physical-lab gating is not claimed.",
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2604.22541"
    ]
  },
  {
    "id": "asa-openfoamgpt",
    "date_added": "2026-06-16",
    "name": "OpenFOAMGPT",
    "category": "physics",
    "domain": "Computational fluid dynamics",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2501.06327"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Natural-language CFD/OpenFOAM task",
    "outputs": "Case setup, boundary/model changes, code translation, simulation iterations",
    "autonomy": "A2-A3",
    "notes": "RAG-augmented single-agent LLM for OpenFOAM CFD; human oversight still important.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2501.06327"
    ]
  },
  {
    "id": "asa-foam-agent",
    "date_added": "2026-06-16",
    "name": "Foam-Agent",
    "category": "physics",
    "domain": "CFD multi-agent workflows",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.04997"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/csml-rpi/Foam-Agent"
      }
    ],
    "access": "open-source",
    "inputs": "Natural-language OpenFOAM/CFD objective",
    "outputs": "Pre-processing, solver setup, HPC scripts, post-processing visualization",
    "autonomy": "A3",
    "notes": "Multi-agent end-to-end OpenFOAM CFD framework (Architect/Input Writer/Runner/Reviewer); 88.2% success on 110 tasks.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2505.04997",
      "https://github.com/csml-rpi/Foam-Agent"
    ]
  },
  {
    "id": "asa-metaopenfoam-sim-cli",
    "date_added": "2026-06-16",
    "name": "MetaOpenFOAM / sim-cli",
    "category": "physics",
    "domain": "CFD automation",
    "paper_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Terry-cyx/MetaOpenFOAM"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Terry-cyx/MetaOpenFOAM"
      }
    ],
    "access": "open-source",
    "inputs": "CFD setup request",
    "outputs": "OpenFOAM setup and workflow automation",
    "autonomy": "A2-A3",
    "notes": "LLM multi-agent CFD framework; repo deprecated, capabilities migrated to sim-cli (github.com/svd-ai-lab/sim-cli).",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/Terry-cyx/MetaOpenFOAM"
    ]
  },
  {
    "id": "asa-mechagents",
    "date_added": "2026-06-16",
    "name": "MechAgents",
    "category": "physics",
    "domain": "Mechanics and finite-element problems",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2311.08166"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/lamm-mit/MechAgents"
      }
    ],
    "access": "open-source",
    "inputs": "Elasticity/mechanics problem",
    "outputs": "FEM code (FEniCS), execution, self-correction, plots/results",
    "autonomy": "A3",
    "notes": "Multi-agent LLM collaboration for mechanics/FEM problems (Ni & Buehler, MIT); published in Extreme Mechanics Letters 2024.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2311.08166",
      "https://github.com/lamm-mit/MechAgents"
    ]
  },
  {
    "id": "asa-mooseagent",
    "date_added": "2026-06-16",
    "name": "MooseAgent",
    "category": "physics",
    "domain": "Multiphysics/FEM automation",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.08621"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/taozhan18/MooseAgent"
      }
    ],
    "access": "open-source",
    "inputs": "MOOSE/FEM multiphysics task in natural language",
    "outputs": "Auto-generated MOOSE input files, solver configuration, post-processing",
    "autonomy": "A3",
    "notes": "LLM multi-agent framework automating MOOSE finite-element simulations via task decomposition and iterative verification.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2504.08621",
      "https://github.com/taozhan18/MooseAgent"
    ]
  },
  {
    "id": "asa-vfeagent",
    "date_added": "2026-06-16",
    "name": "VFEAgent",
    "category": "physics",
    "domain": "End-to-end finite-element analysis",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2605.28978"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Multimodal FEA problem (input images plus problem description)",
    "outputs": "Complete, physically valid finite-element simulations",
    "autonomy": "A3",
    "notes": "Vision-language multi-agent framework for end-to-end automated finite-element analysis with self-debugging code synthesis.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2605.28978"
    ]
  },
  {
    "id": "asa-get-physics-done-gpd",
    "date_added": "2026-06-16",
    "name": "Get Physics Done (GPD)",
    "category": "physics",
    "domain": "General physics research workflow",
    "paper_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/psi-oss/get-physics-done"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/psi-oss/get-physics-done"
      }
    ],
    "access": "open-source",
    "inputs": "Physics question / research objective",
    "outputs": "Scoped workflow, derivations, verification scripts, packaged research artifacts",
    "autonomy": "A3",
    "notes": "Open-source agentic AI physicist (PSI PBC); four-stage formulate/plan/execute/verify workflow installable into Claude Code, Gemini CLI, Codex, etc.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/psi-oss/get-physics-done"
    ]
  },
  {
    "id": "asa-anubuddhi",
    "date_added": "2026-06-16",
    "name": "Anubuddhi",
    "category": "physics",
    "domain": "Quantum optics experiment design",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2512.15736"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Natural-language quantum-optics prompt",
    "outputs": "Designed and simulated quantum-optics experiment (optical layout + physics simulation)",
    "autonomy": "A3",
    "notes": "Multi-agent AI system designing/simulating quantum-optics experiments (HOM, Bell states, teleportation, boson sampling) from natural language.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2512.15736"
    ]
  },
  {
    "id": "asa-llm-sr",
    "date_added": "2026-06-16",
    "name": "LLM-SR",
    "category": "physics",
    "domain": "Scientific equation discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2404.18400"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/deep-symbolic-mathematics/LLM-SR"
      }
    ],
    "access": "open-source",
    "inputs": "Observational data/problem",
    "outputs": "Candidate symbolic equations/programs",
    "autonomy": "A2-A3",
    "notes": "Scientific equation discovery combining LLM code generation with evolutionary search; ICLR 2025 Oral.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2404.18400",
      "https://github.com/deep-symbolic-mathematics/LLM-SR"
    ]
  },
  {
    "id": "asa-gravity-bench",
    "date_added": "2026-06-16",
    "name": "Gravity-Bench",
    "category": "benchmark",
    "domain": "Gravitational physics discovery benchmark",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2501.18411"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/NolanKoblischke/GravityBench"
      }
    ],
    "access": "open-data",
    "inputs": "Agent benchmark tasks (gravitational dynamics simulations, experimental budget)",
    "outputs": "Scores on gravitational-physics discovery tasks",
    "autonomy": "B",
    "notes": "Benchmark on gravitational-physics discovery for agents (Gravity-Bench-v1); ICML 2025. Includes HF dataset.",
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2501.18411",
      "https://github.com/NolanKoblischke/GravityBench"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "asa-scienceagentbench",
    "date_added": "2026-06-16",
    "name": "ScienceAgentBench",
    "category": "benchmark",
    "domain": "Cross-domain data-driven discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2410.05080"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/OSU-NLP-Group/ScienceAgentBench"
      }
    ],
    "access": "open-source",
    "inputs": "Real paper-derived data-analysis task with dataset and instructions",
    "outputs": "Self-contained Python program scored on execution/success metrics",
    "autonomy": "B",
    "notes": "Benchmark for agents that generate and execute scientific-analysis code. The official repository released a verified dataset/artifact revision on 30 April 2026 to mitigate false negatives and directs users to the Hugging Face verified split and benchmark_verified.zip.",
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2410.05080",
      "https://github.com/OSU-NLP-Group/ScienceAgentBench"
    ]
  },
  {
    "id": "asa-discoveryworld",
    "date_added": "2026-06-16",
    "name": "DiscoveryWorld",
    "category": "benchmark",
    "domain": "General scientific discovery environment",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2406.06769"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/allenai/discoveryworld"
      }
    ],
    "access": "open-source",
    "inputs": "Simulated discovery-environment challenge task",
    "outputs": "Agent actions through hypothesis/experiment/analysis cycle; task scores",
    "autonomy": "B",
    "notes": "Virtual environment with 120 tasks across 8 topics for evaluating automated scientific-discovery agents.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2406.06769",
      "https://github.com/allenai/discoveryworld"
    ]
  },
  {
    "id": "asa-researchbench",
    "date_added": "2026-06-16",
    "name": "ResearchBench",
    "category": "benchmark",
    "domain": "Hypothesis discovery benchmark",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2503.21248"
      }
    ],
    "repo_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/datasets/ankilok/ResearchBench"
      }
    ],
    "access": "open-data",
    "inputs": "Paper-derived research question, background survey, candidate inspirations",
    "outputs": "Scores on inspiration retrieval, hypothesis composition, hypothesis ranking",
    "autonomy": "B",
    "notes": "Benchmark decomposing scientific discovery into inspiration-based sub-tasks across 12 disciplines.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2503.21248",
      "https://huggingface.co/datasets/ankilok/ResearchBench"
    ]
  },
  {
    "id": "asa-auto-bench",
    "date_added": "2026-06-16",
    "name": "Auto-Bench",
    "category": "benchmark",
    "domain": "Automated scientific discovery benchmark",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2502.15224"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Interactive causal-graph-discovery task (chemistry or social-network setting) with oracle queries",
    "outputs": "Inferred causal structure, intervention decisions, justifications; performance scores",
    "autonomy": "B",
    "notes": "Automated benchmark testing LLMs on causal-graph scientific discovery via iterative intervention.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2502.15224"
    ]
  },
  {
    "id": "asa-scientist-bench",
    "date_added": "2026-06-16",
    "name": "Scientist-Bench",
    "category": "benchmark",
    "domain": "AI research automation benchmark",
    "paper_links": [
      {
        "label": "OpenReview",
        "url": "https://openreview.net/forum?id=kQWyOYUAC4"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/hkuds/ai-researcher"
      }
    ],
    "access": "open-source",
    "inputs": "AI-research paper tasks for guided innovation and open-ended exploration",
    "outputs": "Evaluation of autonomous research outputs (idea, implementation, manuscript)",
    "autonomy": "B",
    "notes": "Benchmark introduced alongside the AI-Researcher system for assessing autonomous AI-research agents.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/hkuds/ai-researcher",
      "https://openreview.net/forum?id=kQWyOYUAC4"
    ]
  },
  {
    "id": "asa-biodsa-1k",
    "date_added": "2026-06-16",
    "name": "BioDSA-1K",
    "category": "benchmark",
    "domain": "Biomedical data-science agents",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.16100"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Biomedical hypothesis plus supporting data tables and analysis plan",
    "outputs": "Hypothesis decision, evidence-conclusion alignment, reasoning, executable analysis code",
    "autonomy": "B",
    "notes": "Benchmark of 1,029 hypothesis-centric biomedical data-science tasks from 300+ published studies.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2505.16100"
    ]
  },
  {
    "id": "asa-genotex",
    "date_added": "2026-06-16",
    "name": "GenoTEX",
    "category": "benchmark",
    "domain": "Gene-expression data-analysis benchmark",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2406.15341"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Liu-Hy/GenoTEX"
      }
    ],
    "access": "open-source",
    "inputs": "Gene-trait association problem with raw gene-expression datasets",
    "outputs": "Dataset selection, preprocessing and statistical-analysis code/results; alignment with expert annotations",
    "autonomy": "B",
    "notes": "Expert-curated benchmark for automated gene-expression analysis with a GenoAgent baseline.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2406.15341",
      "https://github.com/Liu-Hy/GenoTEX"
    ]
  },
  {
    "id": "asa-qmb100-qmp-bench",
    "date_added": "2026-06-16",
    "name": "QMB100 / QMP-Bench",
    "category": "benchmark",
    "domain": "Quantum many-body physics",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.00149"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "End-to-end quantum-many-body simulation task extracted from research articles",
    "outputs": "Simulation code, verification evidence, corrected results; benchmark scores",
    "autonomy": "B",
    "notes": "QMB100: 100 research-level quantum-many-body simulation tasks from 21 articles, introduced with PhysVEC.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2604.00149"
    ]
  },
  {
    "id": "asa-prl-bench",
    "date_added": "2026-06-16",
    "name": "PRL-Bench",
    "category": "benchmark",
    "domain": "Frontier physics research",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.15411"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Frontier-physics research task derived from recent Physical Review Letters papers",
    "outputs": "End-to-end physics-research workflow with verifiable scores across five subfields",
    "autonomy": "B",
    "notes": "Benchmark of 100 curated PRL-derived tasks evaluating LLMs on frontier physics research.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2604.15411"
    ]
  },
  {
    "id": "asa-openclaw-ecosystem",
    "date_added": "2026-06-16",
    "name": "OpenClaw ecosystem",
    "category": "benchmark",
    "domain": "Scientific agent ecosystem",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.03.30.715118v1"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/FreedomIntelligence/OpenClaw-Medical-Skills"
      }
    ],
    "access": "list",
    "inputs": "Curated dataset of scientific-agent projects and reusable skills",
    "outputs": "Catalog of 91 projects / 2,230 skills across 34 categories; hosted platform",
    "autonomy": "B",
    "notes": "Claw4Science dataset/platform cataloging the OpenClaw scientific-agent ecosystem (skills libraries, not a single benchmark).",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/FreedomIntelligence/OpenClaw-Medical-Skills",
      "https://www.biorxiv.org/content/10.64898/2026.03.30.715118v1"
    ]
  },
  {
    "id": "asa-agentic-science-survey-and-list",
    "date_added": "2026-06-16",
    "name": "Agentic Science survey and list",
    "category": "benchmark",
    "domain": "Cross-domain landscape",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2508.14111"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/AgenticScience/Awesome-Agent-Scientists"
      }
    ],
    "access": "list",
    "inputs": "Literature on autonomous scientific-discovery agents",
    "outputs": "Taxonomy/survey and curated paper list across life sciences, chemistry, materials, physics",
    "autonomy": "B",
    "notes": "Survey 'From AI for Science to Agentic Science' with companion curated list, not a runnable benchmark.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2508.14111",
      "https://github.com/AgenticScience/Awesome-Agent-Scientists"
    ]
  },
  {
    "id": "asa-awesome-llm-scientific-discovery",
    "date_added": "2026-06-16",
    "name": "Awesome LLM Scientific Discovery",
    "category": "benchmark",
    "domain": "Cross-domain curated bibliography",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.13259"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/HKUST-KnowComp/Awesome-LLM-Scientific-Discovery"
      }
    ],
    "access": "list",
    "inputs": "Papers/tools at the intersection of LLMs and scientific discovery",
    "outputs": "Curated, autonomy-leveled bibliography of LLM scientific-discovery research",
    "autonomy": "B",
    "notes": "Curated list backing the survey 'From Automation to Autonomy' (EMNLP 2025); not a single system or benchmark.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2505.13259",
      "https://github.com/HKUST-KnowComp/Awesome-LLM-Scientific-Discovery"
    ]
  },
  {
    "id": "asa-awesome-bioagent-papers",
    "date_added": "2026-06-16",
    "name": "Awesome bioagent papers",
    "category": "benchmark",
    "domain": "Biology/medicine curated bibliography",
    "paper_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/aristoteleo/awesome-bioagent-papers"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/aristoteleo/awesome-bioagent-papers"
      }
    ],
    "access": "list",
    "inputs": "Papers on LLM-based agents in biology and medicine",
    "outputs": "Auto-updated curated lists of bio-agent papers, benchmarks, and reviews",
    "autonomy": "B",
    "notes": "Agent-maintained (Pantheon) tracker of bio/medical agent papers; a bibliography, not a system.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/aristoteleo/awesome-bioagent-papers"
    ]
  },
  {
    "id": "asa-kosmos",
    "date_added": "2026-06-16",
    "name": "Kosmos",
    "category": "crossdomain",
    "domain": "Autonomous AI scientist for R&D",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2511.02824"
      }
    ],
    "repo_links": [
      {
        "label": "Link",
        "url": "https://edisonscientific.com/"
      }
    ],
    "access": "platform",
    "inputs": "Open-ended objective plus dataset or organizational data",
    "outputs": "Code, literature search, hypothesis branches, cited report",
    "autonomy": "A4",
    "notes": "Long-running autonomous AI scientist (Edison Scientific/FutureHouse); hosted via platform.edisonscientific.com.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2511.02824",
      "https://edisonscientific.com/"
    ]
  },
  {
    "id": "asa-futurehouse-platform-agents-crow-falcon-owl-phoenix",
    "date_added": "2026-06-16",
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "name": "Edison Scientific platform agents (formerly FutureHouse): Crow, Falcon, Owl, Phoenix",
    "aliases": [
      "FutureHouse platform agents: Crow, Falcon, Owl, Phoenix"
    ],
    "category": "crossdomain",
    "domain": "Hosted scientific research agents",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://www.futurehouse.org/news/launching-futurehouse-platform-ai-agents"
      }
    ],
    "repo_links": [
      {
        "label": "Link",
        "url": "https://platform.edisonscientific.com/"
      }
    ],
    "access": "platform",
    "inputs": "Research question, papers, chemistry task, prior-art search",
    "outputs": "Scholarly cited answers, deep reviews, prior-art checks, chemistry plans",
    "autonomy": "A2-A3",
    "notes": "Named hosted agents launched by FutureHouse on May 1, 2025: Crow (general QA), Falcon (deep review), Owl (prior-work/precedent), and Phoenix (chemistry planning). The hosted platform is now operated by Edison Scientific, a commercial spinout announced in November 2025; FutureHouse continues as a nonprofit research organization.",
    "sources": [
      "https://platform.edisonscientific.com/",
      "https://www.futurehouse.org/news/launching-futurehouse-platform-ai-agents"
    ]
  },
  {
    "id": "asa-paperqa3-edison-literature",
    "date_added": "2026-06-16",
    "name": "PaperQA3 / Edison Literature",
    "category": "crossdomain",
    "domain": "Literature agent and PaperQA successor",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://edisonscientific.com/news/edison-literature-agent"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Future-House/paper-qa"
      }
    ],
    "access": "platform",
    "inputs": "Scientific query plus papers, figures, tables, patents",
    "outputs": "Cited multimodal literature synthesis answers",
    "autonomy": "A2",
    "notes": "PaperQA3 powers the hosted Edison/FutureHouse Literature agent; multimodal successor to PaperQA2.",
    "verified": "2026-06-16",
    "sources": [
      "https://edisonscientific.com/news/edison-literature-agent",
      "https://github.com/Future-House/paper-qa"
    ]
  },
  {
    "id": "asa-microsoft-discovery",
    "date_added": "2026-06-16",
    "name": "Microsoft Discovery",
    "category": "crossdomain",
    "domain": "Enterprise agentic R&D platform",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://azure.microsoft.com/en-us/blog/transforming-rd-with-agentic-ai-introducing-microsoft-discovery/"
      }
    ],
    "repo_links": [
      {
        "label": "Link",
        "url": "https://azure.microsoft.com/en-us/solutions/discovery/"
      }
    ],
    "access": "platform",
    "inputs": "Organizational data, models, tools, research goal",
    "outputs": "Hypotheses, simulations, analyses, knowledge-graph artifacts",
    "autonomy": "A3-A4",
    "notes": "Enterprise agentic R&D platform on Azure; now GA, used for Majorana 2 quantum chip and materials R&D.",
    "verified": "2026-06-16",
    "sources": [
      "https://azure.microsoft.com/en-us/blog/transforming-rd-with-agentic-ai-introducing-microsoft-discovery/",
      "https://azure.microsoft.com/en-us/solutions/discovery/"
    ]
  },
  {
    "id": "asa-bios-bioagent-deep-research",
    "date_added": "2026-06-16",
    "name": "BIOS / BioAgent Deep Research",
    "category": "crossdomain",
    "domain": "Interactive AI scientist for biology",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2601.12542"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/bio-xyz/BioAgents"
      }
    ],
    "access": "platform",
    "inputs": "Biomedical objective and datasets",
    "outputs": "Literature review, data analysis, hypotheses, novelty checks",
    "autonomy": "A3",
    "notes": "bio.xyz biology deep-research system (paper system named 'Deep Research'); SOTA on BixBench; productized as BIOS at chat.bio.xyz.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2601.12542",
      "https://github.com/bio-xyz/BioAgents"
    ]
  },
  {
    "id": "asa-internagent-novelseek",
    "date_added": "2026-06-16",
    "name": "InternAgent / NovelSeek",
    "category": "crossdomain",
    "domain": "Closed-loop scientific discovery system",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.16938"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/InternScience/InternAgent"
      }
    ],
    "access": "open-source",
    "inputs": "Research goal, tools, code or empirical setting",
    "outputs": "Algorithms, experiments, reports",
    "autonomy": "A4-A5",
    "notes": "Closed-loop multi-agent discovery system across 12 tasks; NovelSeek was the original name, renamed InternAgent July 2025.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2505.16938",
      "https://github.com/InternScience/InternAgent"
    ]
  },
  {
    "id": "asa-internagent-1-5",
    "date_added": "2026-06-16",
    "name": "InternAgent-1.5",
    "category": "crossdomain",
    "domain": "Long-horizon autonomous science",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2602.08990"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/InternScience/InternAgent"
      }
    ],
    "access": "open-source",
    "inputs": "Research goal, tools, code/lab setting",
    "outputs": "Algorithms, dry/wet experiments, reports",
    "autonomy": "A4-A5",
    "notes": "Successor to InternAgent: generation/verification/evolution subsystems, long-horizon memory; leads GAIA/HLE/GPQA/FrontierScience.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2602.08990",
      "https://github.com/InternScience/InternAgent"
    ]
  },
  {
    "id": "asa-deepscientist",
    "date_added": "2026-06-16",
    "name": "DeepScientist",
    "category": "crossdomain",
    "domain": "Autonomous discovery system",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2509.26603"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ResearAI/DeepScientist"
      }
    ],
    "access": "open-source",
    "inputs": "Goal, metric, compute budget",
    "outputs": "Validated findings and improved methods",
    "autonomy": "A4",
    "notes": "Month-scale autonomous discovery as Bayesian optimization; ICLR 2026; ~5000 ideas, ~1100 validated, beat SOTA on 3 AI tasks.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2509.26603",
      "https://github.com/ResearAI/DeepScientist"
    ]
  },
  {
    "id": "asa-evoscientist",
    "date_added": "2026-06-16",
    "name": "EvoScientist",
    "category": "crossdomain",
    "domain": "Self-evolving multi-agent AI scientist",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.08127"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/EvoScientist/EvoScientist"
      }
    ],
    "access": "open-source",
    "inputs": "Research goal, code environment, persistent memory",
    "outputs": "Ideas, experiments, code, paper artifacts",
    "autonomy": "A4",
    "notes": "Self-evolving multi-agent AI scientist with ideation/experimentation memory; Researcher/Engineer/Evolution-Manager agents.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2603.08127",
      "https://github.com/EvoScientist/EvoScientist"
    ]
  },
  {
    "id": "asa-evomaster",
    "date_added": "2026-06-16",
    "name": "EvoMaster",
    "category": "crossdomain",
    "domain": "Evolving autonomous scientific agents",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.17406"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/sjtu-sai-agents/EvoMaster"
      }
    ],
    "access": "open-source",
    "inputs": "Scientific-agent population and evaluation tasks; base harness config",
    "outputs": "Evolving autonomous scientific agents and performance results",
    "autonomy": "A4",
    "notes": "Foundational evolving agent framework (~100 LOC to build a scientific agent); SciMaster/ML-Master/X-Master built on it.",
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2604.17406",
      "https://github.com/sjtu-sai-agents/EvoMaster"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "asa-scider",
    "date_added": "2026-06-16",
    "name": "SciDER",
    "category": "crossdomain",
    "domain": "Data-centric end-to-end researcher",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.01421"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/leonardodalinky/SciDER"
      }
    ],
    "access": "open-source",
    "inputs": "Raw scientific data plus research intent",
    "outputs": "Data summaries, hypotheses, experiments, paper draft",
    "autonomy": "A3-A4",
    "notes": "Data-centric end-to-end researcher; 4 sub-agents; releases OpenSciDER-SFT-8K dataset and OpenSciDER-27B model.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2603.01421",
      "https://github.com/leonardodalinky/SciDER"
    ]
  },
  {
    "id": "asa-jr-ai-scientist",
    "date_added": "2026-06-16",
    "name": "Jr. AI Scientist",
    "category": "crossdomain",
    "domain": "Baseline-paper extension agent",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2511.04583"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Agent4Science-UTokyo/Jr.AI-Scientist"
      }
    ],
    "access": "open-source",
    "inputs": "Baseline paper and repository",
    "outputs": "Limitations, hypotheses, implementation, manuscript",
    "autonomy": "A4",
    "notes": "Extends a given baseline paper (analyze limits -> hypothesize -> experiment -> write); UTokyo; TMLR 2026.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2511.04583",
      "https://github.com/Agent4Science-UTokyo/Jr.AI-Scientist"
    ]
  },
  {
    "id": "asa-mars",
    "date_added": "2026-06-16",
    "name": "MARS",
    "category": "crossdomain",
    "domain": "Automated AI research agent",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2602.02660"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/jfc43/MARS"
      }
    ],
    "access": "open-source",
    "inputs": "ML research repo/task plus budget",
    "outputs": "Modular code, experiments, learned search insights",
    "autonomy": "A3-A4",
    "notes": "Budget-aware reflective search for automated AI research; cost-constrained MCTS; competitive on MLE-Bench.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2602.02660",
      "https://github.com/jfc43/MARS"
    ]
  },
  {
    "id": "asa-autosota",
    "date_added": "2026-06-16",
    "name": "AutoSOTA",
    "category": "crossdomain",
    "domain": "Automated model replication and improvement",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.05550"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/tsinghua-fib-lab/AutoSOTA"
      }
    ],
    "access": "open-source",
    "inputs": "Paper and codebase",
    "outputs": "Repaired environment, optimization ideas, patches, improved SOTA results",
    "autonomy": "A4",
    "notes": "Reproduces and empirically improves SOTA AI models; 8-agent system; discovered 105 improved models.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2604.05550",
      "https://github.com/tsinghua-fib-lab/AutoSOTA"
    ]
  },
  {
    "id": "asa-deep-researcher-agent",
    "date_added": "2026-06-16",
    "name": "Deep Researcher Agent",
    "category": "crossdomain",
    "domain": "Continuous deep-learning experiment loop",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.05854"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Xiangyue-Zhang/auto-deep-researcher-24x7"
      }
    ],
    "access": "open-source",
    "inputs": "DL project, logs, baselines",
    "outputs": "Overnight experiments, analyses, refinements",
    "autonomy": "A4",
    "notes": "24/7 deep-learning experimentation loop with zero-cost monitoring and bounded memory; UTokyo; 30+ day runs.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2604.05854",
      "https://github.com/Xiangyue-Zhang/auto-deep-researcher-24x7"
    ]
  },
  {
    "id": "asa-eurekagent",
    "date_added": "2026-06-16",
    "name": "EurekAgent",
    "category": "crossdomain",
    "domain": "Environment-engineered discovery agent",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2606.13662"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/THU-Team-Eureka/EurekAgent"
      }
    ],
    "access": "open-source",
    "inputs": "Metric and executable environment",
    "outputs": "Improved solutions, artifacts, logs",
    "autonomy": "A4",
    "notes": "Environment-engineering approach (permissions/artifacts/budget/human-in-loop) coordinating CLI agents; SOTA on math/kernel/MLE-Bench tasks.",
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2606.13662",
      "https://github.com/THU-Team-Eureka/EurekAgent"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "asa-codescientist",
    "date_added": "2026-06-16",
    "name": "CodeScientist",
    "category": "crossdomain",
    "domain": "Code-based scientific discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2503.22708"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/allenai/codescientist"
      }
    ],
    "access": "open-source",
    "inputs": "Articles, code blocks, experiment environment",
    "outputs": "Candidate discoveries, experiments, soundness/novelty checks",
    "autonomy": "A3-A4",
    "notes": "Code-based semi-automated discovery via genetic search over papers+codeblocks; AI2; ACL 2025 Findings; Apache 2.0.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2503.22708",
      "https://github.com/allenai/codescientist"
    ]
  },
  {
    "id": "asa-asta-agents",
    "date_added": "2026-06-16",
    "name": "Asta agents",
    "category": "crossdomain",
    "domain": "AI2 scientific assistant ecosystem",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://allenai.org/asta/agents"
      }
    ],
    "repo_links": [
      {
        "label": "Link",
        "url": "https://asta.allen.ai/"
      }
    ],
    "access": "platform",
    "inputs": "Search, synthesis, data-analysis tasks",
    "outputs": "Literature/data answers, agent workflow artifacts",
    "autonomy": "A2-A3",
    "notes": "AI2 scientific assistant ecosystem (search + synthesis + data analysis over 100M+ abstracts).",
    "verified": "2026-06-16",
    "sources": [
      "https://allenai.org/asta/agents",
      "https://asta.allen.ai/"
    ]
  },
  {
    "id": "asa-piflow",
    "date_added": "2026-06-16",
    "name": "PiFlow",
    "category": "crossdomain",
    "domain": "Principle-aware multi-agent discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.15047"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/amair-lab/PiFlow"
      }
    ],
    "access": "open-source",
    "inputs": "Discovery objective and uncertainty-reduction principles",
    "outputs": "Agent plans, experiments, optimized candidates",
    "autonomy": "A3-A4",
    "notes": "Principle-aware multi-agent discovery as uncertainty reduction; plug-and-play; tested on nanomaterials, biomolecules, superconductors.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2505.15047",
      "https://github.com/amair-lab/PiFlow"
    ]
  },
  {
    "id": "asa-alpharesearch",
    "date_added": "2026-06-16",
    "name": "AlphaResearch",
    "category": "crossdomain",
    "domain": "Autonomous algorithm discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2511.08522"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/answers111/alpha-research"
      }
    ],
    "access": "open-source",
    "inputs": "Open-ended algorithmic problem and an executable evaluation environment",
    "outputs": "Candidate algorithms, code, execution/performance traces, simulated peer-review scores",
    "autonomy": "A4",
    "notes": "Autonomous algorithm-discovery agent combining execution-verifiable reward with a simulated peer-review reward model.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2511.08522",
      "https://github.com/answers111/alpha-research"
    ]
  },
  {
    "id": "asa-openscientist",
    "date_added": "2026-06-16",
    "name": "OpenScientist",
    "category": "crossdomain",
    "domain": "Open-source biomedical discovery agent",
    "paper_links": [
      {
        "label": "medRxiv",
        "url": "https://www.medrxiv.org/content/10.64898/2026.03.15.26348338v1"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/openscientist-io/openscientist"
      }
    ],
    "access": "open-source",
    "inputs": "Scientist-defined biomedical query plus data (EHR, omics, imaging, histopathology, biomarkers, tabular/text)",
    "outputs": "Autonomous hypotheses, computational analyses, literature-grounded mechanistic insights, verifiable findings",
    "autonomy": "A3-A4",
    "notes": "Open, auditable agentic biomedical co-scientist (Claude Sonnet 4.5, Agent Skills + MCP, Docker UI at localhost:8080).",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/openscientist-io/openscientist",
      "https://www.medrxiv.org/content/10.64898/2026.03.15.26348338v1"
    ]
  },
  {
    "id": "asa-medical-ai-scientist",
    "date_added": "2026-06-16",
    "name": "Medical AI Scientist",
    "category": "crossdomain",
    "domain": "Domain-aware clinical AI discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.28589"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Clinical/medical-AI research task, data modality (images, video, EHR, ECG, reports, multimodal), benchmark case",
    "outputs": "Idea proposals, experimental pipelines, evidence-grounded manuscript drafts, Med-AI-Bench results",
    "autonomy": "A3-A4",
    "notes": "Autonomous clinical-research framework (Idea Proposer, Experimental Executor, Manuscript Composer) with clinician-engineer co-reasoning.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2603.28589"
    ]
  },
  {
    "id": "asa-fars",
    "date_added": "2026-06-16",
    "name": "FARS",
    "category": "crossdomain",
    "domain": "Fully automated research system",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://analemma.ai/blog/introducing-fars/"
      }
    ],
    "repo_links": [],
    "access": "platform",
    "inputs": "Research objective (currently AI/LLM research domain)",
    "outputs": "Autonomous ideation, plans, experiments on GPU cluster, complete written papers",
    "autonomy": "A4",
    "notes": "Analemma Intelligence's Fully Automated Research System; four agents (Ideation, Planning, Experiment, Writing), publicly livestreamed producing 100 papers.",
    "verified": "2026-06-16",
    "sources": [
      "https://analemma.ai/blog/introducing-fars/"
    ]
  },
  {
    "id": "asa-openai-deep-research",
    "date_added": "2026-06-16",
    "name": "OpenAI Deep Research",
    "category": "crossdomain",
    "domain": "General deep-research agent",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://cdn.openai.com/deep-research-system-card.pdf"
      }
    ],
    "repo_links": [
      {
        "label": "Link",
        "url": "https://developers.openai.com/api/docs/guides/deep-research"
      }
    ],
    "access": "platform",
    "inputs": "Natural-language prompt, optional files, web-browsing scope",
    "outputs": "Multi-step, inline-cited research report",
    "autonomy": "A2",
    "notes": "OpenAI's hosted agentic deep-research capability (o3-based, browsing + sandboxed Python); broadly accessible via ChatGPT/API.",
    "verified": "2026-06-16",
    "sources": [
      "https://cdn.openai.com/deep-research-system-card.pdf",
      "https://developers.openai.com/api/docs/guides/deep-research"
    ]
  },
  {
    "id": "asa-lila-sciences",
    "date_added": "2026-06-16",
    "name": "Lila Sciences",
    "category": "crossdomain",
    "domain": "Commercial autonomous-science/lab platform",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://www.flagshippioneering.com/news/press-release/flagship-pioneering-unveils-lila-sciences-to-build-superintelligence-in-science"
      }
    ],
    "repo_links": [],
    "access": "platform",
    "inputs": "Human-guided scientific goals across life, chemical, materials science",
    "outputs": "Hypotheses, autonomous experiment designs, lab results (AI Science Factory)",
    "autonomy": "A5",
    "notes": "Flagship-Pioneering-founded scientific-superintelligence platform with autonomous labs; closed commercial system.",
    "verified": "2026-06-16",
    "sources": [
      "https://www.flagshippioneering.com/news/press-release/flagship-pioneering-unveils-lila-sciences-to-build-superintelligence-in-science"
    ]
  },
  {
    "id": "asa-periodic-labs",
    "date_added": "2026-06-16",
    "name": "Periodic Labs",
    "category": "crossdomain",
    "domain": "Commercial physical-science autonomous-lab platform",
    "paper_links": [
      {
        "label": "official site",
        "url": "https://periodic.com/"
      }
    ],
    "repo_links": [
      {
        "label": "Link",
        "url": "https://periodic.com/"
      }
    ],
    "access": "platform",
    "inputs": "Physical-science / materials objectives (e.g. superconductors, semiconductors)",
    "outputs": "Autonomous experiments, high-quality experimental data, materials insights",
    "autonomy": "A5",
    "notes": "Commercial 'AI scientist' paired with autonomous physical-science labs; founders from ChatGPT and DeepMind GNoME.",
    "verified": "2026-06-16",
    "sources": [
      "https://periodic.com/"
    ]
  },
  {
    "id": "asa-biodiscoveryagent",
    "date_added": "2026-06-16",
    "name": "BioDiscoveryAgent",
    "category": "biology",
    "domain": "Genetic perturbation experiment design",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2405.17631"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/snap-stanford/BioDiscoveryAgent"
      }
    ],
    "access": "open-source",
    "inputs": "Phenotype goal, prior perturbation results, biomedical literature",
    "outputs": "Proposed gene perturbation batches for the next experimental round",
    "autonomy": "A3-A4",
    "notes": "Closed-loop LLM agent for designing genetic (Perturb-seq) perturbation experiments.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2405.17631",
      "https://github.com/snap-stanford/BioDiscoveryAgent"
    ]
  },
  {
    "id": "asa-geneagent",
    "date_added": "2026-06-16",
    "name": "GeneAgent",
    "category": "biology",
    "domain": "Gene-set interpretation",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41592-025-02748-6"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ncbi-nlp/GeneAgent"
      }
    ],
    "access": "open-source",
    "inputs": "Gene sets",
    "outputs": "Functional annotations with self-verification against domain databases",
    "autonomy": "A2",
    "notes": "Self-verifying LLM agent for gene-set analysis using expert-curated databases.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/ncbi-nlp/GeneAgent",
      "https://www.nature.com/articles/s41592-025-02748-6"
    ]
  },
  {
    "id": "asa-autoba",
    "date_added": "2026-06-16",
    "name": "AutoBA",
    "category": "biology",
    "domain": "Multi-omics analysis automation",
    "paper_links": [
      {
        "label": "PMC",
        "url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC11600294/"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/JoshuaChou2018/AutoBA"
      }
    ],
    "access": "open-source",
    "inputs": "Omics data files and natural-language analysis task",
    "outputs": "Analysis plans, code, plots, reports",
    "autonomy": "A3",
    "notes": "Autonomous AI agent for fully automated multi-omic bioinformatics analysis.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/JoshuaChou2018/AutoBA",
      "https://pmc.ncbi.nlm.nih.gov/articles/PMC11600294/"
    ]
  },
  {
    "id": "asa-cellagent",
    "date_added": "2026-06-16",
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "name": "CellAgent",
    "category": "biology",
    "domain": "scRNA-seq and spatial analysis",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2024.05.13.593861v4"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/liu-shiqiang/CellAgent"
      }
    ],
    "access": "open-source",
    "inputs": "scRNA-seq / spatial transcriptomics data and natural-language task",
    "outputs": "Automated Scanpy-style analyses (QC, clustering, annotation, DE, trajectory)",
    "autonomy": "A3",
    "notes": "LLM-driven Planner/Executor/Evaluator multi-agent for single-cell analysis.",
    "sources": [
      "https://github.com/liu-shiqiang/CellAgent",
      "https://www.biorxiv.org/content/10.1101/2024.05.13.593861v4"
    ]
  },
  {
    "id": "asa-biomaster",
    "date_added": "2026-06-16",
    "name": "BioMaster",
    "category": "biology",
    "domain": "Bioinformatics workflows",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.01.23.634608v1"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ai4nucleome/BioMaster"
      }
    ],
    "access": "open-source",
    "inputs": "Bioinformatics task and data (RNA-seq, ChIP-seq, single-cell, Hi-C, nanopore, etc.)",
    "outputs": "Workflow design, executed scripts, validation, interpretation",
    "autonomy": "A3",
    "notes": "Plan/Task/Debug/Check multi-agent system for automated bioinformatics workflows.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/ai4nucleome/BioMaster",
      "https://www.biorxiv.org/content/10.1101/2025.01.23.634608v1"
    ]
  },
  {
    "id": "asa-compbioagent",
    "date_added": "2026-06-16",
    "name": "CompBioAgent",
    "category": "biology",
    "domain": "Single-cell exploration",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.03.17.643771v1"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/interactivereport/CompBioAgent"
      }
    ],
    "access": "open-source",
    "inputs": "scRNA-seq dataset and natural-language query (CellDepot-backed)",
    "outputs": "Structured data requests, filtering, and visualizations (UMAP, violin, heatmap)",
    "autonomy": "A2-A3",
    "notes": "LLM-powered web agent for natural-language single-cell RNA-seq exploration.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/interactivereport/CompBioAgent",
      "https://www.biorxiv.org/content/10.1101/2025.03.17.643771v1"
    ]
  },
  {
    "id": "asa-biomania",
    "date_added": "2026-06-16",
    "name": "BioMANIA",
    "category": "biology",
    "domain": "Bioinformatics API/tool agent",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2023.10.29.564479v1"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/batmen-lab/BioMANIA"
      }
    ],
    "access": "open-source",
    "inputs": "Omics task and a documented open-source Python tool's APIs/docs",
    "outputs": "Conversational tool/API calls, executed analyses, reports",
    "autonomy": "A2",
    "notes": "Conversational AI pipeline that learns Python tool APIs to automate bioinformatics analysis.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/batmen-lab/BioMANIA",
      "https://www.biorxiv.org/content/10.1101/2023.10.29.564479v1"
    ]
  },
  {
    "id": "asa-scchat",
    "date_added": "2026-06-16",
    "name": "scChat",
    "category": "biology",
    "domain": "Context-aware scRNA-seq copilot",
    "paper_links": [
      {
        "label": "PMC",
        "url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13061372/"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/li-group/scChat"
      }
    ],
    "access": "open-source",
    "inputs": "scRNA-seq data and study/research context",
    "outputs": "Contextualized analysis suggestions, reasoning, and hypotheses",
    "autonomy": "A2-A3",
    "notes": "LLM-powered co-pilot for contextualized single-cell RNA-seq analysis.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/li-group/scChat",
      "https://pmc.ncbi.nlm.nih.gov/articles/PMC13061372/"
    ]
  },
  {
    "id": "asa-flowagent",
    "date_added": "2026-06-16",
    "name": "FlowAgent",
    "category": "biology",
    "domain": "Autonomous bioinformatics workflow planning, execution, recovery, and interpretation",
    "paper_links": [
      {
        "label": "FlowAgent bioRxiv",
        "url": "https://doi.org/10.1101/2025.03.06.641728"
      },
      {
        "label": "FlowBench bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.06.12.731844v1"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/EnteloBio/flowagent"
      }
    ],
    "access": "open-source",
    "inputs": "Bioinformatics goals, datasets, local/HPC environments, tool documentation, and workflow failures",
    "outputs": "Executed shell/Nextflow/Snakemake workflows, QC reports, checkpointed recovery, interpretations, and final results",
    "autonomy": "A3",
    "notes": "Modular agent-based system for automated bioinformatics workflow management.",
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://doi.org/10.1101/2025.03.06.641728",
      "https://github.com/EnteloBio/flowagent",
      "https://www.biorxiv.org/content/10.64898/2026.06.12.731844v1"
    ]
  },
  {
    "id": "asa-transagent",
    "date_added": "2026-06-16",
    "name": "TransAgent",
    "category": "biology",
    "domain": "Transcriptional-regulation and multi-omics analysis",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.04.27.650826v1"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/TOSTRING-Z/transagent"
      }
    ],
    "access": "open-source",
    "inputs": "Epigenomic/transcriptomic multi-omics data and analysis task",
    "outputs": "Transcriptional-regulation analyses and workflows (30+ tools, 20+ data sources)",
    "autonomy": "A3",
    "notes": "Multi-omics-aware LLM agent for transcriptional regulation analysis.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/TOSTRING-Z/transagent",
      "https://www.biorxiv.org/content/10.1101/2025.04.27.650826v1"
    ]
  },
  {
    "id": "asa-tais-genomas",
    "date_added": "2026-06-16",
    "name": "TAIS / GenoMAS",
    "category": "biology",
    "domain": "Gene-expression discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2402.12391"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Liu-Hy/GenoMAS"
      }
    ],
    "access": "open-source",
    "inputs": "Gene-expression datasets (GEO/TCGA)",
    "outputs": "Disease-predictive genes, analysis code, reports",
    "autonomy": "A3-A4",
    "notes": "Multi-agent (PM/data-engineer/domain-expert) framework for gene-expression discovery; GenoMAS is the newer version.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2402.12391",
      "https://github.com/Liu-Hy/GenoMAS"
    ]
  },
  {
    "id": "asa-perturboagent",
    "date_added": "2026-06-16",
    "name": "PerTurboAgent",
    "category": "biology",
    "domain": "Perturb-seq experiment design",
    "paper_links": [
      {
        "label": "PMLR",
        "url": "https://proceedings.mlr.press/v311/hao25b.html"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Initial Perturb-seq data and experimental objective",
    "outputs": "Predicted gene modules and candidate gene panels for next perturbation round",
    "autonomy": "A3",
    "notes": "Self-planning LLM agent for designing iterative/sequential Perturb-seq experiments.",
    "verified": "2026-06-16",
    "sources": [
      "https://proceedings.mlr.press/v311/hao25b.html"
    ]
  },
  {
    "id": "asa-phenograph",
    "date_added": "2026-06-16",
    "name": "PhenoGraph",
    "category": "biology",
    "domain": "Spatial transcriptomics phenotype discovery",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.06.06.658341v1"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Spatial transcriptomics data and a phenotype-driven query",
    "outputs": "Auto-selected/executed analysis pipelines and interpretable phenotype findings",
    "autonomy": "A3",
    "notes": "LLM multi-agent framework for phenotype-driven spatial transcriptomics discovery, augmented with knowledge graphs.",
    "verified": "2026-06-16",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.06.06.658341v1"
    ]
  },
  {
    "id": "asa-proteus",
    "date_added": "2026-06-16",
    "name": "PROTEUS",
    "category": "biology",
    "domain": "Proteomics/multiomics exploratory discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2506.07591"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Raw proteomics/multiomics datasets",
    "outputs": "Research objectives, statistical analyses, and ranked biological hypotheses",
    "autonomy": "A4",
    "notes": "Fully automated LLM system for exploratory proteomics/multiomics hypothesis generation.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2506.07591"
    ]
  },
  {
    "id": "asa-omicsnavigator",
    "date_added": "2026-06-16",
    "name": "OmicsNavigator",
    "category": "biology",
    "domain": "Spatial omics",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.07.21.665821v1"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Spatial omics data (with free-form text queries)",
    "outputs": "Zero-shot structural annotations, pathology assessments, expert-like biological summaries",
    "autonomy": "A3",
    "notes": "LLM-driven multi-agent system for autonomous zero-shot biological analysis in spatial omics.",
    "verified": "2026-06-16",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.07.21.665821v1"
    ]
  },
  {
    "id": "asa-scbasecount-sragent",
    "date_added": "2026-06-16",
    "name": "scBaseCount / SRAgent",
    "category": "biology",
    "domain": "Single-cell repository curation",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.02.27.640494v2.full-text"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ArcInstitute/SRAgent"
      }
    ],
    "access": "open-source",
    "inputs": "SRA records / public scRNA-seq accessions",
    "outputs": "Standardized, uniformly processed h5ad/count single-cell datasets and curated metadata",
    "autonomy": "A3",
    "notes": "Arc Institute AI-agent (SRAgent) that mines SRA to build/maintain the scBaseCount single-cell repository.",
    "verified": "2026-07-13",
    "sources": [
      "https://github.com/ArcInstitute/SRAgent",
      "https://www.biorxiv.org/content/10.1101/2025.02.27.640494v2.full-text"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "asa-mragent",
    "date_added": "2026-06-16",
    "name": "MRAgent",
    "category": "biology",
    "domain": "Mendelian-randomization causal discovery",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://academic.oup.com/bib/article/26/2/bbaf140/8107848"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/xuwei1997/MRAgent"
      }
    ],
    "access": "open-source",
    "inputs": "Disease/topic plus literature and GWAS data",
    "outputs": "Discovered exposure-outcome pairs and Mendelian-randomization causal-inference reports",
    "autonomy": "A3",
    "notes": "LLM agent automating Mendelian-randomization causal knowledge discovery in disease.",
    "verified": "2026-06-16",
    "sources": [
      "https://academic.oup.com/bib/article/26/2/bbaf140/8107848",
      "https://github.com/xuwei1997/MRAgent"
    ]
  },
  {
    "id": "asa-hypogeneagent",
    "date_added": "2026-06-16",
    "name": "HypoGeneAgent",
    "category": "biology",
    "domain": "Perturb-seq cluster hypothesis generation",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2509.09740"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Perturb-seq gene programs / cluster gene sets",
    "outputs": "Ranked GO-based hypotheses with confidence scores and a resolution/agreement score",
    "autonomy": "A2-A3",
    "notes": "LLM hypothesis agent for gene-set cluster resolution selection in Perturb-seq data.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2509.09740"
    ]
  },
  {
    "id": "asa-plantgpt",
    "date_added": "2026-06-16",
    "name": "PlantGPT",
    "category": "biology",
    "domain": "Plant functional genomics",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://advanced.onlinelibrary.wiley.com/doi/10.1002/advs.202503926"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/DrZRX/Plant_LLM"
      }
    ],
    "access": "open-source",
    "inputs": "Natural-language questions about Arabidopsis plant functional genomics",
    "outputs": "RAG-grounded, hallucination-limited functional-genomics answers",
    "autonomy": "A2",
    "notes": "Arabidopsis-focused RAG + fine-tuned LLM (Llama3-8B) agent for plant functional genomics QA.",
    "verified": "2026-06-16",
    "sources": [
      "https://advanced.onlinelibrary.wiley.com/doi/10.1002/advs.202503926",
      "https://github.com/DrZRX/Plant_LLM"
    ]
  },
  {
    "id": "asa-myegpt",
    "date_added": "2026-06-16",
    "name": "MyeGPT",
    "category": "biology",
    "domain": "Multiple-myeloma clinical molecular analysis",
    "paper_links": [
      {
        "label": "medRxiv",
        "url": "https://www.medrxiv.org/content/10.64898/2026.05.14.26353252v5"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/JiaGengChang/MyeGPT"
      }
    ],
    "access": "open-source",
    "inputs": "Natural-language queries over CoMMpass multi-omics multiple-myeloma data",
    "outputs": "De novo analyses, visualizations, and candidate biomarkers",
    "autonomy": "A3",
    "notes": "ReAct AI bioinformatician for multiple-myeloma molecular analysis on the CoMMpass cohort.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/JiaGengChang/MyeGPT",
      "https://www.medrxiv.org/content/10.64898/2026.05.14.26353252v5"
    ]
  },
  {
    "id": "asa-ai-hope",
    "date_added": "2026-06-16",
    "name": "AI-HOPE",
    "category": "biology",
    "domain": "Clinical/genomic precision medicine",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://academic.oup.com/bioinformatics/article/41/7/btaf359/8169327"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Velazquez-Villarreal-Lab/AI-HOPE"
      }
    ],
    "access": "open-source",
    "inputs": "Clinical and genomic data (TCGA via cBioPortal/UCSC Xena, or user-uploaded)",
    "outputs": "Integrated precision-medicine analyses: case-control studies, survival curves, odds ratios, variable scans",
    "autonomy": "A2-A3",
    "notes": "LLM conversational agent that converts natural-language instructions into executable clinical-genomic analyses.",
    "verified": "2026-06-16",
    "sources": [
      "https://academic.oup.com/bioinformatics/article/41/7/btaf359/8169327",
      "https://github.com/Velazquez-Villarreal-Lab/AI-HOPE"
    ]
  },
  {
    "id": "asa-trialgenie",
    "date_added": "2026-06-16",
    "name": "TrialGenie",
    "category": "biology",
    "domain": "Clinical-trial design",
    "paper_links": [
      {
        "label": "medRxiv",
        "url": "https://www.medrxiv.org/content/10.1101/2025.04.17.25326033v1.full-text"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Trial concept plus real-world data (e.g., EHR / MIMIC-IV)",
    "outputs": "Generated trial protocols, eligibility criteria, cohort phenotypes, and RWE analysis reports",
    "autonomy": "A3",
    "notes": "Agentic multi-agent framework using real-world data to support clinical-trial design.",
    "verified": "2026-06-16",
    "sources": [
      "https://www.medrxiv.org/content/10.1101/2025.04.17.25326033v1.full-text"
    ]
  },
  {
    "id": "asa-clinagent",
    "date_added": "2026-06-16",
    "name": "ClinAgent",
    "category": "biology",
    "domain": "Clinical-trial statistical programming",
    "paper_links": [
      {
        "label": "medRxiv",
        "url": "https://www.medrxiv.org/content/10.64898/2026.01.09.26343542v1"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/yanmingyu92/ClinAgent"
      }
    ],
    "access": "open-source",
    "inputs": "CDISC/ADaM specifications and SAS datasets",
    "outputs": "CDISC-compliant TLF / statistical-programming code artifacts",
    "autonomy": "A3",
    "notes": "Five-layer / MCP-tool skill layer augmenting AI coding agents for clinical-trial statistical programming.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/yanmingyu92/ClinAgent",
      "https://www.medrxiv.org/content/10.64898/2026.01.09.26343542v1"
    ]
  },
  {
    "id": "asa-agentmd",
    "date_added": "2026-06-16",
    "name": "AgentMD",
    "category": "biology",
    "domain": "Clinical calculator tool-use agent",
    "paper_links": [
      {
        "label": "PMC",
        "url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC12549800/"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ncbi-nlp/Clinical-Tool-Learning"
      }
    ],
    "access": "open-source",
    "inputs": "Clinical context / patient notes",
    "outputs": "Selected clinical calculators (RiskCalcs) and computed individual/population risk estimates",
    "autonomy": "A2-A3",
    "notes": "NCBI language agent that curates and applies 2,164 clinical calculators for risk prediction.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/ncbi-nlp/Clinical-Tool-Learning",
      "https://pmc.ncbi.nlm.nih.gov/articles/PMC12549800/"
    ]
  },
  {
    "id": "asa-clinicalagent",
    "date_added": "2026-06-16",
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "name": "ClinicalAgent",
    "category": "biology",
    "domain": "Clinical-trial reasoning",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2404.14777"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/lingyue404/clinical-agent"
      }
    ],
    "access": "open-source",
    "inputs": "Clinical-trial query/context (drug, condition, enrollment data)",
    "outputs": "Trial outcome-prediction reasoning and supporting analyses",
    "autonomy": "A3",
    "notes": "GPT-4 multi-agent system (ReAct + LEAST-TO-MOST) for clinical-trial reasoning; ACM-BCB 2024.",
    "sources": [
      "https://arxiv.org/abs/2404.14777",
      "https://github.com/lingyue404/clinical-agent"
    ]
  },
  {
    "id": "asa-protagents",
    "date_added": "2026-06-16",
    "name": "ProtAgents",
    "category": "biology",
    "domain": "Protein discovery/design",
    "paper_links": [
      {
        "label": "RSC",
        "url": "https://pubs.rsc.org/en/content/articlelanding/2024/dd/d4dd00013g"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/lamm-mit/ProtAgents"
      }
    ],
    "access": "open-source",
    "inputs": "Protein-design objective / target properties",
    "outputs": "De novo protein designs, structure analyses (AlphaFold/physics), and reports",
    "autonomy": "A3",
    "notes": "LAMM-MIT LLM multi-agent platform combining physics and ML for de novo protein discovery.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/lamm-mit/ProtAgents",
      "https://pubs.rsc.org/en/content/articlelanding/2024/dd/d4dd00013g"
    ]
  },
  {
    "id": "asa-autoproteinengine",
    "date_added": "2026-06-16",
    "name": "AutoProteinEngine",
    "category": "biology",
    "domain": "Protein-engineering AutoML agent",
    "paper_links": [
      {
        "label": "ACL",
        "url": "https://aclanthology.org/2025.coling-industry.36/"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/tsynbio/AutoPE"
      }
    ],
    "access": "open-source",
    "inputs": "Protein-engineering task and data (sequence/graph modalities)",
    "outputs": "Automated ML pipeline (model selection, HPO, data retrieval) and trained model outputs",
    "autonomy": "A3",
    "notes": "LLM agent for multimodal AutoML in protein engineering (AutoPE), accessible via chat.",
    "verified": "2026-06-16",
    "sources": [
      "https://aclanthology.org/2025.coling-industry.36/",
      "https://github.com/tsynbio/AutoPE"
    ]
  },
  {
    "id": "asa-the-virtual-lab",
    "date_added": "2026-06-16",
    "name": "The Virtual Lab",
    "category": "biology",
    "domain": "Multi-agent nanobody design",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-025-09442-9"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/zou-group/virtual-lab"
      }
    ],
    "access": "open-source",
    "inputs": "Research target / design objective (with high-level human feedback)",
    "outputs": "Designed and experimentally validated nanobody candidates plus a validation pipeline",
    "autonomy": "A4-A5",
    "notes": "Zou-group LLM PI + specialist-agent team; designed/validated SARS-CoV-2 nanobodies.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/zou-group/virtual-lab",
      "https://www.nature.com/articles/s41586-025-09442-9"
    ]
  },
  {
    "id": "asa-origene",
    "date_added": "2026-06-16",
    "name": "OriGene",
    "category": "biology",
    "domain": "Virtual disease biologist",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.06.03.657658v1"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/GENTEL-lab/OriGene"
      }
    ],
    "access": "open-source",
    "inputs": "Disease / therapeutic-target question",
    "outputs": "Mechanism-grounded, prioritized therapeutic-target hypotheses",
    "autonomy": "A3-A4",
    "notes": "Self-evolving multi-agent 'virtual disease biologist' integrating 500+ tools for target discovery.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/GENTEL-lab/OriGene",
      "https://www.biorxiv.org/content/10.1101/2025.06.03.657658v1"
    ]
  },
  {
    "id": "asa-venusrar",
    "date_added": "2026-06-16",
    "name": "VenusRAR",
    "category": "biology",
    "domain": "Protein mutation prediction/ranking",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2602.00197"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Protein sequence/structure context; mutation prediction task (zero-shot)",
    "outputs": "Ranked candidate mutants, audited via chain-of-thought against structural constraints",
    "autonomy": "A3",
    "notes": "Multi-agent (Rank-and-Reason) zero-shot protein mutation prediction; wet-lab validated on a nuclease.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2602.00197"
    ]
  },
  {
    "id": "asa-autobinder-agent",
    "date_added": "2026-06-16",
    "name": "AutoBinder Agent",
    "category": "biology",
    "domain": "Protein binder design",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2602.00019"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Target protein and de novo binder design objective",
    "outputs": "Binder designs via surface analysis, scaffold grafting, sequence optimization, structure prediction",
    "autonomy": "A3-A4",
    "notes": "MCP-based LLM agent orchestrating MaSIF, Rosetta, ProteinMPNN, AlphaFold3 for end-to-end protein binder design.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2602.00019"
    ]
  },
  {
    "id": "asa-molclaw",
    "date_added": "2026-06-16",
    "name": "MolClaw",
    "category": "biology",
    "domain": "Drug-molecule screening and optimization",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.21937"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/InternScience/MolClaw"
      }
    ],
    "access": "open-source",
    "inputs": "Drug molecule screening/optimization task requiring multi-step tool use",
    "outputs": "Molecule evaluation/screening/optimization results across 30+ resources via 70 hierarchical skills",
    "autonomy": "A4",
    "notes": "Autonomous hierarchical-skill agent for drug molecule discovery; introduces MolBench benchmark.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2604.21937",
      "https://github.com/InternScience/MolClaw"
    ]
  },
  {
    "id": "asa-agentd",
    "date_added": "2026-06-16",
    "name": "AgentD",
    "category": "biology",
    "domain": "Modular drug-discovery execution",
    "paper_links": [
      {
        "label": "ACS",
        "url": "https://pubs.acs.org/doi/10.1021/acs.jcim.5c02454"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/hoon-ock/AgentD"
      }
    ],
    "access": "open-source",
    "inputs": "Drug-discovery task: data retrieval, molecule generation, property prediction, refinement",
    "outputs": "Retrieved biomolecular data, generated molecules, 75 predicted properties, 3D protein-ligand structures",
    "autonomy": "A2-A3",
    "notes": "Modular LLM agent for early-stage computational drug discovery; open-source Python package.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/hoon-ock/AgentD",
      "https://pubs.acs.org/doi/10.1021/acs.jcim.5c02454"
    ]
  },
  {
    "id": "asa-drugpilot",
    "date_added": "2026-06-16",
    "name": "DrugPilot",
    "category": "biology",
    "domain": "Drug-discovery tool reasoning",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.13940"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/wzn99/DrugPilot"
      }
    ],
    "access": "open-source",
    "inputs": "Drug-discovery research request; heterogeneous tool/data inputs",
    "outputs": "Multi-stage drug-discovery reasoning and decisions via parameterized memory pool and tool use",
    "autonomy": "A2",
    "notes": "LLM-based parameterized reasoning agent for end-to-end drug discovery workflows.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2505.13940",
      "https://github.com/wzn99/DrugPilot"
    ]
  },
  {
    "id": "asa-frogent",
    "date_added": "2026-06-16",
    "name": "FROGENT",
    "category": "biology",
    "domain": "Full-process drug design",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2508.10760"
      }
    ],
    "repo_links": [
      {
        "label": "Link",
        "url": "https://szuaddg.com/research/"
      }
    ],
    "access": "open-source",
    "inputs": "Drug design objective spanning target ID to synthesis planning",
    "outputs": "Closed-loop pipeline: target identification, molecular generation, peptide optimization, retrosynthesis",
    "autonomy": "A3-A4",
    "notes": "End-to-end full-process drug design multi-agent system using LLMs and MCP over 8 benchmarks.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2508.10760",
      "https://szuaddg.com/research/"
    ]
  },
  {
    "id": "asa-drugsage",
    "date_added": "2026-06-16",
    "name": "DrugSAGE",
    "category": "biology",
    "domain": "Self-evolving drug-discovery ML agent",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2605.15461"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Drug-discovery modeling tasks; accumulated cross-task experience memory",
    "outputs": "SOTA drug-discovery models built by reusing verified skills, strategy evidence, error fixes",
    "autonomy": "A4",
    "notes": "Self-evolving agent that accumulates and reuses experience across drug-discovery tasks.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2605.15461"
    ]
  },
  {
    "id": "asa-molreact",
    "date_added": "2026-06-16",
    "name": "MolReAct",
    "category": "biology",
    "domain": "Lead optimization",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.07669"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Lead molecule and optimization objective under synthesizability constraints",
    "outputs": "Synthesizable optimized molecules via RL over reaction-template-constrained action space",
    "autonomy": "A3",
    "notes": "RL with LLM-guided action spaces for synthesizable lead optimization (MDP over validated reactions).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2604.07669"
    ]
  },
  {
    "id": "asa-mozi",
    "date_added": "2026-06-16",
    "name": "Mozi",
    "category": "biology",
    "domain": "Governed drug-discovery autonomy",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.03655"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Drug-discovery task requiring governed autonomous tool orchestration",
    "outputs": "Governed multi-stage drug-discovery workflow with tool governance and transparency",
    "autonomy": "A3",
    "notes": "Dual-layer (control plane + workflow plane) governed-autonomy architecture for drug-discovery LLM agents.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2603.03655"
    ]
  },
  {
    "id": "asa-pharmaswarm",
    "date_added": "2026-06-16",
    "name": "PharmaSwarm",
    "category": "biology",
    "domain": "Multi-agent drug-discovery hypotheses",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.17967"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Drug-discovery hypothesis question (target/disease)",
    "outputs": "Proposed, validated, refined drug-target hypotheses via genomic/pathway/binding analysis",
    "autonomy": "A3",
    "notes": "Hypothesis-driven multi-agent LLM swarm acting as a drug-discovery copilot with four-tier validation.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2504.17967"
    ]
  },
  {
    "id": "asa-sample",
    "date_added": "2026-06-16",
    "name": "SAMPLE",
    "category": "biology",
    "domain": "Self-driving protein-engineering lab",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s44286-023-00002-4"
      }
    ],
    "repo_links": [
      {
        "label": "Zenodo",
        "url": "https://doi.org/10.5281/zenodo.10048592"
      }
    ],
    "access": "lab-gated",
    "inputs": "Protein engineering objective (e.g. thermostable enzyme); robotic lab loop",
    "outputs": "Designed/tested protein variants; converged thermostable enzymes via closed-loop autonomy",
    "autonomy": "A5",
    "notes": "Self-driving lab (Self-driving Autonomous Machines for Protein Landscape Exploration) navigating protein fitness landscapes.",
    "verified": "2026-06-16",
    "sources": [
      "https://doi.org/10.5281/zenodo.10048592",
      "https://www.nature.com/articles/s44286-023-00002-4"
    ]
  },
  {
    "id": "asa-plmeae",
    "date_added": "2026-06-16",
    "name": "PLMeAE",
    "category": "biology",
    "domain": "PLM plus automatic biofoundry loop",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-025-56751-8"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/HICAI-ZJU/PLMeAE"
      }
    ],
    "access": "open-source",
    "inputs": "Target enzyme to evolve; biofoundry robotic build/test loop",
    "outputs": "PLM-designed variants built and tested by biofoundry; improved enzyme activity over rounds",
    "autonomy": "A5",
    "notes": "Protein language model integrated with automatic biofoundry for closed-loop enzyme evolution.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/HICAI-ZJU/PLMeAE",
      "https://www.nature.com/articles/s41467-025-56751-8"
    ]
  },
  {
    "id": "asa-ai-native-autonomous-biofoundry",
    "date_added": "2026-06-16",
    "name": "AI-native autonomous biofoundry",
    "category": "biology",
    "domain": "Enzyme-engineering DBTL platform",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.02.01.703093v1"
      }
    ],
    "repo_links": [],
    "access": "lab-gated",
    "inputs": "Enzyme engineering objective; active-learning loop with automated experimentation",
    "outputs": "Autonomously engineered enzymes via integrated active learning and robotic experimentation",
    "autonomy": "A5",
    "notes": "AI-native biofoundry for autonomous enzyme engineering combining active learning with automated wet-lab.",
    "verified": "2026-06-16",
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.02.01.703093v1"
    ]
  },
  {
    "id": "asa-bioprovla-agent",
    "date_added": "2026-06-16",
    "name": "BioProVLA-Agent",
    "category": "biology",
    "domain": "Embodied biological lab manipulation",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2605.07306"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/no-guess/BioProVLA-Agent"
      }
    ],
    "access": "lab-gated",
    "inputs": "Lab protocol as task interface; visual state of wet-lab setup",
    "outputs": "Embodied closed-loop execution: protocol parsing, visual verification, robotic manipulation",
    "autonomy": "A5",
    "notes": "Affordable protocol-driven vision-language-action embodied multi-agent system for biological lab manipulation.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2605.07306",
      "https://github.com/no-guess/BioProVLA-Agent"
    ]
  },
  {
    "id": "asa-sparks",
    "date_added": "2026-06-16",
    "name": "Sparks",
    "category": "biology",
    "domain": "Protein design-principle discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.19017"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/lamm-mit/Sparks"
      }
    ],
    "access": "open-source",
    "inputs": "Protein-design research objective (in-silico)",
    "outputs": "Autonomous discovery cycle: hypothesis generation, experiment design, iterative refinement, findings",
    "autonomy": "A4",
    "notes": "Multi-modal multi-agent AI that autonomously discovered protein design principles (Buehler lab).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2504.19017",
      "https://github.com/lamm-mit/Sparks"
    ]
  },
  {
    "id": "asa-chemist-x",
    "date_added": "2026-06-16",
    "name": "Chemist-X",
    "category": "chemistry",
    "domain": "Reaction-condition optimization plus robotic execution",
    "paper_links": [
      {
        "label": "arXiv:2311.10776",
        "url": "https://arxiv.org/abs/2311.10776"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Nikki0526/ChemistX"
      }
    ],
    "access": "lab-gated",
    "inputs": "Reaction/product task",
    "outputs": "Conditions, yield labels, robot-control workflow",
    "autonomy": "A4-A5",
    "notes": "Important early autonomous chemistry/robotics system.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2311.10776",
      "https://github.com/Nikki0526/ChemistX"
    ]
  },
  {
    "id": "asa-chatmof",
    "date_added": "2026-06-16",
    "name": "ChatMOF",
    "category": "chemistry",
    "domain": "MOF prediction and inverse design",
    "paper_links": [
      {
        "label": "Nature Communications",
        "url": "https://www.nature.com/articles/s41467-024-48998-4"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Yeonghun1675/ChatMOF"
      }
    ],
    "access": "platform",
    "inputs": "Natural-language MOF/property query",
    "outputs": "MOF search, property prediction, structures",
    "autonomy": "A2-A3",
    "notes": "Materials-specific natural-language agent.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/Yeonghun1675/ChatMOF",
      "https://www.nature.com/articles/s41467-024-48998-4"
    ]
  },
  {
    "id": "asa-chatgpt-research-group-for-mofs-cofs",
    "date_added": "2026-06-16",
    "name": "ChatGPT Research Group for MOFs/COFs",
    "category": "chemistry",
    "domain": "Reticular-chemistry lab ecosystem",
    "paper_links": [
      {
        "label": "ACS",
        "url": "https://pubs.acs.org/doi/10.1021/acscentsci.3c01087"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "MOF/COF synthesis objective, literature/conditions",
    "outputs": "Synthesis condition suggestions, crystallinity optimization via Bayesian search",
    "autonomy": "A3-A5",
    "notes": "Yaghi-group ChatGPT multi-assistant ecosystem optimizing MOF/COF crystallinity.",
    "verified": "2026-06-16",
    "sources": [
      "https://pubs.acs.org/doi/10.1021/acscentsci.3c01087"
    ]
  },
  {
    "id": "asa-atomagents",
    "date_added": "2026-06-16",
    "name": "AtomAgents",
    "category": "chemistry",
    "domain": "Alloy/materials design with simulations",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://www.pnas.org/doi/10.1073/pnas.2414074122"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/lamm-mit/AtomAgents"
      }
    ],
    "access": "open-source",
    "inputs": "Alloy design query; access to LAMMPS/simulation tools",
    "outputs": "Atomistic simulations, predicted properties, alloy candidates",
    "autonomy": "A3-A4",
    "notes": "Physics-aware multimodal multi-agent system for autonomous alloy design.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/lamm-mit/AtomAgents",
      "https://www.pnas.org/doi/10.1073/pnas.2414074122"
    ]
  },
  {
    "id": "asa-llmatdesign",
    "date_added": "2026-06-16",
    "name": "LLMatDesign",
    "category": "chemistry",
    "domain": "Closed-loop materials design",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2406.13163"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Fung-Lab/LLMatDesign"
      }
    ],
    "access": "open-source",
    "inputs": "Starting composition and target property (e.g. band gap, stability)",
    "outputs": "Modified candidate structures with predicted properties via self-reflective loop",
    "autonomy": "A3",
    "notes": "LLM-driven closed-loop interpretable materials design in the small-data regime.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2406.13163",
      "https://github.com/Fung-Lab/LLMatDesign"
    ]
  },
  {
    "id": "asa-prim",
    "date_added": "2026-06-16",
    "name": "PriM",
    "category": "chemistry",
    "domain": "Principle-guided materials discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.08810"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/amair-lab/PriM"
      }
    ],
    "access": "open-source",
    "inputs": "Materials discovery goal and constraints",
    "outputs": "Principle-guided hypotheses, virtual experiments, optimized parameters",
    "autonomy": "A3-A4",
    "notes": "Principle-inspired multi-agent (roundtable) materials discovery; AI4Mat-ICLR2025.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2504.08810",
      "https://github.com/amair-lab/PriM"
    ]
  },
  {
    "id": "asa-mofgen",
    "date_added": "2026-06-16",
    "name": "MOFGen",
    "category": "chemistry",
    "domain": "Agentic MOF generation and validation",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.14110"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "MOF application targets/property goals",
    "outputs": "Novel MOF compositions, diffusion-generated structures, synthesizable linkers (experimentally validated)",
    "autonomy": "A4-A5",
    "notes": "Multi-agent system (LinkerGen/CrystalGen/QForge/SynthABLE) that generated and synthesized five 'AI-dreamt' MOFs.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2504.14110"
    ]
  },
  {
    "id": "asa-el-agente-q",
    "date_added": "2026-06-16",
    "name": "El Agente Q",
    "category": "chemistry",
    "domain": "Quantum-chemistry workflow agent",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.02484"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Natural-language quantum-chemistry task",
    "outputs": "Generated input files, executed calculations, analysis logs, action traces",
    "autonomy": "A3-A4",
    "notes": "Autonomous multi-agent quantum-chemistry workflow agent (Aspuru-Guzik group); published in Matter.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2505.02484"
    ]
  },
  {
    "id": "asa-matclaw",
    "date_added": "2026-06-16",
    "name": "MatClaw",
    "category": "chemistry",
    "domain": "Code-first autonomous materials computation",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.02688"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/DingyangLyu/MatClaw"
      }
    ],
    "access": "open-source",
    "inputs": "Natural-language materials task",
    "outputs": "Python scripts, multi-code HPC simulations (QE/LAMMPS/RASPA3/VASP), plots, reports",
    "autonomy": "A4",
    "notes": "Code-first autonomous LLM agent for end-to-end materials exploration with 240 built-in skills.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2604.02688",
      "https://github.com/DingyangLyu/MatClaw"
    ]
  },
  {
    "id": "asa-vaspilot",
    "date_added": "2026-06-16",
    "name": "VASPilot",
    "category": "chemistry",
    "domain": "MCP/CrewAI VASP workflow platform",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2508.07035"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/JiaxuanLiu-Arsko/VASPilot"
      }
    ],
    "access": "open-source",
    "inputs": "VASP/DFT task description",
    "outputs": "Input files, Slurm jobs, parsed errors with auto-restart, plots",
    "autonomy": "A3-A4",
    "notes": "CrewAI+MCP multi-agent platform automating end-to-end VASP/DFT workflows; VASP/HPC gated.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2508.07035",
      "https://github.com/JiaxuanLiu-Arsko/VASPilot"
    ]
  },
  {
    "id": "asa-dreams",
    "date_added": "2026-06-16",
    "name": "DREAMS",
    "category": "chemistry",
    "domain": "Hierarchical DFT materials-simulation agent",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2507.14267"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/BattModels/material_agent"
      }
    ],
    "access": "open-source",
    "inputs": "DFT/materials simulation objective",
    "outputs": "Quantum Espresso workflows, convergence tests, adsorption-energy results",
    "autonomy": "A3-A4",
    "notes": "Hierarchical multi-agent DFT research engine (Viswanathan group); HPC/tool gated.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2507.14267",
      "https://github.com/BattModels/material_agent"
    ]
  },
  {
    "id": "asa-autodft",
    "date_added": "2026-06-16",
    "name": "AutoDFT",
    "category": "chemistry",
    "domain": "Closed-loop DFT lifecycle automation",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2605.26179"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "DFT calculation objective",
    "outputs": "Plans, parameters, error recovery, property predictions (30 properties)",
    "autonomy": "A3-A4",
    "notes": "Closed-loop seven-agent framework for autonomous DFT calculations.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2605.26179"
    ]
  },
  {
    "id": "asa-master",
    "date_added": "2026-06-16",
    "name": "MASTER",
    "category": "chemistry",
    "domain": "Catalyst/materials DFT discovery",
    "paper_links": [
      {
        "label": "npj Computational Materials",
        "url": "https://doi.org/10.1038/s41524-026-02139-1"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Catalyst search space and target binding-energy range",
    "outputs": "DFT inputs, adsorption energies, reasoning-guided candidate selection",
    "autonomy": "A4",
    "notes": "Hierarchical multi-agent LLM reasoning for autonomous functional-materials/catalyst discovery (CO on Cu/M-N-C).",
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://doi.org/10.1038/s41524-026-02139-1"
    ]
  },
  {
    "id": "asa-topomas",
    "date_added": "2026-06-16",
    "name": "TopoMAS",
    "category": "chemistry",
    "domain": "Topological-materials discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2507.04053"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Topological-materials query",
    "outputs": "Literature/DB retrieval, candidate structures, first-principles validation, knowledge-graph updates",
    "autonomy": "A3-A4",
    "notes": "LLM-driven multi-agent system for topological-materials discovery; guided SrSbO3 identification.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2507.04053"
    ]
  },
  {
    "id": "asa-genius",
    "date_added": "2026-06-16",
    "name": "GENIUS",
    "category": "chemistry",
    "domain": "Simulation protocol design/execution",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s43246-026-01167-0"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Materials simulation goal (free-form prompt)",
    "outputs": "Validated Quantum Espresso input files, parameters, finite-state error recovery",
    "autonomy": "A3-A4",
    "notes": "Agentic AI framework for autonomous design/execution of DFT simulation protocols.",
    "verified": "2026-06-16",
    "sources": [
      "https://www.nature.com/articles/s43246-026-01167-0"
    ]
  },
  {
    "id": "asa-chemgraph",
    "date_added": "2026-06-16",
    "name": "ChemGraph",
    "category": "chemistry",
    "domain": "Computational-chemistry workflow framework",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s42004-025-01776-9"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/argonne-lcf/ChemGraph"
      }
    ],
    "access": "open-source",
    "inputs": "Molecular/materials simulation request (natural language)",
    "outputs": "Structures, geometry optimizations, thermochemistry via GNN foundation models + ASE",
    "autonomy": "A3",
    "notes": "LangGraph/ASE agentic framework for computational chemistry workflows (Argonne).",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/argonne-lcf/ChemGraph",
      "https://www.nature.com/articles/s42004-025-01776-9"
    ]
  },
  {
    "id": "asa-dynamate",
    "date_added": "2026-06-16",
    "name": "DynaMate",
    "category": "chemistry",
    "domain": "Molecular-dynamics workflow agent",
    "paper_links": [
      {
        "label": "RSC",
        "url": "https://pubs.rsc.org/en/content/articlehtml/2025/me/d5me00062a"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/omendibleba/DynaMate"
      }
    ],
    "access": "open-source",
    "inputs": "MD/protein-ligand task; optional custom Python tools",
    "outputs": "Prepared systems, LAMMPS/MD runs, MDAnalysis post-processing, binding-affinity (MM/PB(GB)SA)",
    "autonomy": "A3",
    "notes": "Modular multi-agent framework for autonomous molecular-dynamics workflows.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/omendibleba/DynaMate",
      "https://pubs.rsc.org/en/content/articlehtml/2025/me/d5me00062a"
    ]
  },
  {
    "id": "asa-dynamate2",
    "date_added": "2026-06-16",
    "name": "DynaMate2",
    "category": "chemistry",
    "domain": "Runtime tool-registration harness",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2605.20819"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Expert Python functions/source/plain-language tool descriptions and a workflow goal",
    "outputs": "Runtime-registered tools and a supervised multi-agent pipeline (persisted across sessions)",
    "autonomy": "A3",
    "notes": "Hierarchical agentic template that turns expert functions into AI-callable tools; LLM only routes, never writes scientific code.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2605.20819"
    ]
  },
  {
    "id": "asa-adsorb-agent",
    "date_added": "2026-06-16",
    "name": "Adsorb-Agent",
    "category": "chemistry",
    "domain": "Catalyst adsorption configuration search",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2410.16658"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/hoon-ock/CatalystAIgent"
      }
    ],
    "access": "open-source",
    "inputs": "Adsorbate and catalyst surface",
    "outputs": "Stable adsorption configurations and global-minimum adsorption-energy candidates",
    "autonomy": "A2-A3",
    "notes": "LLM agent that strategically enumerates adsorption configurations, reducing required DFT setups.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2410.16658",
      "https://github.com/hoon-ock/CatalystAIgent"
    ]
  },
  {
    "id": "asa-polymer-agent",
    "date_added": "2026-06-16",
    "name": "Polymer-Agent",
    "category": "chemistry",
    "domain": "Polymer design",
    "paper_links": [
      {
        "label": "ACS",
        "url": "https://pubs.acs.org/doi/10.1021/acs.jcim.6c00343"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/BaratiLab/Polymer-Agent"
      }
    ],
    "access": "open-source",
    "inputs": "Target polymer property/constraints in natural language",
    "outputs": "Candidate polymer SMILES with predicted properties; structure modifications",
    "autonomy": "A2-A3",
    "notes": "LLM MCP-server agent for closed-loop polymer property prediction and inverse design.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/BaratiLab/Polymer-Agent",
      "https://pubs.acs.org/doi/10.1021/acs.jcim.6c00343"
    ]
  },
  {
    "id": "asa-osda-agent",
    "date_added": "2026-06-16",
    "name": "OSDA Agent",
    "category": "chemistry",
    "domain": "Organic structure-directing-agent design",
    "paper_links": [
      {
        "label": "OpenReview",
        "url": "https://openreview.net/forum?id=9YNyiCJE3k"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Zeolite/OSDA design goal",
    "outputs": "Candidate OSDAs with evaluator scores and self-reflection refinements",
    "autonomy": "A3",
    "notes": "Actor/Evaluator/Self-reflector LLM framework for de novo organic structure-directing agent design.",
    "verified": "2026-06-16",
    "sources": [
      "https://openreview.net/forum?id=9YNyiCJE3k"
    ]
  },
  {
    "id": "asa-matpilot",
    "date_added": "2026-06-16",
    "name": "MatPilot",
    "category": "chemistry",
    "domain": "AI materials scientist",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2411.08063"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Materials objective and experimental constraints",
    "outputs": "Hypotheses, experimental schemes, predictive-model/optimization outputs",
    "autonomy": "A3-A5",
    "notes": "LLM-enabled AI materials scientist under human-machine collaboration with automated experimental platform.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2411.08063"
    ]
  },
  {
    "id": "asa-mapps",
    "date_added": "2026-06-16",
    "name": "MAPPS",
    "category": "chemistry",
    "domain": "Autonomous materials discovery framework",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2506.05616"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Materials task, physics constraints, high-level goals",
    "outputs": "Planned workflows, executable tool code, simulations, scientist-mediated outputs",
    "autonomy": "A3-A4",
    "notes": "Materials Agent unifying Planning, Physics, and Scientists; multi-agent autonomous materials discovery.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2506.05616"
    ]
  },
  {
    "id": "asa-chemhas-chemamp",
    "date_added": "2026-06-16",
    "name": "ChemHAS / ChemAmp",
    "category": "chemistry",
    "domain": "Chemistry tool amplification",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.21569"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Chemistry task and a set of available chemistry tools",
    "outputs": "Optimized hierarchical agent-tool stacks; improved chemistry-task answers",
    "autonomy": "A2-A3",
    "notes": "Hierarchical agent stacking to reduce chemistry-tool prediction errors across four chemistry tasks.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2505.21569"
    ]
  },
  {
    "id": "asa-eunomia",
    "date_added": "2026-06-16",
    "name": "Eunomia",
    "category": "chemistry",
    "domain": "Materials literature extraction",
    "paper_links": [
      {
        "label": "RSC",
        "url": "https://pubs.rsc.org/en/content/articlehtml/2024/dd/d4dd00252k"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/AI4ChemS/Eunomia"
      }
    ],
    "access": "open-source",
    "inputs": "Materials papers/text and an extraction schema",
    "outputs": "Structured materials datasets with chain-of-verification traces",
    "autonomy": "A2",
    "notes": "Chemist AI agent for extracting structured materials datasets from literature (MOF stability, doping, etc.).",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/AI4ChemS/Eunomia",
      "https://pubs.rsc.org/en/content/articlehtml/2024/dd/d4dd00252k"
    ]
  },
  {
    "id": "asa-organa",
    "date_added": "2026-06-16",
    "name": "ORGANA",
    "category": "chemistry",
    "domain": "Robotic chemistry assistant",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2401.06949"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ac-rad/organa"
      }
    ],
    "access": "lab-gated",
    "inputs": "Chemistry experiment objective in natural language",
    "outputs": "Robot plans/scheduling, visual feedback, pH/solubility/recrystallization/electrochemistry workflows and logs",
    "autonomy": "A3-A5",
    "notes": "Human-in-the-loop robotic chemistry assistant; also published in Cell Matter.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2401.06949",
      "https://github.com/ac-rad/organa"
    ]
  },
  {
    "id": "asa-chemreasoner",
    "date_added": "2026-06-16",
    "name": "ChemReasoner",
    "category": "chemistry",
    "domain": "Catalyst and chemistry reasoning",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2402.10980"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/pnnl/chemreasoner"
      }
    ],
    "access": "open-source",
    "inputs": "Catalyst-discovery or chemistry-reasoning task",
    "outputs": "Candidate catalysts with quantum-chemistry/GNN-grounded reasoning artifacts",
    "autonomy": "A2-A3",
    "notes": "Heuristic LLM search over chemical knowledge space with quantum-chemical (GNN) feedback for catalyst discovery.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2402.10980",
      "https://github.com/pnnl/chemreasoner"
    ]
  },
  {
    "id": "asa-cascade",
    "date_added": "2026-06-16",
    "name": "CASCADE",
    "category": "chemistry",
    "domain": "Self-evolving scientific skill acquisition for chemistry and materials research",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2512.23880"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Chemistry/materials task, public documentation and external tools",
    "outputs": "Accumulated executable skills, knowledge-graph memory and solved scientific tasks",
    "autonomy": "A4",
    "notes": "Self-evolving agentic framework evaluated on SciSkillBench. No separate public CASCADE implementation was confirmed; the open Figshare artifact belongs to the benchmark record.",
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2512.23880"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "asa-chemspace-copilot",
    "date_added": "2026-06-16",
    "name": "ChemSpace Copilot",
    "category": "chemistry",
    "domain": "Chemical-space exploration app",
    "paper_links": [
      {
        "label": "ChemRxiv",
        "url": "https://chemrxiv.org/doi/full/10.26434/chemrxiv.15000527/v1"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Chemical-space dataset/query (e.g. ChEMBL set)",
    "outputs": "GTM chemography workflows, interactive 2D maps, compound selections/reports",
    "autonomy": "A2",
    "notes": "Agentic chatbot for chemical-space exploration via Generative Topographic Mapping over ChEMBL.",
    "verified": "2026-07-13",
    "sources": [
      "https://chemrxiv.org/doi/full/10.26434/chemrxiv.15000527/v1"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "asa-mars-materials",
    "date_added": "2026-06-16",
    "name": "MARS materials",
    "category": "chemistry",
    "domain": "Robotic materials system",
    "paper_links": [
      {
        "label": "Link",
        "url": "http://www.cityu.edu.hk/phy/appkchu/Publications/2026/26.23.pdf"
      }
    ],
    "repo_links": [],
    "access": "lab-gated",
    "inputs": "Materials research objective",
    "outputs": "Closed-loop robot experiments, optimized candidates (e.g. perovskite nanocrystals), reports",
    "autonomy": "A5",
    "notes": "Knowledge-driven autonomous materials robot: 19 LLM agents + 16 tools; published in Matter (2026).",
    "verified": "2026-06-16",
    "sources": [
      "http://www.cityu.edu.hk/phy/appkchu/Publications/2026/26.23.pdf"
    ]
  },
  {
    "id": "asa-nanominer",
    "date_added": "2026-06-16",
    "name": "nanoMINER",
    "category": "chemistry",
    "domain": "Nanomaterials literature extraction",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41524-025-01674-7"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ai-chem/nanoMINER"
      }
    ],
    "access": "open-source",
    "inputs": "Nanomaterials papers (text, figures, tables)",
    "outputs": "Structured extracted nanomaterial/nanozyme property data",
    "autonomy": "A2",
    "notes": "Multi-agent multimodal (YOLO + GPT-4o, ReAct) information extraction for nanomaterials datasets.",
    "verified": "2026-07-13",
    "sources": [
      "https://github.com/ai-chem/nanoMINER",
      "https://www.nature.com/articles/s41524-025-01674-7"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "asa-dive",
    "date_added": "2026-06-16",
    "name": "DIVE",
    "category": "chemistry",
    "domain": "Hydrogen-storage literature mining",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2508.13251"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Figures/tables from materials literature",
    "outputs": "Structured experimental data for hydrogen-storage materials design",
    "autonomy": "A2",
    "notes": "Multi-agent workflow reading data from figures/tables; ~30k figures from 4k+ papers on solid-state H2 storage.",
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2508.13251"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "asa-larc",
    "date_added": "2026-06-16",
    "name": "LARC",
    "category": "chemistry",
    "domain": "Constrained retrosynthesis planner",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2508.11860"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ninglab/LARC"
      }
    ],
    "access": "open-source",
    "inputs": "Target molecule plus practical synthesis constraints",
    "outputs": "Synthetic routes satisfying the specified constraints",
    "autonomy": "A2-A3",
    "notes": "LLM agentic constrained retrosynthesis with Agent-as-a-Judge constraint evaluation.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2508.11860",
      "https://github.com/ninglab/LARC"
    ]
  },
  {
    "id": "asa-crystalplasticitysim",
    "date_added": "2026-06-16",
    "name": "CrystalPlasticitySim",
    "category": "chemistry",
    "domain": "Crystal-plasticity simulation agent",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://www.tandfonline.com/doi/full/10.1080/27660400.2026.2630445"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ForeverYoungJay/CrystalPlasticitySim"
      }
    ],
    "access": "open-source",
    "inputs": "Natural-language crystal-plasticity task; target experimental data, material configs, load/boundary conditions",
    "outputs": "DAMASK YAML/input setup, simulation execution, post-processed results, iterative parameter/boundary-condition optimization, plots/CSV logs",
    "autonomy": "A3",
    "notes": "LLM multi-agent framework (Supervisor, Simulation, Code agents on LangGraph) automating DAMASK 3.0 crystal-plasticity simulations.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/ForeverYoungJay/CrystalPlasticitySim",
      "https://www.tandfonline.com/doi/full/10.1080/27660400.2026.2630445"
    ]
  },
  {
    "id": "asa-cmbagent-cmbagent",
    "date_added": "2026-06-16",
    "name": "cmbagent / CMBAgent",
    "category": "physics",
    "domain": "Cosmology workflow agent",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2507.07257"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/CMBAgents/cmbagent"
      }
    ],
    "access": "open-source",
    "inputs": "Research task, papers/codebases, datasets",
    "outputs": "Plans, code, plots, cosmology/scientific analysis",
    "autonomy": "A3-A4",
    "notes": "Generalist planning-and-control multi-agent system (~30 LLM agents) with astrophysics origins; applied to PhD-level cosmology tasks.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2507.07257",
      "https://github.com/CMBAgents/cmbagent"
    ]
  },
  {
    "id": "asa-multi-agent-system-for-cosmological-parameter-analysis",
    "date_added": "2026-06-16",
    "name": "Multi-Agent System for Cosmological Parameter Analysis",
    "category": "physics",
    "domain": "ACT likelihood/MCMC workflows",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2412.00431"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/CMBAgents/cmbagent"
      }
    ],
    "access": "open-source",
    "inputs": "Cosmology likelihood task, docs, local code",
    "outputs": "MCMC setup/results, figures, analysis traces",
    "autonomy": "A3",
    "notes": "Early cosmology-specific multi-agent system (Laverick, Surrao et al.); ACT lensing power spectrum via MCMC; predecessor lineage to cmbagent.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2412.00431",
      "https://github.com/CMBAgents/cmbagent"
    ]
  },
  {
    "id": "asa-ai-cosmologist",
    "date_added": "2026-06-16",
    "name": "AI Cosmologist",
    "category": "physics",
    "domain": "Astronomy/cosmology data-analysis and ML research",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.03424"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/adammoss/aicosmologist"
      }
    ],
    "access": "open-data",
    "inputs": "Dataset and task description",
    "outputs": "Code, experiments, analysis, draft papers",
    "autonomy": "A4",
    "notes": "End-to-end agentic system for astronomy/cosmology data analysis and ML research (Adam Moss); repo currently shares only config files and example outputs, full code not yet released.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2504.03424",
      "https://github.com/adammoss/aicosmologist"
    ]
  },
  {
    "id": "asa-aster",
    "date_added": "2026-06-16",
    "name": "ASTER",
    "category": "physics",
    "domain": "Exoplanet research toolkit",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.26953"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Exoplanet target/data task (e.g. archive query, spectrum)",
    "outputs": "Tool workflows, archive retrievals, radiative-transfer and Bayesian retrieval results",
    "autonomy": "A3",
    "notes": "ASTER: Agentic Science Toolkit for Exoplanet Research; LLM orchestration over NASA Exoplanet Archive, TauREx, retrieval (demoed on WASP-39b).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2603.26953"
    ]
  },
  {
    "id": "asa-astroagent",
    "date_added": "2026-06-16",
    "name": "AstroAgent",
    "category": "physics",
    "domain": "Astronomy idea refinement",
    "paper_links": [],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/SandyYuan/astro-agent"
      }
    ],
    "access": "open-source",
    "inputs": "Astronomy research idea or interests/constraints",
    "outputs": "Generated/refined research ideas, literature (Semantic Scholar) checks, peer-review-style feedback",
    "autonomy": "A2-A3",
    "notes": "Three-agent pipeline (Idea, Literature, Reflection) for astronomy idea generation/refinement; Streamlit app plus MCP server.",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/SandyYuan/astro-agent"
    ]
  },
  {
    "id": "asa-camels-agents",
    "date_added": "2026-06-16",
    "name": "CAMELS Agents",
    "category": "physics",
    "domain": "Cosmology simulation-data assistants",
    "paper_links": [
      {
        "label": "CLAPP arXiv:2508.05728",
        "url": "https://arxiv.org/abs/2508.05728"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/franciscovillaescusa/CAMELS_Agents"
      }
    ],
    "access": "open-source",
    "inputs": "CAMELS docs/data/paper queries; coding and paper-writing requests",
    "outputs": "CAMELS coding help, paper search over CAMELS-papers database, methodology section drafts",
    "autonomy": "A2",
    "notes": "Narrow support agents for CAMELS (Cosmology and Astrophysics with ML Simulations) users; Streamlit interface.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2508.05728",
      "https://github.com/franciscovillaescusa/CAMELS_Agents"
    ]
  },
  {
    "id": "asa-clapp",
    "date_added": "2026-06-16",
    "name": "CLAPP",
    "category": "physics",
    "domain": "CLASS Boltzmann-solver coding agent",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2508.05728"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/santiagocasas/clapp"
      }
    ],
    "access": "open-source",
    "inputs": "CLASS/cosmology coding query",
    "outputs": "Conversational answers, generated code, plots, debugged Python scripts",
    "autonomy": "A2-A3",
    "notes": "CLAPP: The CLASS LLM Agent for Pair Programming; RAG over CLASS docs with Python execution, Streamlit app.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2508.05728",
      "https://github.com/santiagocasas/clapp"
    ]
  },
  {
    "id": "asa-gammapy-agent",
    "date_added": "2026-06-16",
    "name": "Gammapy Agent",
    "category": "physics",
    "domain": "Gamma-ray astronomy code generation",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2503.00821"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Gammapy analysis prompt and data location",
    "outputs": "Executable/self-repaired Gammapy analysis scripts, plots",
    "autonomy": "A2-A3",
    "notes": "Agent-based code generation for the Gammapy gamma-ray astronomy framework (CTAO context); writes, executes, validates code.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2503.00821"
    ]
  },
  {
    "id": "asa-grace",
    "date_added": "2026-06-16",
    "name": "GRACE",
    "category": "physics",
    "domain": "Simulation-native particle and nuclear physics experiment design",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2602.15039"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Prompt or paper describing a particle/nuclear physics experiment",
    "outputs": "Structured experiment representation, runnable toy Monte Carlo simulations, Geant4 escalation, detector design modifications",
    "autonomy": "A4",
    "notes": "Simulation-native autonomous design agent. The paper reports an evaluation suite but no stable separately named public benchmark artifact, so no speculative B record is created.",
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2602.15039"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "asa-heptapod",
    "date_added": "2026-06-16",
    "name": "HEPTAPOD",
    "category": "physics",
    "domain": "HEP workflow orchestration",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2512.15867"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/tonymenzo/heptapod"
      }
    ],
    "access": "open-source",
    "inputs": "HEP workflow/task specs in natural language",
    "outputs": "Run cards, tool calls (FeynCalc, MadGraph, Pythia, Sherpa), reproducible workflow artifacts and execution traces",
    "autonomy": "A3",
    "notes": "HEPTAPOD: Orchestrating High Energy Physics Workflows Towards Autonomous Agency; MCP-based toolkit (FERMILAB-PUB-25-0923).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2512.15867",
      "https://github.com/tonymenzo/heptapod"
    ]
  },
  {
    "id": "asa-rooagent",
    "date_added": "2026-06-16",
    "name": "RooAgent",
    "category": "physics",
    "domain": "ROOT-based HEP data-analysis agent",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2605.17318"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/amanmdesai/RooAgent"
      }
    ],
    "access": "open-source",
    "inputs": "Natural-language ROOT/PyROOT analysis prompt",
    "outputs": "ROOT tool calls, histograms, event selection, fits, significance scans, plots",
    "autonomy": "A2-A3",
    "notes": "RooAgent: LLM agent for ROOT-based HEP analysis; LangGraph/MCP/Ollama modes (author Aman Desai).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2605.17318",
      "https://github.com/amanmdesai/RooAgent"
    ]
  },
  {
    "id": "asa-jfc",
    "date_added": "2026-06-16",
    "name": "JFC",
    "category": "physics",
    "domain": "Autonomous experimental HEP analysis",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.20179"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "HEP dataset (e.g. ALEPH/DELPHI/CMS open data), analysis framework, literature corpus",
    "outputs": "Event selection, background estimation, uncertainty quantification, statistical inference, draft paper",
    "autonomy": "A4",
    "notes": "Proof-of-concept that LLM agents (Claude Code) can autonomously run substantial portions of a collider-data HEP analysis pipeline.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2603.20179"
    ]
  },
  {
    "id": "asa-fermiacc",
    "date_added": "2026-06-16",
    "name": "FERMIACC",
    "category": "physics",
    "domain": "Particle-theory hypothesis agent",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.22538"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "HEP theory/data prompt",
    "outputs": "Autonomously generated and quantitatively validated theory hypotheses for HEP data",
    "autonomy": "A3-A4",
    "notes": "The FERMIACC: Agents for Particle Theory; scaffolded reasoning model on OpenAI agents for hypothesis generation/validation.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2603.22538"
    ]
  },
  {
    "id": "asa-darkagents",
    "date_added": "2026-06-16",
    "name": "DarkAgents",
    "category": "physics",
    "domain": "Astroparticle theory pipeline",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2606.11157"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/PhysicsZandi/DarkAgents"
      }
    ],
    "access": "open-source",
    "inputs": "Dark-sector/cosmology model task (e.g. first-order phase transitions, GW/NANOGrav signatures)",
    "outputs": "Orchestrated pipelines: model building, best-fit parameters, experimental constraints, assumption audit report",
    "autonomy": "A3-A4",
    "notes": "DarkAgents: multi-agent LLM system for theoretical astroparticle physics (FOPT-PTA pipeline); Univ. Bologna/INFN, GPL-3.0.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2606.11157",
      "https://github.com/PhysicsZandi/DarkAgents"
    ]
  },
  {
    "id": "asa-gwagent",
    "date_added": "2026-06-16",
    "name": "GWAgent",
    "category": "physics",
    "domain": "Gravitational-wave surrogate discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2605.11280"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/tousifislam/GWAgent"
      }
    ],
    "access": "open-source",
    "inputs": "Gravitational-wave simulation data / modeling objective",
    "outputs": "Interpretable analytical surrogate/symbolic models with validation (LIGO mismatch metrics)",
    "autonomy": "A3",
    "notes": "Discovery of Interpretable Surrogates via Agentic AI; demoed on eccentric BBH waveforms and GW200129 (author Tousif Islam).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2605.11280",
      "https://github.com/tousifislam/GWAgent"
    ]
  },
  {
    "id": "asa-mitra",
    "date_added": "2026-06-16",
    "name": "MITRA",
    "category": "physics",
    "domain": "Physics-collaboration RAG assistant",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.09800"
      }
    ],
    "repo_links": [],
    "access": "platform",
    "inputs": "Internal CMS-style collaboration documents and natural-language queries",
    "outputs": "Context-aware retrieved answers grounded in internal docs (RAG)",
    "autonomy": "A1-A2",
    "notes": "MITRA: AI Assistant for Knowledge Retrieval in Physics Collaborations; on-premise RAG (Selenium+OCR, two-tier vector DB) for CMS/CERN.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2603.09800"
    ]
  },
  {
    "id": "asa-sr-scientist",
    "date_added": "2026-06-16",
    "name": "SR-Scientist",
    "category": "physics",
    "domain": "Scientific equation discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2510.11661"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/GAIR-NLP/SR-Scientist"
      }
    ],
    "access": "open-source",
    "inputs": "Equation-discovery problem, observational data, code tools",
    "outputs": "Symbolic expressions implemented/evaluated as code, validation traces, optimized equations",
    "autonomy": "A3",
    "notes": "SR-Scientist: Scientific Equation Discovery With Agentic AI; LLM writes/runs code to fit and optimize equations across disciplines.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2510.11661",
      "https://github.com/GAIR-NLP/SR-Scientist"
    ]
  },
  {
    "id": "asa-accelerator-assistant",
    "date_added": "2026-06-16",
    "name": "Accelerator Assistant",
    "category": "physics",
    "domain": "Multi-stage physics-experiment operations",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2509.17255"
      }
    ],
    "repo_links": [],
    "access": "lab-gated",
    "inputs": "Natural-language accelerator/beamline/machine-physics task",
    "outputs": "Execution plans, archive retrieval, control-channel resolution, generated scripts, safe machine actions, analysis",
    "autonomy": "A3-A5",
    "notes": "Agentic AI for multi-stage physics experiments at the Advanced Light Source synchrotron; safety-gated, published in Phys. Rev. Research (2026).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2509.17255"
    ]
  },
  {
    "id": "asa-autonomous-quantum-simulation-llm-agents",
    "date_added": "2026-06-16",
    "name": "Autonomous Quantum Simulation LLM Agents",
    "category": "physics",
    "domain": "Tensor-network quantum simulation",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2601.10194"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Quantum many-body simulation task and method documentation",
    "outputs": "Code and tensor-network simulation results (quantum phase transitions, open-system dynamics, photochemistry)",
    "autonomy": "A3",
    "notes": "Autonomous Quantum Simulation through Large Language Model Agents; multi-agent tensor-network simulations at ~90% accuracy.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2601.10194"
    ]
  },
  {
    "id": "asa-qagent",
    "date_added": "2026-06-16",
    "name": "QAgent",
    "category": "physics",
    "domain": "OpenQASM programming system",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2508.20134"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Quantum-programming task in natural language",
    "outputs": "OpenQASM code with improved compilation/correctness (RAG + few-shot + tools)",
    "autonomy": "A2-A3",
    "notes": "QAgent: LLM-based Multi-Agent System for Autonomous OpenQASM Programming; ~71.6% accuracy improvement over static LLM baselines.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2508.20134"
    ]
  },
  {
    "id": "asa-qcopilot",
    "date_added": "2026-06-16",
    "name": "QCopilot",
    "category": "physics",
    "domain": "Quantum sensor design/diagnosis",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2508.05421"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Quantum sensor design/diagnosis task and experiment parameters",
    "outputs": "Optimization choices, anomaly diagnosis, experiment settings",
    "autonomy": "A4-A5",
    "notes": "LLM-based multi-agent copilot for quantum sensor design and diagnosis; ~100x speedup on atom cooling.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2508.05421"
    ]
  },
  {
    "id": "asa-ai-agents-for-variational-quantum-circuit-design",
    "date_added": "2026-06-16",
    "name": "AI Agents for Variational Quantum Circuit Design",
    "category": "physics",
    "domain": "VQC architecture design",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2602.19387"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Variational quantum circuit design objective",
    "outputs": "Candidate VQC architectures and evaluations",
    "autonomy": "A3",
    "notes": "LLM/AI agents that design and evaluate variational quantum circuits.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2602.19387"
    ]
  },
  {
    "id": "asa-jutulgpt",
    "date_added": "2026-06-16",
    "name": "JutulGPT",
    "category": "physics",
    "domain": "Reservoir simulation",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.00214"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/SINTEF-agentlab/JutulGPT"
      }
    ],
    "access": "open-source",
    "inputs": "Natural-language reservoir model description",
    "outputs": "JutulDarcy models, validated simulation runs, solver diagnostics/logs",
    "autonomy": "A3",
    "notes": "Execution-grounded reservoir-simulation model construction on the Julia JutulDarcy simulator.",
    "verified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2603.00214",
      "https://github.com/SINTEF-agentlab/JutulGPT"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "asa-mcp-sim",
    "date_added": "2026-06-16",
    "name": "MCP-SIM",
    "category": "physics",
    "domain": "Self-correcting physics simulation",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s44387-025-00057-z"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/KAIST-M4/MCP-SIM"
      }
    ],
    "access": "open-source",
    "inputs": "Natural-language physics/FEA prompt",
    "outputs": "Generated code, executed simulation, localized explanatory report",
    "autonomy": "A3",
    "notes": "Self-correcting multi-agent LLM framework for language-based physics simulation and explanation (KAIST).",
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/KAIST-M4/MCP-SIM",
      "https://www.nature.com/articles/s44387-025-00057-z"
    ]
  },
  {
    "id": "asa-flamepilot",
    "date_added": "2026-06-16",
    "name": "FlamePilot",
    "category": "physics",
    "domain": "Combustion CFD agent",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2601.01357"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Research paper or combustion CFD task",
    "outputs": "OpenFOAM/DeepFlame case setup, results, evidence-based refinements",
    "autonomy": "A3-A4",
    "notes": "Literature-aware, self-corrective LLM agent for combustion CFD workflows.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2601.01357"
    ]
  },
  {
    "id": "asa-autosurrogate",
    "date_added": "2026-06-16",
    "name": "AutoSurrogate",
    "category": "physics",
    "domain": "Subsurface-flow surrogate modeling",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.11945"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Simulation data and user preferences (natural language)",
    "outputs": "Trained deep-learning surrogate, QA assessment, optimized hyperparameters",
    "autonomy": "A3",
    "notes": "LLM-driven multi-agent framework for autonomous deep-learning surrogate construction in subsurface flow.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2604.11945"
    ]
  },
  {
    "id": "asa-phia-lp-comda",
    "date_added": "2026-06-16",
    "name": "PHIA / LP-COMDA",
    "category": "physics",
    "domain": "Power-electronics design",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2411.14214"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Power-converter modulation design requirements (conversational)",
    "outputs": "Modulation parameters, performance metrics, design charts",
    "autonomy": "A3",
    "notes": "Physics-informed autonomous LLM agent for power-electronics modulation design; named LP-COMDA (v1) / PHIA (AAAI 2026).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2411.14214"
    ]
  },
  {
    "id": "asa-gas-turbine-domain-specific-react",
    "date_added": "2026-06-16",
    "name": "Gas-turbine Domain-specific ReAct",
    "category": "physics",
    "domain": "Power engineering diagnostics",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2406.07572"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Gas path analysis task and gas-turbine data",
    "outputs": "Tool-mediated diagnostics and physics-integrated modeling outputs",
    "autonomy": "A2-A3",
    "notes": "Dual-agent domain-specific ReAct/tool workflow for gas-turbine gas path analysis.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2406.07572"
    ]
  },
  {
    "id": "asa-llm-agent-for-chemical-process-simulations",
    "date_added": "2026-06-16",
    "name": "LLM Agent for Chemical Process Simulations",
    "category": "physics",
    "domain": "Process simulation",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2601.11650"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Natural-language process-simulation prompts (AVEVA Process Simulation via MCP)",
    "outputs": "Flowsheet analysis, optimization, data extraction, autonomous flowsheet synthesis",
    "autonomy": "A3",
    "notes": "User-friendly LLM agent driving AVEVA Process Simulation through Model Context Protocol.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2601.11650"
    ]
  },
  {
    "id": "asa-llm-guided-chemical-process-optimization",
    "date_added": "2026-06-16",
    "name": "LLM-guided Chemical Process Optimization",
    "category": "physics",
    "domain": "Process optimization",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2506.20921"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Minimal natural-language process description",
    "outputs": "Inferred operating constraints and optimized process settings",
    "autonomy": "A3",
    "notes": "Multi-agent (AutoGen/o3) approach that infers operating constraints and optimizes chemical processes.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2506.20921"
    ]
  },
  {
    "id": "asa-llm-mechanical-designer",
    "date_added": "2026-06-16",
    "name": "LLM Mechanical Designer",
    "category": "physics",
    "domain": "Structural/mechanical design",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2404.17525"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Performance specifications for a truss/structure",
    "outputs": "Candidate structural designs with iterative FEM evaluations",
    "autonomy": "A3",
    "notes": "LLM that autonomously generates/refines 2D truss designs in an FEM feedback loop (outperforms NSGA-II).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2404.17525"
    ]
  },
  {
    "id": "asa-mechatronics-design-framework",
    "date_added": "2026-06-16",
    "name": "Mechatronics Design Framework",
    "category": "physics",
    "domain": "Autonomous mechatronics design",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.14681"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Mechatronics design challenge and constraints",
    "outputs": "Mechanical, electronics, control, and software designs for functional prototypes",
    "autonomy": "A3-A4",
    "notes": "LLM-enabled multi-agent framework for autonomous mechatronics design (e.g. water-quality monitoring vessel).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2504.14681"
    ]
  },
  {
    "id": "asa-tem-agent",
    "date_added": "2026-06-16",
    "name": "TEM Agent",
    "category": "physics",
    "domain": "Electron microscopy control",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2511.08819"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/foundry-mcp/team05-mcp-server"
      }
    ],
    "access": "open-source",
    "inputs": "Text-based TEM operation prompts plus microscope/data/HPC APIs",
    "outputs": "Microscope subsystem actions, acquisition workflows, metadata-driven summaries",
    "autonomy": "A3-A5",
    "notes": "MCP-based agent controlling a TEAM 0.5 transmission electron microscope and related subsystems.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2511.08819",
      "https://github.com/foundry-mcp/team05-mcp-server"
    ]
  },
  {
    "id": "asa-scilink",
    "date_added": "2026-06-16",
    "name": "SciLink",
    "category": "physics",
    "domain": "Materials characterization",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2508.06569"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ziatdinovmax/SciLink"
      }
    ],
    "access": "open-source",
    "inputs": "Microscopy/hyperspectral data, literature, expert guidance",
    "outputs": "Testable scientific claims, novelty scores, DFT/follow-up experiment proposals",
    "autonomy": "A3-A4",
    "notes": "Theory-in-the-loop multi-agent framework linking materials characterization to literature and DFT.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2508.06569",
      "https://github.com/ziatdinovmax/SciLink"
    ]
  },
  {
    "id": "asa-automat-stem2mat-bench",
    "date_added": "2026-06-16",
    "name": "AutoMat",
    "category": "physics",
    "domain": "Agent-assisted microscopy-to-atomistic structure reconstruction",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.12650"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/yyt-2378/AutoMat"
      }
    ],
    "access": "open-source",
    "inputs": "Scanning transmission electron microscopy images",
    "outputs": "Reconstructed crystal structures, relaxed structures and predicted physical properties",
    "autonomy": "A3",
    "notes": "Agent-assisted pipeline coordinating denoising, template retrieval, atomic reconstruction, relaxation and property prediction. The accompanying STEM2Mat-Bench is represented separately.",
    "verified": "2026-07-13",
    "aliases": [
      "AutoMat / STEM2Mat-Bench"
    ],
    "sources": [
      "https://arxiv.org/abs/2505.12650",
      "https://github.com/yyt-2378/AutoMat"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "asa-astabench",
    "date_added": "2026-06-16",
    "name": "AstaBench",
    "category": "benchmark",
    "domain": "Scientific agents",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2510.21652"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/allenai/asta-bench"
      }
    ],
    "access": "open-source",
    "inputs": "Scientific-research problems spanning the discovery process, with controlled search tools",
    "outputs": "Agent solutions and standardized scores across the suite of 2400+ problems",
    "autonomy": "B",
    "notes": "Rigorous scientific-research benchmark suite with production-grade environment and baseline agent classes.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2510.21652",
      "https://github.com/allenai/asta-bench"
    ]
  },
  {
    "id": "asa-airs-bench",
    "date_added": "2026-06-16",
    "name": "AIRS-Bench",
    "category": "benchmark",
    "domain": "AI research agents",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2602.06855"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/facebookresearch/airs-bench"
      }
    ],
    "access": "open-source",
    "inputs": "Full-lifecycle AI-research task from recent ML publications (20 tasks, 7 domains)",
    "outputs": "Agent research artifacts scored against human-SOTA and theoretical-optimal targets",
    "autonomy": "B",
    "notes": "Frontier AI-research-science benchmark spanning ideation, implementation, analysis, and refinement.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2602.06855",
      "https://github.com/facebookresearch/airs-bench"
    ]
  },
  {
    "id": "asa-researchgym",
    "date_added": "2026-06-16",
    "name": "ResearchGym",
    "category": "benchmark",
    "domain": "End-to-end research benchmark",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2602.15112"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Anikethh/ResearchGym"
      }
    ],
    "access": "open-source",
    "inputs": "Containerized research environment from real ICML/ICLR/ACL papers (method withheld)",
    "outputs": "Agent hypotheses, experiments, and metric improvements over human baselines; sub-task scores",
    "autonomy": "B",
    "notes": "End-to-end AI-research benchmark of 5 containerized environments with 39 sub-tasks (ICLR 2026).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2602.15112",
      "https://github.com/Anikethh/ResearchGym"
    ]
  },
  {
    "id": "asa-paperbench",
    "date_added": "2026-06-16",
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "name": "PaperBench",
    "category": "benchmark",
    "domain": "AI-paper replication benchmark",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.01848"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/openai/frontier-evals/tree/main/project/paperbench"
      }
    ],
    "access": "open-source",
    "inputs": "ICML 2024 paper to replicate from scratch (paper text, no author code)",
    "outputs": "Replicated codebase and experiments graded against 8000+ rubric subtasks",
    "autonomy": "B",
    "notes": "Benchmark evaluating agents' ability to replicate 20 ICML 2024 Spotlight/Oral papers.",
    "sources": [
      "https://arxiv.org/abs/2504.01848",
      "https://github.com/openai/frontier-evals/tree/main/project/paperbench"
    ]
  },
  {
    "id": "asa-bioagent-bench",
    "date_added": "2026-06-16",
    "name": "BioAgent Bench",
    "category": "benchmark",
    "domain": "Bioinformatics agents",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2601.21800"
      }
    ],
    "repo_links": [],
    "access": "open-data",
    "inputs": "Bioinformatics task prompts (RNA-seq, variant calling, metagenomics) with specified output artifacts",
    "outputs": "Pipeline-progress and outcome-validity scores, robustness/perturbation results",
    "autonomy": "B",
    "notes": "Evaluation suite for AI agents on end-to-end bioinformatics workflows; LLM-graded.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2601.21800"
    ]
  },
  {
    "id": "asa-single-cell-omics-agent-benchmark",
    "date_added": "2026-06-16",
    "name": "Single-cell omics agent benchmark",
    "category": "benchmark",
    "domain": "Single-cell omics agents",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2508.13201"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/lyyang01/bioagent-benchmark"
      }
    ],
    "access": "open-source",
    "inputs": "Single-cell omics tasks across multi-omics, species, sequencing technologies; agent frameworks/LLMs",
    "outputs": "Multidimensional capability metrics (code synthesis, collaboration, efficiency, knowledge, completion)",
    "autonomy": "B",
    "notes": "Open benchmark of 50 single-cell omics tasks; also published in Genome Biology.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2508.13201",
      "https://github.com/lyyang01/bioagent-benchmark"
    ]
  },
  {
    "id": "asa-lab-bench",
    "date_added": "2026-06-16",
    "name": "LAB-Bench",
    "category": "benchmark",
    "domain": "Biology research language agents",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2407.10362"
      },
      {
        "label": "arXiv (LAB-Bench2)",
        "url": "https://arxiv.org/abs/2604.09554"
      }
    ],
    "repo_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/datasets/futurehouse/lab-bench"
      },
      {
        "label": "GitHub (labbench2)",
        "url": "https://github.com/EdisonScientific/labbench2"
      }
    ],
    "access": "open-data",
    "inputs": "Practical biology-research questions (literature, figures, database navigation, sequences)",
    "outputs": "Multiple-choice accuracy across biology-research capability tasks",
    "autonomy": "B",
    "notes": "FutureHouse benchmark of 2,400+ biology research questions (public HF subset); LAB-Bench2 (2026) expands to ~1,900 authentic, non-multiple-choice biology research tasks.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2407.10362",
      "https://arxiv.org/abs/2604.09554",
      "https://github.com/EdisonScientific/labbench2",
      "https://huggingface.co/datasets/futurehouse/lab-bench"
    ]
  },
  {
    "id": "asa-agentclinic",
    "date_added": "2026-06-16",
    "name": "AgentClinic",
    "category": "benchmark",
    "domain": "Multimodal clinical agents",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2405.07960"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/SamuelSchmidgall/AgentClinic"
      }
    ],
    "access": "open-source",
    "inputs": "Simulated clinical cases; multimodal data under incomplete information; tools",
    "outputs": "Diagnostic accuracy and tool-use behavior in simulated clinical environments",
    "autonomy": "B",
    "notes": "Multimodal agent benchmark for AI in simulated clinical environments.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2405.07960",
      "https://github.com/SamuelSchmidgall/AgentClinic"
    ]
  },
  {
    "id": "asa-agentrx",
    "date_added": "2026-06-16",
    "name": "AgentRx",
    "category": "benchmark",
    "domain": "Clinical prediction agents",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2605.10286"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Multimodal clinical data: EHR, radiology reports, chest X-rays, clinical notes",
    "outputs": "Clinical prediction accuracy/calibration across unimodal and multimodal settings",
    "autonomy": "B",
    "notes": "Benchmark of LLM agents for multimodal clinical prediction (CHIL 2026).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2605.10286"
    ]
  },
  {
    "id": "asa-rwe-bench",
    "date_added": "2026-06-16",
    "name": "RWE-bench",
    "category": "benchmark",
    "domain": "Real-world evidence agents",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.22767"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "Observational-study protocols grounded in MIMIC-IV; medical-database access",
    "outputs": "Tree-structured evidence bundles; question-level and end-to-end task metrics (162 tasks)",
    "autonomy": "B",
    "notes": "Benchmark testing whether LLM agents can execute real-world-evidence observational studies.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2603.22767"
    ]
  },
  {
    "id": "asa-chime",
    "date_added": "2026-06-16",
    "name": "CHIME",
    "category": "benchmark",
    "domain": "Biomedical literature organization",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2407.16148"
      }
    ],
    "repo_links": [],
    "access": "open-data",
    "inputs": "Collections of scientific/biomedical studies and a topic",
    "outputs": "Hierarchical category tree with study assignments; corrector improvements",
    "autonomy": "B",
    "notes": "ACL 2024 Findings benchmark for LLM-assisted hierarchical literature organization.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2407.16148"
    ]
  },
  {
    "id": "asa-do-challenge",
    "date_added": "2026-06-16",
    "name": "DO Challenge",
    "category": "benchmark",
    "domain": "Drug-discovery agents",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://deeporigin.com/pdfs/papers/ai-agents-drug-discovery.pdf"
      }
    ],
    "repo_links": [
      {
        "label": "Zenodo",
        "url": "https://zenodo.org/records/15296510"
      }
    ],
    "access": "open-data",
    "inputs": "Library of 1M molecular conformations (SDF); label-request and submission budget",
    "outputs": "Top-candidate selections; agent decision-making/resource-allocation scores",
    "autonomy": "B",
    "notes": "DeepOrigin benchmark for autonomous drug-discovery agents (virtual screening).",
    "verified": "2026-06-16",
    "sources": [
      "https://deeporigin.com/pdfs/papers/ai-agents-drug-discovery.pdf",
      "https://zenodo.org/records/15296510"
    ]
  },
  {
    "id": "asa-collider-bench",
    "date_added": "2026-06-16",
    "name": "Collider-Bench",
    "category": "benchmark",
    "domain": "HEP analysis agents",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2605.13950"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "LHC search papers, task specs, containerized HEP software environment",
    "outputs": "Executable simulation/analysis pipelines; predicted signal yields vs published targets",
    "autonomy": "B",
    "notes": "Benchmark for autonomous reproduction of LHC particle-physics analyses.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2605.13950"
    ]
  },
  {
    "id": "asa-celloai-benchmarks",
    "date_added": "2026-06-16",
    "name": "CelloAI Benchmarks",
    "category": "benchmark",
    "domain": "HEP assistants",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.01051"
      }
    ],
    "repo_links": [],
    "access": "paper-only",
    "inputs": "HEP/HPC coding tasks: documentation, GPU-kernel code generation, graphical data analysis",
    "outputs": "Repeatable automated scores across three evaluation tracks",
    "autonomy": "B",
    "notes": "Benchmark for LLM assistants on HEP/HPC software-development tasks (BNL et al.).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2603.01051"
    ]
  },
  {
    "id": "asa-discoverphysics",
    "date_added": "2026-06-16",
    "name": "DiscoverPhysics",
    "category": "benchmark",
    "domain": "Physics-law discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2605.26087"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/SampsonML/DiscoverPhysics"
      }
    ],
    "access": "open-source",
    "inputs": "Interactive simulated worlds with non-standard physics; rounds of experiments",
    "outputs": "Natural-language law explanation plus Python implementation; discovery scores",
    "autonomy": "B",
    "notes": "Benchmark of out-of-the-box physics-law discovery in simulated worlds; has leaderboard.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2605.26087",
      "https://github.com/SampsonML/DiscoverPhysics"
    ]
  },
  {
    "id": "asa-physgym",
    "date_added": "2026-06-16",
    "name": "PhysGym",
    "category": "benchmark",
    "domain": "Physics discovery simulation",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2507.15550"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/principia-ai/PhysGym"
      }
    ],
    "access": "open-source",
    "inputs": "Interactive physics simulations with controlled prior-knowledge levels; experiment budget",
    "outputs": "Hypotheses about physical laws; model-fidelity/discovery metrics",
    "autonomy": "B",
    "notes": "NeurIPS 2025 D&B benchmark for interactive physics discovery with controlled priors.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2507.15550",
      "https://github.com/principia-ai/PhysGym"
    ]
  },
  {
    "id": "asa-newtonbench",
    "date_added": "2026-06-16",
    "name": "NewtonBench",
    "category": "benchmark",
    "domain": "Counterfactual physics-law discovery",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2510.07172"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/HKUST-KnowComp/NewtonBench"
      }
    ],
    "access": "open-source",
    "inputs": "324 law-discovery tasks across 12 physics domains with counterfactual law shifts; interactive experiments",
    "outputs": "Discovery accuracy and noise-robustness scores in altered-law tasks",
    "autonomy": "B",
    "notes": "Benchmark for generalizable scientific-law discovery via counterfactual shifts.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2510.07172",
      "https://github.com/HKUST-KnowComp/NewtonBench"
    ]
  },
  {
    "id": "asa-scienceboard",
    "date_added": "2026-06-16",
    "name": "ScienceBoard",
    "category": "benchmark",
    "domain": "Multimodal scientific workflows",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.19897"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/OS-Copilot/ScienceBoard"
      }
    ],
    "access": "open-source",
    "inputs": "GUI/CLI scientific-software workflows in a virtual environment across 6 domains",
    "outputs": "Task success across 169 validated real-world computer-use tasks",
    "autonomy": "B",
    "notes": "ICLR 2026 multimodal computer-using-agent benchmark for scientific software workflows.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2505.19897",
      "https://github.com/OS-Copilot/ScienceBoard"
    ]
  },
  {
    "id": "asa-llm-srbench",
    "date_added": "2026-06-16",
    "name": "LLM-SRBench",
    "category": "benchmark",
    "domain": "Symbolic-regression benchmark",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.10415"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/deep-symbolic-mathematics/llm-srbench"
      }
    ],
    "access": "open-source",
    "inputs": "239 equation-discovery problems across 4 scientific domains (LSR-Transform, LSR-Synth)",
    "outputs": "Symbolic-accuracy and equation-discovery performance scores",
    "autonomy": "B",
    "notes": "ICML 2025 Oral benchmark for LLM-based scientific equation discovery / symbolic regression.",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2504.10415",
      "https://github.com/deep-symbolic-mathematics/llm-srbench"
    ]
  },
  {
    "id": "asa-openclaw-claw-ai-lab-labclaw",
    "date_added": "2026-06-16",
    "name": "OpenClaw / Claw AI Lab / LabClaw",
    "category": "benchmark",
    "domain": "Scientific agent lab infrastructure",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2605.22662"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Claw-AI-Lab/Claw-AI-Lab"
      }
    ],
    "access": "open-source",
    "inputs": "Single prompt to instantiate a multi-agent research team; roles, codebases, datasets, skills",
    "outputs": "Research-team dashboards, runnable experiments, execution artifacts, paper artifacts",
    "autonomy": "B",
    "notes": "Lab-native autonomous multi-agent research platform; primary system is Claw AI Lab (arXiv:2605.22662).",
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2605.22662",
      "https://github.com/Claw-AI-Lab/Claw-AI-Lab"
    ]
  },
  {
    "id": "asa-alphaevolve",
    "date_added": "2026-06-16",
    "name": "AlphaEvolve",
    "category": "crossdomain",
    "domain": "Algorithm/scientific discovery via evolutionary coding",
    "access": "open-data",
    "autonomy": "A4",
    "inputs": "Problem with an automated evaluator (code skeleton + scoring function)",
    "outputs": "Evolved/optimized code, novel algorithms, improved solutions to open math/compute problems",
    "notes": "Gemini-powered evolutionary coding agent that autonomously discovers/optimizes algorithms; found a faster matrix-multiplication scheme and new math constructions, deployed in Google datacenter/chip workflows. Repo is results-only (Colab notebook verifying the math discoveries); full agent code is not released.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2506.13131"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/google-deepmind/alphaevolve_results"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2506.13131",
      "https://github.com/google-deepmind/alphaevolve_results"
    ]
  },
  {
    "id": "asa-carl-autoscience",
    "date_added": "2026-06-16",
    "name": "Carl (Autoscience)",
    "category": "crossdomain",
    "domain": "Autonomous AI research lab (ML research)",
    "access": "platform",
    "autonomy": "A4",
    "inputs": "Research direction/area",
    "outputs": "Literature analysis, hypotheses, experiments, full-length manuscript",
    "notes": "Autoscience's autonomous research scientist; produced work accepted to ICLR 2025 workshop/Tiny Papers track (later withdrawn pending AI-authorship policy). Venture-backed ($14M seed), commercial/closed; no public code.",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://www.autoscience.ai/blog/meet-carl"
      }
    ],
    "repo_links": [],
    "verified": "2026-06-16",
    "sources": [
      "https://www.autoscience.ai/blog/meet-carl"
    ]
  },
  {
    "id": "asa-zochi-intology",
    "date_added": "2026-06-16",
    "name": "Zochi (Intology)",
    "category": "crossdomain",
    "domain": "End-to-end autonomous scientific discovery (NLP/ML)",
    "access": "platform",
    "autonomy": "A4",
    "inputs": "Research area; literature corpus",
    "outputs": "Hypotheses, experiment designs, code, manuscripts (e.g., Tempest, CS-ReFT)",
    "notes": "Intology's artificial-scientist system; its paper 'Tempest' was accepted to ACL 2025 main proceedings (top ~8.2%). Repo (MIT) holds project artifacts/papers; core system is closed/platform.",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://www.intology.ai/blog/zochi-tech-report"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/IntologyAI/Zochi"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/IntologyAI/Zochi",
      "https://www.intology.ai/blog/zochi-tech-report"
    ]
  },
  {
    "id": "asa-sciscigpt",
    "date_added": "2026-06-16",
    "name": "SciSciGPT",
    "category": "crossdomain",
    "domain": "Science of science / metascience research assistant",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "Natural-language research request over science-of-science databases/literature",
    "outputs": "Literature synthesis, database queries, analytics, evaluation, visualizations",
    "notes": "Multi-agent assistant (ResearchManager orchestrating Literature/Database/Analytics/Evaluation specialists) for the science-of-science field; published in Nature Computational Science (vol. 6, 2026). Open-source under AGPL-3.0.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s43588-025-00906-6"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Northwestern-CSSI/SciSciGPT"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/Northwestern-CSSI/SciSciGPT",
      "https://www.nature.com/articles/s43588-025-00906-6"
    ]
  },
  {
    "id": "asa-darwin-godel-machine-dgm",
    "date_added": "2026-06-16",
    "name": "Darwin Godel Machine (DGM)",
    "category": "crossdomain",
    "domain": "Self-improving coding agent / open-ended AI R&D",
    "access": "open-source",
    "autonomy": "A4",
    "inputs": "Coding benchmark/task suite (e.g., SWE-bench, Polyglot)",
    "outputs": "Self-modified agent code, an evolving archive of improved agent variants, benchmark gains",
    "notes": "Sakana AI + UBC self-improving agent that rewrites its own codebase via open-ended evolution; improved SWE-bench 20->50% and Polyglot 14.2->30.7%. Repo (Apache-2.0) confirmed.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.22954"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/jennyzzt/dgm"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2505.22954",
      "https://github.com/jennyzzt/dgm"
    ]
  },
  {
    "id": "asa-ai-cuda-engineer-sakana",
    "date_added": "2026-06-16",
    "name": "AI CUDA Engineer (Sakana)",
    "category": "crossdomain",
    "domain": "Automated GPU kernel discovery/optimization",
    "access": "open-data",
    "autonomy": "A4",
    "inputs": "PyTorch module/operation to accelerate",
    "outputs": "Optimized CUDA kernels with profiling data; verified kernel archive (~30,600 kernels)",
    "notes": "Agentic framework using evolutionary search + RAG + profiling/verification feedback to convert PyTorch ops to CUDA. Released a ~30k-kernel archive (CC-BY-4.0) and robust-kbench; full agent code not open. Early speedup claims drew scrutiny over reward hacking, prompting added correctness verification.",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://sakana.ai/ai-cuda-engineer/"
      }
    ],
    "repo_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/datasets/SakanaAI/AI-CUDA-Engineer-Archive"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://huggingface.co/datasets/SakanaAI/AI-CUDA-Engineer-Archive",
      "https://sakana.ai/ai-cuda-engineer/"
    ]
  },
  {
    "id": "asa-aide-weco-ai",
    "date_added": "2026-06-16",
    "name": "AIDE (Weco AI)",
    "category": "crossdomain",
    "domain": "Machine-learning-engineering / data-science automation",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "ML task / dataset / competition spec",
    "outputs": "Drafted, debugged, and refined solution code; submissions; tree of solution variants",
    "notes": "Tree-search-in-code-space ML-engineering agent; canonical scaffolding used in MLE-bench evaluations and AI-Scientist-style pipelines. Repo (MIT) is the open-source reference build.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2502.13138"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/WecoAI/aideml"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2502.13138",
      "https://github.com/WecoAI/aideml"
    ]
  },
  {
    "id": "asa-popper",
    "date_added": "2026-06-16",
    "name": "POPPER",
    "category": "crossdomain",
    "domain": "Automated hypothesis validation via sequential falsification",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "Free-form natural-language hypothesis plus accessible data",
    "outputs": "Designed falsification experiments, p-values, statistically-controlled validity verdict",
    "notes": "Stanford SNAP agentic framework (Popperian sequential falsification) with strict Type-I error control; matched human scientists on hypothesis validation across six domains at ~10x speed.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2502.09858"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/snap-stanford/POPPER"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2502.09858",
      "https://github.com/snap-stanford/POPPER"
    ]
  },
  {
    "id": "asa-dolphin",
    "date_added": "2026-06-16",
    "name": "Dolphin",
    "category": "crossdomain",
    "domain": "Closed-loop auto-research (ML methods)",
    "access": "open-source",
    "autonomy": "A4",
    "inputs": "Research topic/task plus relevant papers and code templates",
    "outputs": "Generated ideas, implemented/debugged experiment code, auto-analyzed results fed back into next iteration",
    "notes": "ACL 2025 closed-loop think/practice/feedback auto-research framework; iteratively improves idea quality (e.g., on 3D point classification). From InternScience; distinct from InternAgent/NovelSeek already listed.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2501.03916"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/InternScience/Dolphin"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2501.03916",
      "https://github.com/InternScience/Dolphin"
    ]
  },
  {
    "id": "asa-asta-datavoyager-ai2",
    "date_added": "2026-06-16",
    "name": "Asta DataVoyager (Ai2)",
    "category": "crossdomain",
    "domain": "Data-driven scientific discovery and analysis",
    "access": "platform",
    "autonomy": "A3",
    "inputs": "Structured dataset (CSV/Excel/JSON/HDF5/TSV/Parquet) + optional natural-language research objective",
    "outputs": "Explainable, cited analyses, statistical tests, runnable code, visualizations, documented step record",
    "notes": "Ai2 data-driven discovery agent in the Asta ecosystem, launched Oct 2025 (used by 70+ orgs and the Cancer AI Alliance). Lineage traces to the 2024 DataVoyager position-paper proof-of-concept (arXiv:2402.13610, Majumder et al., Ai2). Listed separately from the generic 'Asta agents' wiki row, mirroring how named FutureHouse agents are split out.",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://allenai.org/blog/asta-datavoyager"
      }
    ],
    "repo_links": [],
    "verified": "2026-06-16",
    "sources": [
      "https://allenai.org/blog/asta-datavoyager"
    ]
  },
  {
    "id": "asa-freephdlabor",
    "date_added": "2026-06-16",
    "name": "freephdlabor",
    "category": "crossdomain",
    "domain": "Customizable autonomous research lab (full lifecycle)",
    "access": "open-source",
    "autonomy": "A4",
    "inputs": "User's scientific problem/field; configurable agent roles and tools",
    "outputs": "Continuous (24/7) ideation, experiments, and LaTeX/publication-grade report drafts via dynamic multi-agent workflows",
    "notes": "Open-source (MIT) customizable multi-agent framework automating the full research lifecycle; specialized ManagerAgent/IdeationAgent/ExperimentationAgent with automatic prompt optimization and human-in-the-loop interruption.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2510.15624"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ltjed/freephdlabor"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2510.15624",
      "https://github.com/ltjed/freephdlabor"
    ]
  },
  {
    "id": "asa-researchtown",
    "date_added": "2026-06-16",
    "name": "ResearchTown",
    "category": "crossdomain",
    "domain": "Research-community simulation (idea generation/review)",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "Agent-data graph of researchers and papers",
    "outputs": "Simulated paper writing, review writing, and interdisciplinary research ideas",
    "notes": "Multi-agent simulator (TextGNN message-passing over an agent-data graph) modeling paper reading/writing/review; ICML 2025. Repo (Apache-2.0) confirmed.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2412.17767"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ulab-uiuc/research-town"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2412.17767",
      "https://github.com/ulab-uiuc/research-town"
    ]
  },
  {
    "id": "asa-scimaster-x-master",
    "date_added": "2026-06-16",
    "name": "SciMaster / X-Master",
    "category": "crossdomain",
    "domain": "General-purpose tool-augmented scientific reasoning agent",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "Open-ended scientific question requiring tool-augmented reasoning",
    "outputs": "Tool-mediated multi-step reasoning traces and answers",
    "notes": "SJTU SII general-purpose scientific agent; X-Master scored 32.1% on Humanity's Last Exam (first to exceed 30%, surpassing OpenAI/Google Deep Research). Repo (sjtu-sai-agents/X-Master, Python) confirmed.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2507.05241"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/sjtu-sai-agents/X-Master"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2507.05241",
      "https://github.com/sjtu-sai-agents/X-Master"
    ]
  },
  {
    "id": "asa-storm-stanford-oval",
    "date_added": "2026-06-16",
    "name": "STORM (Stanford OVAL)",
    "category": "crossdomain",
    "domain": "Knowledge curation / grounded long-form report writing",
    "access": "open-source",
    "autonomy": "A2",
    "inputs": "A topic/question",
    "outputs": "Outline plus full-length, citation-grounded Wikipedia-style article/report (Co-STORM adds interactive collaboration)",
    "notes": "NAACL 2024 multi-perspective, retrieval-grounded report-writing system (Synthesis of Topic Outlines through Retrieval and multi-perspective Question Asking), with Co-STORM successor. Widely adopted open-source literature-synthesis baseline.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2402.14207"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/stanford-oval/storm"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2402.14207",
      "https://github.com/stanford-oval/storm"
    ]
  },
  {
    "id": "asa-ether0-futurehouse",
    "date_added": "2026-06-16",
    "name": "ether0 (FutureHouse)",
    "category": "chemistry",
    "domain": "Chemistry scientific-reasoning model for molecular design",
    "access": "open-source",
    "autonomy": "A2",
    "inputs": "Natural-language chemistry questions (e.g., design constraints)",
    "outputs": "Reasoning traces and valid molecular structures (SMILES)",
    "notes": "24B RL-trained reasoning model (from Mistral-Small-24B) on 640k+ chemistry problems; open weights (futurehouse/ether0 on HuggingFace) plus repo with reward functions/benchmark. A model/tool used inside FutureHouse scientific agents rather than a standalone multi-step agent.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2506.17238"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Future-House/ether0"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2506.17238",
      "https://github.com/Future-House/ether0"
    ]
  },
  {
    "id": "asa-aviary-futurehouse",
    "date_added": "2026-06-16",
    "name": "Aviary (FutureHouse)",
    "category": "benchmark",
    "domain": "Gymnasium/training environment for scientific language agents",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Agent policies; language-grounded environments (DNA manipulation, literature QA, protein stability, math, etc.)",
    "outputs": "Environment rollouts, rewards, trained/evaluated agent performance",
    "notes": "Open-source (Apache-2.0) gymnasium formalizing agents as policies over language decision processes; ships challenging scientific environments and shows open models matching/exceeding frontier agents at up to 100x lower inference cost. The substrate behind FutureHouse agents (PaperQA/Robin).",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2412.21154"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Future-House/aviary"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2412.21154",
      "https://github.com/Future-House/aviary"
    ]
  },
  {
    "id": "asa-txgemma-agentic-tx",
    "date_added": "2026-06-16",
    "name": "TxGemma / Agentic-Tx",
    "category": "biology",
    "domain": "Therapeutics development (drug-discovery agentic system)",
    "access": "open-source",
    "autonomy": "A2",
    "inputs": "Therapeutic-entity tasks (molecules, proteins, diseases); TDC-style queries",
    "outputs": "Property predictions, reasoning, multi-step therapeutic workflow results",
    "notes": "Open Gemma-based therapeutics models (2B/9B/27B); Agentic-Tx wraps TxGemma into a Gemini-2.5-powered agent. Distinct from Harvard TxAgent already in wiki.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.06196"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/google-gemini/gemma-cookbook/tree/main/TxGemma"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2504.06196",
      "https://github.com/google-gemini/gemma-cookbook/tree/main/TxGemma"
    ]
  },
  {
    "id": "asa-bioresearcher",
    "date_added": "2026-06-16",
    "name": "BioResearcher",
    "category": "biology",
    "domain": "End-to-end automated biomedical (dry-lab) research",
    "access": "paper-only",
    "autonomy": "A4",
    "inputs": "Research objective/intention",
    "outputs": "Literature review, experimental protocol design, code implementation, with LLM reviewer quality control",
    "notes": "Modular multi-agent system (search, literature, design, programming); ~63% execution success on 8 unmet objectives.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2412.09429"
      }
    ],
    "repo_links": [],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2412.09429"
    ]
  },
  {
    "id": "asa-genegpt",
    "date_added": "2026-06-16",
    "name": "GeneGPT",
    "category": "biology",
    "domain": "Genomics question answering via NCBI Web APIs",
    "access": "open-source",
    "autonomy": "A2",
    "inputs": "Genomics questions (gene names, sequences, aliases, SNPs)",
    "outputs": "Answers grounded in live NCBI E-utilities/BLAST API calls",
    "notes": "Teaches LLMs to call NCBI Web APIs via in-context learning; SOTA (0.83) on GeneTuring. From NCBI. Distinct from GeneAgent already in wiki.",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://academic.oup.com/bioinformatics/article/40/2/btae075/7606338"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ncbi/GeneGPT"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://academic.oup.com/bioinformatics/article/40/2/btae075/7606338",
      "https://github.com/ncbi/GeneGPT"
    ]
  },
  {
    "id": "asa-scagent",
    "date_added": "2026-06-16",
    "name": "scAgent",
    "category": "biology",
    "domain": "Single-cell RNA-seq universal cell-type annotation",
    "access": "paper-only",
    "autonomy": "A3",
    "inputs": "scRNA-seq data and natural-language query",
    "outputs": "Cell-type annotations and novel cell-type discovery across tissues",
    "notes": "LLM-agent (planning/action/memory) for universal annotation; evaluated across 160 cell types and 35 tissues. Zhejiang Univ./Harvard. Distinct from CellVoyager/CellTypeAgent.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.04698"
      }
    ],
    "repo_links": [],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2504.04698"
    ]
  },
  {
    "id": "asa-celltypeagent",
    "date_added": "2026-06-16",
    "name": "CellTypeAgent",
    "category": "biology",
    "domain": "Trustworthy single-cell cell-type annotation",
    "access": "open-source",
    "autonomy": "A2",
    "inputs": "scRNA-seq marker genes / cluster data",
    "outputs": "Verified cell-type labels with reduced hallucination via CellxGene database cross-referencing",
    "notes": "Integrates LLMs with database verification; evaluated on 9 datasets, 303 cell types, 36 tissues. UNC Chapel Hill.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.08844"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/jianghao-zhang/CellTypeAgent"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2505.08844",
      "https://github.com/jianghao-zhang/CellTypeAgent"
    ]
  },
  {
    "id": "asa-cellforge",
    "date_added": "2026-06-16",
    "name": "CellForge",
    "category": "biology",
    "domain": "Agentic design of virtual-cell (perturbation-response) models",
    "access": "open-source",
    "autonomy": "A4",
    "inputs": "Raw multi-omics data (scRNA-seq/scATAC-seq/CITE-seq) plus task description",
    "outputs": "Autonomously designed neural-network architectures and executable code for perturbation prediction",
    "notes": "Multi-agent framework (task analysis, method design, code generation) evaluated on 6 perturbation datasets. Gerstein lab.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2508.02276"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/gersteinlab/CellForge"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2508.02276",
      "https://github.com/gersteinlab/CellForge"
    ]
  },
  {
    "id": "asa-colacare",
    "date_added": "2026-06-16",
    "name": "ColaCare",
    "category": "biology",
    "domain": "EHR modeling and clinical outcome prediction (multi-agent)",
    "access": "open-source",
    "autonomy": "A2",
    "inputs": "Structured EHR data (labs, diagnoses, time series)",
    "outputs": "Mortality / readmission predictions with collaborative DoctorAgent + MetaAgent reasoning reports",
    "notes": "MDT-inspired framework bridging expert models and LLM reasoning over EHR; RAG over guidelines. WWW 2025.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2410.02551"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/PKU-AICare/ColaCare"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2410.02551",
      "https://github.com/PKU-AICare/ColaCare"
    ]
  },
  {
    "id": "asa-pathfinder",
    "date_added": "2026-06-16",
    "name": "PathFinder",
    "category": "biology",
    "domain": "Computational pathology / histopathology diagnosis (multi-agent)",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "Whole-slide histopathology images and diagnostic task",
    "outputs": "Triage, ROI navigation, patch description, and synthesized diagnosis with natural-language explanation",
    "notes": "Four-agent (Triage/Navigation/Description/Diagnosis) WSI system; surpasses pathologists on skin-melanoma by ~9%. ICCV 2025. Data/code/models on project page.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2502.08916"
      }
    ],
    "repo_links": [
      {
        "label": "Link",
        "url": "https://pathfinder-dx.github.io/"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2502.08916",
      "https://pathfinder-dx.github.io/"
    ]
  },
  {
    "id": "asa-polaris-hippocratic-ai",
    "date_added": "2026-06-16",
    "name": "Polaris (Hippocratic AI)",
    "category": "biology",
    "domain": "Real-time patient-facing healthcare conversational agents",
    "access": "platform",
    "autonomy": "A3",
    "inputs": "Multi-turn spoken/text patient conversations and clinical context",
    "outputs": "Safety-checked conversational guidance with specialist-agent support (medication, labs, etc.)",
    "notes": "Safety-focused LLM 'constellation': primary conversational agent plus specialist support agents; evaluated with 1100+ nurses and 130+ physicians. Commercial (Hippocratic AI).",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2403.13313"
      }
    ],
    "repo_links": [],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2403.13313"
    ]
  },
  {
    "id": "asa-chai-2",
    "date_added": "2026-06-16",
    "name": "Chai-2",
    "category": "biology",
    "domain": "De novo antibody / protein-binder design",
    "access": "platform",
    "autonomy": "A4",
    "inputs": "Target structure/epitope defined by a few residues; desired modality (scFv, VHH, minibinder)",
    "outputs": "Zero-shot de novo antibody/nanobody/minibinder candidates",
    "notes": "Chai Discovery multimodal all-atom generative design; ~16-20% de novo antibody hit rate, ~68% minibinder hit rate; early-access only. Borderline 'agent' but central to autonomous biologics design.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.1101/2025.07.05.663018v1"
      }
    ],
    "repo_links": [],
    "verified": "2026-06-16",
    "sources": [
      "https://www.biorxiv.org/content/10.1101/2025.07.05.663018v1"
    ]
  },
  {
    "id": "asa-bixbench",
    "date_added": "2026-06-16",
    "name": "BixBench",
    "category": "benchmark",
    "domain": "Bioinformatics / computational-biology agent benchmark",
    "access": "open-data",
    "autonomy": "B",
    "inputs": "Real-world analysis capsules (data + ~205 open-ended questions from ~60 published Jupyter notebooks)",
    "outputs": "Scores on multi-step bioinformatics analysis, code execution, hypothesis generation and interpretation",
    "notes": "FutureHouse + ScienceMachine benchmark; frontier models reach only ~17% open-answer accuracy.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2503.00096"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Future-House/BixBench"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2503.00096",
      "https://github.com/Future-House/BixBench"
    ]
  },
  {
    "id": "asa-medagentbench",
    "date_added": "2026-06-16",
    "name": "MedAgentBench",
    "category": "benchmark",
    "domain": "Medical LLM agents in a virtual FHIR/EHR environment",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "300 physician-written clinical tasks; ~100 simulated patients with 700k+ data elements; FHIR-compliant interactive environment",
    "outputs": "Task success rates on agentic EHR interactions (data retrieval and write-back via FHIR APIs)",
    "notes": "Stanford ML Group; built on AgentBench/Docker FHIR server; best model (Claude 3.5 Sonnet v2) ~69.67%. NEJM AI.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2501.14654"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/stanfordmlgroup/MedAgentBench"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2501.14654",
      "https://github.com/stanfordmlgroup/MedAgentBench"
    ]
  },
  {
    "id": "asa-medagentsbench",
    "date_added": "2026-06-16",
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "name": "MedAgentsBench",
    "category": "benchmark",
    "domain": "Complex medical reasoning for thinking models and agent frameworks",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Curated hard subset drawn from multiple established medical QA datasets requiring multi-step reasoning",
    "outputs": "Performance/cost/inference-time comparisons across reasoning models and multi-agent frameworks",
    "notes": "Gerstein lab; aggregates ~10 datasets, focuses on <50% accuracy items; baselines MDAgents/MedAgents/MedPrompt. Distinct from Stanford's MedAgentBench.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2503.07459"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/gersteinlab/MedicalAgentsBench"
      }
    ],
    "sources": [
      "https://arxiv.org/abs/2503.07459",
      "https://github.com/gersteinlab/MedicalAgentsBench"
    ]
  },
  {
    "id": "asa-bioprobench",
    "date_added": "2026-06-16",
    "name": "BioProBench",
    "category": "benchmark",
    "domain": "Biological wet-lab protocol understanding and reasoning",
    "access": "open-data",
    "autonomy": "B",
    "inputs": "~27k human-written protocols; ~556k task instances across 5 subtasks",
    "outputs": "Scores on protocol QA, step ordering, generation, error correction, reasoning",
    "notes": "First large-scale multi-task protocol-understanding benchmark; 16 subdomains, 12 models evaluated; ships ProAgent baseline. ICML 2026.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.07889"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/YuyangSunshine/bioprotocolbench"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2505.07889",
      "https://github.com/YuyangSunshine/bioprotocolbench"
    ]
  },
  {
    "id": "asa-biokgbench",
    "date_added": "2026-06-16",
    "name": "BioKGBench",
    "category": "benchmark",
    "domain": "Knowledge-graph checking benchmark for biomedical agents",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Scientific-claim-verification items, KGQA items, and a KGCheck agent task (225 annotated examples)",
    "outputs": "Agent scores on literature understanding and detection of factual errors in biomedical knowledge graphs",
    "notes": "Westlake AutoLab; novel KGCheck task; baseline BKGAgent found 90+ factual errors in a popular KG.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2407.00466"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/westlake-autolab/BioKGBench"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2407.00466",
      "https://github.com/westlake-autolab/BioKGBench"
    ]
  },
  {
    "id": "asa-a-lab-berkeley",
    "date_added": "2026-06-16",
    "name": "A-Lab (Berkeley)",
    "category": "chemistry",
    "domain": "Autonomous inorganic-materials synthesis lab",
    "access": "lab-gated",
    "autonomy": "A5",
    "inputs": "Target compositions from Materials Project / computational screening",
    "outputs": "Autonomously synthesized 36 compounds from 57 targets in the corrected Nature record, with phase identification and campaign data",
    "notes": "Berkeley A-Lab closed the inorganic-synthesis loop across planning, robotics, characterization, and active learning. Nature issued an author correction on 19 January 2026 revising the campaign result to 36 compounds from 57 targets; novelty remains separately contested and is not asserted here.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-023-06734-w"
      }
    ],
    "repo_links": [],
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41586-023-06734-w"
    ]
  },
  {
    "id": "asa-chemos-2-0",
    "date_added": "2026-06-16",
    "name": "ChemOS 2.0",
    "category": "chemistry",
    "domain": "Self-driving-lab orchestration architecture",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "Lab hardware modules, experiment-planner config, optimization objective",
    "outputs": "Coordinated robotic workflows, SiLA2 device control, data exchange, closed-loop experiment scheduling",
    "notes": "Modular open-source orchestration software ('lab as OS') coordinating instruments, planners, and HPC; Sim et al., Matter 2024 (Aspuru-Guzik group). Successor to original ChemOS.",
    "paper_links": [
      {
        "label": "Link",
        "url": "https://www.cell.com/matter/fulltext/S2590-2385(24)00195-4"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/malcolmsimgithub/ChemOS2.0"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/malcolmsimgithub/ChemOS2.0",
      "https://www.cell.com/matter/fulltext/S2590-2385(24)00195-4"
    ]
  },
  {
    "id": "asa-honegumi",
    "date_added": "2026-06-16",
    "name": "Honegumi",
    "category": "chemistry",
    "domain": "Bayesian-optimization code-skeleton interface for experimental science",
    "access": "open-source",
    "autonomy": "A1-A2",
    "inputs": "Interactive selection of BO options (objectives, constraints, parameter types)",
    "outputs": "Ready-to-run, unit-tested Ax/BoTorch optimization Python scripts",
    "notes": "Interactive Bayesian-optimization assistant that generates optimization skeleton code; it does not independently execute end-to-end scientific workflows and is not a benchmark.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2502.06815"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/sgbaird/honegumi"
      }
    ],
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2502.06815",
      "https://github.com/sgbaird/honegumi"
    ]
  },
  {
    "id": "asa-ada-thin-film-self-driving-lab",
    "date_added": "2026-06-16",
    "name": "Ada (thin-film self-driving lab)",
    "category": "chemistry",
    "domain": "Autonomous thin-film / optoelectronic materials optimization",
    "access": "lab-gated",
    "autonomy": "A5",
    "inputs": "Material/process parameter space (composition, anneal conditions), property target",
    "outputs": "Robotically fabricated thin films, measured optoelectronic properties, optimized recipes",
    "notes": "Berlinguette/Hein/Aspuru-Guzik, Science Advances (May 2020). Modular robotic SDL that autonomously optimized hole mobility of spiro-OMeTAD films via model-based optimization in a closed loop.",
    "paper_links": [
      {
        "label": "Science",
        "url": "https://www.science.org/doi/10.1126/sciadv.aaz8867"
      }
    ],
    "repo_links": [],
    "verified": "2026-06-16",
    "sources": [
      "https://www.science.org/doi/10.1126/sciadv.aaz8867"
    ]
  },
  {
    "id": "asa-roborxn-ibm-rxn-for-chemistry",
    "date_added": "2026-06-16",
    "name": "RoboRXN / IBM RXN for Chemistry",
    "category": "chemistry",
    "domain": "Cloud-accessible AI synthesis-planning platform with robotic execution",
    "access": "platform",
    "autonomy": "A4",
    "inputs": "Target molecule structure (drawn/SMILES) via web browser",
    "outputs": "Predicted retrosynthetic routes, reaction conditions, robot-executable synthesis action sequences",
    "notes": "IBM Research RXN/RoboRXN: Molecular Transformer retrosynthesis plus inferred experimental procedures (Vaucher et al., Nat. Commun. 2021, 10.1038/s41467-021-22951-1) feeding cloud robotic execution. Original platform URL rxn.res.ibm.com now redirects to rxn.app.accelerate.science.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-021-22951-1"
      }
    ],
    "repo_links": [
      {
        "label": "Link",
        "url": "https://rxn.app.accelerate.science/rxn/"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://rxn.app.accelerate.science/rxn/",
      "https://www.nature.com/articles/s41467-021-22951-1"
    ]
  },
  {
    "id": "asa-mobile-robotic-chemist-liverpool-cooper",
    "date_added": "2026-06-16",
    "name": "Mobile Robotic Chemist (Liverpool / Cooper)",
    "category": "chemistry",
    "domain": "Free-roaming robot scientist for photocatalysis discovery",
    "access": "lab-gated",
    "autonomy": "A5",
    "inputs": "10-variable experimental search space, photocatalytic H2-evolution objective",
    "outputs": "688 autonomous experiments over 8 days; identified photocatalyst mixtures ~6x more active",
    "notes": "Burger et al. (Cooper group, Liverpool), Nature 2020 (10.1038/s41586-020-2442-2). Humanoid-scale mobile robot operating standard instruments, driven by batched Bayesian search ~21.5 h/day.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-020-2442-2"
      }
    ],
    "repo_links": [],
    "verified": "2026-06-16",
    "sources": [
      "https://www.nature.com/articles/s41586-020-2442-2"
    ]
  },
  {
    "id": "asa-chemputer-chempiler-cronin",
    "date_added": "2026-06-16",
    "name": "Chemputer / Chempiler (Cronin)",
    "category": "chemistry",
    "domain": "Universal programmable robotic organic synthesis",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "Synthesis procedure in XDL / ChASM chemical programming language",
    "outputs": "Compiled low-level hardware instructions executing multistep synthesis on a modular robotic platform",
    "notes": "Steiner et al. (Cronin group, Glasgow), Science 2019 (science.aav2211). Modular synthesis robot plus Chempiler compiling a chemical-descriptive language to robot code; basis of Chemify. Chempiler class lives in croningp/ChemputerSoftware.",
    "paper_links": [
      {
        "label": "Science",
        "url": "https://www.science.org/doi/10.1126/science.aav2211"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/croningp/ChemputerSoftware"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/croningp/ChemputerSoftware",
      "https://www.science.org/doi/10.1126/science.aav2211"
    ]
  },
  {
    "id": "asa-alphaflow",
    "date_added": "2026-06-16",
    "name": "AlphaFlow",
    "category": "chemistry",
    "domain": "RL-guided self-driven fluidic lab for multi-step chemistry",
    "access": "lab-gated",
    "autonomy": "A5",
    "inputs": "Reagent set, reaction-sequence search space (up to ~40 parameters), property objective",
    "outputs": "Discovered/optimized multi-step reaction routes; core-shell nanoparticle syntheses with in-situ spectral feedback",
    "notes": "Volk et al. (Abolhasani group, NC State), Nature Communications 2023 (10.1038/s41467-023-37139-y). Modular microdroplet reactor with RL that discovered cALD-inspired shell-growth routes outperforming conventional sequences.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-023-37139-y"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/AbolhasaniLab/AlphaFlow"
      }
    ],
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://github.com/AbolhasaniLab/AlphaFlow",
      "https://www.nature.com/articles/s41467-023-37139-y"
    ]
  },
  {
    "id": "asa-synbot-ai-driven-robotic-chemist",
    "date_added": "2026-06-16",
    "name": "Synbot (AI-driven robotic chemist)",
    "category": "chemistry",
    "domain": "Autonomous organic synthesis with closed-loop recipe refinement",
    "access": "lab-gated",
    "autonomy": "A5",
    "inputs": "Target organic molecule",
    "outputs": "AI-planned synthetic routes/conditions, robot-executed batch reactions, iteratively optimized recipes",
    "notes": "Samsung SAIT et al., Science Advances 2023 (sciadv.adj0461). Three-layer system (AI S/W: retrosynthesis+DoE+optimization / robot S/W / robot) that plans then refines synthesis via experimental feedback.",
    "paper_links": [
      {
        "label": "Science",
        "url": "https://www.science.org/doi/10.1126/sciadv.adj0461"
      }
    ],
    "repo_links": [],
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://www.science.org/doi/10.1126/sciadv.adj0461"
    ]
  },
  {
    "id": "asa-polybot-argonne",
    "date_added": "2026-06-16",
    "name": "Polybot (Argonne)",
    "category": "chemistry",
    "domain": "Autonomous electronic-polymer / flexible-electronics materials lab",
    "access": "lab-gated",
    "autonomy": "A5",
    "inputs": "7-dimensional processing/composition parameter space; film conductivity/defect targets",
    "outputs": "Robotically synthesized/processed/characterized polymer films; scale-up recipes (>4500 S/cm transparent conductors)",
    "notes": "Argonne CNM (Jie Xu group), Nature Communications 2025 'Autonomous platform for solution processing of electronic polymers' (10.1038/s41467-024-55655-3). Importance-guided Bayesian optimization over a 7-D processing space.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-024-55655-3"
      }
    ],
    "repo_links": [],
    "verified": "2026-06-16",
    "sources": [
      "https://www.nature.com/articles/s41467-024-55655-3"
    ]
  },
  {
    "id": "asa-crest-copilot-for-real-world-experimental-scientists",
    "date_added": "2026-06-16",
    "name": "CRESt (Copilot for Real-world Experimental Scientists)",
    "category": "chemistry",
    "domain": "Multimodal AI copilot driving robotic materials experiments",
    "access": "lab-gated",
    "autonomy": "A5",
    "inputs": "Natural-language goals plus literature, compositions, microstructural images, live electrochemical data",
    "outputs": "Proposed experiments, robotic execution, active-learning hypothesis updates; discovered low-Pd formate fuel-cell catalyst",
    "notes": "MIT (Ju Li group), Nature 2025 'A multimodal robotic platform for multi-element electrocatalyst discovery' (10.1038/s41586-025-09640-5). Fused text/image/recipe/live-data with robotics+active learning; 900+ chemistries, 3,500 electrochemical trials, record formate-fuel-cell performance.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-025-09640-5"
      }
    ],
    "repo_links": [],
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://www.nature.com/articles/s41586-025-09640-5"
    ]
  },
  {
    "id": "asa-honeycomb",
    "date_added": "2026-06-16",
    "name": "HoneyComb",
    "category": "chemistry",
    "domain": "LLM agent system for materials science",
    "access": "paper-only",
    "autonomy": "A3",
    "inputs": "Materials-science question or computational task",
    "outputs": "Tool-augmented answers using curated knowledge base (MatSciKB) and constructed API tools (ToolHub)",
    "notes": "Zhang et al., 'HoneyComb: A Flexible LLM-Based Agent System for Materials Science', Findings of EMNLP 2024 (arXiv 2409.00135). Combines MatSciKB + inductively constructed ToolHub + adaptive retriever to cut hallucination.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2409.00135"
      }
    ],
    "repo_links": [],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2409.00135"
    ]
  },
  {
    "id": "asa-autolabs",
    "date_added": "2026-06-16",
    "name": "AutoLabs",
    "category": "chemistry",
    "domain": "Self-correcting multi-agent system for liquid-handler experimentation",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "Natural-language experimental goals / protocols",
    "outputs": "Stoichiometric calculations and hardware-ready protocol files for high-throughput liquid handlers, with iterative self-correction",
    "notes": "PNNL, 'AutoLabs: Cognitive Multi-Agent Systems with Self-Correction for Autonomous Chemical Experimentation' (arXiv 2509.25651, 2025). Translates NL instructions into executable Big Kahuna liquid-handler protocols; reasoning cuts nRMSE >85%, F1>0.89.",
    "paper_links": [
      {
        "label": "Scientific Reports",
        "url": "https://doi.org/10.1038/s41598-026-45593-z"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/pnnl/AutoLabs"
      }
    ],
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://doi.org/10.1038/s41598-026-45593-z",
      "https://github.com/pnnl/AutoLabs"
    ]
  },
  {
    "id": "asa-k-agents-self-driving-labs-for-quantum-computing",
    "date_added": "2026-06-16",
    "name": "k-agents (self-driving labs for quantum computing)",
    "category": "physics",
    "domain": "Agent framework for autonomous physical-lab experimentation",
    "access": "lab-gated",
    "autonomy": "A5",
    "inputs": "Lab knowledge documentation, available operations, multistep experimental procedure",
    "outputs": "Agent-based state machines executing multistep experiments with closed-loop feedback and result interpretation",
    "notes": "Cao et al. (Oxford/Toronto), 'Agents for self-driving laboratories applied to quantum computing' (arXiv 2412.07978; Patterns 2025, 10.1016/j.patter.2025.101372). LLM agents calibrated/operated a superconducting quantum processor at near-human level.",
    "paper_links": [
      {
        "label": "Patterns",
        "url": "https://doi.org/10.1016/j.patter.2025.101372"
      },
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2412.07978"
      }
    ],
    "repo_links": [],
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2412.07978",
      "https://doi.org/10.1016/j.patter.2025.101372"
    ]
  },
  {
    "id": "asa-matsciagent-modular-llm-agents-for-materials",
    "date_added": "2026-06-16",
    "name": "MatSciAgent / Modular LLM Agents for Materials",
    "category": "chemistry",
    "domain": "Multi-task computational materials-science agent",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "Natural-language materials query / task",
    "outputs": "Materials data retrieval (Materials Project/MatWeb), crystal-structure generation, continuum and MD simulations via delegated agents",
    "notes": "Chaudhari, Ock, Barati Farimani, 'Modular large language model agents for multi-task computational materials science', Communications Materials 2025 (10.1038/s43246-025-00994-x). Master agent delegates to Material-Extraction/Continuum/Crystal-Generation/MD agents.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s43246-025-00994-x"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/cakshat/MatSci-LLM-Agents"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/cakshat/MatSci-LLM-Agents",
      "https://www.nature.com/articles/s43246-025-00994-x"
    ]
  },
  {
    "id": "asa-ai-newton",
    "date_added": "2026-06-16",
    "name": "AI-Newton",
    "category": "physics",
    "domain": "Concept-driven physical-law discovery (classical mechanics)",
    "access": "open-source",
    "autonomy": "A3-A4",
    "inputs": "Collections of physics experiments (observations) with no prior physical knowledge",
    "outputs": "Autonomously formulated symbolic concepts and general laws stored in a knowledge base",
    "notes": "Concept-driven autonomous discovery system (Fang, Jian, Li, Ma) with a physics DSL + knowledge base; rediscovers Newton's second law, gravitation and conservation laws unsupervised. MIT-licensed repo, 80 stars.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.01538"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Science-Discovery/AI-Newton"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2504.01538",
      "https://github.com/Science-Discovery/AI-Newton"
    ]
  },
  {
    "id": "asa-mephisto",
    "date_added": "2026-06-16",
    "name": "Mephisto",
    "category": "physics",
    "domain": "Astronomy / galaxy SED modeling and interpretation",
    "access": "open-source",
    "autonomy": "A3-A4",
    "inputs": "Multi-band galaxy photometry/observations and CIGALE SED model library",
    "outputs": "Iteratively refined physical SED models, reasoning traces, interpretations of galaxy populations",
    "notes": "Self-improving multi-agent LLM framework interfacing with CIGALE; uses tree search + self-play to interpret multi-band galaxy observations incl. JWST 'Little Red Dot' galaxies. MIT-licensed repo.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2510.08354"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ZechangSun/mephisto"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2510.08354",
      "https://github.com/ZechangSun/mephisto"
    ]
  },
  {
    "id": "asa-starwhisper-telescope",
    "date_added": "2026-06-16",
    "name": "StarWhisper Telescope",
    "category": "physics",
    "domain": "Astronomy / end-to-end autonomous telescope observation",
    "access": "open-source",
    "autonomy": "A4-A5",
    "inputs": "Natural-language observation goals; live telescope/instrument control APIs and survey context",
    "outputs": "Site-specific observation lists, telescope control sequences, real-time image-analysis pipelines, transient follow-up proposals",
    "notes": "LLM-agent framework automating end-to-end observations for the Nearby Galaxy Supernovae Survey (10-telescope network); plans, controls and processes data. Communications Engineering 4, 184 (2025); also arXiv:2412.06412. Apache-2.0 repo.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s44172-025-00520-4"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Yu-Yang-Li/StarWhisper"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/Yu-Yang-Li/StarWhisper",
      "https://www.nature.com/articles/s44172-025-00520-4"
    ]
  },
  {
    "id": "asa-cmbevolve-cosmoevolve",
    "date_added": "2026-06-16",
    "name": "CMBEvolve / CosmoEvolve",
    "category": "physics",
    "domain": "Autonomous cosmology discovery (CMB / weak lensing / ACT data)",
    "access": "paper-only",
    "autonomy": "A4",
    "inputs": "Cosmology datasets (e.g., ACT DR6, weak-lensing maps) and research objectives",
    "outputs": "Evolved analysis code, out-of-distribution detectors, analysis-grade diagnostics and findings",
    "notes": "Two complementary agentic systems (Xu & Borrett): CMBEvolve = LLM-guided code evolution + tree search for objective-driven tasks; CosmoEvolve = virtual multi-agent cosmology research lab. No public repo confirmed (MadEvolve-Cosmo is a different system).",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2605.14791"
      }
    ],
    "repo_links": [],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2605.14791"
    ]
  },
  {
    "id": "asa-astroreview",
    "date_added": "2026-06-16",
    "name": "AstroReview",
    "category": "physics",
    "domain": "Astronomy / telescope proposal authoring and peer review",
    "access": "paper-only",
    "autonomy": "A3",
    "inputs": "Telescope proposal drafts/abstracts (e.g., HST Proposal Abstracts Catalog)",
    "outputs": "Drafted/revised proposals, automated reviews, actionable suggestions, reliability checks",
    "notes": "Multi-agent framework (Wang, Xiao, Tian et al.) with a Proposal Authoring Agent + Review Agent simulating the review-refinement cycle; ~87% accuracy identifying accepted proposals, +66% acceptance after two iterations. No public repo found.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2512.24754"
      }
    ],
    "repo_links": [],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2512.24754"
    ]
  },
  {
    "id": "asa-replicationbench",
    "date_added": "2026-06-16",
    "name": "ReplicationBench",
    "category": "benchmark",
    "domain": "Astrophysics research-paper replication for agents",
    "access": "paper-only",
    "autonomy": "B",
    "inputs": "Astrophysics papers decomposed into expert-validated, author-co-developed replication tasks",
    "outputs": "Scores on replicating setup, derivations, data analysis, and codebases",
    "notes": "20 reproducible astrophysics papers (111 tasks) + ReplicationBench-Plus (11 papers / 58 tasks); tasks co-developed with original authors; frontier models score <20%. No public code/data repo confirmed at search time.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2510.24591"
      }
    ],
    "repo_links": [],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2510.24591"
    ]
  },
  {
    "id": "asa-astromlab-astrosage",
    "date_added": "2026-06-16",
    "name": "AstroMLab / AstroSage",
    "category": "benchmark",
    "domain": "Astronomy domain-specialized LLMs and Q&A benchmark",
    "access": "open-data",
    "autonomy": "B",
    "inputs": "Astronomy multiple-choice and Q&A tasks (AstroMLab-1, 4,425 MCQs); astronomical literature for training",
    "outputs": "Benchmark accuracy scores; AstroSage domain-specialized model assistants",
    "notes": "arXiv:2505.17592 is 'AstroMLab 4' introducing AstroSage-Llama-3.1-70B (top-tier astronomy Q&A); AstroMLab-1 benchmark + AstroSage-70B/8B models released on HuggingFace. astromlab.org is the project hub.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.17592"
      }
    ],
    "repo_links": [
      {
        "label": "Link",
        "url": "https://astromlab.org/"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2505.17592",
      "https://astromlab.org/"
    ]
  },
  {
    "id": "asa-climateagent",
    "date_added": "2026-06-16",
    "name": "ClimateAgent",
    "category": "crossdomain",
    "domain": "Climate / earth-system data-science workflows",
    "access": "paper-only",
    "autonomy": "A3-A4",
    "inputs": "Natural-language climate-science questions; climate data APIs",
    "outputs": "Decomposed sub-tasks, data-download scripts, Python analysis, visualizations, final reports",
    "notes": "Multi-agent orchestration (Orchestrate/Plan/Data/Coding agents, Kim et al., HKUST/B. Yuan) with self-correction; ships Climate-Agent-Bench-85 (atmospheric rivers, drought, heat waves, SST, cyclones); 100% task completion. Published TMLR 2026. No public repo found.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2511.20109"
      }
    ],
    "repo_links": [],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2511.20109"
    ]
  },
  {
    "id": "asa-tritondft",
    "date_added": "2026-06-16",
    "name": "TritonDFT",
    "category": "physics",
    "domain": "Condensed-matter / materials DFT automation (Quantum ESPRESSO)",
    "access": "open-source",
    "autonomy": "A3-A4",
    "inputs": "Natural-language DFT/materials task with Quantum ESPRESSO toolchain",
    "outputs": "Planned and executed DFT workflows, parameter inference, refined runs; DFTBench scores",
    "notes": "Multi-agent framework (Hu, Talit, Wang et al.) with expert-curated workflow design, Pareto-aware parameter inference and multi-source knowledge augmentation; bundles Quantum ESPRESSO; ships DFTBench. Repo public with QE submodule + benchmark folder.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.03372"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Leo9660/TritonDFT"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2603.03372",
      "https://github.com/Leo9660/TritonDFT"
    ]
  },
  {
    "id": "asa-caddesigner",
    "date_added": "2026-06-16",
    "name": "CADDesigner",
    "category": "physics",
    "domain": "Engineering / conceptual CAD model generation",
    "access": "paper-only",
    "autonomy": "A2-A3",
    "inputs": "Textual descriptions and/or sketches of a desired part; interactive clarification dialogue",
    "outputs": "Parametric CAD modeling code (extrude/revolve/fillet/sweep/loft, standard components) with iterative visual feedback",
    "notes": "LLM agent (Fan, Ni, Yin et al.) using an Explicit Context Imperative Paradigm + visual feedback loop and a structured knowledge base for early-stage conceptual CAD. No public repo confirmed.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2508.01031"
      }
    ],
    "repo_links": [],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2508.01031"
    ]
  },
  {
    "id": "asa-chatvis",
    "date_added": "2026-06-16",
    "name": "ChatVis",
    "category": "physics",
    "domain": "Scientific visualization automation (ParaView)",
    "access": "open-source",
    "autonomy": "A2-A3",
    "inputs": "Natural-language scientific-visualization task and data",
    "outputs": "ParaView Python scripts with error-detection/iterative correction, rendered visualizations",
    "notes": "LLM assistant (Mallick, Yildiz, Lenz, Peterka; Argonne) generating ParaView Python code with an error detection/correction loop; ships a ChatVis benchmark of canonical visualization scenarios. Repo confirmed (ChatVis_agent + ChatVis_benchmark).",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2410.11863"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/tpeterka/chatvis"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2410.11863",
      "https://github.com/tpeterka/chatvis"
    ]
  },
  {
    "id": "asa-paraview-mcp",
    "date_added": "2026-06-16",
    "name": "ParaView-MCP",
    "category": "physics",
    "domain": "Autonomous scientific-visualization agent with direct tool use",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "Natural-language and multimodal visualization goals over scientific data",
    "outputs": "Direct in-situ control of ParaView's interface during interactive exploration/analysis",
    "notes": "MCP-based autonomous agent (Liu, Miao, Bremer; LLNL) controlling ParaView's Python API via multimodal LLMs with visual feedback; IEEE VIS 2025 short paper. Public repo github.com/llnl/paraview_mcp (BSD-3) found.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.07064"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/llnl/paraview_mcp"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2505.07064",
      "https://github.com/llnl/paraview_mcp"
    ]
  },
  {
    "id": "asa-scivisagentbench",
    "date_added": "2026-06-16",
    "name": "SciVisAgentBench",
    "category": "benchmark",
    "domain": "Scientific data-analysis and visualization agents",
    "access": "open-data",
    "autonomy": "B",
    "inputs": "Expert-crafted scientific-visualization/analysis tasks across domains and tools",
    "outputs": "Outcome- and process-based evaluation scores (LLM-as-judge plus deterministic evaluators)",
    "notes": "108 expert-designed cases across application domain/data type/complexity/visualization operation (Ai, Miao, Tang et al.; Notre Dame, LLNL, others); multimodal outcome-centric eval pipeline. Project page links GitHub + HuggingFace 'SciVisAgentBench-tasks' dataset.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.29139"
      }
    ],
    "repo_links": [
      {
        "label": "Link",
        "url": "https://scivisagentbench.github.io/"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2603.29139",
      "https://scivisagentbench.github.io/"
    ]
  },
  {
    "id": "asa-scicode",
    "date_added": "2026-06-16",
    "name": "SciCode",
    "category": "benchmark",
    "domain": "Research coding across physics, chemistry, biology, materials, math",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Scientist-curated research coding problems decomposed into subproblems with optional background",
    "outputs": "Scored code solutions against gold solutions and test cases",
    "notes": "338 subproblems from 80 main problems across 16 natural-science subfields; best model ~4.6% on the realistic setting (NeurIPS 2024 D&B).",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2407.13168"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/scicode-bench/SciCode"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2407.13168",
      "https://github.com/scicode-bench/SciCode"
    ]
  },
  {
    "id": "asa-core-bench",
    "date_added": "2026-06-16",
    "name": "CORE-Bench",
    "category": "benchmark",
    "domain": "Computational reproducibility of published research (CS, social science, medicine)",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Paper code/data capsules; task = reproduce reported results",
    "outputs": "Reproduction accuracy across 270 tasks / 90 papers at 3 difficulty levels",
    "notes": "Princeton (Siegel, Kapoor, Stroebl, Narayanan); ships AutoGPT and CORE-Agent baselines; best ~21% on hardest level.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2409.11363"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/siegelz/core-bench"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2409.11363",
      "https://github.com/siegelz/core-bench"
    ]
  },
  {
    "id": "asa-discoverybench",
    "date_added": "2026-06-16",
    "name": "DiscoveryBench",
    "category": "benchmark",
    "domain": "Data-driven scientific discovery (6 domains)",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "264 real + 903 synthetic tasks; each = dataset(s) + metadata + NL discovery goal",
    "outputs": "Faceted evaluation of discovered hypotheses/workflows; best system ~25%",
    "notes": "AllenAI (Majumder et al.); formalizes multi-step data-driven discovery from published-paper workflows.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2407.01725"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/allenai/discoverybench"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2407.01725",
      "https://github.com/allenai/discoverybench"
    ]
  },
  {
    "id": "asa-dsbench",
    "date_added": "2026-06-16",
    "name": "DSBench",
    "category": "benchmark",
    "domain": "Data-science agents (analysis + modeling)",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "466 data-analysis + 74 data-modeling tasks from ModelOff and Kaggle (long/multimodal, multi-table)",
    "outputs": "Task success and performance-gap metrics; best agent ~34% of analysis tasks",
    "notes": "Jing et al.; realistic end-to-end data-science-agent benchmark (540 tasks total).",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2409.07703"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/LiqiangJing/DSBench"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2409.07703",
      "https://github.com/LiqiangJing/DSBench"
    ]
  },
  {
    "id": "asa-blade",
    "date_added": "2026-06-16",
    "name": "BLADE",
    "category": "benchmark",
    "domain": "Language-model agents for data-driven science (analysis decisions)",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "12 datasets + research questions from scientific literature; expert ground-truth analyses",
    "outputs": "Evaluation of conceptual-variable choices, data transforms, and statistical-model selection vs expert decisions",
    "notes": "EMNLP 2024 (Gu et al., UW); focuses on diversity/quality of analytic decision-making.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2408.09667"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/behavioral-data/BLADE"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2408.09667",
      "https://github.com/behavioral-data/BLADE"
    ]
  },
  {
    "id": "asa-super",
    "date_added": "2026-06-16",
    "name": "SUPER",
    "category": "benchmark",
    "domain": "Setting up and executing tasks from research repositories (ML/NLP)",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "45 end-to-end + 152 sub-problems + 602 auto-generated tasks from GitHub research repos",
    "outputs": "Success at configuring/running real research codebases; GPT-4o ~16.3% end-to-end",
    "notes": "AllenAI (Bogin et al.), EMNLP 2024; tests the 'get the repo to run' capability central to reproducing research.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2409.07440"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/allenai/super-benchmark"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2409.07440",
      "https://github.com/allenai/super-benchmark"
    ]
  },
  {
    "id": "asa-aaar-1-0",
    "date_added": "2026-06-16",
    "name": "AAAR-1.0",
    "category": "benchmark",
    "domain": "Expertise-intensive research tasks (equation inference, experiment design, paper weakness, review critique)",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Paper submissions/contexts for four research-oriented tasks",
    "outputs": "Scores on EquationInference, ExperimentDesign, PaperWeakness, ReviewCritique",
    "notes": "Penn State (Lou et al.), ICML 2025; research-oriented tasks requiring deep domain expertise.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2410.22394"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/RenzeLou/AAAR-1.0"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2410.22394",
      "https://github.com/RenzeLou/AAAR-1.0"
    ]
  },
  {
    "id": "asa-researcherbench",
    "date_added": "2026-06-16",
    "name": "ResearcherBench",
    "category": "benchmark",
    "domain": "Deep AI research systems on frontier AI-research questions",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "65 frontier AI-research questions across 35 subjects (Technical Details, Literature Review, Open Consulting)",
    "outputs": "Rubric assessment + faithfulness/groundedness scores for deep-research systems",
    "notes": "GAIR-NLP (Xu et al.); evaluates insight generation on unsolved frontier questions vs OpenAI/Gemini Deep Research.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2507.16280"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/GAIR-NLP/ResearcherBench"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2507.16280",
      "https://github.com/GAIR-NLP/ResearcherBench"
    ]
  },
  {
    "id": "asa-massw",
    "date_added": "2026-06-16",
    "name": "MASSW",
    "category": "benchmark",
    "domain": "AI-assisted scientific workflows (CS literature)",
    "access": "open-data",
    "autonomy": "B",
    "inputs": "152k+ peer-reviewed CS papers structured into context/key-idea/method/outcome/impact",
    "outputs": "Benchmark tasks for idea generation, outcome prediction, workflow expansion",
    "notes": "Zhang et al.; 152k+ papers from 17 CS conferences over 50 years; multi-aspect summarization dataset + benchmark.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2406.06357"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/xingjian-zhang/massw"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2406.06357",
      "https://github.com/xingjian-zhang/massw"
    ]
  },
  {
    "id": "asa-humanity-s-last-exam-hle-science-subset",
    "date_added": "2026-06-16",
    "name": "Humanity's Last Exam (HLE) — science subset",
    "category": "benchmark",
    "domain": "Frontier closed-ended academic knowledge incl. natural-science subsets",
    "access": "open-data",
    "autonomy": "B",
    "inputs": "2,500 expert-written questions across dozens of subjects (incl. math/physics/chem/bio), multimodal subset",
    "outputs": "Accuracy + calibration on frontier-knowledge MCQ/short-answer; auto-gradable",
    "notes": "CAIS + Scale AI; multimodal frontier-knowledge benchmark whose natural-science subset is a standard reasoning yardstick.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2501.14249"
      }
    ],
    "repo_links": [
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/datasets/cais/hle"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2501.14249",
      "https://huggingface.co/datasets/cais/hle"
    ]
  },
  {
    "id": "asa-mlgym-mlgym-bench",
    "date_added": "2026-06-16",
    "name": "MLGym / MLGym-Bench",
    "category": "benchmark",
    "domain": "AI research agents (Gym environment for ML research tasks)",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "13 open-ended AI-research tasks (CV, NLP, RL, game theory) via a shell environment",
    "outputs": "Agent performance scores plus an RL-trainable Gym interface (Agents/Environment/Datasets/Tasks)",
    "notes": "Meta GenAI/FAIR + UCSB NLP. First Gym environment for ML research tasks; the framework and MLGym-Bench evaluate LM agents on real research skills. License CC-BY-NC 4.0.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2502.14499"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/facebookresearch/MLGym"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2502.14499",
      "https://github.com/facebookresearch/MLGym"
    ]
  },
  {
    "id": "asa-mlagentbench",
    "date_added": "2026-06-16",
    "name": "MLAgentBench",
    "category": "benchmark",
    "domain": "ML experimentation agents",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "13 ML experimentation tasks (CIFAR-10 improvement to BabyLM) with dataset + task description",
    "outputs": "Competence (>=10% improvement over baseline) and efficiency (time/tokens) metrics",
    "notes": "Stanford SNAP (Huang, Vora, Liang, Leskovec), ICML 2024. Early end-to-end ML-experimentation agent benchmark and widely reused scaffold; MIT-licensed repo (~342 stars).",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2310.03302"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/snap-stanford/MLAgentBench"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2310.03302",
      "https://github.com/snap-stanford/MLAgentBench"
    ]
  },
  {
    "id": "asa-mlrc-bench",
    "date_added": "2026-06-16",
    "name": "MLRC-Bench",
    "category": "benchmark",
    "domain": "Machine-learning research competition challenges (novel methods)",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Tasks adapted from recent ML conference competitions (LLM safety, multimodal, few-shot)",
    "outputs": "Objective metrics on proposing+implementing novel methods; best agent closes only ~9.3% of the human gap",
    "notes": "Zhang et al. (Michigan/LG AI). NeurIPS 2025 Datasets & Benchmarks. Dynamic benchmark with objective scoring rather than LLM-judge; HF leaderboard.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.09702"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/yunx-z/MLRC-Bench"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2504.09702",
      "https://github.com/yunx-z/MLRC-Bench"
    ]
  },
  {
    "id": "asa-scireplicate-bench",
    "date_added": "2026-06-16",
    "name": "SciReplicate-Bench",
    "category": "benchmark",
    "domain": "Agent-driven algorithmic reproduction from research papers (NLP)",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "100 tasks from 36 NLP papers (2024) with algorithm descriptions, test cases, repo dependencies",
    "outputs": "Execution accuracy and reasoning-graph accuracy; best (Sci-Reproducer) ~39% execution accuracy",
    "notes": "Xiang et al. (King's College London / Yulan He). COLM 2025. Ships the Sci-Reproducer dual-agent (Paper Agent + Code Agent) framework.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.00255"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/xyzCS/SciReplicate-Bench"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2504.00255",
      "https://github.com/xyzCS/SciReplicate-Bench"
    ]
  },
  {
    "id": "asa-chembench",
    "date_added": "2026-06-16",
    "name": "ChemBench",
    "category": "benchmark",
    "domain": "Chemistry knowledge and reasoning of LLMs vs chemists",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "2,700+ curated QA pairs across chemistry sub-areas (incl. toxicity/safety, medicinal preference)",
    "outputs": "Accuracy vs expert chemists plus confidence-calibration analysis; leaderboard at chembench.lamalab.org",
    "notes": "LamaLab (Jablonka, FSU Jena), Nature Chemistry 2025. Best models on average beat the best human chemists in the study but fail some basic tasks and are overconfident. Note: the 'Are large language models superhuman chemists?' companion is at s41557-025-01865-1.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41557-025-01815-x"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/lamalab-org/chembench"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://github.com/lamalab-org/chembench",
      "https://www.nature.com/articles/s41557-025-01815-x"
    ]
  },
  {
    "id": "asa-rexbench",
    "date_added": "2026-06-16",
    "name": "RExBench",
    "category": "benchmark",
    "domain": "Autonomous implementation of AI/NLP research extensions",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Realistic extensions of 12 research papers with expert-written instructions and codebases; hypothesis-guided",
    "outputs": "Success at implementing novel research-extension experiments; best agent ~31-33% (under hints <44%)",
    "notes": "Edwards, Lee, Mao, Qin, Schuster, Kim (TIN Lab / Boston Univ.); ACL 2026. Tested with aider, Claude Code, and OpenHands. Site rexbench.com; HF dataset tin-lab/RExBench.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2506.22598"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/tinlaboratory/RExBench"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2506.22598",
      "https://github.com/tinlaboratory/RExBench"
    ]
  },
  {
    "id": "asa-innovatorbench",
    "date_added": "2026-06-16",
    "name": "InnovatorBench",
    "category": "benchmark",
    "domain": "Innovative LLM research (end-to-end, runnable artifacts)",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "20 tasks across data construction/filtering/augmentation, loss design, reward design, scaffold construction",
    "outputs": "Correctness, performance, output quality, uncertainty; runs in the ResearchGym research environment",
    "notes": "GAIR-NLP (Pengfei Liu et al.), ICLR 2026. Ships its own ResearchGym environment and a ReAct agent; Apache-2.0; HF dataset. Distinct from the wiki's separately-listed ResearchGym entry.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2510.27598"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/GAIR-NLP/InnovatorBench"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2510.27598",
      "https://github.com/GAIR-NLP/InnovatorBench"
    ]
  },
  {
    "id": "asa-fml-bench",
    "date_added": "2026-06-16",
    "name": "FML-bench",
    "category": "benchmark",
    "domain": "Automatic ML research agents on fundamental ML problems",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "8 fundamental ML research problems built on real-world codebases (low coding barrier, extensible)",
    "outputs": "Performance plus an Exploration-Diversity metric measuring variance of proposals across iterations",
    "notes": "Zou et al. 'FML-bench: Benchmarking Machine Learning Agents for Scientific Research'. Evaluates the research process (exploration breadth), finding broad exploration beats narrow-deep.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2510.10472"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/qrzou/FML-bench"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2510.10472",
      "https://github.com/qrzou/FML-bench"
    ]
  },
  {
    "id": "asa-researcharena",
    "date_added": "2026-06-16",
    "name": "ResearchArena",
    "category": "benchmark",
    "domain": "Research agents collecting and organizing information (survey-writing)",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Survey-style topics requiring discovery, selection, and organization of literature over a 12M-paper + 7.9k-survey corpus (S2ORC)",
    "outputs": "Information-discovery, selection, and organization scores (survey/mind-map construction)",
    "notes": "Hao Kang & Chenyan Xiong (CMU). Decomposes research into information discovery, selection, and organization; finds LLM agents underperform keyword retrieval baselines.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2406.10291"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/cxcscmu/ResearchArena"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2406.10291",
      "https://github.com/cxcscmu/ResearchArena"
    ]
  },
  {
    "id": "asa-ideabench",
    "date_added": "2026-06-16",
    "name": "IdeaBench",
    "category": "benchmark",
    "domain": "Research idea generation (multi-domain)",
    "access": "open-data",
    "autonomy": "B",
    "inputs": "Titles/abstracts of influential papers plus their referenced works across multiple domains",
    "outputs": "Insight-Score ranking of generated ideas on novelty/feasibility (GPT-4o-as-judge, two-stage)",
    "notes": "Guo, Shariatmadari, Xiong, Huang, Xie, Bekiranov, Zhang (University of Virginia). KDD 2025. Quantifies the novelty/feasibility trade-off for LLM-generated research ideas.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2411.02429"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/amir-hassan25/IdeaBench"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2411.02429",
      "https://github.com/amir-hassan25/IdeaBench"
    ]
  },
  {
    "id": "asa-sciassess",
    "date_added": "2026-06-16",
    "name": "SciAssess",
    "category": "benchmark",
    "domain": "Scientific literature analysis (biology, chemistry, materials, medicine)",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Scientific-literature tasks across fields at three levels: Memorization (L1), Comprehension (L2), Analysis & Reasoning (L3)",
    "outputs": "Per-field proficiency scores for literature-analysis capability (11 LLMs evaluated)",
    "notes": "Cai et al. (DP Technology / DeepModeling). Multimodal-aware literature-analysis benchmark; LGPL-3.0 repo (~88 stars).",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2403.01976"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/deepmodeling/SciAssess"
      }
    ],
    "verified": "2026-06-16",
    "sources": [
      "https://arxiv.org/abs/2403.01976",
      "https://github.com/deepmodeling/SciAssess"
    ]
  },
  {
    "id": "asa-omniscientist",
    "date_added": "2026-06-20",
    "name": "OmniScientist",
    "category": "crossdomain",
    "domain": "Full-cycle autonomous scientific research",
    "access": "open-source",
    "autonomy": "A3-A4",
    "inputs": "Research topics, literature corpora, experimental data, hypotheses",
    "outputs": "Literature reviews, hypotheses, experimental plans, manuscripts, peer-review feedback",
    "notes": "OmniScientist encodes the full human research lifecycle into a co-evolving multi-agent ecosystem, introducing the OSP collaboration protocol and ScienceArena benchmark to coordinate and evaluate agents across data, literature, hypothesis, experiment, writing, and review stages.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2511.16931"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/tsinghua-fib-lab/OmniScientist"
      }
    ],
    "verified": "2026-06-20",
    "sources": [
      "https://arxiv.org/abs/2511.16931",
      "https://github.com/tsinghua-fib-lab/OmniScientist"
    ]
  },
  {
    "id": "asa-safescientist",
    "date_added": "2026-06-20",
    "name": "SafeScientist",
    "category": "crossdomain",
    "domain": "Safe autonomous AI scientist across scientific domains",
    "access": "open-source",
    "autonomy": "A3-A4",
    "inputs": "Research task specifications, scientific hypotheses, tool-use requests",
    "outputs": "Research outputs with safety monitoring reports, ethical review decisions, flagged high-risk actions",
    "notes": "Introduces SciSafetyBench, a benchmark of 240 high-risk scientific tasks across 6 domains, and integrates four safety mechanisms (prompt, agent-collaboration, tool-use monitoring, and ethical reviewer) into the full AI research pipeline.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.23559"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ulab-uiuc/SafeScientist"
      }
    ],
    "verified": "2026-06-20",
    "sources": [
      "https://arxiv.org/abs/2505.23559",
      "https://github.com/ulab-uiuc/SafeScientist"
    ]
  },
  {
    "id": "asa-scp-science-context-protocol",
    "date_added": "2026-06-20",
    "name": "SCP (Science Context Protocol)",
    "category": "crossdomain",
    "domain": "Federated autonomous scientific agent infrastructure",
    "access": "open-source",
    "autonomy": "A3-A4",
    "inputs": "Scientific resource descriptions, experiment requests, tool/model/dataset/instrument specifications",
    "outputs": "Standardized API calls to scientific resources, experiment lifecycle management, federated agent coordination",
    "notes": "SCP defines a universal open standard and federated Hub-and-Server architecture that enables autonomous scientific agents to discover, describe, and invoke heterogeneous scientific resources (tools, models, datasets, instruments) across a global web of participating servers.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2512.24189"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/InternScience/scp"
      }
    ],
    "verified": "2026-06-20",
    "sources": [
      "https://arxiv.org/abs/2512.24189",
      "https://github.com/InternScience/scp"
    ]
  },
  {
    "id": "asa-saga-scientific-autonomous-goal-evolving-agent",
    "date_added": "2026-06-20",
    "name": "SAGA (Scientific Autonomous Goal-evolving Agent)",
    "category": "crossdomain",
    "domain": "Autonomous multi-domain scientific optimization",
    "access": "open-source",
    "autonomy": "A4",
    "inputs": "Scientific optimization task description, domain-specific data (molecular sequences, material properties, etc.)",
    "outputs": "Evolved objective functions and optimized solutions (antibiotics, nanobodies, DNA sequences, materials, chemical compounds)",
    "notes": "SAGA employs a bi-level loop where an outer LLM layer autonomously proposes and refines objective functions while an inner loop optimizes solutions under those objectives, demonstrated across five scientific domains.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2512.21782"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/btyu/SAGA"
      }
    ],
    "verified": "2026-06-20",
    "sources": [
      "https://arxiv.org/abs/2512.21782",
      "https://github.com/btyu/SAGA"
    ]
  },
  {
    "id": "asa-asi-evolve",
    "date_added": "2026-06-20",
    "name": "ASI-Evolve",
    "category": "crossdomain",
    "domain": "Autonomous AI-for-AI research and algorithm discovery",
    "access": "open-source",
    "autonomy": "A4",
    "inputs": "Research objectives, existing codebases, experimental results, and a cognition base of prior knowledge",
    "outputs": "Novel neural architectures, data curation pipelines, and RL algorithms that outperform human-designed baselines",
    "notes": "ASI-Evolve implements a learn-design-experiment-analyze loop with a dedicated analyzer and cognition base, enabling autonomous discovery across neural architecture search, data curation, and reinforcement learning algorithm design.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.29640"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/GAIR-NLP/ASI-Evolve"
      }
    ],
    "verified": "2026-06-20",
    "sources": [
      "https://arxiv.org/abs/2603.29640",
      "https://github.com/GAIR-NLP/ASI-Evolve"
    ]
  },
  {
    "id": "asa-coral",
    "date_added": "2026-06-20",
    "name": "CORAL",
    "category": "crossdomain",
    "domain": "Autonomous multi-agent open-ended problem solving",
    "access": "open-source",
    "autonomy": "A4",
    "inputs": "Open-ended problem specifications, shared persistent memory, agent task queues",
    "outputs": "Evolved solutions, collaborative reasoning traces, performance improvement metrics",
    "notes": "CORAL replaces fixed heuristics with long-running agents that explore, reflect, and collaborate through shared persistent memory and asynchronous execution, achieving 3-10x improvement rates over evolutionary baselines across open-ended tasks.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.01658"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Human-Agent-Society/CORAL"
      }
    ],
    "verified": "2026-06-20",
    "sources": [
      "https://arxiv.org/abs/2604.01658",
      "https://github.com/Human-Agent-Society/CORAL"
    ]
  },
  {
    "id": "asa-paperorchestra",
    "date_added": "2026-06-20",
    "name": "PaperOrchestra",
    "category": "crossdomain",
    "domain": "Autonomous scientific manuscript writing",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "Pre-writing materials (idea summaries, experimental logs, results)",
    "outputs": "Submission-ready LaTeX manuscripts with literature synthesis and generated figures",
    "notes": "Evaluated on PaperWritingBench across 200 top-tier AI conference papers, providing a structured benchmark for automated scientific writing.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.05018"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/google-research/paper-orchestra"
      }
    ],
    "verified": "2026-06-20",
    "sources": [
      "https://arxiv.org/abs/2604.05018",
      "https://github.com/google-research/paper-orchestra"
    ]
  },
  {
    "id": "asa-cassia",
    "date_added": "2026-06-20",
    "name": "CASSIA",
    "category": "biology",
    "domain": "Automated single-cell RNA-seq cell-type annotation",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "scRNA-seq marker gene lists per cluster",
    "outputs": "Cell-type annotations with confidence scores, validation reports, and formatted annotation summaries",
    "notes": "CASSIA coordinates five LLM agents for reference-free cell-type annotation of scRNA-seq data, producing interpretable outputs with quality scores and natural-language reports across diverse tissues and species.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-025-67084-x"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ElliotXie/CASSIA"
      }
    ],
    "verified": "2026-06-20",
    "sources": [
      "https://github.com/ElliotXie/CASSIA",
      "https://www.nature.com/articles/s41467-025-67084-x"
    ]
  },
  {
    "id": "asa-mira",
    "date_added": "2026-06-20",
    "name": "MIRA",
    "category": "biology",
    "domain": "Autonomous clinical diagnosis and treatment planning",
    "access": "paper-only",
    "autonomy": "A3-A4",
    "inputs": "Patient EHR data, clinical history, diagnostic test orders and results",
    "outputs": "Differential diagnoses, ordered tests, treatment plans",
    "notes": "Published in Nature, MIRA outperformed physicians in diagnostic accuracy operating autonomously within a sandboxed EHR environment.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41586-026-10675-5"
      }
    ],
    "repo_links": [],
    "verified": "2026-06-20",
    "sources": [
      "https://www.nature.com/articles/s41586-026-10675-5"
    ]
  },
  {
    "id": "asa-a-lab-gpss",
    "date_added": "2026-06-20",
    "name": "A-Lab GPSS",
    "category": "chemistry",
    "domain": "Autonomous solid-state inorganic materials synthesis",
    "access": "open-source",
    "autonomy": "A5",
    "inputs": "Synthesis targets, experimental parameters, and prior campaign results fed to an LLM reasoning agent controlling a glovebox robotic platform",
    "outputs": "Synthesized air-sensitive solid-state materials, characterization data, and iteratively updated synthesis plans",
    "notes": "Demonstrated autonomous discovery of lithium halide spinel ionic conductors across 352 synthesis campaigns using a self-driving glovebox robotic laboratory with integrated LLM-based experimental planning.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.11957"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/CederGroupHub/alab_gpss_public"
      }
    ],
    "verified": "2026-06-20",
    "sources": [
      "https://arxiv.org/abs/2604.11957",
      "https://github.com/CederGroupHub/alab_gpss_public"
    ]
  },
  {
    "id": "asa-rainbow",
    "date_added": "2026-06-20",
    "name": "Rainbow",
    "category": "chemistry",
    "domain": "Autonomous metal halide perovskite quantum dot synthesis",
    "access": "open-data",
    "autonomy": "A5",
    "inputs": "Synthesis parameters (precursor ratios, temperatures, reaction conditions) and real-time spectroscopic measurements from parallelized batch reactors",
    "outputs": "Optimized perovskite quantum dot nanocrystals with target optical properties; ML-updated experimental protocols",
    "notes": "Rainbow demonstrated closed-loop autonomous optimization of metal halide perovskite quantum dots across multiple robotic synthesis stations with real-time spectroscopic feedback, published in Nature Communications 2025.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://www.nature.com/articles/s41467-025-63209-4"
      }
    ],
    "repo_links": [
      {
        "label": "Zenodo",
        "url": "https://doi.org/10.5281/zenodo.15866667"
      }
    ],
    "verified": "2026-06-20",
    "sources": [
      "https://doi.org/10.5281/zenodo.15866667",
      "https://www.nature.com/articles/s41467-025-63209-4"
    ]
  },
  {
    "id": "asa-matterix",
    "date_added": "2026-06-20",
    "name": "MATTERIX",
    "category": "chemistry",
    "domain": "Robotics-assisted chemistry laboratory automation simulation",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "Robotic lab workflow specifications, material properties, reaction parameters",
    "outputs": "Multiscale simulations of robotic manipulation, powder/liquid dynamics, heat transfer, and reaction kinetics",
    "notes": "Agentic materials workflow with simulation/digital-twin closure and demonstrated sim-to-real transfer; it is not an autonomous physical laboratory.",
    "paper_links": [
      {
        "label": "Nature Computational Science",
        "url": "https://doi.org/10.1038/s43588-025-00924-4"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/AccelerationConsortium/Matterix"
      }
    ],
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://doi.org/10.1038/s43588-025-00924-4",
      "https://github.com/AccelerationConsortium/Matterix"
    ]
  },
  {
    "id": "asa-mada",
    "date_added": "2026-06-20",
    "name": "MADA",
    "category": "physics",
    "domain": "Autonomous inertial confinement fusion capsule design",
    "access": "paper-only",
    "autonomy": "A4",
    "inputs": "Fusion capsule design parameters, simulation code outputs (MARBL), ML surrogate model predictions",
    "outputs": "Optimized ICF capsule designs, multiphysics simulation results, design recommendations",
    "notes": "LLNL/NNSA computational agent that couples LLM reasoning, multiphysics codes, and ML surrogates on restricted HPC systems to explore inertial-confinement-fusion capsule designs. It does not operate a physical fusion experiment.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2510.17830"
      }
    ],
    "repo_links": [],
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2510.17830"
    ]
  },
  {
    "id": "asa-ai-cfd-scientist",
    "date_added": "2026-06-20",
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "name": "AI CFD Scientist",
    "category": "physics",
    "domain": "Autonomous computational fluid dynamics discovery",
    "access": "open-source",
    "autonomy": "A4",
    "inputs": "Scientific literature, CFD problem specifications, OpenFOAM source code",
    "outputs": "Validated CFD simulation results, modified OpenFOAM source, autonomous manuscript",
    "notes": "An end-to-end CFD research agent from RPI that chains literature-grounded ideation, hypothesis generation, OpenFOAM execution, vision-based physics verification, and autonomous manuscript writing in a single inspectable workflow.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2605.06607"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/csml-rpi/AI-CFD-Scientist"
      }
    ],
    "sources": [
      "https://arxiv.org/abs/2605.06607",
      "https://github.com/csml-rpi/AI-CFD-Scientist"
    ]
  },
  {
    "id": "asa-metachat",
    "date_added": "2026-06-20",
    "name": "MetaChat",
    "category": "physics",
    "domain": "Autonomous photonics and metasurface design",
    "access": "open-source",
    "autonomy": "A3-A4",
    "inputs": "Semantic natural-language design goals for photonic metasurfaces",
    "outputs": "Optimized freeform metasurface layouts with simulated performance metrics",
    "notes": "MetaChat uses a multi-agent Agentic Iterative Monologue paradigm with Maxwell surrogate solvers to autonomously design metalenses and beam deflectors at expert-level performance, demonstrated at Stanford on the Jon Fan lab's photonics platform.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2503.20479"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/jonfanlab/metachat"
      }
    ],
    "verified": "2026-06-20",
    "sources": [
      "https://arxiv.org/abs/2503.20479",
      "https://github.com/jonfanlab/metachat"
    ]
  },
  {
    "id": "asa-exp-bench",
    "date_added": "2026-06-20",
    "name": "EXP-Bench",
    "category": "benchmark",
    "domain": "Autonomous AI research experimentation",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Research question and incomplete starter code from AI papers",
    "outputs": "Hypothesis, experiment design, implementation, execution results, and analysis",
    "notes": "Covers 461 tasks derived from 51 top-tier AI papers, evaluating agents across the full experimental lifecycle from hypothesis formulation through result analysis.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.24785"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Just-Curieous/Curie/tree/main/benchmark/exp_bench"
      }
    ],
    "verified": "2026-06-20",
    "sources": [
      "https://arxiv.org/abs/2505.24785",
      "https://github.com/Just-Curieous/Curie/tree/main/benchmark/exp_bench"
    ]
  },
  {
    "id": "asa-mlr-bench",
    "date_added": "2026-06-20",
    "name": "MLR-Bench",
    "category": "benchmark",
    "domain": "Autonomous machine-learning research",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "ML research tasks drawn from NeurIPS/ICLR/ICML workshop papers; agent-generated ideas, proposals, experiment code, and paper drafts",
    "outputs": "Automated evaluation scores across four research stages via MLR-Judge",
    "notes": "A 201-task benchmark spanning four stages of the ML research lifecycle - idea generation, proposal formulation, experimentation, and paper writing - evaluated by an automated MLR-Judge grader.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.19955"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/chchenhui/mlrbench"
      }
    ],
    "verified": "2026-06-20",
    "sources": [
      "https://arxiv.org/abs/2505.19955",
      "https://github.com/chchenhui/mlrbench"
    ]
  },
  {
    "id": "asa-autoresearchbench",
    "date_added": "2026-06-20",
    "name": "AutoResearchBench",
    "category": "benchmark",
    "domain": "Scientific literature discovery and retrieval",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "arXiv paper corpus (3M+ papers), natural language research queries specifying target papers or criteria",
    "outputs": "Benchmark scores for Deep Research (locating specific papers) and Wide Research (comprehensively gathering matching papers) tasks",
    "notes": "A 1,000-instance benchmark spanning over 3 million arXiv papers that evaluates AI agents on two complementary scientific literature discovery modes: targeted retrieval of specific papers and comprehensive collection of papers matching given criteria.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.25256"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/CherYou/AutoResearchBench"
      }
    ],
    "verified": "2026-06-20",
    "sources": [
      "https://arxiv.org/abs/2604.25256",
      "https://github.com/CherYou/AutoResearchBench"
    ]
  },
  {
    "id": "asa-mls-bench",
    "date_added": "2026-06-20",
    "name": "MLS-Bench",
    "category": "benchmark",
    "domain": "Machine learning methods invention and generalization",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "ML task specifications across 12 domains, baseline implementations, and target components to improve",
    "outputs": "Agent performance scores on 140 tasks measuring ability to invent generalizable and scalable ML methods",
    "notes": "MLS-Bench evaluates AI systems across 140 tasks in 12 ML domains on whether they can invent genuinely new methods that generalize beyond the training context, rather than merely applying known techniques.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2605.08678"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Imbernoulli/MLS-Bench"
      }
    ],
    "verified": "2026-06-20",
    "sources": [
      "https://arxiv.org/abs/2605.08678",
      "https://github.com/Imbernoulli/MLS-Bench"
    ]
  },
  {
    "id": "asa-abc-bench",
    "date_added": "2026-06-20",
    "name": "ABC-Bench",
    "category": "benchmark",
    "domain": "Agentic biosecurity capabilities evaluation",
    "access": "paper-only",
    "autonomy": "B",
    "inputs": "LLM agent prompted with biosecurity-relevant tasks (liquid-handling robot code generation, DNA fragment design, synthesis screening evasion)",
    "outputs": "Task completion scores validated against wet-lab experimental outcomes",
    "notes": "ICML 2026 benchmark evaluating LLM agents on agentic biological capabilities relevant to biosecurity, covering robotic protocol generation, in vitro DNA assembly design, and synthesis screening evasion, with wet-lab validation.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2606.11150"
      }
    ],
    "repo_links": [],
    "verified": "2026-06-20",
    "sources": [
      "https://arxiv.org/abs/2606.11150"
    ]
  },
  {
    "id": "asa-researchcodebench",
    "date_added": "2026-06-20",
    "name": "ResearchCodeBench",
    "category": "benchmark",
    "domain": "Machine learning research implementation in code",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "ML research papers (2024-2025), LLM-generated code implementations",
    "outputs": "Pass/fail scores on 212 executable coding challenges derived from novel research contributions",
    "notes": "Benchmarks 30+ LLMs on implementing novel ML research contributions in executable code, with best models achieving under 40% accuracy across 212 challenges from top 2024-2025 papers.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2506.02314"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/PatrickHua/ResearchCodeBench"
      }
    ],
    "verified": "2026-06-20",
    "sources": [
      "https://arxiv.org/abs/2506.02314",
      "https://github.com/PatrickHua/ResearchCodeBench"
    ]
  },
  {
    "id": "asa-hypobench",
    "date_added": "2026-06-20",
    "name": "HypoBench",
    "category": "benchmark",
    "domain": "LLM-based hypothesis generation evaluation",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "LLM-generated hypotheses across 194 datasets spanning 7 real-world and 5 synthetic scientific tasks",
    "outputs": "Scores for discovery rate, generalizability, and practical utility of hypothesis generation systems",
    "notes": "Covers 194 datasets across 12 tasks to evaluate hypothesis generation systems on discovery rate, generalizability, and practical utility.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2504.11524"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ChicagoHAI/HypoBench-code"
      },
      {
        "label": "Project",
        "url": "https://chicagohai.github.io/HypoBench/"
      }
    ],
    "verified": "2026-07-13",
    "date_modified": "2026-07-13",
    "sources": [
      "https://arxiv.org/abs/2504.11524",
      "https://chicagohai.github.io/HypoBench/",
      "https://github.com/ChicagoHAI/HypoBench-code"
    ]
  },
  {
    "id": "asa-posttrainbench",
    "date_added": "2026-06-20",
    "name": "PostTrainBench",
    "category": "benchmark",
    "domain": "Autonomous LLM post-training (data curation, fine-tuning, RLHF)",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Base LLM, compute budget (10 h on H100), target benchmark (e.g. AIME, HumanEval)",
    "outputs": "Post-training pipeline performance scores on target benchmarks",
    "notes": "Evaluates whether frontier AI agents can autonomously execute the full post-training workflow - data curation, fine-tuning, and RLHF - within a fixed compute budget and without predefined strategies.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2603.08640"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/aisa-group/PostTrainBench"
      }
    ],
    "verified": "2026-06-20",
    "sources": [
      "https://arxiv.org/abs/2603.08640",
      "https://github.com/aisa-group/PostTrainBench"
    ]
  },
  {
    "id": "asa-aido-harness",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "AIDO.Harness",
    "category": "biology",
    "domain": "Autonomous biomedical machine-learning model development",
    "access": "paper-only",
    "autonomy": "A4",
    "inputs": "Natural-language biomedical modeling task, target metric, and dataset",
    "outputs": "Executable training and evaluation pipelines, trained models, and benchmark results",
    "notes": "Agentic system that selects modeling strategies, executes experiments, and iteratively revises code and configurations across biomedical benchmarks; no public implementation was confirmed.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.04.20.719735v1"
      }
    ],
    "repo_links": [],
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.04.20.719735v1"
    ]
  },
  {
    "id": "asa-microgrowagents",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "MicroGrowAgents",
    "category": "biology",
    "domain": "Microbial cultivation engineering and growth-media design",
    "access": "paper-only",
    "autonomy": "A3",
    "inputs": "Organism/genome context, structured biological knowledge, candidate ingredients, and cultivation outcomes",
    "outputs": "Evidence-backed media candidates, candidate or optimized experimental designs, and interpretable cultivation analyses",
    "notes": "Modular cultivation/media-design system combining knowledge graphs, metabolic modeling, and statistical design. Its author-owned repository was checked in a separate audit lane but returned repository-not-found during the final link gate, so no currently public implementation is claimed.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://www.biorxiv.org/content/10.64898/2026.06.04.729985v2"
      }
    ],
    "repo_links": [],
    "date_modified": "2026-07-13",
    "sources": [
      "https://www.biorxiv.org/content/10.64898/2026.06.04.729985v2"
    ]
  },
  {
    "id": "asa-scitrace",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "SciTrace",
    "category": "crossdomain",
    "domain": "Trajectory-aware safety framework integrated into scientific-agent pipelines",
    "access": "paper-only",
    "autonomy": "A4",
    "inputs": "Scientific-agent reasoning trajectories and proposed multi-step tool chains",
    "outputs": "Cumulative risk state and pre-execution safety decisions inside an autonomous research pipeline",
    "notes": "Safety framework woven through a Thinker/Experimenter/Writer/Reviewer scientific-agent pipeline. The paper evaluates risk tasks, but does not release a reusable named benchmark; this row represents the framework, not a B-class benchmark.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2606.08234"
      }
    ],
    "repo_links": [
      {
        "label": "Project",
        "url": "https://opensciagent.github.io/SciTrace/"
      }
    ],
    "sources": [
      "https://arxiv.org/abs/2606.08234",
      "https://opensciagent.github.io/SciTrace/"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "asa-agentbuild-rietveld",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "AgentBuild for Rietveld refinement",
    "category": "chemistry",
    "domain": "Provenance-aware construction of crystallography agents",
    "access": "paper-only",
    "autonomy": "A3",
    "inputs": "Scientist-authored rubric, difficulty-graded curriculum, knowledge base, model handle, and GSAS-II tools",
    "outputs": "Versioned A2A-packaged refinement agent and construction provenance trail",
    "notes": "Workflow stage that builds and gates a Rietveld-refinement agent from a scientist-authored contract; demonstrated on an LLZO signal-to-noise curriculum. No public implementation was confirmed.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2606.12834"
      }
    ],
    "repo_links": [],
    "sources": [
      "https://arxiv.org/abs/2606.12834"
    ]
  },
  {
    "id": "asa-self-evolving-fluid-control-agent",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "Self-Evolving Scientific Agent for fluid control",
    "category": "physics",
    "domain": "Autonomous physically reasoned controller discovery",
    "access": "paper-only",
    "autonomy": "A4",
    "inputs": "Seed control policy, simulation observations, target-reaching objectives, and multimodal simulation evidence",
    "outputs": "Iteratively revised source-code controllers and an auditable evolution log",
    "notes": "LLM-driven workflow that proposes, simulates, diagnoses, and refines interpretable control policies for an underactuated dogfish swimmer; no public implementation was confirmed.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2606.08405"
      }
    ],
    "repo_links": [],
    "sources": [
      "https://arxiv.org/abs/2606.08405"
    ]
  },
  {
    "id": "asa-ahois",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "AHOIS",
    "category": "physics",
    "domain": "Socratic autonomous discovery on high-dimensional optical systems",
    "access": "lab-gated",
    "autonomy": "A5",
    "inputs": "High-level research goal, shared scientific state, multimode-fibre measurements, and instrument controls",
    "outputs": "Falsifiable physical hypotheses, adaptive experiments, failure diagnoses, and experimentally validated findings",
    "notes": "Five-agent closed loop with a Socratic physics critic, hardware abstraction, integrity monitoring, modeling, and supervised decision gates; demonstrated on a real multimode-fibre optical platform. Human supervision remained at predefined decision points, and researchers performed repetitive full-scale acquisition after AHOIS validated the workflow at pilot scale.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2606.26722"
      }
    ],
    "repo_links": [],
    "sources": [
      "https://arxiv.org/abs/2606.26722"
    ]
  },
  {
    "id": "asa-aira-dojo",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "AIRA-dojo",
    "category": "crossdomain",
    "domain": "Autonomous machine-learning research and experimentation",
    "access": "source-available",
    "autonomy": "A4",
    "inputs": "MLE-bench task specification, datasets, compute budget, and model backend",
    "outputs": "Iteratively searched, executed, and scored ML solutions with full trajectories",
    "notes": "Meta/UCL framework formalizing research agents as search policies over candidate solutions, including greedy, MCTS, and evolutionary search. Public code is CC BY-NC 4.0 and therefore non-commercial.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2507.02554"
      },
      {
        "label": "OpenReview",
        "url": "https://openreview.net/forum?id=RwfrdKSgCE"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/facebookresearch/aira-dojo"
      }
    ],
    "access_evidence": {
      "software_license": "CC-BY-NC-4.0",
      "commercial_use": false,
      "source_url": "https://github.com/facebookresearch/aira-dojo"
    },
    "sources": [
      "https://arxiv.org/abs/2507.02554",
      "https://github.com/facebookresearch/aira-dojo",
      "https://openreview.net/forum?id=RwfrdKSgCE"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "asa-rd-agent",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "R&D-Agent (RD-Agent)",
    "category": "crossdomain",
    "domain": "Autonomous data-driven AI research and development",
    "access": "open-source",
    "autonomy": "A4",
    "inputs": "Data-science or ML task, datasets, evaluation feedback, and model backends",
    "outputs": "Research hypotheses, implemented solutions, experiments, and iteratively improved models",
    "notes": "Microsoft Research dual-agent framework in which a Researcher proposes ideas from performance feedback and a Developer implements and refines code across parallel evolving traces. MIT-licensed.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.14738"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/microsoft/RD-Agent"
      }
    ],
    "aliases": [
      "RD-Agent"
    ],
    "sources": [
      "https://arxiv.org/abs/2505.14738",
      "https://github.com/microsoft/RD-Agent"
    ]
  },
  {
    "id": "asa-mle-bench",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "MLE-bench",
    "category": "benchmark",
    "domain": "Machine-learning engineering agents",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Prepared Kaggle competition environments and agent submissions",
    "outputs": "Competition scores, medal rates, grading reports, and agent trajectories",
    "notes": "OpenAI benchmark covering 75 real-world ML engineering competitions. MIT-licensed code, task preparation, and graders; competition data retain Kaggle/provider terms. The official leaderboard is public but paused new submissions on 24 April 2026 while comparison procedures are revised.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2410.07095"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/openai/mle-bench"
      }
    ],
    "sources": [
      "https://arxiv.org/abs/2410.07095",
      "https://github.com/openai/mle-bench"
    ]
  },
  {
    "id": "asa-re-bench",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "RE-Bench",
    "category": "benchmark",
    "domain": "Frontier AI research-and-development agents",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Open-ended ML research-engineering environments with fixed time budgets",
    "outputs": "Objective task scores, human-expert comparisons, and agent trajectories",
    "notes": "METR benchmark comparing frontier agents with human experts on realistic long-horizon AI R&D tasks. MIT-licensed suite with protected solution material to limit contamination.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2411.15114"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/METR/RE-Bench"
      }
    ],
    "sources": [
      "https://arxiv.org/abs/2411.15114",
      "https://github.com/METR/RE-Bench"
    ]
  },
  {
    "id": "asa-researchclawbench",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "ResearchClawBench",
    "category": "benchmark",
    "domain": "End-to-end autonomous scientific research across ten disciplines",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Raw data, related literature, and a research goal with the target paper hidden",
    "outputs": "Executed analyses, figures, publication-style reports, and expert-rubric scores",
    "notes": "Forty real-science tasks requiring agents to work from raw data to research reports; includes a lightweight ResearchHarness baseline. MIT-licensed public repository.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2606.07591"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/InternScience/ResearchClawBench"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/datasets/InternScience/ResearchClawBench"
      }
    ],
    "sources": [
      "https://arxiv.org/abs/2606.07591",
      "https://github.com/InternScience/ResearchClawBench",
      "https://huggingface.co/datasets/InternScience/ResearchClawBench"
    ]
  },
  {
    "id": "asa-deepresearch-bench",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "DeepResearch Bench",
    "category": "benchmark",
    "domain": "Long-form evidence-grounded research agents across many fields",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Expert-authored research tasks requiring web evidence and long-form reports",
    "outputs": "Report-quality, citation-grounding, information-recall, analysis, and presentation scores",
    "notes": "Original long-form deep-research benchmark with PhD-level tasks and RACE/FACT evaluation. Split from DeepResearch Bench II because the projects have distinct papers, repositories and task/evaluation identities.",
    "paper_links": [
      {
        "label": "arXiv I",
        "url": "https://arxiv.org/abs/2506.11763"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub I",
        "url": "https://github.com/Ayanami0730/deep_research_bench"
      }
    ],
    "sources": [
      "https://arxiv.org/abs/2506.11763",
      "https://github.com/Ayanami0730/deep_research_bench"
    ],
    "date_modified": "2026-07-13"
  },
  {
    "id": "asa-prescience",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "PreScience",
    "category": "benchmark",
    "domain": "Forecasting scientific contributions and research trajectories",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Temporally aligned paper, author, reference, and citation histories",
    "outputs": "Collaborator, prior-work, contribution, and impact forecasts plus science-trajectory simulations",
    "notes": "Ai2/UChicago benchmark built from large-scale paper histories and a scientific graph. It evaluates forecasting and simulated scientific trajectories rather than laboratory action. Code is Apache-2.0 and data are ODC-BY 1.0.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2602.20459"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/allenai/prescience"
      },
      {
        "label": "Hugging Face",
        "url": "https://huggingface.co/datasets/allenai/prescience"
      }
    ],
    "sources": [
      "https://arxiv.org/abs/2602.20459",
      "https://github.com/allenai/prescience",
      "https://huggingface.co/datasets/allenai/prescience"
    ]
  },
  {
    "id": "asa-mle-dojo",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "MLE-Dojo",
    "category": "benchmark",
    "domain": "Interactive machine-learning engineering environments",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Consolidated Kaggle tasks, agent actions, execution environments, and competition feedback",
    "outputs": "Comparable interactive MLE trajectories and competition outcomes",
    "notes": "NeurIPS 2025 environment for evaluating machine-learning engineering agents. Code is MIT; benchmark data are non-commercial and each Kaggle competition retains its own terms.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2505.07782"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/MLE-Dojo/MLE-Dojo"
      }
    ],
    "sources": [
      "https://arxiv.org/abs/2505.07782",
      "https://github.com/MLE-Dojo/MLE-Dojo"
    ]
  },
  {
    "id": "asa-automedbench",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "AutoMedBench",
    "category": "benchmark",
    "domain": "Autonomous medical-AI research workflows",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Medical imaging and multimodal research task capsules",
    "outputs": "Executable pipelines, predictions, and stage-level workflow scores",
    "notes": "Benchmark of planning, setup, validation, inference, and submission across 48 sandboxed tasks and seven tracks. MIT-licensed harness; source datasets retain provider licences.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2606.01961"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/AutoMedBench/AutoMedBench"
      },
      {
        "label": "Project",
        "url": "https://automedbench.github.io/"
      }
    ],
    "sources": [
      "https://arxiv.org/abs/2606.01961",
      "https://automedbench.github.io/",
      "https://github.com/AutoMedBench/AutoMedBench"
    ]
  },
  {
    "id": "asa-biomedarena",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "BioMedArena",
    "category": "benchmark",
    "domain": "Biomedical deep-research agent evaluation",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Registered benchmarks, model backends, tools, and harness configurations",
    "outputs": "Scores, execution traces, and comparable agent evaluations",
    "notes": "MIT-licensed reproducible toolkit spanning benchmark loading, tool exposure, agent execution, context management, and scoring.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2605.06177"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/AI-in-Health/BioMedArena"
      }
    ],
    "sources": [
      "https://arxiv.org/abs/2605.06177",
      "https://github.com/AI-in-Health/BioMedArena"
    ]
  },
  {
    "id": "asa-flowbench-bioinformatics",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "FlowBench (agentic bioinformatics)",
    "category": "benchmark",
    "domain": "Autonomous bioinformatics workflow execution",
    "access": "open-source",
    "autonomy": "B",
    "inputs": "Bioinformatics tasks, datasets, workflow specifications, and induced failures",
    "outputs": "Planning, recovery, interpretation, and output-fidelity scores",
    "notes": "Agentic-bioinformatics benchmark sharing FlowAgent's GPL-3.0 repository and reproducible corpus. Distinct from the unrelated 2024 benchmark also named FlowBench and from the FlowAgent system row.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://doi.org/10.64898/2026.06.12.731844"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/EnteloBio/flowagent"
      }
    ],
    "sources": [
      "https://doi.org/10.64898/2026.06.12.731844",
      "https://github.com/EnteloBio/flowagent"
    ]
  },
  {
    "id": "asa-autozyme",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "AutoZyme",
    "category": "biology",
    "domain": "Autonomous bioinformatics software optimization",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "Scientific software functions, tests, datasets, and optimization objectives",
    "outputs": "Benchmarked, output-preserving optimized code and packaged patches",
    "notes": "MIT-licensed workflow that iteratively profiles, edits, tests, and retains performance improvements. It optimizes scientific software rather than autonomously generating biological hypotheses.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://doi.org/10.64898/2026.06.12.731250"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ElliotXie/autozyme"
      },
      {
        "label": "Project",
        "url": "https://autozyme.com/"
      },
      {
        "label": "Dataset",
        "url": "https://huggingface.co/datasets/elliotxie/autozyme-datasets"
      }
    ],
    "sources": [
      "https://autozyme.com/",
      "https://doi.org/10.64898/2026.06.12.731250",
      "https://github.com/ElliotXie/autozyme",
      "https://huggingface.co/datasets/elliotxie/autozyme-datasets"
    ]
  },
  {
    "id": "asa-simple-agent-optimization-biomedical-imaging",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "Simple Agent Optimization (biomedical imaging)",
    "category": "biology",
    "domain": "Autonomous biomedical-imaging workflow optimization",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "Imaging datasets, validation metrics, and wrapped scientific models",
    "outputs": "Iteratively generated preprocessing/postprocessing code and optimized workflows",
    "notes": "CVPR 2026 system for Polaris, Cellpose, and MedSAM workflows. Runnable implementation and tests are Apache-2.0.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2512.06006"
      },
      {
        "label": "CVPR",
        "url": "https://openaccess.thecvf.com/content/CVPR2026/papers/Wang_Simple_Agents_Outperform_Experts_in_Biomedical_Imaging_Workflow_Optimization_CVPR_2026_paper.pdf"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/xuefei-wang/simple-agent-opt"
      }
    ],
    "sources": [
      "https://arxiv.org/abs/2512.06006",
      "https://github.com/xuefei-wang/simple-agent-opt",
      "https://openaccess.thecvf.com/content/CVPR2026/papers/Wang_Simple_Agents_Outperform_Experts_in_Biomedical_Imaging_Workflow_Optimization_CVPR_2026_paper.pdf"
    ]
  },
  {
    "id": "asa-medgenesis",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "MedGenesis",
    "category": "biology",
    "domain": "Autonomous clinical and translational research",
    "access": "paper-only",
    "autonomy": "A4",
    "inputs": "Research questions, clinical datasets, evidence resources, and institutional data handles",
    "outputs": "Hypotheses, cohort analyses, evidence products, reproductions, and reports",
    "notes": "Eight-agent in-silico closed loop with a large skill library. Some functions depend on protected EHRs and institutional wrappers; code and benchmarks are promised only upon publication, and the demonstrated wet-lab handoff does not justify A5.",
    "paper_links": [
      {
        "label": "medRxiv",
        "url": "https://doi.org/10.64898/2026.06.14.26355612"
      }
    ],
    "repo_links": [],
    "sources": [
      "https://doi.org/10.64898/2026.06.14.26355612"
    ]
  },
  {
    "id": "asa-behaveagent",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "BehaveAgent",
    "category": "biology",
    "domain": "Autonomous multimodal organism-behavior analysis",
    "access": "paper-only",
    "autonomy": "A3",
    "inputs": "Behavioral videos and research questions",
    "outputs": "Tracking, temporal segmentation, generated analyses, and reports",
    "notes": "Cross-species behavior-analysis system that selects strategies, generates and executes analysis code, and reports results. The official repository is MIT-licensed but explicitly says end-user code is still forthcoming, so it is not classified as runnable open source.",
    "paper_links": [
      {
        "label": "bioRxiv",
        "url": "https://doi.org/10.1101/2025.05.15.653585"
      }
    ],
    "repo_links": [
      {
        "label": "Project repository",
        "url": "https://github.com/LiuLab-Bioelectronics-Harvard/BehaveAgent"
      }
    ],
    "sources": [
      "https://doi.org/10.1101/2025.05.15.653585",
      "https://github.com/LiuLab-Bioelectronics-Harvard/BehaveAgent"
    ]
  },
  {
    "id": "asa-autonomous-mobile-robots-exploratory-synthesis",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "Autonomous mobile robots for exploratory synthetic chemistry",
    "category": "chemistry",
    "domain": "Modular mobile-robot exploratory synthesis",
    "access": "lab-gated",
    "autonomy": "A5",
    "inputs": "Scientist-selected building blocks, reaction space, and general success criteria",
    "outputs": "Robot-executed synthesis, NMR/UPLC-MS interpretation, reproducibility checks, and functional assays",
    "notes": "Distinct 2024 multi-robot successor to the 2020 Liverpool mobile robotic chemist. It ran a real multi-instrument physical workflow for four days; humans handled restocking and final anomaly interpretation. Author-owned control artifacts and workflow data are public on Zenodo.",
    "paper_links": [
      {
        "label": "Nature",
        "url": "https://doi.org/10.1038/s41586-024-08173-7"
      }
    ],
    "repo_links": [
      {
        "label": "Decision/instrument code",
        "url": "https://doi.org/10.5281/zenodo.11209893"
      },
      {
        "label": "Data/workflows",
        "url": "https://doi.org/10.5281/zenodo.11197259"
      }
    ],
    "sources": [
      "https://doi.org/10.1038/s41586-024-08173-7",
      "https://doi.org/10.5281/zenodo.11197259",
      "https://doi.org/10.5281/zenodo.11209893"
    ]
  },
  {
    "id": "asa-flex-cat",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "Flex-Cat",
    "category": "chemistry",
    "domain": "Autonomous homogeneous-catalyst discovery and optimization",
    "access": "lab-gated",
    "autonomy": "A5",
    "inputs": "Ligand library, reaction-variable ranges, and regioselectivity/turnover objectives",
    "outputs": "Robotically executed pressurized reactions, adaptive Bayesian campaigns, catalyst-condition maps, and scale-up validation",
    "notes": "Glovebox-integrated physical closed loop completing 680 experiments across three campaigns. Public Zenodo materials include data, logs, digital twin, and optimization code under CC BY 4.0; the paper-declared GitHub was unavailable at verification time.",
    "paper_links": [
      {
        "label": "Nature Communications",
        "url": "https://doi.org/10.1038/s41467-026-74425-x"
      }
    ],
    "repo_links": [
      {
        "label": "Zenodo",
        "url": "https://doi.org/10.5281/zenodo.18930287"
      }
    ],
    "sources": [
      "https://doi.org/10.1038/s41467-026-74425-x",
      "https://doi.org/10.5281/zenodo.18930287"
    ]
  },
  {
    "id": "asa-robochem-flex",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "RoboChem-Flex",
    "category": "chemistry",
    "domain": "Modular self-driving reaction-optimization laboratory",
    "access": "lab-gated",
    "autonomy": "A5",
    "inputs": "Reaction, controllable parameter space, analytical method, and optimization target",
    "outputs": "Automated experiments, online analysis, Bayesian optimization, and optimized conditions",
    "notes": "Real physical closed-loop reaction-optimization platform demonstrated across multiple chemistry cases; human-in-the-loop mode is optional. Apache-2.0 repository includes control software, firmware, CAD/PCB assets, and examples.",
    "paper_links": [
      {
        "label": "Nature Synthesis",
        "url": "https://doi.org/10.1038/s44160-026-01053-0"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/Noel-Research-Group/Robochem_Flex"
      }
    ],
    "sources": [
      "https://doi.org/10.1038/s44160-026-01053-0",
      "https://github.com/Noel-Research-Group/Robochem_Flex"
    ]
  },
  {
    "id": "asa-self-driving-pvd",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "Self-driving physical vapor deposition system",
    "category": "chemistry",
    "domain": "Autonomous sample-adaptive thin-film deposition",
    "access": "lab-gated",
    "autonomy": "A5",
    "inputs": "Requested optical properties, deposition variables, and per-sample calibration measurements",
    "outputs": "Robotically deposited silver films, in-situ spectra, updated models, and sample-specific decisions",
    "notes": "Genuine autonomous physical PVD loop demonstrated across 72 samples without human intervention during campaigns. Code and data are available only from the authors on request.",
    "paper_links": [
      {
        "label": "npj Computational Materials",
        "url": "https://doi.org/10.1038/s41524-025-01805-0"
      }
    ],
    "repo_links": [],
    "aliases": [
      "Self-driving PVD system"
    ],
    "sources": [
      "https://doi.org/10.1038/s41524-025-01805-0"
    ]
  },
  {
    "id": "asa-sparksmatter",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "SparksMatter",
    "category": "chemistry",
    "domain": "Autonomous in-silico inorganic-materials discovery",
    "access": "open-source",
    "autonomy": "A4",
    "inputs": "Materials objective, Materials Project data, generative/predictive tools, and validation constraints",
    "outputs": "Candidate structures, simulated-property evidence, critiques, validation plans, and research reports",
    "notes": "Apache-2.0 computational system that generates candidates and executes generative-model and surrogate workflows. DFT and physical synthesis are proposed follow-ups rather than performed closure. Distinct from the protein-design system Sparks.",
    "paper_links": [
      {
        "label": "npj Computational Materials",
        "url": "https://doi.org/10.1038/s41524-026-02205-8"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/lamm-mit/SparksMatter"
      }
    ],
    "sources": [
      "https://doi.org/10.1038/s41524-026-02205-8",
      "https://github.com/lamm-mit/SparksMatter"
    ]
  },
  {
    "id": "asa-ai-x-ray-scientist",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "AI X-ray Scientist",
    "category": "physics",
    "domain": "Autonomous synchrotron diffraction alignment",
    "access": "lab-gated",
    "autonomy": "A5",
    "inputs": "Crystal information, beamline state, detector images, and experiment goal",
    "outputs": "Reflection selection, motor-scan commands, anomaly diagnosis, and crystal-orientation matrix",
    "notes": "Developed in a virtual six-circle diffractometer and transferred to the real SSRL BL17-2 beamline, where it aligned crystals and adapted to an unexpected motor offset. A human relayed commands as a passive safety intermediary. Code MIT; data CC BY 4.0.",
    "paper_links": [
      {
        "label": "Nature Machine Intelligence",
        "url": "https://doi.org/10.1038/s42256-026-01261-5"
      }
    ],
    "repo_links": [
      {
        "label": "Code",
        "url": "https://doi.org/10.5281/zenodo.20017991"
      },
      {
        "label": "Data",
        "url": "https://doi.org/10.5281/zenodo.20017861"
      }
    ],
    "sources": [
      "https://doi.org/10.1038/s42256-026-01261-5",
      "https://doi.org/10.5281/zenodo.20017861",
      "https://doi.org/10.5281/zenodo.20017991"
    ]
  },
  {
    "id": "asa-calms-instrument-agents",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "CALMS scientific-instrument agents",
    "category": "physics",
    "domain": "Teachable multi-agent operation of scientific user-facility instruments",
    "access": "lab-gated",
    "autonomy": "A5",
    "inputs": "User task, instrument APIs/protocols, multimodal observations, and optional human corrections",
    "outputs": "Generated and executed instrument workflows, persistent operational memories, and experiment results",
    "notes": "Human-teachable physical agents validated on an X-ray nanoprobe beamline and an autonomous materials-design robot. They close real instrument loops but are not fully self-motivated scientists. DOE software record identifies BSD-3-Clause.",
    "paper_links": [
      {
        "label": "npj Computational Materials",
        "url": "https://doi.org/10.1038/s41524-026-02005-0"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/AdvancedPhotonSource/CALMS/tree/sdl_agents"
      },
      {
        "label": "DOE software record",
        "url": "https://doi.org/10.11578/dc.20240410.1"
      }
    ],
    "sources": [
      "https://doi.org/10.1038/s41524-026-02005-0",
      "https://doi.org/10.11578/dc.20240410.1",
      "https://github.com/AdvancedPhotonSource/CALMS/tree/sdl_agents"
    ]
  },
  {
    "id": "asa-argoloom",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "ArgoLOOM",
    "category": "physics",
    "domain": "Cross-frontier fundamental-physics computation",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "Physics goal, theory constraints, literature knowledge base, and tool configurations",
    "outputs": "Planned and executed cosmology, collider, deep-inelastic-scattering, and nuclear-physics analyses",
    "notes": "MIT-licensed computational pilot spanning several fundamental-physics domains. Demonstrations are small-scale and do not include physical experiment control.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2510.02426"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/ML4HEP-Theory/ArgoLOOM"
      }
    ],
    "sources": [
      "https://arxiv.org/abs/2510.02426",
      "https://github.com/ML4HEP-Theory/ArgoLOOM"
    ]
  },
  {
    "id": "asa-pde-agents",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "PDE-Agents",
    "category": "physics",
    "domain": "Knowledge-graph-grounded autonomous PDE/FEM simulation",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "Natural-language simulation task, material data, scientific references, and prior-run graph",
    "outputs": "Validated DOLFINx simulations, analyses, provenance-linked records, and reports",
    "notes": "MIT-licensed, containerized computational system with Simulation, Analytics, and Database agents. It performs runtime validation and debugging but does not control physical experiments.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2606.07850"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/MatPro-IFE/pde-agents"
      }
    ],
    "sources": [
      "https://arxiv.org/abs/2606.07850",
      "https://github.com/MatPro-IFE/pde-agents"
    ]
  },
  {
    "id": "asa-llmsat",
    "date_added": "2026-07-13",
    "verified": "2026-07-13",
    "name": "LLMSat",
    "category": "physics",
    "domain": "Goal-oriented autonomous spacecraft control",
    "access": "open-source",
    "autonomy": "A3",
    "inputs": "Mission goal, spacecraft state, available actions, and operational constraints",
    "outputs": "Mission plans, replanning decisions, and simulated spacecraft commands",
    "notes": "MIT-licensed implementation evaluated entirely in Kerbal Space Program. It demonstrates simulated spacecraft closure and has not been flight-tested.",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2405.01392"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/DM1122/LLMSat"
      }
    ],
    "sources": [
      "https://arxiv.org/abs/2405.01392",
      "https://github.com/DM1122/LLMSat"
    ]
  },
  {
    "id": "asa-sciskillbench",
    "date_added": "2026-07-13",
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "name": "SciSkillBench",
    "category": "benchmark",
    "domain": "Materials-science and chemistry agent skill benchmark",
    "paper_links": [
      {
        "label": "CASCADE paper",
        "url": "https://arxiv.org/abs/2512.23880"
      }
    ],
    "repo_links": [
      {
        "label": "Figshare dataset",
        "url": "https://figshare.com/articles/dataset/SkillSciBench_CASCADE_Benchmark_for_Evaluating_LLM_Agents_on_Scientific_Tasks/30924998"
      }
    ],
    "access": "open-data",
    "inputs": "116 materials-science and chemistry research tasks",
    "outputs": "Task-success and acquired-skill evaluation results",
    "autonomy": "B",
    "notes": "Named benchmark used to evaluate CASCADE; split from the scientific-agent system so B is reserved for the harness.",
    "sources": [
      "https://arxiv.org/abs/2512.23880",
      "https://figshare.com/articles/dataset/SkillSciBench_CASCADE_Benchmark_for_Evaluating_LLM_Agents_on_Scientific_Tasks/30924998"
    ]
  },
  {
    "id": "asa-stem2mat-bench",
    "date_added": "2026-07-13",
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "name": "STEM2Mat-Bench",
    "category": "benchmark",
    "domain": "Microscopy-to-atomistic reconstruction benchmark",
    "paper_links": [
      {
        "label": "AutoMat paper",
        "url": "https://arxiv.org/abs/2505.12650"
      }
    ],
    "repo_links": [
      {
        "label": "Official repository and dataset",
        "url": "https://github.com/yyt-2378/AutoMat"
      }
    ],
    "access": "open-data",
    "inputs": "STEM images paired with reference crystal structures",
    "outputs": "Lattice RMSD, formation-energy MAE and structure-matching success rate",
    "autonomy": "B",
    "notes": "Dedicated benchmark introduced with AutoMat; split from the A3 pipeline so system capability and evaluation evidence are not flattened into one row.",
    "sources": [
      "https://arxiv.org/abs/2505.12650",
      "https://github.com/yyt-2378/AutoMat"
    ]
  },
  {
    "id": "asa-deepresearch-bench-ii",
    "date_added": "2026-07-13",
    "date_modified": "2026-07-13",
    "verified": "2026-07-13",
    "name": "DeepResearch Bench II",
    "category": "benchmark",
    "domain": "Long-form evidence-grounded deep-research agents",
    "paper_links": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2601.08536"
      }
    ],
    "repo_links": [
      {
        "label": "GitHub",
        "url": "https://github.com/imlrz/DeepResearch-Bench-II"
      }
    ],
    "access": "open-source",
    "inputs": "Expert-derived deep-research tasks and binary rubrics",
    "outputs": "Rubric-level research-report quality and evidence-grounding scores",
    "autonomy": "B",
    "notes": "Second independently released DeepResearch Bench project; separated from the original to preserve paper, code and benchmark identity.",
    "sources": [
      "https://arxiv.org/abs/2601.08536",
      "https://github.com/imlrz/DeepResearch-Bench-II"
    ]
  }
]
