{
  "$schema_doc": "https://gtmstacker.com/registry/schema/entry.schema.json",
  "stability": "emerging",
  "generator": "agentic-media-registry",
  "generated_at": "2026-09-11T00:00:00Z",
  "id": "com.gtmstacker.registry/tool/hyper-tau-bench",
  "type": "tool",
  "slug": "hyper-tau-bench",
  "canonical_url": "https://gtmstacker.com/registry/tool/hyper-tau-bench/",
  "title": "Hyper-τ-Bench",
  "description": "Open-source benchmark (MIT, Python) from Sierra Research that tests whether an AI coding agent can BUILD a working customer-service agent — not just operate one: the developer agent gets a sandbox with business records, a codebase and a production API, and must design, wire and ship a functioning agent scored against simulated customers.",
  "category": "mcp-agents",
  "tags": [
    "mcp-agents",
    "agent-evaluation",
    "benchmark",
    "agent-construction",
    "customer-service"
  ],
  "status": "active",
  "revision": 1,
  "content_hash": "b239b77ec1152b058d6a73e2fe4cfe60a1ed0b633fbd4d078a651fef633a7214",
  "date_published": "2026-09-11T00:00:00Z",
  "date_modified": "2026-09-11T00:00:00Z",
  "source": {
    "name": "github · sierra-research/hyper-tau-bench",
    "url": "https://github.com/sierra-research/hyper-tau-bench"
  },
  "license": "MIT",
  "one_liner": "Hyper-τ-Bench is Sierra's open-source benchmark for whether an AI agent can research, design and build a working customer-service agent end to end.",
  "open_source": "yes",
  "self_hostable": "yes",
  "pricing_model": "free",
  "who_its_for": "Teams deciding whether 'have an agent build our support/service agent' is real yet — the benchmark quantifies the gap between automated construction and expert-built systems.",
  "aliases": [
    "hyper-tau-bench",
    "Hyper-τ-Bench"
  ],
  "alternatives": [],
  "secondary_categories": [
    "ai-infrastructure"
  ],
  "last_verified": "2026-09-11",
  "evidence": {
    "claim_type": "mixed",
    "source_id": "https://github.com/sierra-research/hyper-tau-bench",
    "note": "MIT + activity WebFetch-verified 2026-09-11 (20★, 4 commits, arXiv paper Sep 4 2026); pass-rate figures are Sierra's reported results."
  },
  "caveats": "EARLY as a repo (20★) but from an established research group (τ-bench lineage). Sierra's own numbers show the ceiling: best automated setup passed 23.9% of simulations vs 82.2% for an expert-built reference — read it as evidence agent-built agents are not turnkey yet. Requires Docker + Python 3.12.",
  "lead": "Open-source benchmark (MIT, Python) from Sierra Research that tests whether an AI coding agent can BUILD a working customer-service agent — not just operate one: the developer agent gets a sandbox with business records, a codebase and a production API, and must design, wire and ship a functioning agent scored against simulated customers.",
  "chunks": [
    {
      "index": 0,
      "heading_path": [],
      "est_tokens": 85,
      "text": "Open-source benchmark (MIT, Python) from Sierra Research that tests whether an AI coding agent can BUILD a working customer-service agent — not just operate one: the developer agent gets a sandbox with business records, a codebase and a production API, and must design, wire and ship a functioning agent scored against simulated customers."
    },
    {
      "index": 1,
      "heading_path": [
        null,
        "Provenance"
      ],
      "est_tokens": 201,
      "text": "Python) from Sierra Research that tests whether an AI coding agent can BUILD a working customer-service agent — not just operate one: the developer agent gets a sandbox with business records, a codebase and a production API, and must design, wire and ship a functioning agent scored against simulated customers.\n\n- MIT independently WebFetch-verified 2026-09-11 (20★, 4 commits, active; supports Codex, Claude Code, OpenCode harnesses; public leaderboard). Companion 41-page paper submitted to arXiv 2026-09-04. Pass-rate figures (23.9% best automated vs 82.2% expert reference) are Sierra's reported results, not independently reproduced.\n- Surfaced via the 2026-09-11 viral-posts brief; curated from the GTM Stacker signal registry (2026-09-11 pass); license independently WebFetch-verified 2026-09-11."
    },
    {
      "index": 2,
      "heading_path": [
        null,
        "Why it matters for a GTM stack"
      ],
      "est_tokens": 177,
      "text": "Codex, Claude Code, OpenCode harnesses; public leaderboard). Companion 41-page paper submitted to arXiv 2026-09-04. Pass-rate figures (23.9% best automated vs 82.2% expert reference) are Sierra's reported results, not independently reproduced. - Surfaced via the 2026-09-11 viral-posts brief; curated from the GTM Stacker signal registry (2026-09-11 pass); license independently WebFetch-verified 2026-09-11.\n\n\"An agent will build your support agent\" is the pitch behind a wave of GTM tooling. This is the first open yardstick for that exact claim, and the current numbers argue for buying expert-built or building carefully — useful evidence in any build-vs-buy conversation about agentic customer service."
    }
  ],
  "alternates": {
    "markdown": "https://gtmstacker.com/registry/tool/hyper-tau-bench/index.md",
    "html": "https://gtmstacker.com/registry/tool/hyper-tau-bench/",
    "json": "https://gtmstacker.com/registry/tool/hyper-tau-bench/index.json",
    "server_json": "https://gtmstacker.com/registry/tool/hyper-tau-bench/server.json"
  },
  "jsonld": {
    "@context": "https://schema.org",
    "@graph": [
      {
        "@type": "WebSite",
        "@id": "https://gtmstacker.com/#website",
        "url": "https://gtmstacker.com/",
        "name": "GTM Stacker Agent Registry",
        "description": "A daily-updated, agent-native registry of open-source tool discoveries, tool updates, and curated news for the go-to-market / RevOps engineering niche. Machine-readable first: agents can discover, parse, page, and delta-sync it without scraping HTML.",
        "inLanguage": "en",
        "publisher": {
          "@id": "https://gtmstacker.com/#organization"
        }
      },
      {
        "@type": "Organization",
        "@id": "https://gtmstacker.com/#organization",
        "name": "GTM Stacker",
        "url": "https://gtmstacker.com",
        "description": "The growth-systems practice of Theo Popov: AI-native enrichment, outbound, content engines and internal tooling for startups and venture programs. Its agent-native media property, the GTM Stacker Agent Registry, maintains a daily-updated catalog of open-source go-to-market and RevOps tools that both people and AI engines can discover, compare, and cite.",
        "foundingDate": "2024-08",
        "knowsAbout": [
          "go-to-market engineering",
          "RevOps",
          "sales automation",
          "marketing operations",
          "open-source software",
          "AI agents"
        ],
        "founder": {
          "@type": "Person",
          "@id": "https://gtmstacker.com/#founder",
          "name": "Theo Popov",
          "jobTitle": "Growth Operations & GTM Systems",
          "url": "https://gtmstacker.com/about/",
          "sameAs": [
            "https://www.linkedin.com/in/theo-popov",
            "https://x.com/Theo_Popov",
            "https://github.com/theopopov"
          ],
          "worksFor": {
            "@id": "https://gtmstacker.com/#organization"
          }
        },
        "sameAs": [
          "https://www.linkedin.com/company/gtmstacker"
        ],
        "mainEntityOfPage": "https://gtmstacker.com/registry/about/"
      },
      {
        "@type": "SoftwareApplication",
        "@id": "https://gtmstacker.com/registry/tool/hyper-tau-bench/#software",
        "name": "Hyper-τ-Bench",
        "identifier": "io.github.sierra-research/hyper-tau-bench",
        "description": "Open-source benchmark (MIT, Python) from Sierra Research that tests whether an AI coding agent can BUILD a working customer-service agent — not just operate one: the developer agent gets a sandbox with business records, a codebase and a production API, and must design, wire and ship a functioning agent scored against simulated customers.",
        "applicationCategory": "DeveloperApplication",
        "url": "https://gtmstacker.com/registry/tool/hyper-tau-bench/",
        "datePublished": "2026-09-11T00:00:00Z",
        "dateModified": "2026-09-11T00:00:00Z",
        "isPartOf": {
          "@id": "https://gtmstacker.com/#website"
        },
        "license": "https://spdx.org/licenses/MIT.html",
        "codeRepository": "https://github.com/sierra-research/hyper-tau-bench",
        "keywords": "mcp-agents, ai-infrastructure, agent-evaluation, benchmark, agent-construction, customer-service",
        "author": {
          "@type": "Organization",
          "name": "sierra-research",
          "url": "https://github.com/sierra-research",
          "sameAs": [
            "https://github.com/sierra-research/hyper-tau-bench"
          ]
        },
        "offers": {
          "@type": "Offer",
          "price": 0,
          "priceCurrency": "USD"
        }
      },
      {
        "@type": "BreadcrumbList",
        "@id": "https://gtmstacker.com/registry/tool/hyper-tau-bench/#breadcrumb",
        "itemListElement": [
          {
            "@type": "ListItem",
            "position": 1,
            "name": "GTM Stacker Registry",
            "item": "https://gtmstacker.com/registry/"
          },
          {
            "@type": "ListItem",
            "position": 2,
            "name": "MCP Agents",
            "item": "https://gtmstacker.com/registry/category/mcp-agents/"
          },
          {
            "@type": "ListItem",
            "position": 3,
            "name": "Hyper-τ-Bench",
            "item": "https://gtmstacker.com/registry/tool/hyper-tau-bench/"
          }
        ]
      }
    ]
  },
  "tool": {
    "name": "io.github.sierra-research/hyper-tau-bench",
    "repository": {
      "url": "https://github.com/sierra-research/hyper-tau-bench",
      "source": "github"
    }
  }
}
