{
  "$schema_doc": "https://gtmstacker.com/registry/schema/entry.schema.json",
  "stability": "emerging",
  "generator": "agentic-media-registry",
  "generated_at": "2026-09-14T00:00:00Z",
  "id": "com.gtmstacker.registry/tool/voicebox",
  "type": "tool",
  "slug": "voicebox",
  "canonical_url": "https://gtmstacker.com/registry/tool/voicebox/",
  "title": "Voicebox",
  "description": "Open-source, fully-local AI voice studio (MIT, Python + React) from Jamie Pine: clone a voice from a short sample, generate speech across 7 TTS engines and 23 languages, and dictate into any text field with a global hotkey (a Wispr Flow alternative). The GTM hook is one MCP tool call, voicebox.speak, that lets any MCP-aware agent talk back in a voice you cloned. Models, voice data and captures never leave the machine; runs on Apple Silicon (MLX) or CUDA/ROCm/CPU (PyTorch), macOS and Windows.",
  "category": "content-copywriting",
  "tags": [
    "content-copywriting",
    "local-first",
    "voice-cloning",
    "mcp-agents",
    "dictation"
  ],
  "status": "active",
  "revision": 1,
  "content_hash": "650c9d87bc96ce2137838ccdaf7c303fa14cf5be8fcc4c5f88007ea35bf2ca89",
  "date_published": "2026-09-14T00:00:00Z",
  "date_modified": "2026-09-14T00:00:00Z",
  "source": {
    "name": "github · jamiepine/voicebox",
    "url": "https://github.com/jamiepine/voicebox"
  },
  "license": "MIT",
  "one_liner": "Voicebox clones a voice, dictates into any field, and gives any MCP-aware agent a voice via a single voicebox.speak tool call, all running locally.",
  "open_source": "yes",
  "self_hostable": "yes",
  "pricing_model": "free",
  "who_its_for": "Founders and operators who want narration, dictation and an agent's spoken output on their own hardware with no per-character meter, and agent builders who want to give an MCP agent a real voice with one tool call.",
  "aliases": [
    "Voicebox",
    "voicebox"
  ],
  "alternatives": [
    "voicestudio"
  ],
  "secondary_categories": [
    "mcp-agents",
    "productivity-knowledge"
  ],
  "last_verified": "2026-09-14",
  "evidence": {
    "claim_type": "mixed",
    "source_id": "https://github.com/jamiepine/voicebox",
    "note": "MIT license, local-first design, the 7-engine / 23-language range, dictation, and the voicebox.speak MCP tool call confirmed on the repo (WebFetch 2026-09-14, ~53.2k stars, 638 commits, by Jamie Pine / jamiepine, the Spacedrive founder). Engine list and Whisper transcription are from project docs. Distinct project from VoiceStudio (different repo, MIT vs AGPL-3.0); the viral 'ElevenLabs + Wispr Flow' post named Voicebox specifically."
  },
  "caveats": "Distinct from VoiceStudio in the registry — do not conflate them. Local voice quality depends on your hardware and the engine you pick, and the bundled TTS models carry their own licenses, so check the ones your pipeline uses. Cloning a voice needs the consent of its owner; that obligation is yours, not the software's.",
  "lead": "Open-source, fully-local AI voice studio (MIT, Python + React) from Jamie Pine: clone a voice from a short sample, generate speech across 7 TTS engines and 23 languages, and dictate into any text field with a global hotkey (a Wispr Flow alternative). The GTM hook is one MCP tool call,…",
  "chunks": [
    {
      "index": 0,
      "heading_path": [],
      "est_tokens": 124,
      "text": "Open-source, fully-local AI voice studio (MIT, Python + React) from Jamie Pine: clone a voice from a short sample, generate speech across 7 TTS engines and 23 languages, and dictate into any text field with a global hotkey (a Wispr Flow alternative). The GTM hook is one MCP tool call, voicebox.speak, that lets any MCP-aware agent talk back in a voice you cloned. Models, voice data and captures never leave the machine; runs on Apple Silicon (MLX) or CUDA/ROCm/CPU (PyTorch), macOS and Windows."
    },
    {
      "index": 1,
      "heading_path": [
        null,
        "Provenance"
      ],
      "est_tokens": 306,
      "text": "field with a global hotkey (a Wispr Flow alternative). The GTM hook is one MCP tool call, voicebox.speak, that lets any MCP-aware agent talk back in a voice you cloned. Models, voice data and captures never leave the machine; runs on Apple Silicon (MLX) or CUDA/ROCm/CPU (PyTorch), macOS and Windows.\n\n- MIT license, local-first design, the 7-engine / 23-language range, dictation, and the voicebox.speak MCP tool independently WebFetch-verified on the repo 2026-09-14 (github.com/jamiepine/voicebox, ~53.2k stars, 638 commits, published by Jamie Pine, the Spacedrive founder). Engine and transcription details are from the project docs.\n- Surfaced via the 2026-09-14 viral-posts brief (\"52,000+ stars, open-source alternative to ElevenLabs + Wispr Flow, it's called Voicebox\"). Note: many mirror repos of the name exist; the canonical one verified here is jamiepine/voicebox.\n- Distinct from VoiceStudio (upd0912-voicestudio, AGPL-3.0). Both are local ElevenLabs alternatives; Voicebox adds Wispr-Flow-style dictation and the MCP agent voice, under a more permissive MIT license.\n- Curated from the GTM Stacker signal registry (2026-09-14 pass: daily pull + viral-posts brief); license independently verified 2026-09-14."
    },
    {
      "index": 2,
      "heading_path": [
        null,
        "Why it matters for a GTM stack"
      ],
      "est_tokens": 257,
      "text": "one verified here is jamiepine/voicebox. - Distinct from VoiceStudio (upd0912-voicestudio, AGPL-3.0). Both are local ElevenLabs alternatives; Voicebox adds Wispr-Flow-style dictation and the MCP agent voice, under a more permissive MIT license. - Curated from the GTM Stacker signal registry (2026-09-14 pass: daily pull + viral-posts brief); license independently verified 2026-09-14.\n\nTwo things make this more than another voice toy for a GTM operator. First, MIT plus fully local means narration and dictation stop being a metered line item: docs, demos and voice notes become a build step you run as often as the product changes, on hardware you own. Second, voicebox.speak turns an agent's output audible with one tool call, so a research or monitoring agent can brief you in a voice instead of a wall of text. The honest part: it shares the voice lane with VoiceStudio, so pick on license and dictation, not hype; quality tracks your hardware and the chosen engine; and the consent to clone a voice is on you, every time."
    }
  ],
  "alternates": {
    "markdown": "https://gtmstacker.com/registry/tool/voicebox/index.md",
    "html": "https://gtmstacker.com/registry/tool/voicebox/",
    "json": "https://gtmstacker.com/registry/tool/voicebox/index.json",
    "server_json": "https://gtmstacker.com/registry/tool/voicebox/server.json"
  },
  "jsonld": {
    "@context": "https://schema.org",
    "@graph": [
      {
        "@type": "WebSite",
        "@id": "https://gtmstacker.com/#website",
        "url": "https://gtmstacker.com/",
        "name": "GTM Stacker Agent Registry",
        "description": "A daily-updated, agent-native registry of open-source tool discoveries, tool updates, and curated news for the go-to-market / RevOps engineering niche. Machine-readable first: agents can discover, parse, page, and delta-sync it without scraping HTML.",
        "inLanguage": "en",
        "publisher": {
          "@id": "https://gtmstacker.com/#organization"
        }
      },
      {
        "@type": "Organization",
        "@id": "https://gtmstacker.com/#organization",
        "name": "GTM Stacker",
        "url": "https://gtmstacker.com",
        "description": "The growth-systems practice of Theo Popov: AI-native enrichment, outbound, content engines and internal tooling for startups and venture programs. Its agent-native media property, the GTM Stacker Agent Registry, maintains a daily-updated catalog of open-source go-to-market and RevOps tools that both people and AI engines can discover, compare, and cite.",
        "foundingDate": "2024-08",
        "knowsAbout": [
          "go-to-market engineering",
          "RevOps",
          "sales automation",
          "marketing operations",
          "open-source software",
          "AI agents"
        ],
        "founder": {
          "@type": "Person",
          "@id": "https://gtmstacker.com/#founder",
          "name": "Theo Popov",
          "jobTitle": "Growth Operations & GTM Systems",
          "url": "https://gtmstacker.com/about/",
          "sameAs": [
            "https://www.linkedin.com/in/theo-popov",
            "https://x.com/Theo_Popov",
            "https://github.com/theopopov"
          ],
          "worksFor": {
            "@id": "https://gtmstacker.com/#organization"
          }
        },
        "sameAs": [
          "https://www.linkedin.com/company/gtmstacker",
          "https://www.youtube.com/@gtmstacker",
          "https://www.instagram.com/gtmstacker/",
          "https://www.tiktok.com/@gtmstacker"
        ],
        "mainEntityOfPage": "https://gtmstacker.com/registry/about/"
      },
      {
        "@type": "SoftwareApplication",
        "@id": "https://gtmstacker.com/registry/tool/voicebox/#software",
        "name": "Voicebox",
        "identifier": "io.github.jamiepine/voicebox",
        "description": "Open-source, fully-local AI voice studio (MIT, Python + React) from Jamie Pine: clone a voice from a short sample, generate speech across 7 TTS engines and 23 languages, and dictate into any text field with a global hotkey (a Wispr Flow alternative). The GTM hook is one MCP tool call, voicebox.speak, that lets any MCP-aware agent talk back in a voice you cloned. Models, voice data and captures never leave the machine; runs on Apple Silicon (MLX) or CUDA/ROCm/CPU (PyTorch), macOS and Windows.",
        "applicationCategory": "DeveloperApplication",
        "url": "https://gtmstacker.com/registry/tool/voicebox/",
        "datePublished": "2026-09-14T00:00:00Z",
        "dateModified": "2026-09-14T00:00:00Z",
        "isPartOf": {
          "@id": "https://gtmstacker.com/#website"
        },
        "license": "https://spdx.org/licenses/MIT.html",
        "codeRepository": "https://github.com/jamiepine/voicebox",
        "keywords": "content-copywriting, mcp-agents, productivity-knowledge, local-first, voice-cloning, dictation",
        "author": {
          "@type": "Organization",
          "name": "jamiepine",
          "url": "https://github.com/jamiepine",
          "sameAs": [
            "https://github.com/jamiepine/voicebox"
          ]
        },
        "offers": {
          "@type": "Offer",
          "price": 0,
          "priceCurrency": "USD"
        },
        "isSimilarTo": [
          {
            "@type": "SoftwareApplication",
            "name": "VoiceStudio",
            "url": "https://gtmstacker.com/registry/tool/voicestudio/",
            "applicationCategory": "DeveloperApplication",
            "offers": {
              "@type": "Offer",
              "price": 0,
              "priceCurrency": "USD"
            }
          }
        ]
      },
      {
        "@type": "BreadcrumbList",
        "@id": "https://gtmstacker.com/registry/tool/voicebox/#breadcrumb",
        "itemListElement": [
          {
            "@type": "ListItem",
            "position": 1,
            "name": "GTM Stacker Registry",
            "item": "https://gtmstacker.com/registry/"
          },
          {
            "@type": "ListItem",
            "position": 2,
            "name": "Content Copywriting",
            "item": "https://gtmstacker.com/registry/category/content-copywriting/"
          },
          {
            "@type": "ListItem",
            "position": 3,
            "name": "Voicebox",
            "item": "https://gtmstacker.com/registry/tool/voicebox/"
          }
        ]
      }
    ]
  },
  "tool": {
    "name": "io.github.jamiepine/voicebox",
    "repository": {
      "url": "https://github.com/jamiepine/voicebox",
      "source": "github"
    }
  }
}
