{
  "$schema_doc": "https://gtmstacker.com/registry/schema/entry.schema.json",
  "stability": "emerging",
  "generator": "agentic-media-registry",
  "generated_at": "2026-09-08T00:00:00Z",
  "id": "com.gtmstacker.registry/tool/freetoken",
  "type": "tool",
  "slug": "freetoken",
  "canonical_url": "https://gtmstacker.com/registry/tool/freetoken/",
  "title": "FreeToken",
  "description": "Edge-native Mixture-of-Experts inference engine (UC Berkeley) that runs frontier open-weight models on consumer hardware — e.g. a 35B model on an 8GB GPU — by treating GPU/CPU/host-memory as one bandwidth-adaptive platform; 2–4× faster than Ollama on MoE. Desktop app + Python CLI.",
  "category": "ai-infrastructure",
  "tags": [
    "ai-infrastructure",
    "local-inference",
    "moe",
    "self-hosting"
  ],
  "status": "active",
  "revision": 1,
  "content_hash": "4b2a2a1e0ee2e4594c9d568de4761611bd86de06bd218658bf639fe68354e0b4",
  "date_published": "2026-09-01T11:00:27Z",
  "date_modified": "2026-09-01T11:00:27Z",
  "source": {
    "name": "github · FlashML-org/FreeToken",
    "url": "https://github.com/FlashML-org/FreeToken"
  },
  "license": "Apache-2.0",
  "one_liner": "Edge-native Mixture-of-Experts inference engine (UC Berkeley) that runs frontier open-weight models on consumer hardware — e.g. a 35B model on an 8GB GPU — by…",
  "open_source": "yes",
  "self_hostable": "yes",
  "pricing_model": "free",
  "who_its_for": "Individuals with consumer/gaming hardware who want to run frontier-scale open-weight MoE models locally.",
  "aliases": [],
  "alternatives": [],
  "secondary_categories": [],
  "last_verified": "2026-09-04",
  "evidence": {
    "claim_type": "vendor-claim",
    "source_id": "https://github.com/FlashML-org/FreeToken"
  },
  "lead": "Edge-native Mixture-of-Experts inference engine (UC Berkeley) that runs frontier open-weight models on consumer hardware — e.g. a 35B model on an 8GB GPU — by treating GPU/CPU/host-memory as one bandwidth-adaptive platform; 2–4× faster than Ollama on MoE. Desktop app + Python CLI.",
  "chunks": [
    {
      "index": 0,
      "heading_path": [],
      "est_tokens": 71,
      "text": "Edge-native Mixture-of-Experts inference engine (UC Berkeley) that runs frontier open-weight models on consumer hardware — e.g. a 35B model on an 8GB GPU — by treating GPU/CPU/host-memory as one bandwidth-adaptive platform; 2–4× faster than Ollama on MoE. Desktop app + Python CLI."
    },
    {
      "index": 1,
      "heading_path": [
        null,
        "Provenance"
      ],
      "est_tokens": 124,
      "text": "Edge-native Mixture-of-Experts inference engine (UC Berkeley) that runs frontier open-weight models on consumer hardware — e.g. a 35B model on an 8GB GPU — by treating GPU/CPU/host-memory as one bandwidth-adaptive platform; 2–4× faster than Ollama on MoE. Desktop app + Python CLI.\n\n- Apache-2.0 verified (11.4k★, active; arXiv 2608.16157).\n- Surfaced via the GTM Stacker X/Twitter signal reports (the \"35 marketing repos\" list + named drops); license independently WebFetch-verified 2026-09-03."
    }
  ],
  "alternates": {
    "markdown": "https://gtmstacker.com/registry/tool/freetoken/index.md",
    "html": "https://gtmstacker.com/registry/tool/freetoken/",
    "json": "https://gtmstacker.com/registry/tool/freetoken/index.json",
    "server_json": "https://gtmstacker.com/registry/tool/freetoken/server.json"
  },
  "jsonld": {
    "@context": "https://schema.org",
    "@graph": [
      {
        "@type": "WebSite",
        "@id": "https://gtmstacker.com/#website",
        "url": "https://gtmstacker.com/",
        "name": "GTM Stacker Agent Registry",
        "description": "A daily-updated, agent-native registry of open-source tool discoveries, tool updates, and curated news for the go-to-market / RevOps engineering niche. Machine-readable first: agents can discover, parse, page, and delta-sync it without scraping HTML.",
        "inLanguage": "en",
        "publisher": {
          "@id": "https://gtmstacker.com/#organization"
        }
      },
      {
        "@type": "Organization",
        "@id": "https://gtmstacker.com/#organization",
        "name": "GTM Stacker",
        "url": "https://gtmstacker.com",
        "alternateName": "Agent-native registry of open-source go-to-market tools",
        "description": "GTM Stacker is an agent-native media property that maintains a daily-updated registry of open-source go-to-market, RevOps, sales, and marketing tools, engineered so both people and AI engines can discover, compare, and cite each tool.",
        "foundingDate": "2024-08",
        "knowsAbout": [
          "go-to-market engineering",
          "RevOps",
          "sales automation",
          "marketing operations",
          "open-source software",
          "AI agents"
        ],
        "founder": {
          "@type": "Person",
          "@id": "https://gtmstacker.com/#founder",
          "name": "Theo Popov",
          "jobTitle": "Founder",
          "url": "https://gtmstacker.com/about/",
          "sameAs": [
            "https://www.linkedin.com/in/theo-popov/",
            "https://x.com/Theo_Popov",
            "https://github.com/theopopov"
          ],
          "worksFor": {
            "@id": "https://gtmstacker.com/#organization"
          }
        },
        "mainEntityOfPage": "https://gtmstacker.com/registry/about/"
      },
      {
        "@type": "SoftwareApplication",
        "@id": "https://gtmstacker.com/registry/tool/freetoken/#software",
        "name": "FreeToken",
        "identifier": "io.github.flashml-org/freetoken",
        "description": "Edge-native Mixture-of-Experts inference engine (UC Berkeley) that runs frontier open-weight models on consumer hardware — e.g. a 35B model on an 8GB GPU — by treating GPU/CPU/host-memory as one bandwidth-adaptive platform; 2–4× faster than Ollama on MoE. Desktop app + Python CLI.",
        "applicationCategory": "DeveloperApplication",
        "url": "https://gtmstacker.com/registry/tool/freetoken/",
        "datePublished": "2026-09-01T11:00:27Z",
        "dateModified": "2026-09-01T11:00:27Z",
        "isPartOf": {
          "@id": "https://gtmstacker.com/#website"
        },
        "license": "Apache-2.0",
        "codeRepository": "https://github.com/FlashML-org/FreeToken",
        "keywords": "ai-infrastructure, local-inference, moe, self-hosting",
        "author": {
          "@type": "Organization",
          "name": "FlashML-org",
          "url": "https://github.com/FlashML-org",
          "sameAs": [
            "https://github.com/FlashML-org/FreeToken"
          ]
        },
        "offers": {
          "@type": "Offer",
          "price": 0,
          "priceCurrency": "USD"
        }
      },
      {
        "@type": "BreadcrumbList",
        "@id": "https://gtmstacker.com/registry/tool/freetoken/#breadcrumb",
        "itemListElement": [
          {
            "@type": "ListItem",
            "position": 1,
            "name": "GTM Stacker Registry",
            "item": "https://gtmstacker.com/registry/"
          },
          {
            "@type": "ListItem",
            "position": 2,
            "name": "AI Infrastructure",
            "item": "https://gtmstacker.com/registry/category/ai-infrastructure/"
          },
          {
            "@type": "ListItem",
            "position": 3,
            "name": "FreeToken",
            "item": "https://gtmstacker.com/registry/tool/freetoken/"
          }
        ]
      }
    ]
  },
  "tool": {
    "name": "io.github.flashml-org/freetoken",
    "repository": {
      "url": "https://github.com/FlashML-org/FreeToken",
      "source": "github"
    }
  }
}
