{
  "$schema_doc": "https://gtmstacker.com/registry/schema/entry.schema.json",
  "stability": "emerging",
  "generator": "agentic-media-registry",
  "generated_at": "2026-09-08T00:00:00Z",
  "category": "data-scraping",
  "name": "Data Scraping",
  "count": 25,
  "html": "https://gtmstacker.com/registry/category/data-scraping/",
  "items": [
    {
      "id": "com.gtmstacker.registry/tool/browser-harness",
      "type": "tool",
      "slug": "browser-harness",
      "url": "https://gtmstacker.com/registry/tool/browser-harness/",
      "title": "Browser Harness",
      "description": "Pipeline inventory: Stage 5 strong acquisition fallback (fit 8/10) and Stage 2 partial enrichment extractor (fit 6/10). Self-healing harness connects LLMs to a real Chrome browser via CDP/Playwright and writes reusable helpers at runtime; complements Agent Reach/Crawl4AI/Crawlee/Lightpanda; ~16.7k stars. Expected to reduce interactive public-web extraction gaps by roughly 40-60%, but does not supply proprietary firmographic/contact data. CAVEATS: target-site ToS and…",
      "one_liner": "Pipeline inventory: Stage 5 strong acquisition fallback (fit 8/10) and Stage 2 partial enrichment extractor (fit 6/10). Self-healing harness connects LLMs to…",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "data-mining-scraping",
        "generic-web-desktop-automation-engines"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "freemium",
      "status": "active",
      "revision": 1,
      "content_hash": "daa0a13bdfb1a4c710f9a6bb3e71e3464cda69114f8f3e54553b952ebed3ee6c",
      "date_published": "2026-09-03T12:00:00Z",
      "date_modified": "2026-09-03T12:00:00Z",
      "json": "https://gtmstacker.com/registry/tool/browser-harness/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/browser-harness/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/browser-use",
      "type": "tool",
      "slug": "browser-use",
      "url": "https://gtmstacker.com/registry/tool/browser-use/",
      "title": "browser-use",
      "description": "Flagship 'make websites accessible for AI agents'; ~108k stars; CDP/Playwright; complements Crawl4AI/Browser Harness/Lightpanda; CAVEAT target-site ToS + agent runtime actions",
      "one_liner": "Flagship 'make websites accessible for AI agents'; ~108k stars; CDP/Playwright; complements Crawl4AI/Browser Harness/Lightpanda; CAVEAT target-site ToS +…",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "data-mining-scraping",
        "generic-web-desktop-automation-engines"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "freemium",
      "status": "active",
      "revision": 1,
      "content_hash": "8816e347c612cc8ae28df2a4fb62d370bfbb90759c7eb00b3bef76e777d6fabf",
      "date_published": "2026-09-03T12:00:00Z",
      "date_modified": "2026-09-03T12:00:00Z",
      "json": "https://gtmstacker.com/registry/tool/browser-use/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/browser-use/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/browser-use-web-ui",
      "type": "tool",
      "slug": "browser-use-web-ui",
      "url": "https://gtmstacker.com/registry/tool/browser-use-web-ui/",
      "title": "browser-use web-ui",
      "description": "Web GUI to run browser-use agents; ~16k stars; complements browser-use",
      "one_liner": "Web GUI to run browser-use agents; ~16k stars; complements browser-use",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "data-mining-scraping",
        "generic-web-desktop-automation-engines"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "6a1345173af1c01485ec12dc9be65e2b0784116a80c9e35e19c3c682c441a584",
      "date_published": "2026-09-03T12:00:00Z",
      "date_modified": "2026-09-03T12:00:00Z",
      "json": "https://gtmstacker.com/registry/tool/browser-use-web-ui/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/browser-use-web-ui/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/cdp-use",
      "type": "tool",
      "slug": "cdp-use",
      "url": "https://gtmstacker.com/registry/tool/cdp-use/",
      "title": "cdp-use",
      "description": "Pure Chrome DevTools Protocol, type-safe in Python; browser-automation primitive; ~308 stars",
      "one_liner": "Pure Chrome DevTools Protocol, type-safe in Python; browser-automation primitive; ~308 stars",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "data-mining-scraping"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "2be5dc93866303189f7e81cbe145fa92bf6e4b474b114e1c38a7be2d55bbdc9f",
      "date_published": "2026-09-03T12:00:00Z",
      "date_modified": "2026-09-03T12:00:00Z",
      "json": "https://gtmstacker.com/registry/tool/cdp-use/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/cdp-use/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/crawl4ai",
      "type": "tool",
      "slug": "crawl4ai",
      "url": "https://gtmstacker.com/registry/tool/crawl4ai/",
      "title": "Crawl4AI",
      "description": "OSS alt to Browserbase; ~77347 stars; via openalternative.co",
      "one_liner": "OSS alt to Browserbase; ~77347 stars; via openalternative.co",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "data-mining-scraping"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "freemium",
      "status": "active",
      "revision": 1,
      "content_hash": "80a5c61635bdc9860c426e07799858bf33ab05096697425ffca1b9b534aecc6e",
      "date_published": "2026-09-03T12:00:00Z",
      "date_modified": "2026-09-03T12:00:00Z",
      "json": "https://gtmstacker.com/registry/tool/crawl4ai/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/crawl4ai/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/crawlee",
      "type": "tool",
      "slug": "crawlee",
      "url": "https://gtmstacker.com/registry/tool/crawlee/",
      "title": "Crawlee",
      "description": "Generic scraping/browser-automation library",
      "one_liner": "Generic scraping/browser-automation library",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "prospecting-enrichment",
        "data-mining-scraping"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "eadaa4aeebf7f9be725858b54aec77f742ba4901d3e4ebeb7c4798ca89eb78a3",
      "date_published": "2026-09-03T12:00:00Z",
      "date_modified": "2026-09-03T12:00:00Z",
      "json": "https://gtmstacker.com/registry/tool/crawlee/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/crawlee/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/darts",
      "type": "tool",
      "slug": "darts",
      "url": "https://gtmstacker.com/registry/tool/darts/",
      "title": "Darts",
      "description": "~9487 stars; via ossinsight.io",
      "one_liner": "~9487 stars; via ossinsight.io",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "data-mining-scraping"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "cccc0bd92ef3d8b8f8fe5f3d9902b71e6198f6b4ea15f390d9a9174750692621",
      "date_published": "2026-09-03T12:00:00Z",
      "date_modified": "2026-09-03T12:00:00Z",
      "json": "https://gtmstacker.com/registry/tool/darts/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/darts/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/dlt",
      "type": "tool",
      "slug": "dlt",
      "url": "https://gtmstacker.com/registry/tool/dlt/",
      "title": "dlt",
      "description": "Pure OSI Python EL library; schema inference",
      "one_liner": "Pure OSI Python EL library; schema inference",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "data-mining-scraping"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "2475687fe7279af1e6af0eca9d7f5afb2c0078518b5c30ed8a8bf18b97f8946f",
      "date_published": "2026-09-03T12:00:00Z",
      "date_modified": "2026-09-03T12:00:00Z",
      "json": "https://gtmstacker.com/registry/tool/dlt/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/dlt/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/openserp",
      "type": "tool",
      "slug": "openserp",
      "url": "https://gtmstacker.com/registry/tool/openserp/",
      "title": "OpenSERP",
      "description": "OSS alt to SearchAPI; ~1235 stars; via openalternative.co",
      "one_liner": "OSS alt to SearchAPI; ~1235 stars; via openalternative.co",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "data-mining-scraping"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "22fbbe20abb3816f664656ab8c5f7bc2a2b9177709c4d3a09408de038d4bda84",
      "date_published": "2026-09-03T12:00:00Z",
      "date_modified": "2026-09-03T12:00:00Z",
      "json": "https://gtmstacker.com/registry/tool/openserp/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/openserp/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/pyod",
      "type": "tool",
      "slug": "pyod",
      "url": "https://gtmstacker.com/registry/tool/pyod/",
      "title": "PyOD",
      "description": "~9953 stars; via ossinsight.io",
      "one_liner": "~9953 stars; via ossinsight.io",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "data-mining-scraping"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "89b3461f2261f8153841408b1a77ef80acdae9f4d06765248e0f78a82454a8cd",
      "date_published": "2026-09-03T12:00:00Z",
      "date_modified": "2026-09-03T12:00:00Z",
      "json": "https://gtmstacker.com/registry/tool/pyod/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/pyod/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/scrapegraphai",
      "type": "tool",
      "slug": "scrapegraphai",
      "url": "https://gtmstacker.com/registry/tool/scrapegraphai/",
      "title": "ScrapeGraphAI",
      "description": "OSS alt to Diffbot; ~29238 stars; via opensource.builders",
      "one_liner": "OSS alt to Diffbot; ~29238 stars; via opensource.builders",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "data-mining-scraping"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "1a432142dd32c6f2d315f8bb2f1a5bba1157dbb53ab9529aeefaec5e42f956f6",
      "date_published": "2026-09-03T12:00:00Z",
      "date_modified": "2026-09-03T12:00:00Z",
      "json": "https://gtmstacker.com/registry/tool/scrapegraphai/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/scrapegraphai/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/scrapy",
      "type": "tool",
      "slug": "scrapy",
      "url": "https://gtmstacker.com/registry/tool/scrapy/",
      "title": "Scrapy",
      "description": "Mature; Zyte-backed; no built-in JS render",
      "one_liner": "Mature; Zyte-backed; no built-in JS render",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "data-mining-scraping"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "04dd5357257c69a127b4e1e7a533e5fda65ceceacefb407a639cd97618fd6f16",
      "date_published": "2026-09-03T12:00:00Z",
      "date_modified": "2026-09-03T12:00:00Z",
      "json": "https://gtmstacker.com/registry/tool/scrapy/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/scrapy/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/spacy",
      "type": "tool",
      "slug": "spacy",
      "url": "https://gtmstacker.com/registry/tool/spacy/",
      "title": "spaCy",
      "description": "Production NLP; Explosion-maintained",
      "one_liner": "Production NLP; Explosion-maintained",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "data-mining-scraping"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "7fa32108f307b555b02a71d5822c776473c46b0fd5fd34e89c6f044f34fb8550",
      "date_published": "2026-09-03T12:00:00Z",
      "date_modified": "2026-09-03T12:00:00Z",
      "json": "https://gtmstacker.com/registry/tool/spacy/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/spacy/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/stanza",
      "type": "tool",
      "slug": "stanza",
      "url": "https://gtmstacker.com/registry/tool/stanza/",
      "title": "Stanza",
      "description": "SOTA neural accuracy; 60+ languages",
      "one_liner": "SOTA neural accuracy; 60+ languages",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "data-mining-scraping"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "7543b9740620394ed1048167ca1d8d07019499139d584d73a8277e206577abea",
      "date_published": "2026-09-03T12:00:00Z",
      "date_modified": "2026-09-03T12:00:00Z",
      "json": "https://gtmstacker.com/registry/tool/stanza/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/stanza/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/stumpy",
      "type": "tool",
      "slug": "stumpy",
      "url": "https://gtmstacker.com/registry/tool/stumpy/",
      "title": "STUMPY",
      "description": "~4140 stars; via ossinsight.io",
      "one_liner": "~4140 stars; via ossinsight.io",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "data-mining-scraping"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "b7213b4a7f34c7942ab7111caae578ef675004987278d3fd5a2d7427fcfed4d2",
      "date_published": "2026-09-03T12:00:00Z",
      "date_modified": "2026-09-03T12:00:00Z",
      "json": "https://gtmstacker.com/registry/tool/stumpy/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/stumpy/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/firecrawl-anydoc",
      "type": "tool",
      "slug": "firecrawl-anydoc",
      "url": "https://gtmstacker.com/registry/tool/firecrawl-anydoc/",
      "title": "Firecrawl anydoc",
      "description": "Open-source document-to-Markdown converter from the Firecrawl team (Rust with multi-language bindings): Word, PowerPoint, Excel, OpenDocument, RTF, EPUB, CSV and PDF into clean GitHub-Flavoured Markdown — an ingest layer for agent and RAG pipelines.",
      "one_liner": "Open-source document-to-Markdown converter from the Firecrawl team (Rust with multi-language bindings): Word, PowerPoint, Excel, OpenDocument, RTF, EPUB, CSV…",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "markdown",
        "document-parsing",
        "rag",
        "agents"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "800fccead13781a976e16d128e5568698612ec807270e119d1e9ba4772307240",
      "date_published": "2026-09-03T00:00:00Z",
      "date_modified": "2026-09-03T00:00:00Z",
      "json": "https://gtmstacker.com/registry/tool/firecrawl-anydoc/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/firecrawl-anydoc/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/obscura-headless-browser",
      "type": "tool",
      "slug": "obscura-headless-browser",
      "url": "https://gtmstacker.com/registry/tool/obscura-headless-browser/",
      "title": "Obscura",
      "description": "Open-source headless browser for AI agents and web scraping, built in Rust: runs V8 JavaScript, speaks Chrome DevTools Protocol, drop-in for Puppeteer/Playwright, ~30MB, with stealth/anti-fingerprinting and MCP support — no Chromium.",
      "one_liner": "Open-source headless browser for AI agents and web scraping, built in Rust: runs V8 JavaScript, speaks Chrome DevTools Protocol, drop-in for…",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "headless-browser",
        "rust",
        "scraping",
        "stealth"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "81af1c6e9357357eacff90a2d82dc0bc2f4172268a5e01c78475309acb8eeb26",
      "date_published": "2026-08-31T22:57:41Z",
      "date_modified": "2026-08-31T22:57:41Z",
      "json": "https://gtmstacker.com/registry/tool/obscura-headless-browser/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/obscura-headless-browser/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/maxun",
      "type": "tool",
      "slug": "maxun",
      "url": "https://gtmstacker.com/registry/tool/maxun/",
      "title": "Maxun",
      "description": "Open-source, no-code web-data platform: extract structured data (recorded actions or natural language), scrape pages to Markdown/HTML with screenshots, crawl whole sites, and run automated searches — self-hostable, with an SDK and CLI.",
      "one_liner": "Open-source, no-code web-data platform: extract structured data (recorded actions or natural language), scrape pages to Markdown/HTML with screenshots, crawl…",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "no-code",
        "web-data",
        "self-hostable"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "subscription",
      "status": "active",
      "revision": 1,
      "content_hash": "4f1be45d1905afdc8adbf32b9a89abe9f82805c326abb4bdff3eeee812fca292",
      "date_published": "2026-08-31T12:10:14Z",
      "date_modified": "2026-08-31T12:10:14Z",
      "json": "https://gtmstacker.com/registry/tool/maxun/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/maxun/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/chrome-devtools-mcp",
      "type": "tool",
      "slug": "chrome-devtools-mcp",
      "url": "https://gtmstacker.com/registry/tool/chrome-devtools-mcp/",
      "title": "Chrome DevTools MCP",
      "description": "MCP server that lets an agent inspect and control a live Chrome browser — performance traces, network requests, console errors, rendered pages, and screenshots.",
      "one_liner": "MCP server that lets an agent inspect and control a live Chrome browser — performance traces, network requests, console errors, rendered pages, and screenshots.",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "browser-automation",
        "mcp",
        "debugging"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "9337db77439c0f48d58b2d900c36cf7ea8171d50b172187410d8a5d2858e5a2f",
      "date_published": "2026-08-25T21:34:12Z",
      "date_modified": "2026-08-25T21:34:12Z",
      "json": "https://gtmstacker.com/registry/tool/chrome-devtools-mcp/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/chrome-devtools-mcp/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/docling",
      "type": "tool",
      "slug": "docling",
      "url": "https://gtmstacker.com/registry/tool/docling/",
      "title": "Docling",
      "description": "Document-processing toolkit with advanced PDF/DOCX/PPTX/XLSX understanding (tables, complex layouts) and generative-AI integrations — for when simple PDF-to-text loses structure.",
      "one_liner": "Document-processing toolkit with advanced PDF/DOCX/PPTX/XLSX understanding (tables, complex layouts) and generative-AI integrations — for when simple…",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "document-processing",
        "pdf",
        "rag"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "b179344d8e00ce4b97edb31fd4d4c12dab297f4debd58e10c6458431f8f9e474",
      "date_published": "2026-08-25T21:34:12Z",
      "date_modified": "2026-08-25T21:34:12Z",
      "json": "https://gtmstacker.com/registry/tool/docling/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/docling/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/firecrawl",
      "type": "tool",
      "slug": "firecrawl",
      "url": "https://gtmstacker.com/registry/tool/firecrawl/",
      "title": "Firecrawl",
      "description": "The context API to search, scrape, and crawl the web at scale and turn sites into clean, LLM-ready data. Core is AGPL-3.0; SDKs/UI are MIT.",
      "one_liner": "The context API to search, scrape, and crawl the web at scale and turn sites into clean, LLM-ready data. Core is AGPL-3.0; SDKs/UI are MIT.",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "crawler",
        "scraping",
        "llm-ready",
        "oss-alternative"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "freemium",
      "status": "active",
      "revision": 1,
      "content_hash": "05a892b7d05b1ccc6fac1a0eb07040b8ebae5ec525497ca394901fed01b8068f",
      "date_published": "2026-08-25T21:34:12Z",
      "date_modified": "2026-08-25T21:34:12Z",
      "json": "https://gtmstacker.com/registry/tool/firecrawl/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/firecrawl/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/firecrawl-mcp-server",
      "type": "tool",
      "slug": "firecrawl-mcp-server",
      "url": "https://gtmstacker.com/registry/tool/firecrawl-mcp-server/",
      "title": "Firecrawl MCP Server",
      "description": "Official Firecrawl MCP server that brings web search, scraping, crawling, and structured extraction to MCP-compatible agents as clean, agent-ready context.",
      "one_liner": "Official Firecrawl MCP server that brings web search, scraping, crawling, and structured extraction to MCP-compatible agents as clean, agent-ready context.",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "mcp",
        "scraping",
        "web-search"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "freemium",
      "status": "active",
      "revision": 1,
      "content_hash": "e16d534a6b275e31bf5ba77de3e00ba38ae49f07af2dec16df4a8a6863eb7a11",
      "date_published": "2026-08-25T21:34:12Z",
      "date_modified": "2026-08-25T21:34:12Z",
      "json": "https://gtmstacker.com/registry/tool/firecrawl-mcp-server/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/firecrawl-mcp-server/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/markitdown",
      "type": "tool",
      "slug": "markitdown",
      "url": "https://gtmstacker.com/registry/tool/markitdown/",
      "title": "MarkItDown",
      "description": "Microsoft utility that converts PDFs, Word, PowerPoint, Excel, HTML, images and more into clean Markdown for LLM pipelines — structure-preserving.",
      "one_liner": "Microsoft utility that converts PDFs, Word, PowerPoint, Excel, HTML, images and more into clean Markdown for LLM pipelines — structure-preserving.",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "document-conversion",
        "markdown",
        "microsoft"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "875fd5f10c7a8ebe34e31946d008e5a35376c8fb5362ad44ecf7be0fcdddf182",
      "date_published": "2026-08-25T21:34:12Z",
      "date_modified": "2026-08-25T21:34:12Z",
      "json": "https://gtmstacker.com/registry/tool/markitdown/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/markitdown/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/playwright-mcp",
      "type": "tool",
      "slug": "playwright-mcp",
      "url": "https://gtmstacker.com/registry/tool/playwright-mcp/",
      "title": "Playwright MCP",
      "description": "Microsoft MCP server that lets agents automate a browser via structured accessibility snapshots (not screenshots) — for rendered-page checks, forms, funnels, and landing-page QA.",
      "one_liner": "Microsoft MCP server that lets agents automate a browser via structured accessibility snapshots (not screenshots) — for rendered-page checks, forms, funnels,…",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "browser-automation",
        "mcp",
        "microsoft"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "c91b02d242b067f4ff82e6a476ea6ef66ff4b72f53c1005ecc258e5b15da1bb2",
      "date_published": "2026-08-25T21:34:12Z",
      "date_modified": "2026-08-25T21:34:12Z",
      "json": "https://gtmstacker.com/registry/tool/playwright-mcp/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/playwright-mcp/index.md"
    },
    {
      "id": "com.gtmstacker.registry/tool/oc-only-cli",
      "type": "tool",
      "slug": "oc-only-cli",
      "url": "https://gtmstacker.com/registry/tool/oc-only-cli/",
      "title": "oc (only-cli)",
      "description": "Open-source CLI that fetches a web page and returns a compact numbered view instead of raw HTML — a cheap web-read layer for scraping/enrichment agents. Installs as a CLI and as an agent skill (npx skills add).",
      "one_liner": "Open-source CLI that fetches a web page and returns a compact numbered view instead of raw HTML — a cheap web-read layer for scraping/enrichment agents.…",
      "category": "data-scraping",
      "tags": [
        "data-scraping",
        "agent-tools",
        "web-content",
        "cli",
        "mcp"
      ],
      "open_source": "yes",
      "self_hostable": "yes",
      "pricing_model": "free",
      "status": "active",
      "revision": 1,
      "content_hash": "ef028c94345f2d4c3e0212aa5fab830466ed27f91cc0e66533ed7e9d648b7c04",
      "date_published": "2026-08-25T14:46:42Z",
      "date_modified": "2026-08-25T14:46:42Z",
      "json": "https://gtmstacker.com/registry/tool/oc-only-cli/index.json",
      "markdown": "https://gtmstacker.com/registry/tool/oc-only-cli/index.md"
    }
  ]
}
