{
  "site": {
    "name": "Openbenchmarks",
    "url": "https://openbenchmarks.com",
    "description": "Open benchmarks - verified benchmarks for build vs buy decisions on AI agent tooling."
  },
  "last_updated": "2026-09-21T16:00:00.000Z",
  "benchmarks": [
    {
      "slug": "lookalikes",
      "name": "Lookalike Benchmark",
      "description": "Head-to-head benchmark for company lookalike APIs (Ocean.io, Exa, Parallel, Seltz, PredictLeads, Extruct, CUFinder, Discolike, and ZoomInfo). Each vendor returns its top-K lookalikes per seed company; an LLM judge scores every returned company for relevance. Cell value = Precision@K, headline metric = avg Precision@K across the seed cohort.",
      "page_url": "https://openbenchmarks.com/lookalikes",
      "api_url": "https://openbenchmarks.com/api/benchmarks/lookalikes",
      "source_repo": "https://github.com/openbenchmarks-labs/lookalikes",
      "status": "live",
      "provider_count": 9,
      "canonical_categories": [
        "b2b-saas",
        "devtools",
        "ecommerce",
        "healthtech",
        "home-services",
        "trades",
        "real-estate",
        "fintech",
        "cybersecurity",
        "industrial",
        "logistics",
        "hospitality",
        "energy"
      ],
      "winners": {
        "highest_avg_precision_at_k": {
          "provider": "Seltz",
          "avg_precision_at_k": 63.42,
          "seeds_judged": 48,
          "k": 100
        }
      }
    },
    {
      "slug": "company-funding",
      "name": "Company Funding Benchmark",
      "description": "Head-to-head benchmark for company funding data providers. Each provider is asked for the latest funding stage of the same company domains and judged against funding events verified from official sources. Two boards ranked separately: freshness for rounds announced in the last 30 days, enrichment for older rounds.",
      "page_url": "https://openbenchmarks.com/company-funding",
      "api_url": "https://openbenchmarks.com/api/benchmarks/company-funding",
      "source_repo": "https://github.com/openbenchmarks-labs/company-funding",
      "status": "live",
      "provider_count": 23,
      "boards": [
        "freshness",
        "enrichment"
      ],
      "winners": {
        "highest_stage_correct_yield": {
          "provider": "Firecrawl",
          "board": "enrichment",
          "stage_correct_yield_pct": 92.33,
          "companies_measured": 300
        }
      }
    },
    {
      "slug": "voice-agent-latency",
      "name": "Voice Agent Latency Benchmark",
      "description": "First-party benchmark measuring Time To First Audio Byte (TTFAB) of voice AI agent platforms over real phone calls, from saved call audio — never from platform-reported timestamps. Includes a per-turn latency curve (does the agent slow down as the conversation grows?).",
      "page_url": "https://openbenchmarks.com/voice-agent-latency",
      "api_url": "https://openbenchmarks.com/api/benchmarks/voice-agent-latency",
      "status": "live",
      "provider_count": 5,
      "winners": {
        "lowest_median_ttfab": {
          "provider": "Telnyx",
          "ttfab_onset_p50_ms": 1296,
          "turns_usable": 419,
          "is_comparison": true
        }
      }
    },
    {
      "slug": "inference",
      "name": "LLM Inference Provider Benchmark",
      "description": "Independent benchmark of LLM inference providers all serving the same GLM 5.3 Flash model: end-to-end latency p50/p95/p99, time to first token, tokens per second, exact task success and failure rate over 600 structured-output requests per provider from one client at concurrency one. Failures stay in the denominator.",
      "page_url": "https://openbenchmarks.com/inference",
      "api_url": "https://openbenchmarks.com/api/benchmarks/inference",
      "status": "live",
      "provider_count": 10,
      "model": "GLM 5.3 Flash",
      "reviewed_at": "2026-09-03T03:03:46.01458+00:00",
      "verdict": "Measured Sep 3, 2026 across 10 providers, 600 requests each: lowest end-to-end p99 latency: Baseten (3.23 seconds); most tokens per second: Nebius (269.3 tokens per second median); highest task success: Fireworks AI (99.5%)."
    },
    {
      "slug": "text-to-speech-benchmark-by-coval",
      "name": "Text-to-Speech Benchmark (by Coval)",
      "description": "Independent text-to-speech benchmark by Coval: Time to First Audio (TTFA) latency and Word Error Rate across providers. Coval sells voice-agent evaluation infrastructure. Rolling-window aggregates, mirrored here with attribution.",
      "page_url": "https://openbenchmarks.com/text-to-speech-benchmark-by-coval",
      "data_url": "https://raw.githubusercontent.com/openbenchmarks-labs/tts-stt-benchmarks-by-coval/main/coval-benchmarks.json",
      "attribution": "Coval",
      "source_url": "https://benchmarks.coval.ai",
      "source_repo": "https://github.com/openbenchmarks-labs/tts-stt-benchmarks-by-coval",
      "methodology_url": "https://github.com/coval-ai/benchmarks",
      "status": "live",
      "provider_count": 20,
      "last_synced": "2026-09-21T11:32:42.362Z",
      "winners": {
        "lowest_median_latency": {
          "provider": "fluxions",
          "model": "vui",
          "metric": "TTFA",
          "median_ms": 49.22821350003801
        },
        "lowest_word_error_rate": {
          "provider": "soniox",
          "model": "tts-rt-v1",
          "wer_percent": 4.247705510089129
        }
      }
    },
    {
      "slug": "speech-to-text-benchmark-by-coval",
      "name": "Live Speech-to-Text Benchmark (by Coval)",
      "description": "Live independent speech-to-text benchmark by Coval: rolling 7-day Time to Final Segment (TTFS) and Word Error Rate. Coval sells voice-agent evaluation infrastructure. Mirrored here with attribution.",
      "page_url": "https://openbenchmarks.com/speech-to-text-benchmark-by-coval",
      "data_url": "https://raw.githubusercontent.com/openbenchmarks-labs/tts-stt-benchmarks-by-coval/main/coval-benchmarks.json",
      "attribution": "Coval",
      "source_url": "https://benchmarks.coval.ai",
      "source_repo": "https://github.com/openbenchmarks-labs/tts-stt-benchmarks-by-coval",
      "methodology_url": "https://github.com/coval-ai/benchmarks",
      "status": "live",
      "provider_count": 19,
      "last_synced": "2026-09-21T11:32:42.362Z",
      "winners": {
        "lowest_median_latency": {
          "provider": "baseten",
          "model": "qwen3-asr-1.7b",
          "metric": "TTFS",
          "median_ms": 21.68099000037138
        },
        "lowest_word_error_rate": {
          "provider": "assemblyai",
          "model": "universal-3.5-pro",
          "wer_percent": 2.823333005990376
        }
      }
    }
  ],
  "docs": {
    "openapi": "https://openbenchmarks.com/openapi.json",
    "llms": "https://openbenchmarks.com/llms.txt",
    "methodology": "https://openbenchmarks.com/lookalikes#methodology"
  }
}