{
  "title": "Million-request citation-first RAG benchmark 2026",
  "executionDate": "2026-08-24",
  "niche": "Industrial maintenance and field-service troubleshooting",
  "scope": "Hybrid retrieval and extractive cited responses; LLM generation excluded",
  "corpusChunks": 20000,
  "quality": {
    "id": "exact-code-plus-hybrid-final",
    "region": "us-east1",
    "queries": 500,
    "recallAt1": 1.0,
    "recallAt3": 1.0,
    "latencyMs": {
      "mean": 3.649,
      "p50": 3.533,
      "p90": 4.17,
      "p95": 4.447,
      "p99": 5.39,
      "p999": 8.622,
      "max": 14.439
    },
    "passed": true
  },
  "finalRun": {
    "runId": "mixed-1m-1000rps",
    "mode": "mixed",
    "targetRequests": 1000000,
    "attemptedRequests": 1000000,
    "successful": 1000000,
    "errors": 0,
    "errorRate": 0,
    "durationSeconds": 1023.76162366,
    "achievedRps": 976.7898863262221,
    "configuredRps": 1000,
    "workers": 200,
    "cacheHits": 949697,
    "cacheMisses": 50303,
    "cacheHitRate": 0.949697,
    "responseBytes": 2886661768,
    "latencyMs": {
      "mean": 9.335,
      "p50": 8.695,
      "p90": 11.923,
      "p95": 13.448,
      "p99": 18.069,
      "p999": 39.701,
      "max": 816.281
    },
    "histogram": {
      "<=5ms": 685,
      "<=10ms": 845601,
      "<=20ms": 147915,
      "<=30ms": 3770,
      "<=50ms": 1517,
      "<=75ms": 331,
      "<=100ms": 90,
      "<=150ms": 57,
      "<=250ms": 26,
      "<=500ms": 7,
      "<=1000ms": 1
    },
    "statusCodes": {
      "200": 1000000
    },
    "errorTypes": {},
    "passed": true
  },
  "monitoring": {
    "requestCount": 1000000,
    "cpuAllocationVcpuSeconds": 4234.784491184029,
    "memoryAllocationGibSeconds": 2247.557643730896,
    "billableInstanceSeconds": 2168.3728015295046,
    "maxActiveInstances": 5,
    "meanActiveInstances": 2.263157894736842,
    "cloudSqlCpuUtilizationMean": 0.1118474068780811,
    "cloudSqlCpuUtilizationMax": 0.1336809882478825,
    "cloudSqlMemoryUtilizationMean": 0.4707800652260452,
    "cloudSqlMemoryUtilizationMax": 0.4743997954650571
  },
  "marginalCostPerMillion": 0.8513706204344261,
  "productionBaselineMonthly": 238.606,
  "monthlyScenarios": [
    {
      "millionRequests": 1,
      "total": 239.45737062043446,
      "costPerMillion": 239.45737062043446
    },
    {
      "millionRequests": 5,
      "total": 242.86285310217215,
      "costPerMillion": 48.57257062043443
    },
    {
      "millionRequests": 50,
      "total": 281.17453102172135,
      "costPerMillion": 5.623490620434427
    }
  ],
  "gates": {
    "recallAt3Minimum": 0.95,
    "mixedP95MaximumMs": 50,
    "forcedMissP95MaximumMs": 150,
    "errorRateMaximum": 0.001,
    "actualRequestsMinimum": 1000000
  },
  "limitations": [
    "The corpus and gold queries are synthetic; 100% source recovery does not imply real maintenance answer quality or safety.",
    "The result measures a retrieval and extractive citation layer, not uncached LLM generation. Model tokens and latency are excluded.",
    "The temporary database was zonal. The production cost model uses regional HA pricing, but HA failover was not tested.",
    "The per-instance cache produced a failed scale-out warmup run; a shared cache or deliberate prewarming is required for a strict burst SLO.",
    "Internet egress is an upper-bound list-price estimate; actual routing, free tiers, discounts and response compression change billed cost.",
    "The million-request run used a same-region Cloud Run job through an external load balancer, not geographically distributed end users."
  ]
}
