{
  "gpu": "NVIDIA GeForce RTX 4090",
  "cuda_version": "12.1",
  "latency_median_ms": 2.352,
  "latency_p95_ms": 2.399,
  "latency_p99_ms": 2.433,
  "throughput": {
    "batch_1": {
      "throughput_per_sec": 422.08,
      "latency_ms": 2.369
    },
    "batch_8": {
      "throughput_per_sec": 3433.34,
      "latency_ms": 0.291
    },
    "batch_16": {
      "throughput_per_sec": 6618.75,
      "latency_ms": 0.151
    },
    "batch_32": {
      "throughput_per_sec": 10477.5,
      "latency_ms": 0.095
    },
    "batch_64": {
      "throughput_per_sec": 11502.58,
      "latency_ms": 0.087
    }
  },
  "provenance": {
    "kind": "synthetic_stub",
    "not_c4_onnx": true,
    "note": "TransformerEncoder stub latency proxy \u2014 not shipping C4 Dual/ONNX classifier throughput"
  }
}
