{
  "boundary": "role_completion",
  "cases": [
    {
      "case_id": "request-01",
      "duration_ms": 3923.377,
      "failure_stage": null,
      "outcome": "pass",
      "primary_outcome": "pass",
      "summary": null,
      "title": "Synthetic request 01"
    },
    {
      "case_id": "request-02",
      "duration_ms": 6931.26,
      "failure_stage": "answer_checks",
      "outcome": "fail",
      "primary_outcome": "fail",
      "summary": null,
      "title": "Synthetic request 02"
    },
    {
      "case_id": "request-03",
      "duration_ms": 3801.035,
      "failure_stage": null,
      "outcome": "pass",
      "primary_outcome": "pass",
      "summary": null,
      "title": "Synthetic request 03"
    },
    {
      "case_id": "request-04",
      "duration_ms": 4393.57,
      "failure_stage": null,
      "outcome": "pass",
      "primary_outcome": "pass",
      "summary": null,
      "title": "Synthetic request 04"
    },
    {
      "case_id": "request-05",
      "duration_ms": 4387.226,
      "failure_stage": "answer_checks",
      "outcome": "fail",
      "primary_outcome": "fail",
      "summary": null,
      "title": "Synthetic request 05"
    },
    {
      "case_id": "request-06",
      "duration_ms": 3203.412,
      "failure_stage": null,
      "outcome": "pass",
      "primary_outcome": "pass",
      "summary": null,
      "title": "Synthetic request 06"
    },
    {
      "case_id": "request-07",
      "duration_ms": 3635.69,
      "failure_stage": null,
      "outcome": "pass",
      "primary_outcome": "pass",
      "summary": null,
      "title": "Synthetic request 07"
    },
    {
      "case_id": "request-08",
      "duration_ms": 6869.321,
      "failure_stage": null,
      "outcome": "pass",
      "primary_outcome": "pass",
      "summary": null,
      "title": "Synthetic request 08"
    }
  ],
  "comparison_key": null,
  "completed_at": "2026-09-20T14:32:32.769010+00:00",
  "conditions": [
    {
      "label": "Timing observer SHA-256",
      "value": "8a29437cb3e4b806f4be78454483721e9c31b09a7063f55bd9ec50081ffc3bef"
    },
    {
      "label": "Dataset SHA-256",
      "value": "53ddfba9e794833f0ddb92f9785dc1c012c566ef6d438df0b4dd153ca074860f"
    },
    {
      "label": "Correctness scorer",
      "value": "frozen-evidence-fields-v1"
    },
    {
      "label": "Peak-overlap basis",
      "value": "Not recorded; serving-client peak estimates are excluded"
    },
    {
      "label": "Runtime image digest",
      "value": "sha256:f5de20f9b4a806a94012d405f5cb8734c129eb69a7b3875d5362210527bad575"
    },
    {
      "label": "Weight verification",
      "value": "All checkpoint files rehashed against the pinned artifact on all four replicas"
    },
    {
      "label": "Serving layout",
      "value": "TP4, four hosts; prefill chunk 2048; scheduler cap 8; static memory fraction 0.80"
    },
    {
      "label": "Observed KV token pool",
      "value": "1204416 tokens total; not eight full-length sessions"
    },
    {
      "label": "Parsers",
      "value": "glm45 reasoning; glm47 tools"
    },
    {
      "label": "Speculation",
      "value": "Adaptive NEXTN: launch 5 steps/6 draft tokens; observed 3/4 at readiness"
    },
    {
      "label": "Thinking request",
      "value": "High reasoning; thinking enabled"
    },
    {
      "label": "Request settings",
      "value": "Temperature 0; seed 20260920; high reasoning; completion cap 4096 tokens; streaming with server usage"
    },
    {
      "label": "Input preparation",
      "value": "Identical frozen UTF-8 documents across tokenizers; approximate 4K target;8 independent development tasks"
    },
    {
      "label": "Cache state",
      "value": "Unique document prefixes first used on an already warm engine; no warmup requests or cache flush"
    },
    {
      "label": "Measurement path",
      "value": "CPU-only client over local network HTTP; no SSH tunnel in the measured inference path"
    },
    {
      "label": "Available scorer identity",
      "value": "Named scorer version; original grading-script source hash was not recorded"
    }
  ],
  "counts": {
    "declared": 8,
    "outcomes": {
      "fail": 2,
      "pass": 6
    },
    "primary_passes": 6
  },
  "deployment": {
    "concurrency": 8,
    "context_limit": 262144,
    "hardware": "Four NVIDIA DGX Spark systems, GB10, 128 GB unified memory per system",
    "id": "glm53-flash-stock-spark-tp4-high",
    "model": "GLM-5.3 Flash",
    "node_count": 4,
    "profile": "Warm-engine unique-prefix pilot; about 4K input, client C1",
    "quantization": "Native FP8 weights; FP8 E4M3 KV cache",
    "revision": "eb9eb208eb0d988989d07a6a12d0fdeb5f52574a",
    "runtime": "SGLang",
    "runtime_version": "0.0.0.dev1+gd6ab04bdf1",
    "source_url": "https://huggingface.co/zai-org/GLM-5.3-Flash",
    "variant": "stock",
    "weights_verified": true
  },
  "evaluator_sha256": "1f2d6733c17c647cba8599f8b560fdb2529b2fc9e710f8dd6ad71c68d606bf40",
  "evidence": [],
  "evidence_mode": "actual_inference",
  "limitations": [
    "Prompt-only development workloads; these do not exercise native agent loops or establish held-out quality.",
    "Input-size labels are targets. Actual server token usage differs across tokenizers and does not establish a maximum usable context window.",
    "Client concurrency limits and configured server slots are not observed overlap or maximum sustainable capacity.",
    "Observed overlap, when available, counts client request spans; it is not a count of active GPU sequences.",
    "No throughput ranking or comparison delta is authorized by this record; matching conditions have not been established.",
    "Transport completion and independently graded answer correctness are separate. Missing attempts and grades stay in the denominator.",
    "First final-content arrival is a channel timing, not proof that the content is useful or correct.",
    "Final-content timings exclude responses flagged for thinking tags in that channel, or lacking that observation; separate reasoning duration is not inferred.",
    "Eight requests in one run are a pilot, not a throughput ceiling or a stable tail-latency estimate.",
    "Completion-token usage combines reasoning and final output; the server did not report separate token counts.",
    "Actual server input usage is used. The serving client reported different tokenizer counts.",
    "The peak-overlap estimate is omitted because observer v3 lacks request start/end timestamps.",
    "Correctness is taken from the separate frozen-evidence-fields-v1 receipt; original grading-script source identity was not recorded.",
    "The schema task is a textual instruction in this workload; constrained response-format transport was not exercised."
  ],
  "metrics": [
    {
      "definition": "Server-reported completion tokens divided by the entire measured batch duration, including prefill, waiting and failed requests. Token accounting is stated in the serving cell.",
      "id": "completion_throughput_tps",
      "label": "End-to-end completion throughput",
      "samples": 8,
      "unit": "tok/s",
      "value": 21.182677917950876
    },
    {
      "definition": "First nonempty reasoning or content delta; this may precede any visible answer. Includes every response with that timing; missing timings are not zero.",
      "id": "first_generated_median_ms",
      "label": "First generated output, median",
      "samples": 8,
      "unit": "ms",
      "value": 2563.1885
    },
    {
      "definition": "First nonempty final-content delta; arrival does not establish correctness or usefulness. Includes every response with that timing; missing timings are not zero.",
      "id": "first_final_content_median_ms",
      "label": "First final content, median",
      "samples": 8,
      "unit": "ms",
      "value": 3322.5505000000003
    },
    {
      "definition": "Full request duration, including reasoning; includes failures with recorded duration. Includes every response with that timing; missing timings are not zero.",
      "id": "completion_median_ms",
      "label": "Request completion, median",
      "samples": 8,
      "unit": "ms",
      "value": 4155.3015
    },
    {
      "definition": "Linear-interpolated p95 over all recorded request durations, including failures. Small samples give an unstable tail estimate.",
      "id": "completion_p95_ms",
      "label": "Request completion, p95",
      "samples": 8,
      "unit": "ms",
      "value": 6909.58135
    },
    {
      "definition": "Observed SSE intervals reported by the serving client. Speculative decoding can emit several tokens per chunk; this is not per-token latency.",
      "id": "stream_chunk_interval_median_ms",
      "label": "Stream chunk interval, median",
      "samples": 198,
      "unit": "ms",
      "value": 87.76121400296688
    },
    {
      "definition": "Client measurement interval, excluding process setup. This is the denominator for aggregate completion throughput.",
      "id": "batch_duration_seconds",
      "label": "Measured batch duration",
      "samples": 1,
      "unit": "s",
      "value": 37.15299845696427
    }
  ],
  "performance": {
    "actual_input_tokens_max": 3735,
    "actual_input_tokens_min": 3704,
    "cache_state": "warm_engine_unique_prefixes",
    "client_concurrency_limit": 1,
    "observed_peak_inflight": null,
    "stream_interval_basis": "sse_chunks",
    "target_input_tokens": 4096,
    "token_accounting": "completion_includes_reasoning",
    "transport_successes": 8,
    "version": "serving-cell-v1"
  },
  "phase": "performance",
  "role": "serving",
  "run_id": "glm53-flash-stock-serving-4k-c1-pilot-119f8a2564a2",
  "schema_version": 1,
  "started_at": "2026-09-20T14:31:36.083421+00:00",
  "status": "completed",
  "suite_sha256": "7604be858e8b928dd7c2acb188dd25ec8b46bf26d544872220e18c20344db1d6",
  "suite_version": "serving-performance-v1"
}
