| { |
| "metadata": { |
| "version": "0.4.29", |
| "engine": "vllm", |
| "model": "glm-5.3-flash-dflash2", |
| "server": "192.168.1.201:8000", |
| "timestamp": "2026-09-02T16:58:14.207747", |
| "decode_mode": "duration", |
| "primary_decode_layer": "sustained_decode", |
| "duration_per_test": 10.0, |
| "request_count": 0, |
| "warmup_request_count": 0, |
| "run_burst": false, |
| "prefill_mode": "skipped", |
| "standalone_prefill": false, |
| "prefill_only": false, |
| "skip_prefill": true, |
| "burst_e2e_status": "not_run_use_--run-burst", |
| "burst_request_count": 0, |
| "burst_warmup_request_count": 0, |
| "burst_requests_per_concurrency": 5, |
| "decode_warmup_seconds": 3.0, |
| "decode_warmup_context": 0, |
| "decode_warmup_concurrency": 1, |
| "cell_warmup_timeout_seconds": 0.0, |
| "cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0", |
| "show_capacity_limited_values": false, |
| "max_tokens": 512, |
| "temperature": 0.0, |
| "ignore_eos": true, |
| "max_total_tokens": 6624000, |
| "dcp_size": 0, |
| "metrics_available": true, |
| "metrics_warning": "", |
| "concurrency_levels": [ |
| 1 |
| ], |
| "context_lengths": [ |
| 0 |
| ], |
| "startup_diagnostics_available": true, |
| "nvidia_p2p_override_effective": false, |
| "p2pmark_status": "not_run", |
| "amd_fabric_status": "not_run" |
| }, |
| "startup_diagnostics": { |
| "version": "0.4.29", |
| "server_url": "http://192.168.1.201:8000", |
| "hostname": "macmini", |
| "uname": "Darwin macmini 25.5.0 Darwin Kernel Version 25.5.0: Tue Jun 9 22:28:34 PDT 2026; root:xnu-12377.121.10~1/RELEASE_ARM64_T6041 arm64", |
| "env": {}, |
| "args": { |
| "concurrency": "1", |
| "contexts": "0", |
| "max_tokens": 512, |
| "duration": 10.0, |
| "request_count": 0, |
| "run_burst": false, |
| "standalone_prefill": false, |
| "prefill_only": false, |
| "skip_prefill": true, |
| "prefill_contexts": "8k,64k,128k", |
| "prefill_metric": "client", |
| "dcp_size": 0, |
| "kv_budget": 0 |
| }, |
| "nvidia_p2p_override": { |
| "effective": false, |
| "configured": false, |
| "params_path": "/proc/driver/nvidia/params", |
| "params_available": false, |
| "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf", |
| "modprobe_available": false, |
| "runtime": { |
| "ForceP2P": "", |
| "RMForceP2PType": "", |
| "RMPcieP2PType": "", |
| "GrdmaPciTopoCheckOverride": "", |
| "EnableResizableBar": "", |
| "DmaRemapPeerMmio": "" |
| }, |
| "expected": { |
| "ForceP2P": "0x11", |
| "RMForceP2PType": "1", |
| "RMPcieP2PType": "2", |
| "GrdmaPciTopoCheckOverride": "1", |
| "EnableResizableBar": "1" |
| }, |
| "missing": [ |
| "ForceP2P", |
| "RMForceP2PType", |
| "RMPcieP2PType", |
| "GrdmaPciTopoCheckOverride", |
| "EnableResizableBar" |
| ], |
| "mismatched": {}, |
| "registry_dwords": "", |
| "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"", |
| "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded" |
| }, |
| "p2pmark": { |
| "status": "not_run" |
| }, |
| "amd_fabric": { |
| "status": "not_run" |
| }, |
| "nvidia_smi_error": "nvidia-smi not found" |
| }, |
| "nvidia_p2p_override": { |
| "effective": false, |
| "configured": false, |
| "params_path": "/proc/driver/nvidia/params", |
| "params_available": false, |
| "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf", |
| "modprobe_available": false, |
| "runtime": { |
| "ForceP2P": "", |
| "RMForceP2PType": "", |
| "RMPcieP2PType": "", |
| "GrdmaPciTopoCheckOverride": "", |
| "EnableResizableBar": "", |
| "DmaRemapPeerMmio": "" |
| }, |
| "expected": { |
| "ForceP2P": "0x11", |
| "RMForceP2PType": "1", |
| "RMPcieP2PType": "2", |
| "GrdmaPciTopoCheckOverride": "1", |
| "EnableResizableBar": "1" |
| }, |
| "missing": [ |
| "ForceP2P", |
| "RMForceP2PType", |
| "RMPcieP2PType", |
| "GrdmaPciTopoCheckOverride", |
| "EnableResizableBar" |
| ], |
| "mismatched": {}, |
| "registry_dwords": "", |
| "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"", |
| "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded" |
| }, |
| "p2pmark": { |
| "status": "not_run" |
| }, |
| "amd_fabric": { |
| "status": "not_run" |
| }, |
| "hardware_run_summary": {}, |
| "event_log": [ |
| "16:57:44 benchmark start engine=vllm", |
| "16:57:44 startup server=http://192.168.1.201:8000 model=glm-5.3-flash-dflash2", |
| "16:57:44 startup decode concurrency=1 contexts=0", |
| "16:57:44 startup NVIDIA P2P override: unknown: /proc/driver/nvidia/params is not readable", |
| "16:57:44 startup engine vLLM 0.1.dev20051+g487ecf187 models=['glm-5.3-flash-dflash2']", |
| "16:57:44 startup KV cache budget from vLLM metrics: 6,624,000 tokens (2875 blocks x 2304)", |
| "16:57:44 startup model context length: 262,144 tokens", |
| "16:57:44 startup prefill tests: skipped", |
| "16:57:44 startup startup preparation done", |
| "16:57:44 startup hardware monitor disabled", |
| "16:57:44 decode warmup start", |
| "16:57:45 decode warmup start C=1 ctx=0 3s", |
| "16:57:45 cell start C=1 ctx=0", |
| "16:57:51 ready C=1 ctx=0 running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s", |
| "16:57:54 cell done C=1 ctx=0 41.7 tok/s | norm 13.5 step/s len=3.10", |
| "16:57:54 decode warmup done C=1 ctx=0", |
| "16:57:56 cell start C=1 ctx=0", |
| "16:58:02 ready C=1 ctx=0 running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s", |
| "16:58:12 cell done C=1 ctx=0 38.3 tok/s | norm 13.1 step/s len=2.92" |
| ], |
| "prefill": {}, |
| "results": [ |
| { |
| "concurrency": 1, |
| "context_tokens": 0, |
| "benchmark_mode": "duration", |
| "request_count_target": 0, |
| "warmup_request_count": 0, |
| "measurement_seconds": 9.979069, |
| "measurement_wall_seconds": 10.00222, |
| "client_output_tokens": 382, |
| "server_output_tokens": 382, |
| "aggregate_source": "openai_continuous_usage", |
| "aggregate_tps": 38.2801233299203, |
| "per_request_avg_tps": 38.2801233299203, |
| "ttft_avg": 0.24859522949554957, |
| "ttft_p50": 0.24859522949554957, |
| "ttft_p90": 0.26818874588352626, |
| "ttft_p99": 0.27259728707082104, |
| "time_to_second_token_avg": 0.07653589600522537, |
| "time_to_second_token_p50": 0.07653589600522537, |
| "time_to_second_token_p90": 0.07764024640491698, |
| "time_to_second_token_p99": 0.07788872524484759, |
| "request_latency_avg": 13.67058420900139, |
| "request_latency_p50": 13.67058420900139, |
| "request_latency_p90": 13.67058420900139, |
| "request_latency_p99": 13.67058420900139, |
| "inter_token_latency_avg": 0.022360881372554525, |
| "inter_token_latency_p50": 0.022360881372554525, |
| "inter_token_latency_p90": 0.025523418348860513, |
| "inter_token_latency_p99": 0.026234989168529357, |
| "output_tps_per_user_avg": 46.16378523847843, |
| "output_tps_per_user_p50": 46.16378523847843, |
| "output_tps_per_user_p90": 52.69280685218498, |
| "output_tps_per_user_p99": 54.16183671526895, |
| "e2e_output_tps_per_user_avg": 37.45267884476172, |
| "e2e_output_tps_per_user_p50": 37.45267884476172, |
| "e2e_output_tps_per_user_p90": 37.45267884476172, |
| "e2e_output_tps_per_user_p99": 37.45267884476172, |
| "chunk_inter_token_latency_avg": 0.07604885094410449, |
| "chunk_inter_token_latency_p50": 0.07604885094410449, |
| "chunk_inter_token_latency_p90": 0.07679192778881462, |
| "chunk_inter_token_latency_p99": 0.0769591200788744, |
| "input_seq_len_avg": 78.0, |
| "output_seq_len_avg": 512.0, |
| "output_seq_len_p50": 512.0, |
| "output_seq_len_p90": 512.0, |
| "output_seq_len_p99": 512.0, |
| "request_count": 2, |
| "completed_request_count": 1, |
| "request_samples": [ |
| { |
| "ttft": 0.2241033340105787, |
| "time_to_second_token": 0.07515545800561085, |
| "latency": 13.67058420900139, |
| "inter_token_latency_avg": 0.02631405259293701, |
| "chunk_inter_token_latency_avg": 0.07512000488821682, |
| "input_tokens": 78, |
| "output_tokens": 512, |
| "output_tps_per_user": 38.00250822134525, |
| "e2e_output_tps_per_user": 37.45267884476172, |
| "completed": true |
| }, |
| { |
| "ttft": 0.27308712498052046, |
| "time_to_second_token": 0.07791633400483988, |
| "latency": 0.0, |
| "inter_token_latency_avg": 0.01840771015217204, |
| "chunk_inter_token_latency_avg": 0.07697769699999216, |
| "input_tokens": 78, |
| "output_tokens": 93, |
| "output_tps_per_user": 54.32506225561161, |
| "e2e_output_tps_per_user": 0.0, |
| "completed": false |
| } |
| ], |
| "total_tokens": 382, |
| "wall_time": 15.714896708988817, |
| "num_completed": 1, |
| "num_errors": 0, |
| "server_gen_throughput": 38.087877857376206, |
| "server_utilization": 0.01217814892136393, |
| "server_spec_accept_rate": 0.27582417582417584, |
| "server_spec_accept_length": 2.930769230769231, |
| "server_spec_drafts": 130, |
| "server_spec_draft_tokens": 910, |
| "server_spec_accepted_tokens": 251, |
| "server_spec_pos_accept": [ |
| 0.7154, |
| 0.4692, |
| 0.2846, |
| 0.1846, |
| 0.1154, |
| 0.0923, |
| 0.0692 |
| ], |
| "server_engine_steps": 131.0, |
| "server_steps_per_s": 13.127476848742301, |
| "server_accept_len_effective": 2.9160305343511452, |
| "accept_norm_tps": 0.0, |
| "accept_norm_ref_len": 0.0, |
| "avg_running_reqs": 1, |
| "max_running_reqs": 1, |
| "effective_concurrency": 1, |
| "avg_queue_reqs": 0, |
| "max_queue_reqs": 0, |
| "queue_fraction": 0.0, |
| "underfilled": false, |
| "warmup_timed_out": false, |
| "warmup_duration": 5.66, |
| "ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s", |
| "timeout_reason": "", |
| "capacity_limited": false, |
| "hardware_summary": {} |
| } |
| ], |
| "summary_table": { |
| "0": { |
| "1": 38.2801233299203 |
| } |
| }, |
| "burst_results": [], |
| "burst_summary_table": {}, |
| "methodology": { |
| "prefill": { |
| "name": "Prefill", |
| "present": false, |
| "mode": "skipped", |
| "formula": "prompt_tokens / TTFT", |
| "notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation." |
| }, |
| "sustained_decode": { |
| "name": "Sustained Decode", |
| "present": true, |
| "formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable", |
| "notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline." |
| }, |
| "burst_e2e_decode": { |
| "name": "Burst / E2E Decode", |
| "present": false, |
| "status": "not run; use --run-burst", |
| "formula": "sum(completion_tokens) / profiling_wall_time", |
| "notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion." |
| }, |
| "acceptance_normalization": { |
| "name": "Acceptance-normalized decode (MTP / speculative)", |
| "present": true, |
| "formula": "engine_steps = spec_drafts + max(0, output_tokens - (accepted_tokens + spec_drafts)); accept_len_effective = output_tokens / engine_steps; steps_per_s = aggregate_tps / accept_len_effective", |
| "notes": "With speculative decoding tok/s = steps_per_s * accept_len, so raw tok/s mixes engine speed with data-dependent acceptance. steps_per_s (target-model forward passes per second) is the acceptance-independent speed used to compare runs; server_spec_pos_accept holds per-draft-position acceptance probabilities. Counters are vLLM window deltas; SGLang falls back to its lifetime accept-length gauge." |
| }, |
| "coding_peak": { |
| "name": "Coding Peak", |
| "present": true, |
| "formula": "usage.completion_tokens / (last_stream_time - first_token_time)", |
| "notes": "Sequential cc1 Sieve-of-Eratosthenes coding prompt, matching /mnt/test.py throughput semantics. Uses OpenAI stream usage with continuous_usage_stats when the server supports it." |
| } |
| }, |
| "coding_peak": { |
| "mode": "coding_peak", |
| "prompt": "Write a Python script that implements the Sieve of Eratosthenes.", |
| "runs_requested": 3, |
| "runs_ok": 3, |
| "max_tokens": 2000, |
| "temperature": 0.0, |
| "summary": { |
| "mean_generation_tok_s": 73.16228317941676, |
| "median_generation_tok_s": 73.68997497304997, |
| "max_generation_tok_s": 74.74192110738443, |
| "min_generation_tok_s": 71.0549534578159, |
| "cjk_runs": 0 |
| }, |
| "samples": [ |
| { |
| "run": 1, |
| "ok": true, |
| "finish_reason": "stop", |
| "completion_tokens": 1874, |
| "ttft": 0.24347924999892712, |
| "gen_elapsed": 26.373952958994778, |
| "total_elapsed": 26.617432208993705, |
| "generation_tok_s": 71.0549534578159, |
| "total_tok_s": 70.40498817789036, |
| "content_chars": 7012, |
| "reasoning_chars": 0, |
| "cjk_chars": 0, |
| "content_preview": "The user wants a Python script implementing the Sieve of Eratosthenes. This is a classic algorithm for finding all prime numbers up to a given limit. Let me think about what makes a good implementation.\n\nThe Sieve of Eratosthenes works as follows:\n1. Create a boolean array of size n+1, initialized to True (assuming all numbers are prime initially)\n2. Mark 0 and 1 as not prime\n3. Starting from 2, for each number that is still marked as prime, mark all its multiples as not prime\n4. The key optimiz" |
| }, |
| { |
| "run": 2, |
| "ok": true, |
| "finish_reason": "length", |
| "completion_tokens": 2000, |
| "ttft": 0.19760791698354296, |
| "gen_elapsed": 27.14073387501412, |
| "total_elapsed": 27.338341791997664, |
| "generation_tok_s": 73.68997497304997, |
| "total_tok_s": 73.15732663000905, |
| "content_chars": 7295, |
| "reasoning_chars": 0, |
| "cjk_chars": 0, |
| "content_preview": "The user wants a Python script implementing the Sieve of Eratosthenes. This is a classic algorithm for finding all prime numbers up to a given limit. Let me think about what makes a good implementation:\n\n1. **The algorithm basics:**\n - Create a boolean array of size n+1, initialized to True\n - Mark 0 and 1 as not prime\n - For each number p starting from 2, if p is still marked prime, mark all multiples of p (starting from p²) as not prime\n - Continue until p² > n\n - Collect all indices" |
| }, |
| { |
| "run": 3, |
| "ok": true, |
| "finish_reason": "stop", |
| "completion_tokens": 1829, |
| "ttft": 0.19229045798419975, |
| "gen_elapsed": 24.470872208010405, |
| "total_elapsed": 24.663162665994605, |
| "generation_tok_s": 74.74192110738443, |
| "total_tok_s": 74.15918326329707, |
| "content_chars": 6561, |
| "reasoning_chars": 0, |
| "cjk_chars": 0, |
| "content_preview": "The user wants a Python script implementing the Sieve of Eratosthenes. This is a classic algorithm for finding all prime numbers up to a given limit. Let me think about what makes a good response here.\n\nThe Sieve of Eratosthenes algorithm:\n1. Create a boolean list of size n+1, initialized to True (assuming all numbers are prime)\n2. Mark 0 and 1 as not prime\n3. Starting from 2, for each number p that is still marked prime, mark all multiples of p (starting from p²) as not prime\n4. Continue until " |
| } |
| ] |
| } |
| } |
|
|