GLM-5.3-Flash-DFlash2-TP4-Spark / glm53-dflash2-v2-code-fixed.json
cfontes's picture
Upload folder using huggingface_hub
fe78d4c verified
Raw
History Blame Contribute Delete
16.2 kB
{
"metadata": {
"version": "0.4.29",
"engine": "vllm",
"model": "glm-5.3-flash-dflash2",
"server": "192.168.1.201:8000",
"timestamp": "2026-09-02T16:58:14.207747",
"decode_mode": "duration",
"primary_decode_layer": "sustained_decode",
"duration_per_test": 10.0,
"request_count": 0,
"warmup_request_count": 0,
"run_burst": false,
"prefill_mode": "skipped",
"standalone_prefill": false,
"prefill_only": false,
"skip_prefill": true,
"burst_e2e_status": "not_run_use_--run-burst",
"burst_request_count": 0,
"burst_warmup_request_count": 0,
"burst_requests_per_concurrency": 5,
"decode_warmup_seconds": 3.0,
"decode_warmup_context": 0,
"decode_warmup_concurrency": 1,
"cell_warmup_timeout_seconds": 0.0,
"cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
"show_capacity_limited_values": false,
"max_tokens": 512,
"temperature": 0.0,
"ignore_eos": true,
"max_total_tokens": 6624000,
"dcp_size": 0,
"metrics_available": true,
"metrics_warning": "",
"concurrency_levels": [
1
],
"context_lengths": [
0
],
"startup_diagnostics_available": true,
"nvidia_p2p_override_effective": false,
"p2pmark_status": "not_run",
"amd_fabric_status": "not_run"
},
"startup_diagnostics": {
"version": "0.4.29",
"server_url": "http://192.168.1.201:8000",
"hostname": "macmini",
"uname": "Darwin macmini 25.5.0 Darwin Kernel Version 25.5.0: Tue Jun 9 22:28:34 PDT 2026; root:xnu-12377.121.10~1/RELEASE_ARM64_T6041 arm64",
"env": {},
"args": {
"concurrency": "1",
"contexts": "0",
"max_tokens": 512,
"duration": 10.0,
"request_count": 0,
"run_burst": false,
"standalone_prefill": false,
"prefill_only": false,
"skip_prefill": true,
"prefill_contexts": "8k,64k,128k",
"prefill_metric": "client",
"dcp_size": 0,
"kv_budget": 0
},
"nvidia_p2p_override": {
"effective": false,
"configured": false,
"params_path": "/proc/driver/nvidia/params",
"params_available": false,
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
"modprobe_available": false,
"runtime": {
"ForceP2P": "",
"RMForceP2PType": "",
"RMPcieP2PType": "",
"GrdmaPciTopoCheckOverride": "",
"EnableResizableBar": "",
"DmaRemapPeerMmio": ""
},
"expected": {
"ForceP2P": "0x11",
"RMForceP2PType": "1",
"RMPcieP2PType": "2",
"GrdmaPciTopoCheckOverride": "1",
"EnableResizableBar": "1"
},
"missing": [
"ForceP2P",
"RMForceP2PType",
"RMPcieP2PType",
"GrdmaPciTopoCheckOverride",
"EnableResizableBar"
],
"mismatched": {},
"registry_dwords": "",
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
},
"p2pmark": {
"status": "not_run"
},
"amd_fabric": {
"status": "not_run"
},
"nvidia_smi_error": "nvidia-smi not found"
},
"nvidia_p2p_override": {
"effective": false,
"configured": false,
"params_path": "/proc/driver/nvidia/params",
"params_available": false,
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
"modprobe_available": false,
"runtime": {
"ForceP2P": "",
"RMForceP2PType": "",
"RMPcieP2PType": "",
"GrdmaPciTopoCheckOverride": "",
"EnableResizableBar": "",
"DmaRemapPeerMmio": ""
},
"expected": {
"ForceP2P": "0x11",
"RMForceP2PType": "1",
"RMPcieP2PType": "2",
"GrdmaPciTopoCheckOverride": "1",
"EnableResizableBar": "1"
},
"missing": [
"ForceP2P",
"RMForceP2PType",
"RMPcieP2PType",
"GrdmaPciTopoCheckOverride",
"EnableResizableBar"
],
"mismatched": {},
"registry_dwords": "",
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
},
"p2pmark": {
"status": "not_run"
},
"amd_fabric": {
"status": "not_run"
},
"hardware_run_summary": {},
"event_log": [
"16:57:44 benchmark start engine=vllm",
"16:57:44 startup server=http://192.168.1.201:8000 model=glm-5.3-flash-dflash2",
"16:57:44 startup decode concurrency=1 contexts=0",
"16:57:44 startup NVIDIA P2P override: unknown: /proc/driver/nvidia/params is not readable",
"16:57:44 startup engine vLLM 0.1.dev20051+g487ecf187 models=['glm-5.3-flash-dflash2']",
"16:57:44 startup KV cache budget from vLLM metrics: 6,624,000 tokens (2875 blocks x 2304)",
"16:57:44 startup model context length: 262,144 tokens",
"16:57:44 startup prefill tests: skipped",
"16:57:44 startup startup preparation done",
"16:57:44 startup hardware monitor disabled",
"16:57:44 decode warmup start",
"16:57:45 decode warmup start C=1 ctx=0 3s",
"16:57:45 cell start C=1 ctx=0",
"16:57:51 ready C=1 ctx=0 running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
"16:57:54 cell done C=1 ctx=0 41.7 tok/s | norm 13.5 step/s len=3.10",
"16:57:54 decode warmup done C=1 ctx=0",
"16:57:56 cell start C=1 ctx=0",
"16:58:02 ready C=1 ctx=0 running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
"16:58:12 cell done C=1 ctx=0 38.3 tok/s | norm 13.1 step/s len=2.92"
],
"prefill": {},
"results": [
{
"concurrency": 1,
"context_tokens": 0,
"benchmark_mode": "duration",
"request_count_target": 0,
"warmup_request_count": 0,
"measurement_seconds": 9.979069,
"measurement_wall_seconds": 10.00222,
"client_output_tokens": 382,
"server_output_tokens": 382,
"aggregate_source": "openai_continuous_usage",
"aggregate_tps": 38.2801233299203,
"per_request_avg_tps": 38.2801233299203,
"ttft_avg": 0.24859522949554957,
"ttft_p50": 0.24859522949554957,
"ttft_p90": 0.26818874588352626,
"ttft_p99": 0.27259728707082104,
"time_to_second_token_avg": 0.07653589600522537,
"time_to_second_token_p50": 0.07653589600522537,
"time_to_second_token_p90": 0.07764024640491698,
"time_to_second_token_p99": 0.07788872524484759,
"request_latency_avg": 13.67058420900139,
"request_latency_p50": 13.67058420900139,
"request_latency_p90": 13.67058420900139,
"request_latency_p99": 13.67058420900139,
"inter_token_latency_avg": 0.022360881372554525,
"inter_token_latency_p50": 0.022360881372554525,
"inter_token_latency_p90": 0.025523418348860513,
"inter_token_latency_p99": 0.026234989168529357,
"output_tps_per_user_avg": 46.16378523847843,
"output_tps_per_user_p50": 46.16378523847843,
"output_tps_per_user_p90": 52.69280685218498,
"output_tps_per_user_p99": 54.16183671526895,
"e2e_output_tps_per_user_avg": 37.45267884476172,
"e2e_output_tps_per_user_p50": 37.45267884476172,
"e2e_output_tps_per_user_p90": 37.45267884476172,
"e2e_output_tps_per_user_p99": 37.45267884476172,
"chunk_inter_token_latency_avg": 0.07604885094410449,
"chunk_inter_token_latency_p50": 0.07604885094410449,
"chunk_inter_token_latency_p90": 0.07679192778881462,
"chunk_inter_token_latency_p99": 0.0769591200788744,
"input_seq_len_avg": 78.0,
"output_seq_len_avg": 512.0,
"output_seq_len_p50": 512.0,
"output_seq_len_p90": 512.0,
"output_seq_len_p99": 512.0,
"request_count": 2,
"completed_request_count": 1,
"request_samples": [
{
"ttft": 0.2241033340105787,
"time_to_second_token": 0.07515545800561085,
"latency": 13.67058420900139,
"inter_token_latency_avg": 0.02631405259293701,
"chunk_inter_token_latency_avg": 0.07512000488821682,
"input_tokens": 78,
"output_tokens": 512,
"output_tps_per_user": 38.00250822134525,
"e2e_output_tps_per_user": 37.45267884476172,
"completed": true
},
{
"ttft": 0.27308712498052046,
"time_to_second_token": 0.07791633400483988,
"latency": 0.0,
"inter_token_latency_avg": 0.01840771015217204,
"chunk_inter_token_latency_avg": 0.07697769699999216,
"input_tokens": 78,
"output_tokens": 93,
"output_tps_per_user": 54.32506225561161,
"e2e_output_tps_per_user": 0.0,
"completed": false
}
],
"total_tokens": 382,
"wall_time": 15.714896708988817,
"num_completed": 1,
"num_errors": 0,
"server_gen_throughput": 38.087877857376206,
"server_utilization": 0.01217814892136393,
"server_spec_accept_rate": 0.27582417582417584,
"server_spec_accept_length": 2.930769230769231,
"server_spec_drafts": 130,
"server_spec_draft_tokens": 910,
"server_spec_accepted_tokens": 251,
"server_spec_pos_accept": [
0.7154,
0.4692,
0.2846,
0.1846,
0.1154,
0.0923,
0.0692
],
"server_engine_steps": 131.0,
"server_steps_per_s": 13.127476848742301,
"server_accept_len_effective": 2.9160305343511452,
"accept_norm_tps": 0.0,
"accept_norm_ref_len": 0.0,
"avg_running_reqs": 1,
"max_running_reqs": 1,
"effective_concurrency": 1,
"avg_queue_reqs": 0,
"max_queue_reqs": 0,
"queue_fraction": 0.0,
"underfilled": false,
"warmup_timed_out": false,
"warmup_duration": 5.66,
"ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
"timeout_reason": "",
"capacity_limited": false,
"hardware_summary": {}
}
],
"summary_table": {
"0": {
"1": 38.2801233299203
}
},
"burst_results": [],
"burst_summary_table": {},
"methodology": {
"prefill": {
"name": "Prefill",
"present": false,
"mode": "skipped",
"formula": "prompt_tokens / TTFT",
"notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
},
"sustained_decode": {
"name": "Sustained Decode",
"present": true,
"formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
"notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
},
"burst_e2e_decode": {
"name": "Burst / E2E Decode",
"present": false,
"status": "not run; use --run-burst",
"formula": "sum(completion_tokens) / profiling_wall_time",
"notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
},
"acceptance_normalization": {
"name": "Acceptance-normalized decode (MTP / speculative)",
"present": true,
"formula": "engine_steps = spec_drafts + max(0, output_tokens - (accepted_tokens + spec_drafts)); accept_len_effective = output_tokens / engine_steps; steps_per_s = aggregate_tps / accept_len_effective",
"notes": "With speculative decoding tok/s = steps_per_s * accept_len, so raw tok/s mixes engine speed with data-dependent acceptance. steps_per_s (target-model forward passes per second) is the acceptance-independent speed used to compare runs; server_spec_pos_accept holds per-draft-position acceptance probabilities. Counters are vLLM window deltas; SGLang falls back to its lifetime accept-length gauge."
},
"coding_peak": {
"name": "Coding Peak",
"present": true,
"formula": "usage.completion_tokens / (last_stream_time - first_token_time)",
"notes": "Sequential cc1 Sieve-of-Eratosthenes coding prompt, matching /mnt/test.py throughput semantics. Uses OpenAI stream usage with continuous_usage_stats when the server supports it."
}
},
"coding_peak": {
"mode": "coding_peak",
"prompt": "Write a Python script that implements the Sieve of Eratosthenes.",
"runs_requested": 3,
"runs_ok": 3,
"max_tokens": 2000,
"temperature": 0.0,
"summary": {
"mean_generation_tok_s": 73.16228317941676,
"median_generation_tok_s": 73.68997497304997,
"max_generation_tok_s": 74.74192110738443,
"min_generation_tok_s": 71.0549534578159,
"cjk_runs": 0
},
"samples": [
{
"run": 1,
"ok": true,
"finish_reason": "stop",
"completion_tokens": 1874,
"ttft": 0.24347924999892712,
"gen_elapsed": 26.373952958994778,
"total_elapsed": 26.617432208993705,
"generation_tok_s": 71.0549534578159,
"total_tok_s": 70.40498817789036,
"content_chars": 7012,
"reasoning_chars": 0,
"cjk_chars": 0,
"content_preview": "The user wants a Python script implementing the Sieve of Eratosthenes. This is a classic algorithm for finding all prime numbers up to a given limit. Let me think about what makes a good implementation.\n\nThe Sieve of Eratosthenes works as follows:\n1. Create a boolean array of size n+1, initialized to True (assuming all numbers are prime initially)\n2. Mark 0 and 1 as not prime\n3. Starting from 2, for each number that is still marked as prime, mark all its multiples as not prime\n4. The key optimiz"
},
{
"run": 2,
"ok": true,
"finish_reason": "length",
"completion_tokens": 2000,
"ttft": 0.19760791698354296,
"gen_elapsed": 27.14073387501412,
"total_elapsed": 27.338341791997664,
"generation_tok_s": 73.68997497304997,
"total_tok_s": 73.15732663000905,
"content_chars": 7295,
"reasoning_chars": 0,
"cjk_chars": 0,
"content_preview": "The user wants a Python script implementing the Sieve of Eratosthenes. This is a classic algorithm for finding all prime numbers up to a given limit. Let me think about what makes a good implementation:\n\n1. **The algorithm basics:**\n - Create a boolean array of size n+1, initialized to True\n - Mark 0 and 1 as not prime\n - For each number p starting from 2, if p is still marked prime, mark all multiples of p (starting from p²) as not prime\n - Continue until p² > n\n - Collect all indices"
},
{
"run": 3,
"ok": true,
"finish_reason": "stop",
"completion_tokens": 1829,
"ttft": 0.19229045798419975,
"gen_elapsed": 24.470872208010405,
"total_elapsed": 24.663162665994605,
"generation_tok_s": 74.74192110738443,
"total_tok_s": 74.15918326329707,
"content_chars": 6561,
"reasoning_chars": 0,
"cjk_chars": 0,
"content_preview": "The user wants a Python script implementing the Sieve of Eratosthenes. This is a classic algorithm for finding all prime numbers up to a given limit. Let me think about what makes a good response here.\n\nThe Sieve of Eratosthenes algorithm:\n1. Create a boolean list of size n+1, initialized to True (assuming all numbers are prime)\n2. Mark 0 and 1 as not prime\n3. Starting from 2, for each number p that is still marked prime, mark all multiples of p (starting from p²) as not prime\n4. Continue until "
}
]
}
}