{
  "metadata": {
    "version": "0.4.30",
    "engine": "vllm",
    "model": "DeepSeek-V4-Flash",
    "server": "http://192.168.0.213",
    "timestamp": "2026-07-12T22:51:24.158492",
    "decode_mode": "duration",
    "primary_decode_layer": "sustained_decode",
    "duration_per_test": 30.0,
    "request_count": 0,
    "warmup_request_count": 0,
    "run_burst": false,
    "prefill_mode": "standalone_cold",
    "standalone_prefill": true,
    "prefill_only": true,
    "skip_prefill": false,
    "burst_e2e_status": "not_run_use_--run-burst",
    "burst_request_count": 0,
    "burst_warmup_request_count": 0,
    "burst_requests_per_concurrency": 5,
    "decode_warmup_seconds": 3.0,
    "decode_warmup_context": 0,
    "decode_warmup_concurrency": 1,
    "cell_warmup_timeout_seconds": 0.0,
    "cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
    "show_capacity_limited_values": false,
    "max_tokens": 1,
    "temperature": null,
    "ignore_eos": true,
    "max_total_tokens": 0,
    "dcp_size": 0,
    "metrics_available": true,
    "metrics_warning": "",
    "concurrency_levels": [
      1
    ],
    "context_lengths": [
      1044480
    ],
    "unique_context_percent": 0.0,
    "shared_context_percent": 100.0,
    "context_sharing_mode": "fully_shared",
    "startup_diagnostics_available": true,
    "nvidia_p2p_override_effective": false,
    "p2pmark_status": "not_run",
    "amd_fabric_status": "not_run"
  },
  "startup_diagnostics": {
    "version": "0.4.30",
    "server_url": "http://192.168.0.213:8000",
    "hostname": "DESKTOP-RMAJ6JK",
    "uname": "",
    "env": {},
    "args": {
      "concurrency": "1",
      "contexts": "1020k",
      "max_tokens": 1,
      "duration": 30.0,
      "request_count": 0,
      "run_burst": false,
      "standalone_prefill": true,
      "prefill_only": true,
      "skip_prefill": false,
      "prefill_contexts": "1020k",
      "prefill_metric": "client",
      "dcp_size": 0,
      "kv_budget": 0
    },
    "nvidia_p2p_override": {
      "effective": false,
      "configured": false,
      "params_path": "/proc/driver/nvidia/params",
      "params_available": false,
      "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
      "modprobe_available": false,
      "runtime": {
        "ForceP2P": "",
        "RMForceP2PType": "",
        "RMPcieP2PType": "",
        "GrdmaPciTopoCheckOverride": "",
        "EnableResizableBar": "",
        "DmaRemapPeerMmio": ""
      },
      "expected": {
        "ForceP2P": "0x11",
        "RMForceP2PType": "1",
        "RMPcieP2PType": "2",
        "GrdmaPciTopoCheckOverride": "1",
        "EnableResizableBar": "1"
      },
      "missing": [
        "ForceP2P",
        "RMForceP2PType",
        "RMPcieP2PType",
        "GrdmaPciTopoCheckOverride",
        "EnableResizableBar"
      ],
      "mismatched": {},
      "registry_dwords": "",
      "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
      "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
    },
    "p2pmark": {
      "status": "not_run"
    },
    "amd_fabric": {
      "status": "not_run"
    },
    "nvidia_smi_query": {
      "cmd": [
        "nvidia-smi",
        "--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
        "--format=csv,noheader,nounits"
      ],
      "returncode": 0,
      "stdout": "0, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 595.79, 00000000:01:00.0, 1, 16, 450.00",
      "stderr": ""
    },
    "nvidia_smi_topo": {
      "cmd": [
        "nvidia-smi",
        "topo",
        "-m"
      ],
      "returncode": 255,
      "stdout": "ERROR: Option -m is missing its value. Please run 'nvidia-smi -h' for help.",
      "stderr": ""
    }
  },
  "nvidia_p2p_override": {
    "effective": false,
    "configured": false,
    "params_path": "/proc/driver/nvidia/params",
    "params_available": false,
    "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
    "modprobe_available": false,
    "runtime": {
      "ForceP2P": "",
      "RMForceP2PType": "",
      "RMPcieP2PType": "",
      "GrdmaPciTopoCheckOverride": "",
      "EnableResizableBar": "",
      "DmaRemapPeerMmio": ""
    },
    "expected": {
      "ForceP2P": "0x11",
      "RMForceP2PType": "1",
      "RMPcieP2PType": "2",
      "GrdmaPciTopoCheckOverride": "1",
      "EnableResizableBar": "1"
    },
    "missing": [
      "ForceP2P",
      "RMForceP2PType",
      "RMPcieP2PType",
      "GrdmaPciTopoCheckOverride",
      "EnableResizableBar"
    ],
    "mismatched": {},
    "registry_dwords": "",
    "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
    "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
  },
  "p2pmark": {
    "status": "not_run"
  },
  "amd_fabric": {
    "status": "not_run"
  },
  "hardware_run_summary": {},
  "event_log": [],
  "prefill": {
    "1044480": {
      "ttft_seconds": 261.157,
      "prefill_seconds": 261.157,
      "tok_per_sec": 3999.0,
      "client_ttft_seconds": 261.157,
      "client_tok_per_sec": 3999.0,
      "prompt_tokens": 1044482,
      "samples": 1,
      "method": "client",
      "server_validation": {
        "method": "",
        "tok_per_sec": 0.0,
        "prefill_seconds": 0.0,
        "prompt_tokens": 0,
        "request_prompt_tokens": 0,
        "cached_tokens": 0,
        "token_source": "",
        "samples": 0,
        "invalid_reason": ""
      },
      "hardware_summary": {}
    }
  },
  "results": [],
  "summary_table": {},
  "burst_results": [],
  "burst_summary_table": {},
  "methodology": {
    "prefill": {
      "name": "Prefill",
      "present": true,
      "mode": "standalone_cold",
      "formula": "prompt_tokens / TTFT",
      "notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
    },
    "sustained_decode": {
      "name": "Sustained Decode",
      "present": false,
      "formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
      "notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline. Context sharing is controlled by --unique-context-percent: one scout warms stream 0, then each worker diverges at the requested suffix boundary."
    },
    "burst_e2e_decode": {
      "name": "Burst / E2E Decode",
      "present": false,
      "status": "not run; use --run-burst",
      "formula": "sum(completion_tokens) / profiling_wall_time",
      "notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
    }
  }
}