{
  "machine": "Apple M4, 10 GPU cores, 32 GiB unified memory",
  "mlx": "0.32.2",
  "mlx_lm": "0.31.3",
  "target": "Qwen/Qwen3-4B",
  "target_revision": "1cfa9a7208912126459214e8b04321603b3df60c",
  "draft": "z-lab/Qwen3-4B-DFlash-b16",
  "draft_revision": "b74e3a329c4d963783143b1e970d95b002be72bd",
  "dflash_source_commit": "07ebd93db9f472af339b644bb70221ad8428328a",
  "dtype": "bfloat16",
  "measurement_limits": [
    "shared host; no workload exclusion",
    "no DRAM or GPU arithmetic utilization counters",
    "single DFlash task, greedy fixed output length",
    "logical operand bytes and analytical dense linear FLOPs only"
  ]
}