{
  "generated_at": "2026-09-15T10:45:10.511Z",
  "total": 12,
  "exported": 12,
  "truncated": false,
  "filters": {
    "hardwareId": null,
    "modelId": "unknown:meta-llama-llama-3-1-8b-instruct",
    "quantId": null,
    "frameworkId": null,
    "platform": null,
    "minEvidence": "L0",
    "range": "all",
    "sort": "recent",
    "page": 1,
    "perPage": 50,
    "keyword": null,
    "scope": "all"
  },
  "data_status": "sample",
  "runs": [
    {
      "model": "Llama-3.1-8B-Instruct · 未归一化",
      "quant": "Q8_0",
      "hardware": "Apple M3 Max · 128G 统一内存",
      "framework": "vLLM 0.6.2",
      "decode_tps": 32.7,
      "prefill_tps": 1142.4,
      "ttft_cold_ms": 4322,
      "vram_peak_gb": 11.3,
      "sample_size": 5,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待判定",
      "source_platform": "HuggingFace",
      "source_url": "https://huggingface.co/araoai-demo/llama-3-1-8b-instruct-bench/discussions/2",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Llama-3.1-8B-Instruct · 未归一化",
      "quant": "Q5_K_M",
      "hardware": "Arc A770 · 未归一化",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 65.7,
      "prefill_tps": 2266,
      "ttft_cold_ms": 3113,
      "vram_peak_gb": 8,
      "sample_size": 9,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "GitHub",
      "source_url": "https://github.com/araoai-demo/benchmarks/issues/1021",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Llama-3.1-8B-Instruct · 未归一化",
      "quant": "AWQ 4-bit",
      "hardware": "Arc A770 · 未归一化",
      "framework": "llama.cpp b5980",
      "decode_tps": 74.1,
      "prefill_tps": 2494.7,
      "ttft_cold_ms": 2655,
      "vram_peak_gb": 6.7,
      "sample_size": 12,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "V2EX",
      "source_url": "https://www.v2ex.com/t/araoai-demo-24",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Llama-3.1-8B-Instruct · 未归一化",
      "quant": "Q4_K_M",
      "hardware": "RX 7800 XT · 未归一化",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 85.4,
      "prefill_tps": 2945.8,
      "ttft_cold_ms": 2780,
      "vram_peak_gb": 7.1,
      "sample_size": 16,
      "evidence": "L2",
      "repro_count": 3,
      "repro_chain": "一致 3 次，偏差 0 次，共 3 格",
      "status": "稳定",
      "source_platform": "GitHub",
      "source_url": "https://github.com/araoai-demo/benchmarks/issues/1028",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Llama-3.1-8B-Instruct · 未归一化",
      "quant": "Q4_K_M",
      "hardware": "RX 7800 XT · 未归一化",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 79,
      "prefill_tps": 2673,
      "ttft_cold_ms": 2780,
      "vram_peak_gb": 7.1,
      "sample_size": 17,
      "evidence": "L2",
      "repro_count": 2,
      "repro_chain": "一致 2 次，偏差 0 次，共 3 格",
      "status": "稳定",
      "source_platform": "Reddit",
      "source_url": "https://www.reddit.com/r/araoai_demo/comments/demo2900_llama-3-1-8b-instruct/",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Llama-3.1-8B-Instruct · 未归一化",
      "quant": "Q6_K",
      "hardware": "RX 7900 GRE · 未归一化",
      "framework": "vLLM 0.6.2",
      "decode_tps": 60.6,
      "prefill_tps": 2118.7,
      "ttft_cold_ms": 3530,
      "vram_peak_gb": 9.1,
      "sample_size": 4,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待判定",
      "source_platform": "V2EX",
      "source_url": "https://www.v2ex.com/t/araoai-demo-31",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Llama-3.1-8B-Instruct · 未归一化",
      "quant": "Q5_K_M",
      "hardware": "NVIDIA RTX 3090 · 24G",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 109.8,
      "prefill_tps": 3787.5,
      "ttft_cold_ms": 3113,
      "vram_peak_gb": 8,
      "sample_size": 13,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "B站",
      "source_url": "https://www.bilibili.com/video/araoai-demo-40",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Llama-3.1-8B-Instruct · 未归一化",
      "quant": "AWQ 4-bit",
      "hardware": "RX 7900 GRE · 未归一化",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 77.8,
      "prefill_tps": 2631.8,
      "ttft_cold_ms": 2655,
      "vram_peak_gb": 6.7,
      "sample_size": 17,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "V2EX",
      "source_url": "https://www.v2ex.com/t/araoai-demo-59",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Llama-3.1-8B-Instruct · 未归一化",
      "quant": "AWQ 4-bit",
      "hardware": "RTX 4080 SUPER · 未归一化",
      "framework": "vLLM 0.6.2",
      "decode_tps": 113.6,
      "prefill_tps": 3970.5,
      "ttft_cold_ms": 2655,
      "vram_peak_gb": 6.7,
      "sample_size": 4,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待判定",
      "source_platform": "B站",
      "source_url": "https://www.bilibili.com/video/araoai-demo-61",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Llama-3.1-8B-Instruct · 未归一化",
      "quant": "Q5_K_M",
      "hardware": "Apple M3 Max · 128G 统一内存",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 43.4,
      "prefill_tps": 1468.7,
      "ttft_cold_ms": 3113,
      "vram_peak_gb": 8,
      "sample_size": 6,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待判定",
      "source_platform": "GitHub",
      "source_url": "https://github.com/araoai-demo/benchmarks/issues/1063",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Llama-3.1-8B-Instruct · 未归一化",
      "quant": "Q8_0",
      "hardware": "RX 7900 GRE · 未归一化",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 41.2,
      "prefill_tps": 1393.3,
      "ttft_cold_ms": 4322,
      "vram_peak_gb": 11.3,
      "sample_size": 9,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "V2EX",
      "source_url": "https://www.v2ex.com/t/araoai-demo-66",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Llama-3.1-8B-Instruct · 未归一化",
      "quant": "Q8_0",
      "hardware": "RX 7800 XT · 未归一化",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 44.6,
      "prefill_tps": 1509.4,
      "ttft_cold_ms": 4322,
      "vram_peak_gb": 11.3,
      "sample_size": 14,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "Reddit",
      "source_url": "https://www.reddit.com/r/araoai_demo/comments/demo71_llama-3-1-8b-instruct/",
      "fetched_at": "2026-09-14T00:00:00Z"
    }
  ]
}
