{
  "generated_at": "2026-09-14T20:11:27.997Z",
  "total": 72,
  "exported": 72,
  "truncated": false,
  "filters": {
    "hardwareId": null,
    "modelId": null,
    "quantId": null,
    "frameworkId": null,
    "platform": null,
    "minEvidence": "L0",
    "range": "30d",
    "sort": "recent",
    "page": 1,
    "perPage": 50
  },
  "data_status": "sample",
  "runs": [
    {
      "model": "Qwen3-14B",
      "quant": "Q5_K_M",
      "hardware": "NVIDIA RTX 4090 · 24G",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 62.5,
      "prefill_tps": 2114.9,
      "ttft_cold_ms": 4863,
      "vram_peak_gb": 12.8,
      "sample_size": 4,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "Reddit",
      "source_url": "https://www.reddit.com/r/araoai_demo/comments/demo1_qwen3-14b-gguf/",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "meta-llama/Llama-3.1-8B-Instruct",
      "quant": "Q8_0",
      "hardware": "Apple M3 Max · 128G 统一内存",
      "framework": "vLLM 0.6.2",
      "decode_tps": 32.7,
      "prefill_tps": 1142.4,
      "ttft_cold_ms": 4322,
      "vram_peak_gb": 11.3,
      "sample_size": 5,
      "evidence": "L1",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "HuggingFace",
      "source_url": "https://huggingface.co/araoai-demo/llama-3-1-8b-instruct-bench/discussions/2",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "deepseek-ai/DeepSeek-R1-Distill-Qwen-14B",
      "quant": "Q4_K_M",
      "hardware": "rtx_3060",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 28.2,
      "prefill_tps": 971.2,
      "ttft_cold_ms": 4280,
      "vram_peak_gb": 11.2,
      "sample_size": 6,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "V2EX",
      "source_url": "https://www.v2ex.com/t/araoai-demo-3",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen/Qwen2.5-7B-Instruct-GGUF",
      "quant": "AWQ 4-bit",
      "hardware": "Apple M3 Max · 128G 统一内存",
      "framework": "vLLM 0.6.2",
      "decode_tps": 70.5,
      "prefill_tps": 2466.2,
      "ttft_cold_ms": 2421,
      "vram_peak_gb": 6.1,
      "sample_size": 7,
      "evidence": "L2",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "知乎",
      "source_url": "https://zhuanlan.zhihu.com/p/araoai-demo-4",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "mistralai/Mistral-Nemo-Instruct-2407",
      "quant": "Q5_K_M",
      "hardware": "NVIDIA RTX 4090 · 24G",
      "framework": "llama.cpp b6120",
      "decode_tps": 74.4,
      "prefill_tps": 2529.6,
      "ttft_cold_ms": 4280,
      "vram_peak_gb": 11.2,
      "sample_size": 8,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "B站",
      "source_url": "https://www.bilibili.com/video/araoai-demo-5",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "microsoft/phi-4-GGUF",
      "quant": "Q4_K_M",
      "hardware": "rx_7900_gre",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 45.1,
      "prefill_tps": 1553.8,
      "ttft_cold_ms": 4280,
      "vram_peak_gb": 11.2,
      "sample_size": 9,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "站方自产",
      "source_url": "https://araoai.com/demo/rig-run-6",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "microsoft/phi-4-GGUF",
      "quant": "Q6_K",
      "hardware": "NVIDIA RTX 3090 · 24G",
      "framework": "llama.cpp b5980",
      "decode_tps": 48.2,
      "prefill_tps": 1624.5,
      "ttft_cold_ms": 5592,
      "vram_peak_gb": 14.8,
      "sample_size": 10,
      "evidence": "L1",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "GitHub",
      "source_url": "https://github.com/araoai-demo/benchmarks/issues/1007",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "microsoft/phi-4-GGUF",
      "quant": "Q5_K_M",
      "hardware": "rtx_4080_super",
      "framework": "ollama 0.5.1",
      "decode_tps": 43.3,
      "prefill_tps": 1448.1,
      "ttft_cold_ms": 4863,
      "vram_peak_gb": 12.8,
      "sample_size": 11,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "Reddit",
      "source_url": "https://www.reddit.com/r/araoai_demo/comments/demo8_phi-4-gguf/",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen3-14B",
      "quant": "Q8_0",
      "hardware": "Apple M3 Max · 128G 统一内存",
      "framework": "ollama 0.5.1",
      "decode_tps": 15.5,
      "prefill_tps": 518.5,
      "ttft_cold_ms": 6978,
      "vram_peak_gb": 18.6,
      "sample_size": 12,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "HuggingFace",
      "source_url": "https://huggingface.co/araoai-demo/qwen3-14b-gguf-bench/discussions/9",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "mistralai/Mistral-Nemo-Instruct-2407",
      "quant": "Q4_K_M",
      "hardware": "apple_m2_ultra",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 73,
      "prefill_tps": 2517.8,
      "ttft_cold_ms": 3780,
      "vram_peak_gb": 9.8,
      "sample_size": 13,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "V2EX",
      "source_url": "https://www.v2ex.com/t/araoai-demo-10",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "microsoft/phi-4-GGUF",
      "quant": "Q4_K_M",
      "hardware": "NVIDIA RTX 3090 · 24G",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 73.2,
      "prefill_tps": 2525,
      "ttft_cold_ms": 4280,
      "vram_peak_gb": 11.2,
      "sample_size": 14,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "知乎",
      "source_url": "https://zhuanlan.zhihu.com/p/araoai-demo-11",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen3-32B",
      "quant": "Q4_K_M",
      "hardware": "Apple M3 Max · 128G 统一内存",
      "framework": "llama.cpp b6120",
      "decode_tps": 12.9,
      "prefill_tps": 439.2,
      "ttft_cold_ms": 8780,
      "vram_peak_gb": 23.5,
      "sample_size": 15,
      "evidence": "L1",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "B站",
      "source_url": "https://www.bilibili.com/video/araoai-demo-12",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen/Qwen2.5-7B-Instruct-GGUF",
      "quant": "Q4_K_M",
      "hardware": "AMD Radeon RX 7900 XTX · 24G",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 138.9,
      "prefill_tps": 4699.7,
      "ttft_cold_ms": 2530,
      "vram_peak_gb": 6.4,
      "sample_size": 16,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "站方自产",
      "source_url": "https://araoai.com/demo/rig-run-13",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen/Qwen2.5-7B-Instruct-GGUF",
      "quant": "Q8_0",
      "hardware": "rx_7800_xt",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 51,
      "prefill_tps": 1725.1,
      "ttft_cold_ms": 3879,
      "vram_peak_gb": 10.1,
      "sample_size": 17,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "GitHub",
      "source_url": "https://github.com/araoai-demo/benchmarks/issues/1014",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen/Qwen2.5-7B-Instruct-GGUF",
      "quant": "AWQ 4-bit",
      "hardware": "AMD Radeon RX 7900 XTX · 24G",
      "framework": "vLLM 0.6.2",
      "decode_tps": 169.3,
      "prefill_tps": 5918.8,
      "ttft_cold_ms": 2421,
      "vram_peak_gb": 6.1,
      "sample_size": 3,
      "evidence": "L2",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "Reddit",
      "source_url": "https://www.reddit.com/r/araoai_demo/comments/demo15_qwen2-5-7b-instruct-gguf/",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen/Qwen2.5-7B-Instruct-GGUF",
      "quant": "Q5_K_M",
      "hardware": "rtx_3060",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 44.6,
      "prefill_tps": 1510.6,
      "ttft_cold_ms": 2822,
      "vram_peak_gb": 7.2,
      "sample_size": 4,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "HuggingFace",
      "source_url": "https://huggingface.co/araoai-demo/qwen2-5-7b-instruct-gguf-bench/discussions/16",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen/Qwen2.5-7B-Instruct-GGUF",
      "quant": "Q4_K_M",
      "hardware": "apple_m2_ultra",
      "framework": "ollama 0.5.1",
      "decode_tps": 109.8,
      "prefill_tps": 3672.7,
      "ttft_cold_ms": 2530,
      "vram_peak_gb": 6.4,
      "sample_size": 5,
      "evidence": "L1",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "V2EX",
      "source_url": "https://www.v2ex.com/t/araoai-demo-17",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "mistralai/Mistral-Nemo-Instruct-2407",
      "quant": "AWQ 4-bit",
      "hardware": "rx_7900_gre",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 56.1,
      "prefill_tps": 1933.7,
      "ttft_cold_ms": 3592,
      "vram_peak_gb": 9.3,
      "sample_size": 6,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "知乎",
      "source_url": "https://zhuanlan.zhihu.com/p/araoai-demo-18",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "microsoft/phi-4-GGUF",
      "quant": "AWQ 4-bit",
      "hardware": "arc_a770",
      "framework": "vLLM 0.6.2",
      "decode_tps": 49.4,
      "prefill_tps": 1726.3,
      "ttft_cold_ms": 4061,
      "vram_peak_gb": 10.6,
      "sample_size": 7,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "B站",
      "source_url": "https://www.bilibili.com/video/araoai-demo-19",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "meta-llama/Llama-3.1-8B-Instruct",
      "quant": "Q5_K_M",
      "hardware": "arc_a770",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 65.7,
      "prefill_tps": 2266,
      "ttft_cold_ms": 3113,
      "vram_peak_gb": 8,
      "sample_size": 8,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "站方自产",
      "source_url": "https://araoai.com/demo/rig-run-20",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen3-14B",
      "quant": "Q4_K_M",
      "hardware": "rtx_4080_super",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 53.2,
      "prefill_tps": 1801.6,
      "ttft_cold_ms": 4280,
      "vram_peak_gb": 11.2,
      "sample_size": 9,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "GitHub",
      "source_url": "https://github.com/araoai-demo/benchmarks/issues/1021",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen/Qwen2.5-7B-Instruct-GGUF",
      "quant": "FP16",
      "hardware": "apple_m2_ultra",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 34.7,
      "prefill_tps": 1174.9,
      "ttft_cold_ms": 6613,
      "vram_peak_gb": 17.6,
      "sample_size": 10,
      "evidence": "L1",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "Reddit",
      "source_url": "https://www.reddit.com/r/araoai_demo/comments/demo22_qwen2-5-7b-instruct-gguf/",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "meta-llama/Llama-3.1-8B-Instruct",
      "quant": "AWQ 4-bit",
      "hardware": "arc_a770",
      "framework": "llama.cpp b5980",
      "decode_tps": 74.1,
      "prefill_tps": 2494.7,
      "ttft_cold_ms": 2655,
      "vram_peak_gb": 6.7,
      "sample_size": 11,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "HuggingFace",
      "source_url": "https://huggingface.co/araoai-demo/llama-3-1-8b-instruct-bench/discussions/23",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "microsoft/phi-4-GGUF",
      "quant": "AWQ 4-bit",
      "hardware": "Apple M3 Max · 128G 统一内存",
      "framework": "llama.cpp b5980",
      "decode_tps": 30.2,
      "prefill_tps": 1018.2,
      "ttft_cold_ms": 4061,
      "vram_peak_gb": 10.6,
      "sample_size": 12,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "V2EX",
      "source_url": "https://www.v2ex.com/t/araoai-demo-24",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen/Qwen2.5-7B-Instruct-GGUF",
      "quant": "AWQ 4-bit",
      "hardware": "rtx_4080_super",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 113.6,
      "prefill_tps": 3843.3,
      "ttft_cold_ms": 2421,
      "vram_peak_gb": 6.1,
      "sample_size": 13,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "知乎",
      "source_url": "https://zhuanlan.zhihu.com/p/araoai-demo-25",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen/Qwen2.5-7B-Instruct-GGUF",
      "quant": "Q6_K",
      "hardware": "apple_m2_ultra",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 91,
      "prefill_tps": 3139.1,
      "ttft_cold_ms": 3186,
      "vram_peak_gb": 8.2,
      "sample_size": 14,
      "evidence": "L2",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "B站",
      "source_url": "https://www.bilibili.com/video/araoai-demo-26",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "meta-llama/Llama-3.1-8B-Instruct",
      "quant": "Q4_K_M",
      "hardware": "rx_7800_xt",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 85.4,
      "prefill_tps": 2945.8,
      "ttft_cold_ms": 2780,
      "vram_peak_gb": 7.1,
      "sample_size": 15,
      "evidence": "L1",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "站方自产",
      "source_url": "https://araoai.com/demo/rig-run-27",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "deepseek-ai/DeepSeek-R1-Distill-Qwen-14B",
      "quant": "Q4_K_M",
      "hardware": "arc_a770",
      "framework": "llama.cpp b5980",
      "decode_tps": 39.7,
      "prefill_tps": 1336.4,
      "ttft_cold_ms": 4280,
      "vram_peak_gb": 11.2,
      "sample_size": 16,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "GitHub",
      "source_url": "https://github.com/araoai-demo/benchmarks/issues/1028",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "meta-llama/Llama-3.1-8B-Instruct",
      "quant": "Q6_K",
      "hardware": "rx_7900_gre",
      "framework": "vLLM 0.6.2",
      "decode_tps": 60.6,
      "prefill_tps": 2118.7,
      "ttft_cold_ms": 3530,
      "vram_peak_gb": 9.1,
      "sample_size": 17,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "Reddit",
      "source_url": "https://www.reddit.com/r/araoai_demo/comments/demo29_llama-3-1-8b-instruct/",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "mistralai/Mistral-Nemo-Instruct-2407",
      "quant": "Q4_K_M",
      "hardware": "NVIDIA RTX 4090 · 24G",
      "framework": "ollama 0.5.1",
      "decode_tps": 80.7,
      "prefill_tps": 2699.4,
      "ttft_cold_ms": 3780,
      "vram_peak_gb": 9.8,
      "sample_size": 3,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "HuggingFace",
      "source_url": "https://huggingface.co/araoai-demo/mistral-nemo-instruct-2407-bench/discussions/30",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen3-14B",
      "quant": "FP16",
      "hardware": "apple_m2_ultra",
      "framework": "llama.cpp b6120",
      "decode_tps": 17.7,
      "prefill_tps": 602.3,
      "ttft_cold_ms": 12447,
      "vram_peak_gb": 33.5,
      "sample_size": 4,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "V2EX",
      "source_url": "https://www.v2ex.com/t/araoai-demo-31",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "deepseek-ai/DeepSeek-R1-Distill-Qwen-14B",
      "quant": "Q4_K_M",
      "hardware": "NVIDIA RTX 3090 · 24G",
      "framework": "llama.cpp b5980",
      "decode_tps": 66.3,
      "prefill_tps": 2233.7,
      "ttft_cold_ms": 4280,
      "vram_peak_gb": 11.2,
      "sample_size": 5,
      "evidence": "L1",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "知乎",
      "source_url": "https://zhuanlan.zhihu.com/p/araoai-demo-32",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen/Qwen2.5-7B-Instruct-GGUF",
      "quant": "Q5_K_M",
      "hardware": "Apple M3 Max · 128G 统一内存",
      "framework": "ollama 0.5.1",
      "decode_tps": 47.1,
      "prefill_tps": 1574,
      "ttft_cold_ms": 2822,
      "vram_peak_gb": 7.2,
      "sample_size": 6,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "B站",
      "source_url": "https://www.bilibili.com/video/araoai-demo-33",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen3-14B",
      "quant": "Q6_K",
      "hardware": "rtx_4080_super",
      "framework": "llama.cpp b6120",
      "decode_tps": 39.5,
      "prefill_tps": 1343.3,
      "ttft_cold_ms": 5592,
      "vram_peak_gb": 14.8,
      "sample_size": 7,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "站方自产",
      "source_url": "https://araoai.com/demo/rig-run-34",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "deepseek-ai/DeepSeek-R1-Distill-Qwen-14B",
      "quant": "AWQ 4-bit",
      "hardware": "rx_7800_xt",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 48.1,
      "prefill_tps": 1629.2,
      "ttft_cold_ms": 4061,
      "vram_peak_gb": 10.6,
      "sample_size": 8,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "GitHub",
      "source_url": "https://github.com/araoai-demo/benchmarks/issues/1035",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen3-14B",
      "quant": "Q4_K_M",
      "hardware": "NVIDIA RTX 3090 · 24G",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 67.7,
      "prefill_tps": 2291.1,
      "ttft_cold_ms": 4280,
      "vram_peak_gb": 11.2,
      "sample_size": 9,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "Reddit",
      "source_url": "https://www.reddit.com/r/araoai_demo/comments/demo36_qwen3-14b-gguf/",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "deepseek-ai/DeepSeek-R1-Distill-Qwen-14B",
      "quant": "FP16",
      "hardware": "Apple M3 Max · 128G 统一内存",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 8.7,
      "prefill_tps": 293.7,
      "ttft_cold_ms": 12447,
      "vram_peak_gb": 33.5,
      "sample_size": 10,
      "evidence": "L1",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "HuggingFace",
      "source_url": "https://huggingface.co/araoai-demo/deepseek-r1-distill-qwen-14b-bench/discussions/37",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "meta-llama/Llama-3.1-8B-Instruct",
      "quant": "Q5_K_M",
      "hardware": "NVIDIA RTX 3090 · 24G",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 109.8,
      "prefill_tps": 3787.5,
      "ttft_cold_ms": 3113,
      "vram_peak_gb": 8,
      "sample_size": 11,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "V2EX",
      "source_url": "https://www.v2ex.com/t/araoai-demo-38",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen/Qwen2.5-7B-Instruct-GGUF",
      "quant": "Q8_0",
      "hardware": "NVIDIA RTX 3090 · 24G",
      "framework": "llama.cpp b6120",
      "decode_tps": 78,
      "prefill_tps": 2652.9,
      "ttft_cold_ms": 3879,
      "vram_peak_gb": 10.1,
      "sample_size": 12,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "知乎",
      "source_url": "https://zhuanlan.zhihu.com/p/araoai-demo-39",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen/Qwen2.5-7B-Instruct-GGUF",
      "quant": "Q4_K_M",
      "hardware": "arc_a770",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 87.6,
      "prefill_tps": 3021.4,
      "ttft_cold_ms": 2530,
      "vram_peak_gb": 6.4,
      "sample_size": 13,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "B站",
      "source_url": "https://www.bilibili.com/video/araoai-demo-40",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "microsoft/phi-4-GGUF",
      "quant": "Q4_K_M",
      "hardware": "Apple M3 Max · 128G 统一内存",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 28.9,
      "prefill_tps": 979.1,
      "ttft_cold_ms": 4280,
      "vram_peak_gb": 11.2,
      "sample_size": 14,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "站方自产",
      "source_url": "https://araoai.com/demo/rig-run-41",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen3-14B",
      "quant": "Q6_K",
      "hardware": "rtx_4080_super",
      "framework": "ollama 0.5.1",
      "decode_tps": 36.7,
      "prefill_tps": 1228.7,
      "ttft_cold_ms": 5592,
      "vram_peak_gb": 14.8,
      "sample_size": 15,
      "evidence": "L1",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "GitHub",
      "source_url": "https://github.com/araoai-demo/benchmarks/issues/1042",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen/Qwen2.5-7B-Instruct-GGUF",
      "quant": "AWQ 4-bit",
      "hardware": "NVIDIA RTX 4090 · 24G",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 155.5,
      "prefill_tps": 5263.7,
      "ttft_cold_ms": 2421,
      "vram_peak_gb": 6.1,
      "sample_size": 16,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "Reddit",
      "source_url": "https://www.reddit.com/r/araoai_demo/comments/demo43_qwen2-5-7b-instruct-gguf/",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen/Qwen2.5-7B-Instruct-GGUF",
      "quant": "Q6_K",
      "hardware": "NVIDIA RTX 3090 · 24G",
      "framework": "ollama 0.5.1",
      "decode_tps": 93.5,
      "prefill_tps": 3125.1,
      "ttft_cold_ms": 3186,
      "vram_peak_gb": 8.2,
      "sample_size": 17,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "HuggingFace",
      "source_url": "https://huggingface.co/araoai-demo/qwen2-5-7b-instruct-gguf-bench/discussions/44",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "mistralai/Mistral-Nemo-Instruct-2407",
      "quant": "Q4_K_M",
      "hardware": "rtx_3060",
      "framework": "llama.cpp b6120",
      "decode_tps": 31,
      "prefill_tps": 1054,
      "ttft_cold_ms": 3780,
      "vram_peak_gb": 9.8,
      "sample_size": 3,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "V2EX",
      "source_url": "https://www.v2ex.com/t/araoai-demo-45",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen3-14B",
      "quant": "Q4_K_M",
      "hardware": "rtx_4080_super",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 57.6,
      "prefill_tps": 1985.5,
      "ttft_cold_ms": 4280,
      "vram_peak_gb": 11.2,
      "sample_size": 4,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "知乎",
      "source_url": "https://zhuanlan.zhihu.com/p/araoai-demo-46",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "deepseek-ai/DeepSeek-R1-Distill-Qwen-14B",
      "quant": "AWQ 4-bit",
      "hardware": "rx_7900_gre",
      "framework": "ollama 0.5.1",
      "decode_tps": 42.2,
      "prefill_tps": 1410.3,
      "ttft_cold_ms": 4061,
      "vram_peak_gb": 10.6,
      "sample_size": 5,
      "evidence": "L1",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "B站",
      "source_url": "https://www.bilibili.com/video/araoai-demo-47",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen3-14B",
      "quant": "Q4_K_M",
      "hardware": "rtx_3060",
      "framework": "vLLM 0.6.2",
      "decode_tps": 29.8,
      "prefill_tps": 1040.4,
      "ttft_cold_ms": 4280,
      "vram_peak_gb": 11.2,
      "sample_size": 6,
      "evidence": "L2",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "站方自产",
      "source_url": "https://araoai.com/demo/rig-run-48",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen3-14B",
      "quant": "Q4_K_M",
      "hardware": "rx_7900_gre",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 41.7,
      "prefill_tps": 1409.9,
      "ttft_cold_ms": 4280,
      "vram_peak_gb": 11.2,
      "sample_size": 7,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "GitHub",
      "source_url": "https://github.com/araoai-demo/benchmarks/issues/1049",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen3-14B",
      "quant": "AWQ 4-bit",
      "hardware": "rx_7900_gre",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 44.4,
      "prefill_tps": 1503.9,
      "ttft_cold_ms": 4061,
      "vram_peak_gb": 10.6,
      "sample_size": 8,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "Reddit",
      "source_url": "https://www.reddit.com/r/araoai_demo/comments/demo50_qwen3-14b-gguf/",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "microsoft/phi-4-GGUF",
      "quant": "AWQ 4-bit",
      "hardware": "Apple M3 Max · 128G 统一内存",
      "framework": "ollama 0.5.1",
      "decode_tps": 29.3,
      "prefill_tps": 979.4,
      "ttft_cold_ms": 4061,
      "vram_peak_gb": 10.6,
      "sample_size": 9,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "HuggingFace",
      "source_url": "https://huggingface.co/araoai-demo/phi-4-gguf-bench/discussions/51",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "mistralai/Mistral-Nemo-Instruct-2407",
      "quant": "Q5_K_M",
      "hardware": "rtx_3060",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 28.2,
      "prefill_tps": 971.2,
      "ttft_cold_ms": 4280,
      "vram_peak_gb": 11.2,
      "sample_size": 10,
      "evidence": "L1",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "V2EX",
      "source_url": "https://www.v2ex.com/t/araoai-demo-52",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen/Qwen2.5-7B-Instruct-GGUF",
      "quant": "AWQ 4-bit",
      "hardware": "rtx_3060",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 55.6,
      "prefill_tps": 1879.9,
      "ttft_cold_ms": 2421,
      "vram_peak_gb": 6.1,
      "sample_size": 11,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "知乎",
      "source_url": "https://zhuanlan.zhihu.com/p/araoai-demo-53",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "microsoft/phi-4-GGUF",
      "quant": "Q4_K_M",
      "hardware": "AMD Radeon RX 7900 XTX · 24G",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 75.1,
      "prefill_tps": 2589.7,
      "ttft_cold_ms": 4280,
      "vram_peak_gb": 11.2,
      "sample_size": 12,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "B站",
      "source_url": "https://www.bilibili.com/video/araoai-demo-54",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "deepseek-ai/DeepSeek-R1-Distill-Qwen-14B",
      "quant": "Q6_K",
      "hardware": "arc_a770",
      "framework": "vLLM 0.6.2",
      "decode_tps": 33.7,
      "prefill_tps": 1177,
      "ttft_cold_ms": 5592,
      "vram_peak_gb": 14.8,
      "sample_size": 13,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "站方自产",
      "source_url": "https://araoai.com/demo/rig-run-55",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "microsoft/phi-4-GGUF",
      "quant": "AWQ 4-bit",
      "hardware": "Apple M3 Max · 128G 统一内存",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 33.4,
      "prefill_tps": 1151,
      "ttft_cold_ms": 4061,
      "vram_peak_gb": 10.6,
      "sample_size": 14,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "GitHub",
      "source_url": "https://github.com/araoai-demo/benchmarks/issues/1056",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "deepseek-ai/DeepSeek-R1-Distill-Qwen-14B",
      "quant": "AWQ 4-bit",
      "hardware": "rx_7800_xt",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 52.1,
      "prefill_tps": 1795.6,
      "ttft_cold_ms": 4061,
      "vram_peak_gb": 10.6,
      "sample_size": 15,
      "evidence": "L1",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "Reddit",
      "source_url": "https://www.reddit.com/r/araoai_demo/comments/demo57_deepseek-r1-distill-qwen-14b/",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "microsoft/phi-4-GGUF",
      "quant": "AWQ 4-bit",
      "hardware": "rtx_4080_super",
      "framework": "llama.cpp b6120",
      "decode_tps": 57.9,
      "prefill_tps": 1970.1,
      "ttft_cold_ms": 4061,
      "vram_peak_gb": 10.6,
      "sample_size": 16,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "HuggingFace",
      "source_url": "https://huggingface.co/araoai-demo/phi-4-gguf-bench/discussions/58",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "meta-llama/Llama-3.1-8B-Instruct",
      "quant": "Q6_K",
      "hardware": "rx_7900_gre",
      "framework": "llama.cpp b5980",
      "decode_tps": 51.9,
      "prefill_tps": 1749.5,
      "ttft_cold_ms": 3530,
      "vram_peak_gb": 9.1,
      "sample_size": 17,
      "evidence": "L2",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "V2EX",
      "source_url": "https://www.v2ex.com/t/araoai-demo-59",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "deepseek-ai/DeepSeek-R1-Distill-Qwen-14B",
      "quant": "Q8_0",
      "hardware": "Apple M3 Max · 128G 统一内存",
      "framework": "llama.cpp b5980",
      "decode_tps": 16,
      "prefill_tps": 539.1,
      "ttft_cold_ms": 6978,
      "vram_peak_gb": 18.6,
      "sample_size": 3,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "知乎",
      "source_url": "https://zhuanlan.zhihu.com/p/araoai-demo-60",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "mistralai/Mistral-Nemo-Instruct-2407",
      "quant": "Q5_K_M",
      "hardware": "rx_7900_gre",
      "framework": "vLLM 0.6.2",
      "decode_tps": 47.6,
      "prefill_tps": 1664.7,
      "ttft_cold_ms": 4280,
      "vram_peak_gb": 11.2,
      "sample_size": 4,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "B站",
      "source_url": "https://www.bilibili.com/video/araoai-demo-61",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "meta-llama/Llama-3.1-8B-Instruct",
      "quant": "AWQ 4-bit",
      "hardware": "rx_7900_gre",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 77.8,
      "prefill_tps": 2631.8,
      "ttft_cold_ms": 2655,
      "vram_peak_gb": 6.7,
      "sample_size": 5,
      "evidence": "L1",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "站方自产",
      "source_url": "https://araoai.com/demo/rig-run-62",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "deepseek-ai/DeepSeek-R1-Distill-Qwen-14B",
      "quant": "Q4_K_M",
      "hardware": "rtx_4080_super",
      "framework": "vLLM 0.6.2",
      "decode_tps": 60.8,
      "prefill_tps": 2127.1,
      "ttft_cold_ms": 4280,
      "vram_peak_gb": 11.2,
      "sample_size": 6,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "GitHub",
      "source_url": "https://github.com/araoai-demo/benchmarks/issues/1063",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "meta-llama/Llama-3.1-8B-Instruct",
      "quant": "AWQ 4-bit",
      "hardware": "rtx_4080_super",
      "framework": "vLLM 0.6.2",
      "decode_tps": 113.6,
      "prefill_tps": 3970.5,
      "ttft_cold_ms": 2655,
      "vram_peak_gb": 6.7,
      "sample_size": 7,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "待复现",
      "source_platform": "Reddit",
      "source_url": "https://www.reddit.com/r/araoai_demo/comments/demo64_llama-3-1-8b-instruct/",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen/Qwen2.5-7B-Instruct-GGUF",
      "quant": "AWQ 4-bit",
      "hardware": "rx_7900_gre",
      "framework": "ExLlamaV2 0.2.5",
      "decode_tps": 96.1,
      "prefill_tps": 3314.9,
      "ttft_cold_ms": 2421,
      "vram_peak_gb": 6.1,
      "sample_size": 8,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "HuggingFace",
      "source_url": "https://huggingface.co/araoai-demo/qwen2-5-7b-instruct-gguf-bench/discussions/65",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "meta-llama/Llama-3.1-8B-Instruct",
      "quant": "Q5_K_M",
      "hardware": "Apple M3 Max · 128G 统一内存",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 43.4,
      "prefill_tps": 1468.7,
      "ttft_cold_ms": 3113,
      "vram_peak_gb": 8,
      "sample_size": 9,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "V2EX",
      "source_url": "https://www.v2ex.com/t/araoai-demo-66",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen3-14B",
      "quant": "Q5_K_M",
      "hardware": "apple_m2_ultra",
      "framework": "ollama 0.5.1",
      "decode_tps": 47.1,
      "prefill_tps": 1574,
      "ttft_cold_ms": 4863,
      "vram_peak_gb": 12.8,
      "sample_size": 10,
      "evidence": "L1",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "知乎",
      "source_url": "https://zhuanlan.zhihu.com/p/araoai-demo-67",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "microsoft/phi-4-GGUF",
      "quant": "AWQ 4-bit",
      "hardware": "rx_7800_xt",
      "framework": "llama.cpp b5980",
      "decode_tps": 47.2,
      "prefill_tps": 1588.4,
      "ttft_cold_ms": 4061,
      "vram_peak_gb": 10.6,
      "sample_size": 11,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "B站",
      "source_url": "https://www.bilibili.com/video/araoai-demo-68",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "meta-llama/Llama-3.1-8B-Instruct",
      "quant": "Q8_0",
      "hardware": "rx_7900_gre",
      "framework": "mlx-lm 0.18.2",
      "decode_tps": 41.2,
      "prefill_tps": 1393.3,
      "ttft_cold_ms": 4322,
      "vram_peak_gb": 11.3,
      "sample_size": 12,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "站方自产",
      "source_url": "https://araoai.com/demo/rig-run-69",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen3-32B",
      "quant": "Q4_K_M",
      "hardware": "apple_m2_ultra",
      "framework": "llama.cpp b6120",
      "decode_tps": 25.8,
      "prefill_tps": 878.3,
      "ttft_cold_ms": 8780,
      "vram_peak_gb": 23.5,
      "sample_size": 13,
      "evidence": "L2",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "GitHub",
      "source_url": "https://github.com/araoai-demo/benchmarks/issues/1070",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen3-14B",
      "quant": "Q8_0",
      "hardware": "apple_m2_ultra",
      "framework": "llama.cpp b6120",
      "decode_tps": 33.3,
      "prefill_tps": 1133.7,
      "ttft_cold_ms": 6978,
      "vram_peak_gb": 18.6,
      "sample_size": 14,
      "evidence": "L0",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "Reddit",
      "source_url": "https://www.reddit.com/r/araoai_demo/comments/demo71_qwen3-14b-gguf/",
      "fetched_at": "2026-09-14T00:00:00Z"
    },
    {
      "model": "Qwen3-32B",
      "quant": "Q5_K_M",
      "hardware": "Apple M3 Max · 128G 统一内存",
      "framework": "llama.cpp b5980",
      "decode_tps": 10.6,
      "prefill_tps": 358,
      "ttft_cold_ms": 10113,
      "vram_peak_gb": 27.1,
      "sample_size": 15,
      "evidence": "L1",
      "repro_count": 0,
      "repro_chain": "待复现",
      "status": "稳定",
      "source_platform": "HuggingFace",
      "source_url": "https://huggingface.co/araoai-demo/qwen3-32b-gguf-bench/discussions/72",
      "fetched_at": "2026-09-14T00:00:00Z"
    }
  ]
}
