{
  "title": "I Tested 14 Local AI Setups on a Real App — Signal Stood Out",
  "slug": "14-local-ai-setups-signal",
  "videoId": "UgQ2K_HLzwY",
  "publishedAt": "2026-09-29",
  "duration": "5:35",
  "asOf": "2026-09-24",
  "runs": [
    {
      "id": "qwen38-q6-medium",
      "name": "Qwen3.8 27B",
      "date": "2026-08-24",
      "quant": "Q6_K GGUF",
      "backend": "LM Studio / llama.cpp",
      "reasoning": "medium",
      "status": "Pass",
      "tests": 92,
      "failures": 0,
      "acceptance_failures": 0,
      "build": "Pass",
      "wall_seconds": 3160.240046,
      "elapsed_seconds": null,
      "span_seconds": null,
      "tool_calls": 30,
      "output_tokens": 32109,
      "config": "118,016 effective context · one slot · native MTP, max 3 draft tokens",
      "detail": "Correct two-file patch. Repeated planning and test launches delayed completion; first write took nearly 35 minutes."
    },
    {
      "id": "grug-v1.1-q6",
      "name": "Grug v1.1 27B",
      "date": "2026-08-24",
      "quant": "Q6_K GGUF",
      "backend": "LM Studio / llama.cpp",
      "reasoning": "medium",
      "status": "Fail",
      "tests": 90,
      "failures": 12,
      "acceptance_failures": 8,
      "build": "Pass",
      "wall_seconds": 321.689766,
      "elapsed_seconds": null,
      "span_seconds": null,
      "tool_calls": 9,
      "output_tokens": 2345,
      "config": "118,016 effective context · one slot · native MTP, max 3 draft tokens",
      "detail": "Fast but missed U+2011; inserted U+200B removal and repeated ASCII hyphen. Wrote tests but did not run them."
    },
    {
      "id": "kiwen1.1-q6",
      "name": "Kiwen1.1 27B",
      "date": "2026-08-25",
      "quant": "Q6_K GGUF",
      "backend": "LM Studio / llama.cpp",
      "reasoning": "medium",
      "status": "Pass",
      "tests": 91,
      "failures": 0,
      "acceptance_failures": 0,
      "build": "Pass",
      "wall_seconds": 5323.239488,
      "elapsed_seconds": null,
      "span_seconds": null,
      "tool_calls": 68,
      "output_tokens": 48110,
      "config": "118,016 effective context · one slot · native MTP, max 3 draft tokens",
      "detail": "Eventually correct after a long regex repair loop. 68 tool calls and 88.7 minutes; also expanded support to U+2010 and wrote scratch files outside its worktree."
    },
    {
      "id": "grug-v1.1-q6-xhigh",
      "name": "Grug v1.1 27B",
      "date": "2026-08-25",
      "quant": "Q6_K GGUF",
      "backend": "LM Studio / llama.cpp",
      "reasoning": "xhigh",
      "status": "Fail",
      "tests": 88,
      "failures": 19,
      "acceptance_failures": 12,
      "build": "Pass",
      "wall_seconds": 262.927168,
      "elapsed_seconds": null,
      "span_seconds": null,
      "tool_calls": 8,
      "output_tokens": 1600,
      "config": "118,016 effective context · one slot · native MTP, max 3 draft tokens",
      "detail": "Verified native xhigh did not repair correctness. Recognized U+2011 but failed normalization; did not run its useful regression test."
    },
    {
      "id": "qwen-q4",
      "name": "Qwen3.8 27B",
      "date": "2026-08-26",
      "quant": "4-bit MLX",
      "backend": "mlx_vlm.server",
      "reasoning": "medium",
      "status": "No patch",
      "tests": 87,
      "failures": 16,
      "acceptance_failures": 16,
      "build": "Pass",
      "wall_seconds": 4154.746635,
      "elapsed_seconds": null,
      "span_seconds": null,
      "tool_calls": 18,
      "output_tokens": 23310,
      "config": "262,144 loaded context · MTP max 3 · KV 8-bit / group 64, starts at token 5,000",
      "detail": "Diagnosed the intended fix but never applied it. Two Metal out-of-memory retries, unsupported structured requests, and long non-tool responses. Only wrote a scratch harness."
    },
    {
      "id": "qwopus",
      "name": "Qwopus3.6 35B A3B",
      "date": "2026-08-26",
      "quant": "oQ4 MLX",
      "backend": "mlx_vlm.server",
      "reasoning": "medium",
      "status": "Fail",
      "tests": 94,
      "failures": 8,
      "acceptance_failures": 8,
      "build": "Pass",
      "wall_seconds": 2022.19,
      "elapsed_seconds": null,
      "span_seconds": null,
      "tool_calls": 29,
      "output_tokens": 9304,
      "config": "118,016 Hermes context / 262,144 model context · extracted native MTP max 3 · KV 8-bit / group 64",
      "detail": "Its own 92 tests passed, but production code and tests shared the same wrong character: U+200C instead of U+2011. Independent checks caught it. Two Xcode diagnostic timeouts also increased wall time."
    },
    {
      "id": "muse",
      "name": "Muse Glimmer 30B",
      "date": "2026-08-28",
      "quant": "6-bit MLX",
      "backend": "MLX / Hermes",
      "reasoning": "Not verified",
      "status": "No patch",
      "tests": null,
      "failures": null,
      "acceptance_failures": null,
      "build": "Not recorded",
      "wall_seconds": null,
      "elapsed_seconds": null,
      "span_seconds": 2846.8845932483673,
      "tool_calls": 10,
      "output_tokens": 12501,
      "config": "Base MLX path; no DFlash drafter. Exact effective context and rendered reasoning tier not verified in retained summary.",
      "detail": "Retained usage and session evidence show 16 model calls without a usable patch or final response. The later status audit confirmed a clean benchmark worktree. DFlash was not exercised. Full independent Xcode results were not recovered."
    },
    {
      "id": "dflash-118K",
      "name": "Qwen3.8 + DFlash2 · 118K",
      "date": "2026-08-30",
      "quant": "6-bit MLX",
      "backend": "mlx-dspark / DFlash2",
      "reasoning": "medium",
      "status": "Pass",
      "tests": 91,
      "failures": 0,
      "acceptance_failures": 0,
      "build": "Pass",
      "wall_seconds": 1634.531899,
      "elapsed_seconds": null,
      "span_seconds": null,
      "tool_calls": 24,
      "output_tokens": 30025,
      "config": "118,016 context · DreamFoundries/Qwen3.8-27B-6bit + incoai/Qwen3.8-27B-DFlash2 · medium",
      "detail": "Passed with 24 tool calls, no exact repeated calls, and no recorded context compaction. 48.3% less wall time than the historical Q6 control. About 12.88 GiB observed swap and seven memory-guard cache sheds remained. DFlash draft acceptance: 56.50%. This end-to-end speed difference is not an isolated on/off measurement of speculation."
    },
    {
      "id": "gsq",
      "name": "Qwen3.8 GSQ-RCO 27B",
      "date": "2026-09-13",
      "quant": "IQ3_S GGUF",
      "backend": "LM Studio / llama.cpp 2.37.0",
      "reasoning": "medium",
      "status": "Pass",
      "tests": 91,
      "failures": 0,
      "acceptance_failures": 0,
      "build": "Pass",
      "wall_seconds": 1184.22517,
      "elapsed_seconds": null,
      "span_seconds": null,
      "tool_calls": 22,
      "output_tokens": 16402,
      "config": "118,016 effective context · one slot · native MTP, max 3 draft tokens",
      "detail": "Correct two-line parser change and four authored tests. Recovered from an obsolete Xcode path and a faulty URL assertion; final independent suite passed. Test coverage has minor gaps described in the run report. Previously fastest passing local configuration; Signal later completed this task in 15m56s."
    },
    {
      "id": "signal-iq3s-mtp",
      "name": "Signal 3.8 27B",
      "date": "2026-09-16",
      "quant": "AP-IQ3_S GGUF",
      "backend": "Standalone llama.cpp 2.37.0 / Hermes",
      "reasoning": "medium",
      "status": "Pass",
      "tests": 91,
      "failures": 0,
      "acceptance_failures": 0,
      "build": "Pass",
      "wall_seconds": 956.0120775410323,
      "elapsed_seconds": null,
      "span_seconds": null,
      "tool_calls": 36,
      "output_tokens": 8582,
      "config": "118,016 context · q8_0 K/V · native MTP max 4 · backend draft sampling off · vision projector loaded · temperature .6 / top-p .95 / top-k 20 / min-p .05 / presence penalty 1.0 · Xcode 27 beta 6",
      "detail": "Earlier passing local observation: 15m56s, 19.3% less time and 47.7% fewer output tokens than GSQ-RCO. More tool calls (36 vs 22), so speed did not mean fewer actions. Four authored tests; recovered from an invalid Xcode selector and two URL-encoding assertion failures. Explicit Unicode inspection and exact Markdown links were strengths. Used file search instead of graph; wrote a /tmp scratch file despite the scope constraint; edited production before tests. Independent 91/91 and build passed without repair. Different endpoint wrapper, q8 KV and MTP max 4 mean this is a complete-configuration comparison, not an isolated fine-tune effect."
    },
    {
      "id": "signal-q6-mtp",
      "name": "Signal 3.8 27B · Q6",
      "date": "2026-09-17",
      "quant": "AP-Q6_K GGUF",
      "backend": "Standalone llama.cpp 2.37.0 / Hermes",
      "reasoning": "medium",
      "status": "Pass",
      "tests": 91,
      "failures": 0,
      "acceptance_failures": 0,
      "build": "Pass",
      "wall_seconds": 681.1753717500251,
      "elapsed_seconds": null,
      "span_seconds": null,
      "tool_calls": 24,
      "output_tokens": 7798,
      "config": "118,016 context · q8_0 K/V · native MTP max 4 · backend draft sampling off · BF16 vision projector loaded · temperature .6 / top-p .95 / top-k 20 / min-p .05 / presence penalty 1.0 · same harness/runtime settings as Signal IQ3_S",
      "detail": "Fastest passing local observation: 11m21s, 28.7% less time than Signal IQ3_S and 42.5% less than GSQ-RCO. 24 tool calls versus 36 and 22 respectively. Simpler two-line production patch; four authored tests with all-dash full Markdown links and direct raw-variant URL round-trips. Recovered from four URL-encoding assertion failures; corrected own 89-test suite passed, then independent 91/91 and build passed without repair. File search instead of graph, production-before-test ordering, /tmp diagnostics and piped exit-code handling remain workflow limitations. Unrelated TypeScript check lacked tsc and was disclosed. One run per configuration; cache, machine state and work trajectory can affect timing."
    },
    {
      "id": "qwopus38-v2-q6-mlx",
      "name": "Qwopus3.8 27B Flash V2 · Q6",
      "date": "2026-09-22",
      "quant": "MLX affine 6bit · group 64",
      "backend": "mlx-vlm 0.6.15 / Hermes",
      "reasoning": "medium",
      "status": "Pass",
      "tests": 91,
      "failures": 0,
      "acceptance_failures": 0,
      "build": "Pass",
      "wall_seconds": 1023.2939069999848,
      "elapsed_seconds": null,
      "span_seconds": null,
      "tool_calls": 16,
      "output_tokens": 5619,
      "config": "118,016 context · MLX 8-bit KV group 64 from token 0 · prompt cache on and observed · no MTP drafter · vision included and smoke-tested · temperature .6 / top-p .95 / top-k 20 / min-p .05 / presence penalty 1.0 (64-token window) · verified sampler correction · Xcode 27.0 released",
      "detail": "Passed in 17m03s: 91 independent tests and Debug build, without external patch repair. Two production lines and four authored tests. 16 tools and 5,619 output tokens, fewer than Signal Q6, but 50.2% more elapsed time. Used file search instead of graph and production-before-test ordering; recovered unaided from stale Xcode path and test-without-building before a build. Authored Markdown and URL checks are narrower than Signal Q6; fixed independent acceptance tests passed. Runtime, KV representation, no MTP and released Xcode differ from Signal. Single observation, not an isolated model-quality comparison. Vision passed a basic image smoke test only."
    },
    {
      "id": "thinkingcap-q6",
      "name": "ThinkingCap Qwen3.8 27B",
      "date": "2026-09-24",
      "quant": "Q6_K GGUF",
      "backend": "LM Studio bundled llama.cpp 2.37.0, commit 8172e65; standalone server",
      "reasoning": "medium",
      "status": "Agent failure",
      "tests": 91,
      "failures": 1,
      "acceptance_failures": 0,
      "build": "Pass",
      "wall_seconds": null,
      "elapsed_seconds": 3614.963595832931,
      "span_seconds": null,
      "tool_calls": 52,
      "output_tokens": 38660,
      "config": "Hermes 3aee290899e478c5fdfb6a241ef62758a49829b3; 118,016 context; q8_0 K/V; native MTP max 4; backend draft sampling disabled; temperature 1, top-p 0.95, top-k 20, min-p 0, presence penalty 0. Released Xcode 27; iPhone 17 / iOS 26.5. Same historical baseline, prompt and fixed evaluator. Vision projector loaded and image smoke test passed; coding benchmark was text-only",
      "detail": "Incomplete: stopped by the controller after 60m 15s with no final agent response. Independent evaluation of the unmodified patch passed both fixed acceptance methods (eight dash/spacing cases) and the Debug build. The combined suite passed 90 of 91 tests; the sole failure was a model-authored Markdown test incorrectly expecting colon percent-encoding (%3A). The one-hour cutoff was chosen around minute 48, not preregistered or uniformly applied to earlier runs. Elapsed time at termination is not a successful completion time."
    },
    {
      "id": "ornith-reported",
      "name": "Ornith 1.5 35B-A3B",
      "date": "Date not recovered",
      "quant": "Not recovered",
      "backend": "Not recovered",
      "reasoning": "Not recovered",
      "status": "Reported fail",
      "tests": null,
      "failures": null,
      "acceptance_failures": null,
      "build": "Not recorded",
      "wall_seconds": null,
      "elapsed_seconds": null,
      "span_seconds": null,
      "tool_calls": null,
      "output_tokens": null,
      "config": "Configuration named in the supplied screenshot; detailed runtime settings unavailable",
      "detail": "The user-provided historical comparison lists Ornith 1.5 35B-A3B as failed. Detailed benchmark logs, timing, quantization, context, and test counts were not recovered. This is a reported result, not an independently revalidated measurement. It is not conflated with a separately found Ornith 9B download."
    }
  ],
  "ledgerSha256": "b1fc4463de731a470b42f189f9b51863c036ac38a34f43c57d954e920c2dfded",
  "promptSha256": "a9bd80dad2cf9792ebaacdde38e71a2a0b4c85113f578954c5ced789a9710a9d",
  "evaluatorSha256": "f8c54b591bed62ee5fd7eec5e81e7c9e88b6481511b726369f9f31e768fc7f64",
  "receiptIdentities": [
    {
      "file": "33-signal-benchmark-summary.json",
      "sha256": "94ddeb5fa2d207ddcfc016379604acdeeb01629ef14a63160279ee4146afee33",
      "matches_ledger": true
    },
    {
      "file": "38-signal-q6-benchmark-summary.json",
      "sha256": "9e8cc5cdd0272ff2e0e92c436f52b21d1a9cb39d5ed681fdfde61d7d9e7d3616",
      "matches_ledger": true
    },
    {
      "file": "52-qwopus38-v2-q6-benchmark-summary.json",
      "sha256": "1b59f89b3cc0a86f978e159bf30ce98deae9d97da5cff0742a0a47d7e04df392",
      "matches_ledger": true
    },
    {
      "file": "59-thinkingcap-benchmark-summary.json",
      "sha256": "b28e412cb5533c8ebdf7bc62ca9917c744830969569e38d0dfebc8a09fd41afb",
      "matches_ledger": true
    }
  ]
}
