{ "schemaVersion": 1, "backend": "metal", "sourceOfTruth": "Runtime-ready means a built llama.cpp fork graph route selects the shipped Metal kernel and a numeric smoke test passes. Symbol presence in default.metallib is not enough.", "latestReport": "reports/porting/2026-05-11/remaining-work-ledger.md", "kernels": { "turbo3": { "runtimeCapabilityKey": "turbo3", "status": "runtime-ready", "runtimeReady": true, "shippedSymbols": [ "kernel_turbo3_dot", "kernel_turbo3_dot_multi" ], "graphOp": "GGML_OP_ATTN_SCORE_TBQ", "smokeTarget": "dispatch-smoke", "smokeCommand": "make -C packages/inference/verify dispatch-smoke", "smokeScores": 32, "maxDiff": 4.768e-7, "evidenceDate": "2026-05-11", "notes": "dispatch_smoke links against the built fork's libggml-metal.dylib and drives GGML_OP_ATTN_SCORE_TBQ through kernel_turbo3_dot_multi." }, "turbo4": { "runtimeCapabilityKey": "turbo4", "status": "runtime-ready", "runtimeReady": true, "shippedSymbols": [ "kernel_turbo4_dot", "kernel_turbo4_dot_multi" ], "graphOp": "GGML_OP_ATTN_SCORE_TBQ", "smokeTarget": "dispatch-smoke", "smokeCommand": "make -C packages/inference/verify dispatch-smoke", "smokeScores": 32, "maxDiff": 4.768e-7, "evidenceDate": "2026-05-11", "notes": "kernel_turbo4_dot(_multi) consumes the fork GGML_TYPE_TBQ4_0 layout: four 32-wide block_tbq4_0 records per 128-row (ggml_row_size=72). dispatch_smoke drives GGML_OP_ATTN_SCORE_TBQ through kernel_turbo4_dot_multi and numerically checks the reference dot." }, "turbo3_tcq": { "runtimeCapabilityKey": "turbo3_tcq", "status": "runtime-ready", "runtimeReady": true, "shippedSymbols": [ "kernel_turbo3_tcq_dot", "kernel_turbo3_tcq_dot_multi" ], "graphOp": "GGML_OP_ATTN_SCORE_TBQ", "smokeTarget": "dispatch-smoke", "smokeCommand": "make -C packages/inference/verify dispatch-smoke", "smokeScores": 32, "maxDiff": 4.768e-7, "evidenceDate": "2026-05-11", "notes": "The patch adds GGML_TYPE_TBQ3_TCQ row-size traits and drives GGML_OP_ATTN_SCORE_TBQ through kernel_turbo3_tcq_dot_multi with the TCQ codebook bound." }, "qjl": { "runtimeCapabilityKey": "qjl_full", "status": "runtime-ready", "runtimeReady": true, "shippedSymbols": [ "kernel_attn_score_qjl1_256", "kernel_attn_score_qjl1_256_multi" ], "graphOp": "GGML_OP_ATTN_SCORE_QJL", "smokeTarget": "dispatch-smoke", "smokeCommand": "make -C packages/inference/verify dispatch-smoke", "smokeScores": 32, "maxDiff": 2.384e-7, "evidenceDate": "2026-05-11", "notes": "dispatch_smoke links against the built fork's libggml-metal.dylib and drives GGML_OP_ATTN_SCORE_QJL through kernel_attn_score_qjl1_256_multi. Runtime tokens_per_threadgroup remains default 32, with ELIZA_METAL_QJL_TOKENS_PER_TG available for per-device autotune experiments." }, "polar": { "runtimeCapabilityKey": "polarquant", "status": "runtime-ready", "runtimeReady": true, "shippedSymbols": [ "kernel_get_rows_q4_polar", "kernel_mul_mv_q4_polar_f32", "kernel_mul_mv_q4_polar_preht_f32", "kernel_attn_score_q4_polar_preht_f32" ], "graphOp": "GGML_OP_ATTN_SCORE_POLAR", "smokeTarget": "dispatch-smoke", "smokeCommand": "make -C packages/inference/verify dispatch-smoke", "smokeScores": 128, "maxDiff": 3.815e-6, "evidenceDate": "2026-05-11", "notes": "dispatch_smoke covers raw-q use_qjl=0/1 through kernel_mul_mv_q4_polar_f32 and pre-Hadamard use_qjl=0/1 through ggml_attn_score_polar_preht plus kernel_attn_score_q4_polar_preht_f32. The preHT route is only valid when the graph explicitly supplies H*q." } } }