{ "schema_version": "1.0", "ok": true, "generated_at": "2026-07-02", "skill_dir": ".", "commit": "24bea607a2b16e766bc8f2bc8483f22fae1edc61", "git_status": { "available": true, "dirty": true, "changed_file_count": 37, "generated_dirty": true, "generated_changed_file_count": 36, "source_dirty": true, "source_changed_file_count": 1, "sample": [ " M SKILL.md", " M registry/index.json", " M registry/packages/yao-meta-skill.json", " M reports/adaptation_proposals.json", " M reports/adaptation_proposals.md", " M reports/adoption_drift_report.json", " M reports/benchmark_reproducibility.json", " M reports/benchmark_reproducibility.md", " M reports/compiled_targets.json", " M reports/context_budget.json", " M reports/context_budget.md", " M reports/context_budget_summary.json" ], "source_sample": [ " M SKILL.md" ], "generated_sample": [ " M registry/index.json", " M registry/packages/yao-meta-skill.json", " M reports/adaptation_proposals.json", " M reports/adaptation_proposals.md", " M reports/adoption_drift_report.json", " M reports/benchmark_reproducibility.json", " M reports/benchmark_reproducibility.md", " M reports/compiled_targets.json", " M reports/context_budget.json", " M reports/context_budget.md", " M reports/context_budget_summary.json", " M reports/evidence_consistency.json" ], "generated_dirty_prefixes": [ "dist/", "registry/index.json", "registry/packages/", "reports/", "skill_atlas/", "skill-ir/examples/" ], "scope": "generation-time status before this report is written" }, "summary": { "reproducibility_ready": true, "release_lock_ready": false, "methodology_complete": true, "required_artifact_count": 25, "missing_artifact_count": 0, "evidence_bundle_sha256": "6c95043c2d8513102b37bc70a01647246933c84e8a27c2748cc2cae20e040158", "source_contract_sha256": "db71157184dc649ee72dfcaf399204d44a366d942393d055571e0137d42c9f7c", "archive_sha256": "c56a8519244c0bc8ec17c92d4641c832629cc059f4b51fe177a2c039515253a7", "output_case_count": 5, "failure_disclosure_count": 3, "command_count": 23, "command_executed_count": 10, "timing_observed_count": 10, "model_executed_count": 10, "token_observed_count": 10, "human_review_complete": false, "provider_evidence_complete": true, "world_class_ready": false, "world_class_open_gap_count": 3, "world_class_task_count": 4, "world_class_ledger_pending_count": 4, "world_class_source_check_count": 19, "world_class_source_pass_count": 12, "world_class_source_blocked_count": 7, "beta_test_ready": false, "beta_test_blocker_count": 1, "beta_test_deferred_evidence_count": 4, "public_claim_ready": false, "public_claim_blocker_count": 4, "working_tree_dirty": true, "changed_file_count": 37, "source_tree_dirty": true, "source_changed_file_count": 1, "generated_tree_dirty": true, "generated_changed_file_count": 36 }, "beta_test_release": { "ready": false, "scope": "beta/public test release without superiority, fully-reviewed, or world-class claims", "blockers": [ "release lock is not clean or commit is unavailable" ], "allowed_deferred_evidence": [ { "key": "provider-holdout", "label": "Provider Holdout", "reason": "Provider-backed source evidence exists, but formal ledger submission and reviewer acceptance are still pending before public claims." }, { "key": "human-adjudication", "label": "Human Adjudication", "reason": "Human adjudication evidence is still pending; deferred for beta/public testing and still required before superiority, fully-reviewed, or world-class claims." }, { "key": "native-permission-enforcement", "label": "Native Permission Enforcement", "reason": "Native enforcement proof is still pending; deferred for beta/public testing and still required before world-class claims." }, { "key": "native-client-telemetry", "label": "Native Client Telemetry", "reason": "Real client telemetry is still pending; deferred for beta/public testing and still required before world-class claims." } ], "policy": "Human blind-review, native permission enforcement, real client telemetry, and ledger acceptance may be deferred for beta/public testing, but public claims must remain blocked until those evidence entries are accepted.", "required_wording": "Use beta, public test, or technical preview wording; do not claim world-class readiness, fully reviewed quality, or proven superiority over baseline." }, "public_claim": { "ready": false, "scope": "public benchmark or world-class readiness claim", "blockers": [ "release lock is not clean or commit is unavailable", "human blind-review adjudication is incomplete", "world-class evidence is not accepted yet (3 open gaps, 4 ledger pending)", "world-class source checks are not all accepted (12/19 pass, 7 blocked)" ], "policy": "Local reproducibility can pass before public claims; public claims require provider evidence, human adjudication, clean release lock, accepted world-class evidence, and complete source checks." }, "release_lock": { "ready": false, "commit": "24bea607a2b16e766bc8f2bc8483f22fae1edc61", "status_scope": "generation-time status before this report is written", "source_changed_file_count": 1, "generated_changed_file_count": 36, "reason": "source files were dirty at generation time" }, "evidence_bundle": { "algorithm": "sha256(path,label,exists,artifact_sha256)", "artifact_count": 25, "existing_count": 25, "missing_count": 0, "missing_paths": [], "sha256": "6c95043c2d8513102b37bc70a01647246933c84e8a27c2748cc2cae20e040158" }, "methodology": { "path": "reports/benchmark_methodology.md", "exists": true, "sections": [ { "heading": "## Benchmark Types", "exists": true }, { "heading": "## Sample Sources", "exists": true }, { "heading": "## Evaluation Dimensions", "exists": true }, { "heading": "## Weighting Rule", "exists": true }, { "heading": "## Failure Disclosure", "exists": true }, { "heading": "## Reproduction", "exists": true } ], "missing_sections": [] }, "artifacts_checked": [ { "label": "methodology", "path": "reports/benchmark_methodology.md", "exists": true, "bytes": 2715, "sha256": "57025e0123ce5d10401c5bff376d2eeeac7943c83897ae1ad3fb22cadf790f92" }, { "label": "failure_disclosure", "path": "evals/failure-cases.md", "exists": true, "bytes": 889, "sha256": "28833c0d4a217d612879d193fb5de199880dd5b3093ab5757e4315600fa4fb08" }, { "label": "output_cases", "path": "evals/output/cases.jsonl", "exists": true, "bytes": 6555, "sha256": "a6ae9685711620d7203b73ace4412194ae287689f945b5139ec8ad75b9eefe04" }, { "label": "output_schema", "path": "evals/output/schema.json", "exists": true, "bytes": 2193, "sha256": "8ee340c95064260c5e952be614e19841ac676162c6bf01d21b107e38cb04e0b9" }, { "label": "output_scorecard", "path": "reports/output_quality_scorecard.json", "exists": true, "bytes": 25530, "sha256": "0806258a8e084b27e112537faff0de64a8519ca90cfdc78b57c0e4c08a514cca" }, { "label": "output_execution", "path": "reports/output_execution_runs.json", "exists": true, "bytes": 8449, "sha256": "4df66b63d2e717a559a6881b966fb1b92def7329cc89aa48ee0d8596f87402e5" }, { "label": "blind_review", "path": "reports/output_blind_review_pack.json", "exists": true, "bytes": 7804, "sha256": "bbe2db8ec2776fe289cd7d6bb78d48c2b8baa106ad37f79c15e43669b10c9390" }, { "label": "review_adjudication", "path": "reports/output_review_adjudication.json", "exists": true, "bytes": 14084, "sha256": "91fd88dd9b0f8876f68027a87893d1852dac59772613345bbbfabba658c9227e" }, { "label": "trigger_scorecard", "path": "reports/route_scorecard.json", "exists": true, "bytes": 17042, "sha256": "53fc22d220dc11453c1e66a17ec2d935e069f9bc936952fe4a53afdb4c71b2cc" }, { "label": "runtime_conformance", "path": "reports/conformance_matrix.json", "exists": true, "bytes": 10342, "sha256": "97f9ba949c23a60b00e9ba2ff279ca03ba517845cc7c62aa8a42645d58006c7e" }, { "label": "trust_report", "path": "reports/security_trust_report.json", "exists": true, "bytes": 139090, "sha256": "108aee597a1f245f73d9b69454b6fa8a144545f905231a7c314364c7066541c6" }, { "label": "python_compatibility", "path": "reports/python_compatibility.json", "exists": true, "bytes": 30497, "sha256": "4d82942052a2ca87db7acea451096332d4b8bf6420a29fa9f5596e83464d21ea" }, { "label": "registry_audit", "path": "reports/registry_audit.json", "exists": true, "bytes": 3155, "sha256": "d307675d5c6c3cd2f104ce7d963b37d36113f35334e4c6895b55fbb8b255e915" }, { "label": "package_verification", "path": "reports/package_verification.json", "exists": true, "bytes": 19525, "sha256": "e2ad726ce0486be0f57d88e518ef4d265479a04756419c6f8df50cf7d05b1f14" }, { "label": "install_simulation", "path": "reports/install_simulation.json", "exists": true, "bytes": 8947, "sha256": "65adf51d8c4b1d5ea49edde09004f7e4bfc06127e640baf2de23af695fbe0728" }, { "label": "skill_os2_audit", "path": "reports/skill_os2_audit.json", "exists": true, "bytes": 13824, "sha256": "150c0027deaa3c2dd44db5380e9a9b16f123e8a8e3ff6142c711b2b0c432f81c" }, { "label": "world_class_evidence_plan", "path": "reports/world_class_evidence_plan.json", "exists": true, "bytes": 23314, "sha256": "483ed2b3bd020fa2c5a888300c0d0c1a265eebaf648076581f082f2d527d3669" }, { "label": "world_class_evidence_ledger", "path": "reports/world_class_evidence_ledger.json", "exists": true, "bytes": 26274, "sha256": "b19e7113ec463dbb1cfb87c45d3c51624151aeea1747c8d9c286b6967a385c6a" }, { "label": "world_class_evidence_intake", "path": "reports/world_class_evidence_intake.json", "exists": true, "bytes": 20922, "sha256": "885bff9c1c8546c606f583a982a2283229438dbdcd44f3c7e55aff71f858f8f4" }, { "label": "world_class_evidence_preflight", "path": "reports/world_class_evidence_preflight.json", "exists": true, "bytes": 90854, "sha256": "06e93b1c0c99107002466d9418a4910c82397bce26c3bbdd5b5cc9875eeb65b2" }, { "label": "world_class_submission_review", "path": "reports/world_class_submission_review.json", "exists": true, "bytes": 17292, "sha256": "e8df5c72ade7ad0704b6ee5dd6aedefea98f79a9a370cd4d80655401862e87f4" }, { "label": "world_class_operator_runbook", "path": "reports/world_class_operator_runbook.json", "exists": true, "bytes": 83712, "sha256": "bb67813244d64f8e5e416620bb9d7856d95d85a1ec7f9581b2a97b9fed47e02e" }, { "label": "world_class_operator_runbook_markdown", "path": "reports/world_class_operator_runbook.md", "exists": true, "bytes": 25504, "sha256": "e8d4dee7541323f0e6fa6be4903e7c482b1d00ef1db5594f0c44212088b53d85" }, { "label": "world_class_operator_runbook_html", "path": "reports/world_class_operator_runbook.html", "exists": true, "bytes": 34651, "sha256": "022594013b2469a0e5776e3f5f6a99fa069917c5ba5cb90256b7841b1f5fd1e3" }, { "label": "world_class_claim_guard", "path": "reports/world_class_claim_guard.json", "exists": true, "bytes": 20713, "sha256": "911ba833055ab857250f88d88ccdd0bb652076a81214f725c49d763352035572" } ], "missing_artifacts": [], "reproduction_commands": [ { "label": "source commit", "command": "git rev-parse HEAD", "evidence": "git commit hash" }, { "label": "trigger eval", "command": "make eval-suite", "evidence": "reports/eval_suite.json" }, { "label": "output eval", "command": "python3 scripts/yao.py output-eval", "evidence": "reports/output_quality_scorecard.json" }, { "label": "output execution", "command": "python3 scripts/yao.py output-exec --runner-command '[\"python3\",\"scripts/local_output_eval_runner.py\"]'", "evidence": "reports/output_execution_runs.json" }, { "label": "blind review adjudication", "command": "python3 scripts/yao.py output-review", "evidence": "reports/output_review_adjudication.json" }, { "label": "skill ir", "command": "python3 scripts/yao.py skill-ir . --output-json skill-ir/examples/yao-meta-skill.json", "evidence": "skill-ir/examples/yao-meta-skill.json" }, { "label": "runtime conformance", "command": "python3 scripts/yao.py conformance .", "evidence": "reports/conformance_matrix.json" }, { "label": "trust report", "command": "python3 scripts/yao.py trust .", "evidence": "reports/security_trust_report.json" }, { "label": "python compatibility", "command": "python3 scripts/yao.py python-compat .", "evidence": "reports/python_compatibility.json" }, { "label": "package", "command": "python3 scripts/yao.py package . --platform openai --platform claude --platform generic --platform vscode --expectations evals/packaging_expectations.json --output-dir dist --zip", "evidence": "dist/yao-meta-skill.zip" }, { "label": "package verify", "command": "python3 scripts/yao.py package-verify . --package-dir dist --require-zip", "evidence": "reports/package_verification.json" }, { "label": "install simulate", "command": "python3 scripts/yao.py install-simulate . --package-dir dist", "evidence": "reports/install_simulation.json" }, { "label": "registry audit", "command": "python3 scripts/yao.py registry-audit .", "evidence": "reports/registry_audit.json" }, { "label": "skill os audit", "command": "python3 scripts/yao.py skill-os2-audit .", "evidence": "reports/skill_os2_audit.json" }, { "label": "world-class evidence plan", "command": "python3 scripts/yao.py world-class-evidence .", "evidence": "reports/world_class_evidence_plan.json" }, { "label": "world-class evidence ledger", "command": "python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions", "evidence": "reports/world_class_evidence_ledger.json" }, { "label": "world-class evidence intake", "command": "python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions", "evidence": "reports/world_class_evidence_intake.json" }, { "label": "world-class evidence preflight", "command": "python3 scripts/yao.py world-class-preflight . --submissions-dir evidence/world_class/submissions", "evidence": "reports/world_class_evidence_preflight.json" }, { "label": "world-class submission review", "command": "python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions", "evidence": "reports/world_class_submission_review.json" }, { "label": "world-class operator runbook", "command": "python3 scripts/yao.py world-class-runbook . --submissions-dir evidence/world_class/submissions", "evidence": "reports/world_class_operator_runbook.json" }, { "label": "world-class claim guard", "command": "python3 scripts/yao.py world-class-claim-guard .", "evidence": "reports/world_class_claim_guard.json" }, { "label": "evidence consistency", "command": "python3 scripts/yao.py evidence-consistency .", "evidence": "reports/evidence_consistency.json" }, { "label": "full ci", "command": "make ci-test", "evidence": "CI target output" } ], "failure_disclosure": { "path": "evals/failure-cases.md", "case_count": 3, "policy": "Keep representative failures visible and tied to regression checks." }, "limitations": [ "The git commit and dirty flags are generation-time context; release lock is blocked by source changes, while generated evidence artifacts are tracked separately.", "Provider-backed model holdout source evidence is complete, but ledger acceptance still requires a valid independently reviewed submission packet.", "Pending blind-review decisions are visible but do not count as human adjudication.", "World-class readiness remains false until external and human evidence gaps close.", "Beta/public testing may proceed without human blind-review only when wording avoids superiority, fully-reviewed, or world-class claims." ], "artifacts": { "json": "reports/benchmark_reproducibility.json", "markdown": "reports/benchmark_reproducibility.md" } }