diff --git a/.changeset/amc-a2a-negotiation-methodology.md b/.changeset/amc-a2a-negotiation-methodology.md new file mode 100644 index 000000000..11bc82274 --- /dev/null +++ b/.changeset/amc-a2a-negotiation-methodology.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add A2A-NT-style agent-to-agent negotiation methodology boundaries, migration triggers, and documentation. diff --git a/.changeset/amc-academiclaw-metric-validity.md b/.changeset/amc-academiclaw-metric-validity.md new file mode 100644 index 000000000..78e390c90 --- /dev/null +++ b/.changeset/amc-academiclaw-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add AcademiClaw academic-task metric-validity receipts with fail-closed Score/Shield/Watch surfaces, methodology/docs bindings, and eval-pack proof fields. diff --git a/.changeset/amc-accessibility-regressions.md b/.changeset/amc-accessibility-regressions.md new file mode 100644 index 000000000..d0cb72c6a --- /dev/null +++ b/.changeset/amc-accessibility-regressions.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Fix accessibility regressions flagged in the Batch 5 audit: top-level CLI `--no-color` handling and help text, accessible names for browser playground and generated console chart canvases, generated-dashboard secondary text contrast, published accessibility statement, and OG image asset verification. diff --git a/.changeset/amc-accessibility-release-evidence.md b/.changeset/amc-accessibility-release-evidence.md new file mode 100644 index 000000000..2ff6f5cc8 --- /dev/null +++ b/.changeset/amc-accessibility-release-evidence.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add an accessibility release-evidence generator and runbook for recording Playwright axe run status without overclaiming manual assistive-technology coverage. diff --git a/.changeset/amc-adgen-soc-dataset-replay.md b/.changeset/amc-adgen-soc-dataset-replay.md new file mode 100644 index 000000000..216825899 --- /dev/null +++ b/.changeset/amc-adgen-soc-dataset-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add AD-GEN-style SOC dataset replay receipts, summaries, CI fail-closed fields, and public methodology proof boundaries for ATT&CK-aligned endpoint telemetry benchmark claims. diff --git a/.changeset/amc-adk-runtime-live-drift.md b/.changeset/amc-adk-runtime-live-drift.md new file mode 100644 index 000000000..748c544f9 --- /dev/null +++ b/.changeset/amc-adk-runtime-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add ADK TypeScript runtime live-drift receipts so Watch fails closed on missing runtime, framework, graph, tool-registry, eval dataset/case, runner, session, live-queue, API-route, deployment, metric, signed-evidence, and row-hash proof. diff --git a/.changeset/amc-advanced-rag-notebook-replay.md b/.changeset/amc-advanced-rag-notebook-replay.md new file mode 100644 index 000000000..94bb51e2f --- /dev/null +++ b/.changeset/amc-advanced-rag-notebook-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add Advanced RAG notebook replay receipts with course/lesson identity, retrieval variant, notebook/output hashes, environment and dependency-lock hashes, corpus/index/query/reference-answer hashes, retrieval/generation/eval/observability traces, replay command, deterministic seed, query count, RAG triad metric thresholds, CI failed-row reporting, and r42 methodology docs. diff --git a/.changeset/amc-adversarial-alignment-probes.md b/.changeset/amc-adversarial-alignment-probes.md new file mode 100644 index 000000000..aeb776e5b --- /dev/null +++ b/.changeset/amc-adversarial-alignment-probes.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add an `adversarialAlignmentProbes` assurance/redteam pack with executable deceptive-alignment, reward-model-gaming, and goal-misgeneralization probes, plus regression coverage and catalog discoverability. diff --git a/.changeset/amc-agent-belt-methodology-versioning.md b/.changeset/amc-agent-belt-methodology-versioning.md new file mode 100644 index 000000000..9ef9ab6b7 --- /dev/null +++ b/.changeset/amc-agent-belt-methodology-versioning.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Agent Belt methodology-versioning assurance receipts for reproducible coding-agent evaluation claims. diff --git a/.changeset/amc-agent-bench-java-coding-validity.md b/.changeset/amc-agent-bench-java-coding-validity.md new file mode 100644 index 000000000..9fd3b5b03 --- /dev/null +++ b/.changeset/amc-agent-bench-java-coding-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Agent Bench-style Java coding-agent metric-validity gates so benchmark/source/license, Java task, YAML benchmark, isolated workspace, CLI-agent, cascaded judge, Maven/JUnit/JaCoCo, result, accuracy/pass@k, sample/CI, signed evidence, and row-hash proof fail closed before Java coding-agent benchmark claims are accepted. diff --git a/.changeset/amc-agent-eval-harness-live-drift.md b/.changeset/amc-agent-eval-harness-live-drift.md new file mode 100644 index 000000000..430927bf2 --- /dev/null +++ b/.changeset/amc-agent-eval-harness-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Adds agent-eval-harness live-drift receipt fields, thresholds, row hashing, Watch alerts, methodology docs, and tests for signed trace/evidence coverage, tool-success, hallucination, latency, cost, framework, trace-mode, and metric-context drift. diff --git a/.changeset/amc-agent-eval-observability-live-drift.md b/.changeset/amc-agent-eval-observability-live-drift.md new file mode 100644 index 000000000..5b9a481c7 --- /dev/null +++ b/.changeset/amc-agent-eval-observability-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add fail-closed live-drift proof for agent-evaluation observability rows, including config, telemetry, evidence coverage, metric-set and telemetry distributions, public methodology r133 docs, and source-safe documentation for vladfeigin/llm-agents-evaluation. diff --git a/.changeset/amc-agent-mont-monitoring-replay.md b/.changeset/amc-agent-mont-monitoring-replay.md new file mode 100644 index 000000000..4b6a1b766 --- /dev/null +++ b/.changeset/amc-agent-mont-monitoring-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Agent_Mont-style monitoring replay receipts to the benchmark corpus, including fail-closed evidence checks for monitoring configuration, framework, token/cost/latency/resource/carbon/log/visualization artifacts, summaries, CI receipt fields, public methodology versioning, and documentation. diff --git a/.changeset/amc-agent-reading-test-live-drift.md b/.changeset/amc-agent-reading-test-live-drift.md new file mode 100644 index 000000000..84bd470c3 --- /dev/null +++ b/.changeset/amc-agent-reading-test-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Agent Reading Test-style web-content reading live-drift receipts with source snapshot, license, homepage, answer key, task manifest, score form, live-site, raw content, canary, alert, signed evidence, and row-hash proof. diff --git a/.changeset/amc-agent-security-live-drift.md b/.changeset/amc-agent-security-live-drift.md new file mode 100644 index 000000000..3446a8b66 --- /dev/null +++ b/.changeset/amc-agent-security-live-drift.md @@ -0,0 +1,7 @@ +--- +"agent-maturity-compass": patch +--- + +Add agent-security control live-drift receipts for Watch and Shield. + +Live drift rows can now bind guard identity, policy hashes, taint/proxy/audit/telemetry/eval-pack/classifier proof, origin and taint coverage, policy-decision accuracy, secret-scrub rate, audit integrity, attack-effectiveness rate, false-positive rate, guard latency, signed evidence, and row hashes. Missing or degraded agent-security evidence fails closed through score drift, behavior drift, Watch alerts, Shield verification, and public methodology r68. diff --git a/.changeset/amc-agent-testing-methodology-live-drift.md b/.changeset/amc-agent-testing-methodology-live-drift.md new file mode 100644 index 000000000..c97a44a32 --- /dev/null +++ b/.changeset/amc-agent-testing-methodology-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add agent-testing methodology live-drift receipts for Watch and Shield. Live rows can now carry testing taxonomy, methodology, scenario, fault-injection, observability, safety, standards, category, approach, fault-model, benchmark-family, coverage, resilience, safety-regression, and observability-signal evidence so AMC fails closed when live traffic drifts away from the declared testing methodology despite stable generic scores. diff --git a/.changeset/amc-agent-workflow-kit-replay.md b/.changeset/amc-agent-workflow-kit-replay.md new file mode 100644 index 000000000..df66a1a3d --- /dev/null +++ b/.changeset/amc-agent-workflow-kit-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Agent Workflow Kit-style workflow replay receipts with fail-closed source, policy, approval, verification, docs-check, replay, threshold, CI receipt, methodology, and documentation coverage. diff --git a/.changeset/amc-agentbench-config-replay.md b/.changeset/amc-agentbench-config-replay.md new file mode 100644 index 000000000..0ead011c6 --- /dev/null +++ b/.changeset/amc-agentbench-config-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Adds AgentBench-style config-pinned replay receipts to benchmark corpus runs so source, repository, dataset, agent/global/model-server/environment/dependency, run/replay command, trace, result, metric, seed, sample, shuffle, replay-pass, trace-coverage, signed-evidence, and row-hash proof fail closed. diff --git a/.changeset/amc-agentdefense-provider-drift.md b/.changeset/amc-agentdefense-provider-drift.md new file mode 100644 index 000000000..c1d2f41ef --- /dev/null +++ b/.changeset/amc-agentdefense-provider-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add AgentDefense-Bench provider-drift receipts with source/MCP/security-defense proof, fail-closed Watch/CI alerts, public methodology r200 binding, and docs for the live verified source boundary. diff --git a/.changeset/amc-agentest-scenario-metric-validity.md b/.changeset/amc-agentest-scenario-metric-validity.md new file mode 100644 index 000000000..d1c3b0f39 --- /dev/null +++ b/.changeset/amc-agentest-scenario-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add Agentest-style scenario-test metric-validity proof with signed source, endpoint, scenario, persona, goal, knowledge, tool-mock, scripted-turn, trajectory assertion, LLM judge, comparison, CI reporter, result, sample-size, confidence-interval, and row-hash evidence. Missing or invalid proof now fails closed when `requireAgentScenarioTestProof` is enabled. diff --git a/.changeset/amc-agentic-graph-rag-metric-validity.md b/.changeset/amc-agentic-graph-rag-metric-validity.md new file mode 100644 index 000000000..fd12fdfcf --- /dev/null +++ b/.changeset/amc-agentic-graph-rag-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Agentic Graph RAG metric-validity receipts with fail-closed source/no-license, graph/RAG, vector-store, evaluation, experiment-tracking, UI-question, dependency-lock, owner, confidence-interval, signed-evidence, artifact-hash, and row-hash proof. diff --git a/.changeset/amc-agentic-search-live-drift.md b/.changeset/amc-agentic-search-live-drift.md new file mode 100644 index 000000000..147b09243 --- /dev/null +++ b/.changeset/amc-agentic-search-live-drift.md @@ -0,0 +1,7 @@ +--- +"agent-maturity-compass": minor +--- + +Add agentic-search live drift receipts for baseline-to-live monitoring. + +Rows can now bind benchmark id, dataset family, query type, query/task ids, source and tool-config hashes, planner/search/citation/synthesis trace hashes, result manifest hash, planning/query-decomposition/relevance/synthesis scores, and citation coverage. Score, Watch, and Shield now fail closed on agentic-search score drops, citation or trace coverage gaps, dataset-family drift, query-type drift, tool-context drift, missing signed evidence, or receipt hash mismatches. diff --git a/.changeset/amc-agentkernelarena-gpu-kernel-replay.md b/.changeset/amc-agentkernelarena-gpu-kernel-replay.md new file mode 100644 index 000000000..60f5a0043 --- /dev/null +++ b/.changeset/amc-agentkernelarena-gpu-kernel-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add AgentKernelArena-style GPU-kernel replay receipts with task/config, agent roster, workspace isolation, GPU profile, compile/correctness/performance proof, speedup delta, replay/result coverage, CI receipt, signed evidence, and row-hash gates. diff --git a/.changeset/amc-agentrial-statistical-question-explainability.md b/.changeset/amc-agentrial-statistical-question-explainability.md new file mode 100644 index 000000000..e9347968a --- /dev/null +++ b/.changeset/amc-agentrial-statistical-question-explainability.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add AgentTrial-style statistical question-explainability receipts with repeated trial counts, Wilson confidence intervals, bootstrap cost/latency, failure attribution, regression comparison, CI proof, reliability score, and row-hash gates. diff --git a/.changeset/amc-agentrim-diagnostic-question.md b/.changeset/amc-agentrim-diagnostic-question.md new file mode 100644 index 000000000..8d743eee8 --- /dev/null +++ b/.changeset/amc-agentrim-diagnostic-question.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add an AgenTRIM-backed diagnostic question for per-step least-privilege tool access with status-aware validation evidence gates. diff --git a/.changeset/amc-agentstock-judge-calibration.md b/.changeset/amc-agentstock-judge-calibration.md new file mode 100644 index 000000000..0ee8e29e7 --- /dev/null +++ b/.changeset/amc-agentstock-judge-calibration.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add AgentStock-style future-outcome ranking proof to judge calibration receipts, including source snapshot, leaderboard, PnL, appeal, replay, signed-evidence, and Watch fail-closed gates. diff --git a/.changeset/amc-agiflow-observability-methodology.md b/.changeset/amc-agiflow-observability-methodology.md new file mode 100644 index 000000000..780d6e3db --- /dev/null +++ b/.changeset/amc-agiflow-observability-methodology.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add LLM workflow observability methodology-versioning boundaries for trace, visual-debugger, prompt/model registry, frontend analytics, user-feedback, session-replay, telemetry privacy, migration, badge, and report proof. diff --git a/.changeset/amc-ai-agent-benchmark-comparison-replay.md b/.changeset/amc-ai-agent-benchmark-comparison-replay.md new file mode 100644 index 000000000..df9c19eb4 --- /dev/null +++ b/.changeset/amc-ai-agent-benchmark-comparison-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add AI-agent benchmark comparison replay proof to the benchmark corpus receipt so source, repository, license, agent roster, benchmark dataset, source/pricing/user-report/leaderboard/score manifests, eval-pack, fixture, replay, result, score-delta report, CI receipt, coverage metrics, signed evidence, and row hashes fail closed before comparison claims are accepted. diff --git a/.changeset/amc-ai-coding-landscape-explainability.md b/.changeset/amc-ai-coding-landscape-explainability.md new file mode 100644 index 000000000..134e6a0a4 --- /dev/null +++ b/.changeset/amc-ai-coding-landscape-explainability.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add AI-coding landscape question-explainability lenses. Question receipts can now bind source category, dataset refs and SHA-256 hashes, update cadence, freshness, cohort refs, benchmark/tool/model refs, accepted evidence, rejected-evidence reasons, and repair hints into row hashes, fail-closed status, Studio evidence drilldown, and public methodology r36. diff --git a/.changeset/amc-ai-evaluation-guide-methodology.md b/.changeset/amc-ai-evaluation-guide-methodology.md new file mode 100644 index 000000000..76a796d2c --- /dev/null +++ b/.changeset/amc-ai-evaluation-guide-methodology.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Awesome AI Evaluation Guide public-methodology receipts with source/license, default branch, guide manifests, benchmark/tool taxonomies, metric-selection, threshold, calibration, trace, human-review, cost-control, deprecation, migration, signed evidence, and row-hash proof. diff --git a/.changeset/amc-ai-reputation-claude-live-drift.md b/.changeset/amc-ai-reputation-claude-live-drift.md new file mode 100644 index 000000000..3edc741d6 --- /dev/null +++ b/.changeset/amc-ai-reputation-claude-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add AI Reputation Claude live-drift receipts with source/no-license, agent roster, skill catalog, review-source, sentiment, competitor, response-policy, crisis, report, baseline/live result, drift statistic, alert receipt, brand-safety metric, signed-evidence, and row-hash proof. diff --git a/.changeset/amc-aicrypto-methodology-boundary.md b/.changeset/amc-aicrypto-methodology-boundary.md new file mode 100644 index 000000000..c11230153 --- /dev/null +++ b/.changeset/amc-aicrypto-methodology-boundary.md @@ -0,0 +1,7 @@ +--- +"agent-maturity-compass": patch +--- + +Add an AICrypto-style cryptography benchmark methodology boundary. + +Public cryptography capability claims now require methodology-versioned proof for paper/dataset versions, MCQ/CTF/proof task families, expert baselines, sandbox/toolchain evidence, proof rubrics, scoring formulas, thresholds, signed evidence, and row hashes before AMC reports or badges can use those claims as external evidence. diff --git a/.changeset/amc-alignment-feedback-source-validation.md b/.changeset/amc-alignment-feedback-source-validation.md new file mode 100644 index 000000000..43fc571f0 --- /dev/null +++ b/.changeset/amc-alignment-feedback-source-validation.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add alignment feedback-source validation scoring and a diagnostic question for evaluator source quality, bias, collusion, and signed feedback provenance. diff --git a/.changeset/amc-alignment-index-subcategories.md b/.changeset/amc-alignment-index-subcategories.md new file mode 100644 index 000000000..15082bdaf --- /dev/null +++ b/.changeset/amc-alignment-index-subcategories.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Expose alignment-index subcategory breakdowns for goal misgeneralization, reward hacking, deceptive alignment, feedback source validation, sycophancy, and sabotage. diff --git a/.changeset/amc-api-key-cli.md b/.changeset/amc-api-key-cli.md new file mode 100644 index 000000000..d9b538052 --- /dev/null +++ b/.changeset/amc-api-key-cli.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc api key create`, `amc api key list`, and `amc api key revoke` for local programmatic API key management. Keys are displayed only once on creation; the persisted store keeps hashed secret material and public metadata under `.amc/auth/api-keys.json`. diff --git a/.changeset/amc-api-operational-contracts.md b/.changeset/amc-api-operational-contracts.md new file mode 100644 index 000000000..483249141 --- /dev/null +++ b/.changeset/amc-api-operational-contracts.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Document API rate-limit headers, org SSE reconnect behavior, and webhook retry boundaries; add org SSE event IDs and a 15-second reconnect hint. diff --git a/.changeset/amc-assurance-certificate-threshold-guide.md b/.changeset/amc-assurance-certificate-threshold-guide.md new file mode 100644 index 000000000..ea8d8fc20 --- /dev/null +++ b/.changeset/amc-assurance-certificate-threshold-guide.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Document signed assurance certificate issuance, verification, and policy threshold tuning. diff --git a/.changeset/amc-assurance-demo-nosign.md b/.changeset/amc-assurance-demo-nosign.md new file mode 100644 index 000000000..d6589bc80 --- /dev/null +++ b/.changeset/amc-assurance-demo-nosign.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc assurance run --demo --no-sign` and make single-pack no-sign runs vault-less. diff --git a/.changeset/amc-assurance-lab-index-mjs-docs.md b/.changeset/amc-assurance-lab-index-mjs-docs.md new file mode 100644 index 000000000..bd33f2c3b --- /dev/null +++ b/.changeset/amc-assurance-lab-index-mjs-docs.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add regression coverage for Assurance Lab pack-authoring docs so community packs document `index.mjs` as the scaffolded entry point and `index.js` only as legacy fallback. diff --git a/.changeset/amc-assurance-remediation-priority.md b/.changeset/amc-assurance-remediation-priority.md new file mode 100644 index 000000000..aaca6dbcd --- /dev/null +++ b/.changeset/amc-assurance-remediation-priority.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add a remediation-priority section to text assurance runs so failed scenarios are ordered by severity with reason, fix hint, evidence path, and verbose rerun command. diff --git a/.changeset/amc-assurance-run-nosign-audit.md b/.changeset/amc-assurance-run-nosign-audit.md new file mode 100644 index 000000000..b2ed40e54 --- /dev/null +++ b/.changeset/amc-assurance-run-nosign-audit.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add built-CLI coverage for `assurance run --all --no-sign` and align the UX audit with the current unsigned assurance behavior. diff --git a/.changeset/amc-awesome-agent-memory-live-drift.md b/.changeset/amc-awesome-agent-memory-live-drift.md new file mode 100644 index 000000000..3c4e24e7c --- /dev/null +++ b/.changeset/amc-awesome-agent-memory-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Awesome-Agent-Memory-style memory-catalog live-drift receipts with source snapshot, no-license boundary, README blob, taxonomy, benchmark/eval, drift statistic, alert, signed evidence, and row-hash proof. diff --git a/.changeset/amc-azure-agent-lab-replay-corpus.md b/.changeset/amc-azure-agent-lab-replay-corpus.md new file mode 100644 index 000000000..6ff567608 --- /dev/null +++ b/.changeset/amc-azure-agent-lab-replay-corpus.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Azure Agent Lab replay-corpus receipts with lab/module identity, workshop and notebook hashes, Azure service/project/search/RAG/tool/evaluator configs, cloud-run and identity proof, replay command hashes, deterministic seeds, scenario counts, evaluation scores, groundedness thresholds, CI failed-row ids, Watch alerts, and public methodology/docs coverage. diff --git a/.changeset/amc-backdooragent-stage-backdoor-live-drift.md b/.changeset/amc-backdooragent-stage-backdoor-live-drift.md new file mode 100644 index 000000000..bb23e54da --- /dev/null +++ b/.changeset/amc-backdooragent-stage-backdoor-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add BackdoorAgent-style stage-aware backdoor live-drift receipts so attack success, clean accuracy, trigger persistence, trigger propagation, trajectory coverage, evidence coverage, stage/task/attack-family drift, signed evidence, and row hashes fail closed. diff --git a/.changeset/amc-batch5-audit-score-consistency.md b/.changeset/amc-batch5-audit-score-consistency.md new file mode 100644 index 000000000..a90dcf08a --- /dev/null +++ b/.changeset/amc-batch5-audit-score-consistency.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add a Batch 5 audit consistency regression so persona table scores, section headings, and rating lines stay aligned after follow-up fixes. diff --git a/.changeset/amc-benchloop-local-benchmark-replay.md b/.changeset/amc-benchloop-local-benchmark-replay.md new file mode 100644 index 000000000..76b968ac5 --- /dev/null +++ b/.changeset/amc-benchloop-local-benchmark-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add BenchLoop-style local benchmark replay receipts with fail-closed suite, harness, provider, hardware, run, trace, latency, token, export, and metric evidence bindings. diff --git a/.changeset/amc-benchmark-hackability-audit-replay.md b/.changeset/amc-benchmark-hackability-audit-replay.md new file mode 100644 index 000000000..a823d511d --- /dev/null +++ b/.changeset/amc-benchmark-hackability-audit-replay.md @@ -0,0 +1,7 @@ +--- +"agent-maturity-compass": patch +--- + +Add benchmark-hackability audit replay receipts for replay benchmark corpora. + +Replay rows can now bind scanner identity, target benchmark/task manifests, audit configs, phase traces, static-tool reports, AI-inspection traces, vulnerability finding manifests, dashboard/report artifacts, replay commands, sandbox controls, PoC validation, vulnerability-class coverage, task-count coverage, exploitability thresholds, signed evidence, and row hashes. Missing benchmark-hackability evidence fails closed through the manifest, CI receipt, Shield verification, Watch alerts, and public methodology r67. diff --git a/.changeset/amc-benchmark-submission-question-explainability.md b/.changeset/amc-benchmark-submission-question-explainability.md new file mode 100644 index 000000000..dfb34a9d1 --- /dev/null +++ b/.changeset/amc-benchmark-submission-question-explainability.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add benchmark-submission question explainability receipts with task status, criterion scoring, leaderboard metric views, replay hashes, fail-closed validation, drilldown previews, guide remediation hints, and public methodology r65. diff --git a/.changeset/amc-besttester-replay-corpus.md b/.changeset/amc-besttester-replay-corpus.md new file mode 100644 index 000000000..70d0e3204 --- /dev/null +++ b/.changeset/amc-besttester-replay-corpus.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add BestTester replay-corpus receipts for QA-agent benchmark evidence. AMC now validates source/license snapshots, package and lockfile refs, Playwright and TypeScript proof, test/agent/MCP/security/workflow artifacts, LLM-judge agreement, security coverage, CI coverage, signed evidence, and row hashes before BestTester-style claims can pass Score, Shield, or Watch gates. diff --git a/.changeset/amc-bioagentbench-metric-validity.md b/.changeset/amc-bioagentbench-metric-validity.md new file mode 100644 index 000000000..3e5d536b6 --- /dev/null +++ b/.changeset/amc-bioagentbench-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add BioAgentBench-style bioinformatics agent metric-validity proof for Score/Shield rows. AMC now requires signed benchmark/source, task, input dataset, truth/reference, workflow reproduction, Docker/environment, tool-version, harness, grader, result artifact, perturbation, privacy-boundary, owner, sample-size, confidence-interval, and row-hash evidence before bioinformatics workflow claims can be used externally. diff --git a/.changeset/amc-biokgbench-biomedical-kg-replay.md b/.changeset/amc-biokgbench-biomedical-kg-replay.md new file mode 100644 index 000000000..2ae9d2b84 --- /dev/null +++ b/.changeset/amc-biokgbench-biomedical-kg-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add BioKGBench-style biomedical KG replay receipts so source, repository, paper, license, dataset release, knowledge graph, KGCheck/KGQA/SCV task manifests, agent/RAG/Neo4j configs, evaluation scripts, result manifests, error-discovery reports, replay commands, CI receipts, deterministic seeds, metrics, thresholds, signed evidence, and row hashes fail closed before public biomedical KG benchmark claims. diff --git a/.changeset/amc-biomedarena-biomedical-replay.md b/.changeset/amc-biomedarena-biomedical-replay.md new file mode 100644 index 000000000..96240f3d5 --- /dev/null +++ b/.changeset/amc-biomedarena-biomedical-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add BioMedArena-style biomedical harness replay receipts with source, harness, benchmark-family, tool-mode, adapter/tool/vendor, baseline, result, replay, CI, coverage, sandbox, and row-hash proof. diff --git a/.changeset/amc-board-l3-risk-memo.md b/.changeset/amc-board-l3-risk-memo.md new file mode 100644 index 000000000..d1be19c93 --- /dev/null +++ b/.changeset/amc-board-l3-risk-memo.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add a board-facing L3 business-risk memo and link it from executive surfaces without over-approving production use. diff --git a/.changeset/amc-business-fair-scenario.md b/.changeset/amc-business-fair-scenario.md new file mode 100644 index 000000000..90cc527e9 --- /dev/null +++ b/.changeset/amc-business-fair-scenario.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc business fair-scenario` to run a deterministic FAIR-style scenario loss distribution with explicit frequency and loss-magnitude calibration ranges, maturity-adjusted exposure, P10/P50/P90/P95 outputs, and risk-appetite status without claiming certified Open FAIR or native GRC sync. diff --git a/.changeset/amc-business-grc-export.md b/.changeset/amc-business-grc-export.md new file mode 100644 index 000000000..4594c6fe5 --- /dev/null +++ b/.changeset/amc-business-grc-export.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc business grc-export` to turn portfolio maturity-risk inputs into CSV, JSON, or Markdown GRC treatment-plan exports with owner, due-date, risk appetite, ISO 31000 context, and FAIR-style loss-frequency/loss-magnitude fields. diff --git a/.changeset/amc-business-risk-heatmap.md b/.changeset/amc-business-risk-heatmap.md new file mode 100644 index 000000000..ff5d31c14 --- /dev/null +++ b/.changeset/amc-business-risk-heatmap.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc business heatmap` for portfolio-level monetary risk heatmaps across agents, business units, residual expected annual loss, and risk-appetite breaches. diff --git a/.changeset/amc-business-risk-quantification.md b/.changeset/amc-business-risk-quantification.md new file mode 100644 index 000000000..900e8ee66 --- /dev/null +++ b/.changeset/amc-business-risk-quantification.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc business risk` to estimate residual incident frequency, expected annual loss, expected loss reduction, and risk-appetite status from agent maturity. diff --git a/.changeset/amc-business-roi-calculator.md b/.changeset/amc-business-roi-calculator.md new file mode 100644 index 000000000..ee0f12fbd --- /dev/null +++ b/.changeset/amc-business-roi-calculator.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc business roi` for cost-of-trust-gap ROI estimates backed by the maturity-linked expected annual loss model. diff --git a/.changeset/amc-buyer-packages-pricing-link.md b/.changeset/amc-buyer-packages-pricing-link.md new file mode 100644 index 000000000..1da0721e3 --- /dev/null +++ b/.changeset/amc-buyer-packages-pricing-link.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Link buyer packages prominently from the homepage pricing section for procurement workflows. diff --git a/.changeset/amc-calibra-public-methodology.md b/.changeset/amc-calibra-public-methodology.md new file mode 100644 index 000000000..c33d39f44 --- /dev/null +++ b/.changeset/amc-calibra-public-methodology.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Calibra-style public-methodology receipts with campaign matrix, task, report, dashboard, changelog, migration, and row-hash proof. diff --git a/.changeset/amc-catastrophic-risk-indicators.md b/.changeset/amc-catastrophic-risk-indicators.md new file mode 100644 index 000000000..e1925a0e3 --- /dev/null +++ b/.changeset/amc-catastrophic-risk-indicators.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add ForesightSafety catastrophic-risk scoring for self-replication, resource acquisition, shutdown resistance, persistence, goal-preservation pressure, and cross-system propagation. diff --git a/.changeset/amc-cc-plugin-eval-metric-validity.md b/.changeset/amc-cc-plugin-eval-metric-validity.md new file mode 100644 index 000000000..6dbf24717 --- /dev/null +++ b/.changeset/amc-cc-plugin-eval-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add cc-plugin-eval-style metric-validity proof fields, fail-closed thresholds, public methodology boundaries, and documentation for component-trigger reliability evidence. diff --git a/.changeset/amc-cert-preview-no-sign.md b/.changeset/amc-cert-preview-no-sign.md new file mode 100644 index 000000000..20af84221 --- /dev/null +++ b/.changeset/amc-cert-preview-no-sign.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc cert generate --no-sign` / `--preview` for unsigned trust-certificate previews that work without vault signing, are labeled `UNSIGNED_PREVIEW`, and are intentionally rejected by verifier logic until regenerated as signed certificates. diff --git a/.changeset/amc-chaos-reliability-live-drift.md b/.changeset/amc-chaos-reliability-live-drift.md new file mode 100644 index 000000000..23909a9d4 --- /dev/null +++ b/.changeset/amc-chaos-reliability-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add chaos-reliability live-drift receipts for Watch and Shield. Live rows can now carry benchmark, scenario, chaos profile, injection, mutation, endpoint contract, judge, trace bundle, score ledger, agent-card, improvement-eval, framework, modality, benchmark-family, production-reliability, resilience, chaos-drop, recovery, and failure-trace evidence so AMC fails closed when live agents drift under failure-injection pressure despite stable generic scores. diff --git a/.changeset/amc-chipbenchmark-metric-validity.md b/.changeset/amc-chipbenchmark-metric-validity.md new file mode 100644 index 000000000..487f20917 --- /dev/null +++ b/.changeset/amc-chipbenchmark-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add ChipBenchmark-style hardware benchmark metric-validity receipts across Score, Shield, Watch, public methodology, and docs. AMC now fails closed unless ChipBenchmark claims bind source snapshot, no-license boundary, benchmark/hardware/model/precision manifests, environment and runner/serving scripts, result/frontend/pricing datasets, throughput/latency/cost metrics, regression thresholds, owners, confidence intervals, signed evidence, and row hashes. diff --git a/.changeset/amc-chunking-strategy-replay.md b/.changeset/amc-chunking-strategy-replay.md new file mode 100644 index 000000000..0d397c435 --- /dev/null +++ b/.changeset/amc-chunking-strategy-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add RAG chunking-strategy replay corpus receipts with fail-closed document/question/reference-answer, chunker, retrieval, scoring, export, replay-command, count, score, answer-span, and semantic-focus proof. diff --git a/.changeset/amc-ci-init-nosign.md b/.changeset/amc-ci-init-nosign.md new file mode 100644 index 000000000..69c32f1fb --- /dev/null +++ b/.changeset/amc-ci-init-nosign.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc ci init --no-sign` and `amc gate --no-sign` for explicit unsigned CI setup. diff --git a/.changeset/amc-ci-pretest-build.md b/.changeset/amc-ci-pretest-build.md new file mode 100644 index 000000000..8e8bad6d1 --- /dev/null +++ b/.changeset/amc-ci-pretest-build.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Ensure `npm test` builds the CLI before Vitest so clean CI checkouts can run CLI-focused tests that execute `dist/cli.js`. diff --git a/.changeset/amc-ci-provider-secret-examples.md b/.changeset/amc-ci-provider-secret-examples.md new file mode 100644 index 000000000..fd04ad263 --- /dev/null +++ b/.changeset/amc-ci-provider-secret-examples.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Document provider-specific signed CI secret setup for GitHub Actions, GitLab CI/CD, and CircleCI. diff --git a/.changeset/amc-ci-redteam-gate.md b/.changeset/amc-ci-redteam-gate.md new file mode 100644 index 000000000..498932f89 --- /dev/null +++ b/.changeset/amc-ci-redteam-gate.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc ci redteam`, a fail-closed CI regression gate for red-team plugin scores, vulnerability thresholds, optional Evil MCP scenarios, and score-gaming resistance checks with JSON output. diff --git a/.changeset/amc-citation-metadata.md b/.changeset/amc-citation-metadata.md new file mode 100644 index 000000000..c159d67d0 --- /dev/null +++ b/.changeset/amc-citation-metadata.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add citable BibTeX metadata to the AMC whitepaper and RFC while replacing unsupported arXiv-preprint claims with explicit repository-preprint status. diff --git a/.changeset/amc-clawenvkit-replay-corpus.md b/.changeset/amc-clawenvkit-replay-corpus.md new file mode 100644 index 000000000..f70020dff --- /dev/null +++ b/.changeset/amc-clawenvkit-replay-corpus.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add ClawEnvKit-style environment-generation replay-corpus receipts with generated task YAML, task schema, generation prompt, fixture manifest, mock service catalog/state, audit logs, trajectories, verification/scoring/safety configs, harness tier/id, adapter and MCP/skill-shell config proof, Docker or agent-loop evidence, replay commands, deterministic seeds, service/task/check counts, component scores, final score, safety gate evidence, CI failed-row ids, Watch alerts, and public methodology/docs coverage. diff --git a/.changeset/amc-clbench-continual-learning-explainability.md b/.changeset/amc-clbench-continual-learning-explainability.md new file mode 100644 index 000000000..1f71d15de --- /dev/null +++ b/.changeset/amc-clbench-continual-learning-explainability.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add CL-Bench-style continual-learning question explainability proof for stateful workflow claims, including replayable dataset/state/mutation/conversation/entity/tool/evaluator/result hashes, metric thresholds, drilldown previews, fail-closed guards, methodology r135, docs, and legal source-boundary disclosure. diff --git a/.changeset/amc-clonemem-long-term-memory-replay.md b/.changeset/amc-clonemem-long-term-memory-replay.md new file mode 100644 index 000000000..b37abbcc5 --- /dev/null +++ b/.changeset/amc-clonemem-long-term-memory-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add CloneMem-style long-term-memory replay receipts with digital-trace, persona, question, evidence, bilingual, task-category, replay, score-delta, and fail-closed CI proof. diff --git a/.changeset/amc-cloud-reference-architectures.md b/.changeset/amc-cloud-reference-architectures.md new file mode 100644 index 000000000..373efddf4 --- /dev/null +++ b/.changeset/amc-cloud-reference-architectures.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Document AWS, GCP, and Azure self-hosted cloud reference architectures and add an OpenAPI `https://{host}/api` server template. diff --git a/.changeset/amc-codequest-quality-question-explainability.md b/.changeset/amc-codequest-quality-question-explainability.md new file mode 100644 index 000000000..38b86bbb0 --- /dev/null +++ b/.changeset/amc-codequest-quality-question-explainability.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add CodeQuest-style quality question-explainability receipts for source-backed evaluator/optimizer code-quality claims with dimension deltas, feedback/grounding coverage, replay/CI proof, and no-source-copy evidence boundaries. diff --git a/.changeset/amc-codercup-metric-validity.md b/.changeset/amc-codercup-metric-validity.md new file mode 100644 index 000000000..0efc4b19d --- /dev/null +++ b/.changeset/amc-codercup-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Adds CoderCup metric-validity receipts for continuous public coding-agent benchmark claims, binding source/license/homepage, branch snapshots, README/contributing, CI, package locks, task specs, test suites, runner contracts, score ledgers, live artifacts, methodology/reference pages, cost accounting, reliability, confidence intervals, signed evidence, artifact hashes, and row hashes. diff --git a/.changeset/amc-coding-agent-report-replay.md b/.changeset/amc-coding-agent-report-replay.md new file mode 100644 index 000000000..3d80758c6 --- /dev/null +++ b/.changeset/amc-coding-agent-report-replay.md @@ -0,0 +1,7 @@ +--- +"agent-maturity-compass": patch +--- + +Add comparative coding-agent report replay receipts for replay benchmark corpora. + +Replay rows can now bind report/source identity, source-material proof, standardized prompt hash, agent roster, scoring rubric, category scores, implementation artifacts, screenshot manifests, report artifacts, replay commands, reviewer/test evidence, agent/category coverage, recommendation use cases, normalized score thresholds, signed evidence, and row hashes. Missing comparative-report evidence fails closed through the manifest, CI receipt, Shield verification, Watch alerts, and public methodology r66. diff --git a/.changeset/amc-command-count-metadata-drift.md b/.changeset/amc-command-count-metadata-drift.md new file mode 100644 index 000000000..6ef01bf42 --- /dev/null +++ b/.changeset/amc-command-count-metadata-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Reconcile public CLI command-count and npm metadata drift by regenerating the command inventory from the compiled CLI registry, updating API/reference/pricing surfaces to 1,132 public command paths, and adding regression coverage for command-count and package keyword claims. diff --git a/.changeset/amc-community-demo-kit.md b/.changeset/amc-community-demo-kit.md new file mode 100644 index 000000000..150063901 --- /dev/null +++ b/.changeset/amc-community-demo-kit.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add a community demo kit with a GitHub-shareable terminal SVG, a concise Why AMC one-pager, and DevRel-ready demo script/copy blocks for five-minute AMC walkthroughs. diff --git a/.changeset/amc-community-registry-review-gates.md b/.changeset/amc-community-registry-review-gates.md new file mode 100644 index 000000000..757173ae9 --- /dev/null +++ b/.changeset/amc-community-registry-review-gates.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Surface community pack registry review gates before upload and document moderation rejection criteria. diff --git a/.changeset/amc-compare-badge-docs.md b/.changeset/amc-compare-badge-docs.md new file mode 100644 index 000000000..410c45aaf --- /dev/null +++ b/.changeset/amc-compare-badge-docs.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Document `amc compare --badge` in public onboarding docs and make the two-run comparison path write the comparison SVG badge instead of silently ignoring the flag. diff --git a/.changeset/amc-compliance-legal-review-appendix.md b/.changeset/amc-compliance-legal-review-appendix.md new file mode 100644 index 000000000..95d73e112 --- /dev/null +++ b/.changeset/amc-compliance-legal-review-appendix.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add an export-ready legal-review appendix with framework-specific notes to Markdown compliance reports. diff --git a/.changeset/amc-compliance-report-readability.md b/.changeset/amc-compliance-report-readability.md new file mode 100644 index 000000000..0fcf6c5a8 --- /dev/null +++ b/.changeset/amc-compliance-report-readability.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Shorten Markdown compliance report evidence refs and add config remediation guidance. diff --git a/.changeset/amc-compliance-status-hash-drilldown.md b/.changeset/amc-compliance-status-hash-drilldown.md new file mode 100644 index 000000000..c0cf2dcf5 --- /dev/null +++ b/.changeset/amc-compliance-status-hash-drilldown.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add status definitions and per-evidence JSON drill-down hints to Markdown compliance reports. diff --git a/.changeset/amc-comply-report-framework-picker.md b/.changeset/amc-comply-report-framework-picker.md new file mode 100644 index 000000000..637d48968 --- /dev/null +++ b/.changeset/amc-comply-report-framework-picker.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add an interactive framework picker for `amc comply report` while preserving non-interactive framework listing and usage output. diff --git a/.changeset/amc-comply-risk-classify-docs.md b/.changeset/amc-comply-risk-classify-docs.md new file mode 100644 index 000000000..e0797b0fb --- /dev/null +++ b/.changeset/amc-comply-risk-classify-docs.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Document and regression-test the `amc comply risk-classify` EU AI Act risk-tier command surface. diff --git a/.changeset/amc-console-api-quickstart.md b/.changeset/amc-console-api-quickstart.md new file mode 100644 index 000000000..694eb1107 --- /dev/null +++ b/.changeset/amc-console-api-quickstart.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add demo Console API Quickstart examples with auth headers, curl snippets, and response shapes. diff --git a/.changeset/amc-continual-game-learning-validity.md b/.changeset/amc-continual-game-learning-validity.md new file mode 100644 index 000000000..cde2f0570 --- /dev/null +++ b/.changeset/amc-continual-game-learning-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add typed continual-game learning metric-validity proof for TokenSpire2-style agents with required task sequence, game build, mod manifest, LLM config, prompt language, memory, conversation log, run summary, gameplay log, decision trace, outcome metric, improvement trend, fallback control, run-count, confidence-interval, signed evidence, row-hash, tests, and r45 methodology docs. diff --git a/.changeset/amc-cooperbench-metric-validity.md b/.changeset/amc-cooperbench-metric-validity.md new file mode 100644 index 000000000..cfa05cd84 --- /dev/null +++ b/.changeset/amc-cooperbench-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add CooperBench metric-validity receipts with source/no-license, release, branch, dataset/task, feature-conflict, runner, eval-backend, team-harness, agent-adapter, CI, package-lock, public-report, owner, sample-size, signed-evidence, artifact-hash, and row-hash proof. diff --git a/.changeset/amc-costnav-navigation-replay.md b/.changeset/amc-costnav-navigation-replay.md new file mode 100644 index 000000000..1d25db601 --- /dev/null +++ b/.changeset/amc-costnav-navigation-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add CostNav-style physical navigation replay proof binding for benchmark corpus rows, including source/repository/license refs, benchmark spec, scenario manifest, route graph, economic-cost model, simulator and trajectory evidence, CI receipts, route/scenario coverage, navigation success, replay pass rate, economic-cost deltas, score deltas, methodology r134 docs, and fail-closed tests. diff --git a/.changeset/amc-cpu-agentic-live-drift.md b/.changeset/amc-cpu-agentic-live-drift.md new file mode 100644 index 000000000..cb119d833 --- /dev/null +++ b/.changeset/amc-cpu-agentic-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add CPU-centric agentic workload live-drift receipts with workload/runtime/schedule context, latency, throughput, utilization, memory, bottleneck-share, evidence-coverage, row-hash, Watch alert, and public methodology bindings. diff --git a/.changeset/amc-credence-engine-live-drift.md b/.changeset/amc-credence-engine-live-drift.md new file mode 100644 index 000000000..8def11cdf --- /dev/null +++ b/.changeset/amc-credence-engine-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Credence Engine-style live-drift receipts for Score, Shield, and Watch. Rows now bind source/license/archive proof, README/SPEC/package/lock/results artifacts, experiment and benchmark harness refs, test-suite proof, posterior/VOI/expected-utility policies, baseline/live results, drift statistics, alert receipts, signed evidence, and row hashes before Bayesian decision benchmark drift claims can be used externally. diff --git a/.changeset/amc-critic-rubrics-methodology.md b/.changeset/amc-critic-rubrics-methodology.md new file mode 100644 index 000000000..3b0ff4530 --- /dev/null +++ b/.changeset/amc-critic-rubrics-methodology.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Critic Rubrics methodology assurance to the public scoring methodology so rubric-supervised critic, sparse-outcome, type-safe function-calling judge, reranking, and early-stopping claims require source snapshots, no-license boundaries, arXiv/release refs, signed evidence, and row hashes. diff --git a/.changeset/amc-ctf-agent-benchmark-live-drift.md b/.changeset/amc-ctf-agent-benchmark-live-drift.md new file mode 100644 index 000000000..810f50d68 --- /dev/null +++ b/.changeset/amc-ctf-agent-benchmark-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add FishCodeTech CTF-agent benchmark live-drift receipts with source snapshot, GPL license, challenge/Docker/MCP/sidecar/scoreboard, flag-log, solve, first-flag, contamination, independence, partial-credit, trace, sandbox, signed evidence, and row-hash proof. diff --git a/.changeset/amc-custom-adapter-guide.md b/.changeset/amc-custom-adapter-guide.md new file mode 100644 index 000000000..792a5a0f0 --- /dev/null +++ b/.changeset/amc-custom-adapter-guide.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add a custom adapter authoring guide covering declarative plugin adapters, SDK wrapper adapters, evidence semantics, and validation checks, with links from the adapter documentation. diff --git a/.changeset/amc-darwin-godel-machine-live-drift.md b/.changeset/amc-darwin-godel-machine-live-drift.md new file mode 100644 index 000000000..308a0aec0 --- /dev/null +++ b/.changeset/amc-darwin-godel-machine-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Darwin Godel Machine-style live-drift receipts for self-improving coding-agent score movement. The Watch adapter now requires source snapshot, no-license boundary, README/security/CI, controller/archive/self-modification/evaluation/scorer/sandbox/live-run/benchmark/lineage proof, baseline/live result hashes, drift statistics, alert receipts, signed evidence, and row hashes before DGM-style claims can be used externally. diff --git a/.changeset/amc-dashboard-accessibility-mode.md b/.changeset/amc-dashboard-accessibility-mode.md new file mode 100644 index 000000000..484856d52 --- /dev/null +++ b/.changeset/amc-dashboard-accessibility-mode.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add generated-dashboard high-contrast theme controls and onboarding modal focus trapping with regression coverage. diff --git a/.changeset/amc-dashboard-board-ready-trends.md b/.changeset/amc-dashboard-board-ready-trends.md new file mode 100644 index 000000000..7a3779ada --- /dev/null +++ b/.changeset/amc-dashboard-board-ready-trends.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add board-ready trend, drill-down, evidence, and next-action panels to the first-run dashboard overview. diff --git a/.changeset/amc-dashboard-heatmap-accessibility.md b/.changeset/amc-dashboard-heatmap-accessibility.md new file mode 100644 index 000000000..458e7cb85 --- /dev/null +++ b/.changeset/amc-dashboard-heatmap-accessibility.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add generated-dashboard question heatmap text alternatives, ARIA grid semantics, confidence meters, and regression coverage. diff --git a/.changeset/amc-dashboard-open-browser.md b/.changeset/amc-dashboard-open-browser.md new file mode 100644 index 000000000..d216515ca --- /dev/null +++ b/.changeset/amc-dashboard-open-browser.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Harden `amc dashboard open` browser launching with argument-based spawning and add `--no-open` for headless runs. diff --git a/.changeset/amc-dashboard-trust-topology.md b/.changeset/amc-dashboard-trust-topology.md new file mode 100644 index 000000000..5c7d79245 --- /dev/null +++ b/.changeset/amc-dashboard-trust-topology.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Embed styled trust delegation topology and review actions in dashboard builds. diff --git a/.changeset/amc-db-context-enrichment-replay.md b/.changeset/amc-db-context-enrichment-replay.md new file mode 100644 index 000000000..9bdd32f17 --- /dev/null +++ b/.changeset/amc-db-context-enrichment-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add DB context enrichment replay receipts to benchmark corpus manifests, CI receipts, methodology, and API/benchmark docs. diff --git a/.changeset/amc-dbt-warehouse-llm-eval-replay.md b/.changeset/amc-dbt-warehouse-llm-eval-replay.md new file mode 100644 index 000000000..721f43709 --- /dev/null +++ b/.changeset/amc-dbt-warehouse-llm-eval-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add warehouse-native LLM eval replay receipts with dbt/warehouse/capture/baseline/judge/drift/no-egress proof, fail-closed CI receipt IDs, Watch severity projection, methodology r101 documentation, and source-review handoff notes. diff --git a/.changeset/amc-decibench-voice-live-drift.md b/.changeset/amc-decibench-voice-live-drift.md new file mode 100644 index 000000000..ca22485f4 --- /dev/null +++ b/.changeset/amc-decibench-voice-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Decibench voice live-drift receipts with fail-closed source, license-boundary, CLI, MCP, RAG, evaluator, audio, scenario, no-transcript-copy, Watch alert, row-hash, and public methodology evidence. diff --git a/.changeset/amc-deepmath-math-agent-replay.md b/.changeset/amc-deepmath-math-agent-replay.md new file mode 100644 index 000000000..87bcb044d --- /dev/null +++ b/.changeset/amc-deepmath-math-agent-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add DeepMath-style math-agent replay receipts so sandbox, executor, GRPO, vLLM, dataset, output, metric, replay, signed-evidence, and row-hash proof fail closed. diff --git a/.changeset/amc-deepresearch-replay-corpus.md b/.changeset/amc-deepresearch-replay-corpus.md new file mode 100644 index 000000000..858c83dcb --- /dev/null +++ b/.changeset/amc-deepresearch-replay-corpus.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add DeepResearch-style progressive-search replay receipts to the replay corpus, including workflow/context/search/tool/cross-evaluation/report proof, fail-closed validation, CI receipt fields, Watch alerts, methodology docs, and source-treatment documentation. diff --git a/.changeset/amc-demo-no-vault.md b/.changeset/amc-demo-no-vault.md new file mode 100644 index 000000000..af9a71956 --- /dev/null +++ b/.changeset/amc-demo-no-vault.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc demo run --no-vault` / `--demo` so first-time users and sales engineers can run the live gateway demo without preparing or unlocking the current workspace vault. The no-vault path uses an ephemeral demo workspace and labels output `DEMO_ONLY` so it is not confused with production audit evidence. diff --git a/.changeset/amc-desktop-installers.md b/.changeset/amc-desktop-installers.md new file mode 100644 index 000000000..f73682571 --- /dev/null +++ b/.changeset/amc-desktop-installers.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add reproducible macOS, Linux, and Windows desktop installer archives generated from the local AMC npm tarball. diff --git a/.changeset/amc-diagnostic-run-aliases.md b/.changeset/amc-diagnostic-run-aliases.md new file mode 100644 index 000000000..f52cd0e1c --- /dev/null +++ b/.changeset/amc-diagnostic-run-aliases.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add agent-scoped diagnostic run aliases. Users can now name a run with `amc run-alias set `, resolve reports with `amc report `, and see saved aliases in `amc history`. diff --git a/.changeset/amc-dlp-scan-eu-pii.md b/.changeset/amc-dlp-scan-eu-pii.md new file mode 100644 index 000000000..6e9af1918 --- /dev/null +++ b/.changeset/amc-dlp-scan-eu-pii.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc dlp scan` and `amc vault dlp scan` CLI surfaces, expand DLP detection with validated IBANs, IP addresses, EU VAT IDs, keyword-bound EU national IDs, passport numbers, and health-record identifiers, and document the source-backed GDPR/IBAN basis. diff --git a/.changeset/amc-docker-quickstart-ghcr-boundary.md b/.changeset/amc-docker-quickstart-ghcr-boundary.md new file mode 100644 index 000000000..7a8c4e29a --- /dev/null +++ b/.changeset/amc-docker-quickstart-ghcr-boundary.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Replace unverified GHCR quickstart image install commands with local Docker build/run instructions and document the GHCR visibility verification boundary. diff --git a/.changeset/amc-docthinker-document-rag-replay.md b/.changeset/amc-docthinker-document-rag-replay.md new file mode 100644 index 000000000..91ca51bdc --- /dev/null +++ b/.changeset/amc-docthinker-document-rag-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add DocThinker-style document and multimodal RAG memory replay receipts with fail-closed source, carrier, router, KG, memory, observability, metric, replay, CI, and methodology evidence. diff --git a/.changeset/amc-document-dataset-live-drift.md b/.changeset/amc-document-dataset-live-drift.md new file mode 100644 index 000000000..1b19a52e0 --- /dev/null +++ b/.changeset/amc-document-dataset-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add document-to-dataset live drift receipts and Watch alerts for corpus/index/document/page/cell evidence, generated QA/Summary/RAG samples, exports, numeric integrity, quality metrics, token savings, throughput, memory, and task/source-format/export/pipeline context drift. diff --git a/.changeset/amc-domain-logistics-aliases.md b/.changeset/amc-domain-logistics-aliases.md new file mode 100644 index 000000000..8dbf0d199 --- /dev/null +++ b/.changeset/amc-domain-logistics-aliases.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Expose supply-chain and logistics domain aliases in the domain registry, CLI help/listing, persona docs, and sector-pack docs so operations users can route `supply-chain` to `environment` and `logistics`/`freight`/`3pl`/`warehouse` to `mobility`. diff --git a/.changeset/amc-dsar-cli-persistence.md b/.changeset/amc-dsar-cli-persistence.md new file mode 100644 index 000000000..53a044663 --- /dev/null +++ b/.changeset/amc-dsar-cli-persistence.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add persistent DSAR CLI workflow for `amc vault dsar submit/status/list/complete`, including file-backed request storage, subject-hashed JSONL audit events, legacy `dsar-status` counters, documentation, and focused regression tests. diff --git a/.changeset/amc-earbench-physical-risk-methodology.md b/.changeset/amc-earbench-physical-risk-methodology.md new file mode 100644 index 000000000..4506c923a --- /dev/null +++ b/.changeset/amc-earbench-physical-risk-methodology.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add EARBench-style physical-risk-awareness public methodology versioning boundaries. diff --git a/.changeset/amc-edd-rag-strategy-live-drift.md b/.changeset/amc-edd-rag-strategy-live-drift.md new file mode 100644 index 000000000..ae7c84cdc --- /dev/null +++ b/.changeset/amc-edd-rag-strategy-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add EDD-style RAG strategy proof to live score and behavior drift receipts. Rows can now bind recursive document-agent versus metadata-replacement strategy comparisons to strategy, index, query-set, reference-answer, evaluator, model, and result hashes, with fail-closed Watch alerts for missing strategy proof or strategy-mix drift. diff --git a/.changeset/amc-edge-ai-agent-replay.md b/.changeset/amc-edge-ai-agent-replay.md new file mode 100644 index 000000000..360aadc06 --- /dev/null +++ b/.changeset/amc-edge-ai-agent-replay.md @@ -0,0 +1,11 @@ +--- +"agent-maturity-compass": patch +--- + +Add edge AI agent replay proof for on-device multimodal-agent benchmark claims. + +AMC now fails closed unless edge AI agent replay rows bind source, repository, +license, device profile, runtime, optimization, dataset, task, application +scenario, replay, metric, threshold, signed-evidence, and row-hash proof before +using mobile, embedded, wearable, IoT, offline, privacy, latency, memory, energy, +accuracy, replay pass-rate, or score-delta claims as external evidence. diff --git a/.changeset/amc-effect-autoagent-replay-corpus.md b/.changeset/amc-effect-autoagent-replay-corpus.md new file mode 100644 index 000000000..990c05f7f --- /dev/null +++ b/.changeset/amc-effect-autoagent-replay-corpus.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add effect-autoagent replay-corpus receipts with fail-closed signed evidence, fixed-seed replay manifests, CI receipts, and public methodology bindings. diff --git a/.changeset/amc-embodied-agent-metric-validity.md b/.changeset/amc-embodied-agent-metric-validity.md new file mode 100644 index 000000000..8220dd881 --- /dev/null +++ b/.changeset/amc-embodied-agent-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add embodied-agent metric-validation gates that fail closed unless simulator benchmark metrics bind task-type coverage, simulator config, scene/dataset package, random/human/model baselines, action-observation trajectories, result folders, overall/per-task metric reports, metric owner, sample size, confidence intervals, and signed evidence refs. diff --git a/.changeset/amc-encourage-rag-replay.md b/.changeset/amc-encourage-rag-replay.md new file mode 100644 index 000000000..02e077235 --- /dev/null +++ b/.changeset/amc-encourage-rag-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Encourage-style modular RAG replay-corpus receipts with fail-closed source, package, method, inference-runner, template, vector DB, metric-suite, MLflow, replay, and CI evidence gates. diff --git a/.changeset/amc-enterprise-agent-interop-methodology.md b/.changeset/amc-enterprise-agent-interop-methodology.md new file mode 100644 index 000000000..16f10fa8c --- /dev/null +++ b/.changeset/amc-enterprise-agent-interop-methodology.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add an enterprise agent evaluation interop public-methodology boundary requiring dataset, test-case, agent registration, endpoint contract, evaluation-run, MCP/tool registry, tool-call trace, result metric, persistence/export, signed-evidence, and row-hash proof before public interop claims. diff --git a/.changeset/amc-eval-ai-library-question-explainability.md b/.changeset/amc-eval-ai-library-question-explainability.md new file mode 100644 index 000000000..6ac917510 --- /dev/null +++ b/.changeset/amc-eval-ai-library-question-explainability.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add eval-ai-library question-explainability receipts for question-level metric result, score breakdown, accepted/rejected evidence, repair hint, regression threshold, CI, signed evidence, and no-source-copy proof. diff --git a/.changeset/amc-eval-techniques-live-drift.md b/.changeset/amc-eval-techniques-live-drift.md new file mode 100644 index 000000000..e6229ced7 --- /dev/null +++ b/.changeset/amc-eval-techniques-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add 12-technique evaluator live-drift receipts with signed evidence coverage, metric drops, technique/context drift alerts, public methodology r56 docs, and source-use posture for FareedKhan-dev/ai-agents-eval-techniques. diff --git a/.changeset/amc-evaluator-suite-metric-validity.md b/.changeset/amc-evaluator-suite-metric-validity.md new file mode 100644 index 000000000..0839f0fe1 --- /dev/null +++ b/.changeset/amc-evaluator-suite-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Tribunal-style evaluator-suite metric validity proof with deterministic assertion, LLM judge, safety/red-team, dataset eval, custom judge, reporter, framework, threshold, owner, sample-size, and confidence-interval evidence that is row-hashed and fail-closed. diff --git a/.changeset/amc-evidence-command-references.md b/.changeset/amc-evidence-command-references.md new file mode 100644 index 000000000..38eb71cac --- /dev/null +++ b/.changeset/amc-evidence-command-references.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Fix invalid evidence-ingest command references across score output, REPL routing, dashboard actions, and chain documentation. diff --git a/.changeset/amc-evidence-first-run-capture.md b/.changeset/amc-evidence-first-run-capture.md new file mode 100644 index 000000000..121a26f8b --- /dev/null +++ b/.changeset/amc-evidence-first-run-capture.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add a one-command first-run evidence capture path with dry-run preview and unsigned starter capture. diff --git a/.changeset/amc-evidra-provider-drift.md b/.changeset/amc-evidra-provider-drift.md new file mode 100644 index 000000000..f9979db44 --- /dev/null +++ b/.changeset/amc-evidra-provider-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Evidra provider-drift receipts with source/protocol/evidence-chain proof, fail-closed Watch/API/CI alerts, public methodology r208 binding, and docs for the live verified source boundary. diff --git a/.changeset/amc-executive-board-path.md b/.changeset/amc-executive-board-path.md new file mode 100644 index 000000000..1a85dd635 --- /dev/null +++ b/.changeset/amc-executive-board-path.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add a dedicated executive board-brief website page, link it from the homepage, refresh executive overview counts/pricing, and add regression coverage for the board-facing path and stale executive claims. diff --git a/.changeset/amc-executive-brief.md b/.changeset/amc-executive-brief.md new file mode 100644 index 000000000..29a6f9f8c --- /dev/null +++ b/.changeset/amc-executive-brief.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc executive brief` to generate a print-ready board one-pager from a diagnostic run without requiring certificate signing. diff --git a/.changeset/amc-exploitgym-metric-validity.md b/.changeset/amc-exploitgym-metric-validity.md new file mode 100644 index 000000000..925d8b8ff --- /dev/null +++ b/.changeset/amc-exploitgym-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add ExploitGym-style exploit-development metric-validity proof to pentest benchmark checks, requiring source/license, release, task, target-image, controller, firewall, proxy, execution, success-metric, owner, confidence-interval, signed evidence, and row-hash receipts before security-agent benchmark claims pass. diff --git a/.changeset/amc-external-source-verification-policy.md b/.changeset/amc-external-source-verification-policy.md new file mode 100644 index 000000000..71b781551 --- /dev/null +++ b/.changeset/amc-external-source-verification-policy.md @@ -0,0 +1,7 @@ +--- +"agent-maturity-compass": patch +--- + +Add a public external-source verification policy to the scoring methodology. + +The methodology manifest now states that live or primary-source verification is required for external claims, repository metadata and cached snippets are rejected as parity proof, unavailable sources must be disclosed, and third-party code, commands, prompts, datasets, examples, UI text/assets, README prose, screenshots, benchmark rows, configuration, and implementation details must not be copied without separate license review. Public methodology r70 renders the policy and includes migration guidance for stale or unavailable source-backed claims. diff --git a/.changeset/amc-falcon-evaluate-provider-drift.md b/.changeset/amc-falcon-evaluate-provider-drift.md new file mode 100644 index 000000000..76fca46d9 --- /dev/null +++ b/.changeset/amc-falcon-evaluate-provider-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Falcon Evaluate provider-drift receipts with fail-closed source, metric-family, provider-route, canary-result, Watch alert, eval-pack row-hash, and public methodology evidence. diff --git a/.changeset/amc-fire-fact-checking-replay.md b/.changeset/amc-fire-fact-checking-replay.md new file mode 100644 index 000000000..4111ea98d --- /dev/null +++ b/.changeset/amc-fire-fact-checking-replay.md @@ -0,0 +1,6 @@ +--- +"agent-maturity-compass": patch +--- + +Add FIRE-style atomic-claim fact-checking replay receipts with fail-closed source, +paper, dataset, retrieval, verification, cost, replay, CI, and threshold proof. diff --git a/.changeset/amc-fleet-overview.md b/.changeset/amc-fleet-overview.md new file mode 100644 index 000000000..fc6252319 --- /dev/null +++ b/.changeset/amc-fleet-overview.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc fleet overview` for executive fleet verdict, coverage, drift, weakest agents, and next actions. diff --git a/.changeset/amc-fleet-trust-graph.md b/.changeset/amc-fleet-trust-graph.md new file mode 100644 index 000000000..34e60f728 --- /dev/null +++ b/.changeset/amc-fleet-trust-graph.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc fleet trust-graph` with Mermaid, DOT, and JSON visualization of delegation trust edges. diff --git a/.changeset/amc-fleet-trust-report-nosign.md b/.changeset/amc-fleet-trust-report-nosign.md new file mode 100644 index 000000000..2405c1f7b --- /dev/null +++ b/.changeset/amc-fleet-trust-report-nosign.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Make `amc fleet trust-report --no-sign` generate an explicit unsigned trust composition report when setup or signing prerequisites are unavailable, and align the UX audit text with the current behavior. diff --git a/.changeset/amc-fore-public-methodology-versioning.md b/.changeset/amc-fore-public-methodology-versioning.md new file mode 100644 index 000000000..438a6a165 --- /dev/null +++ b/.changeset/amc-fore-public-methodology-versioning.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add fore public methodology versioning receipts to the public methodology manifest, docs, API surface, and benchmark source-boundary notes. diff --git a/.changeset/amc-freshstack-retrieval-replay.md b/.changeset/amc-freshstack-retrieval-replay.md new file mode 100644 index 000000000..94d430b80 --- /dev/null +++ b/.changeset/amc-freshstack-retrieval-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add FreshStack-style IR/RAG retrieval replay receipts so repository, paper, query/corpus datasets, StackOverflow query and GitHub corpus manifests, licenses, BEIR/qrels, chunking, retriever, index, runfile, evaluator, metrics, leaderboard, replay command, alpha-nDCG, coverage, recall, signed evidence, and row hashes fail closed. diff --git a/.changeset/amc-gage-unified-evaluation-replay.md b/.changeset/amc-gage-unified-evaluation-replay.md new file mode 100644 index 000000000..ef16a2ecb --- /dev/null +++ b/.changeset/amc-gage-unified-evaluation-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add GAGE-style unified evaluation replay receipts with engine/run identity, modality, harness mode, config/registry/dataset/backend/adapter/metric/output contracts, event/sample/summary/artifact/output-dir/environment/dependency/replay hashes, deterministic seed, sample/metric/artifact counts, replay coverage, score thresholds, CI failed-row reporting, fixture-hash binding, and r43 methodology docs. diff --git a/.changeset/amc-gaia-agent-replay-corpus.md b/.changeset/amc-gaia-agent-replay-corpus.md new file mode 100644 index 000000000..22d25bd7e --- /dev/null +++ b/.changeset/amc-gaia-agent-replay-corpus.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add GAIA-agent replay-corpus receipts with source-linked benchmark harness proof, fixed seeds, tool-trace coverage, score-delta summaries, CI receipt fields, and public methodology/docs bindings. diff --git a/.changeset/amc-garage-live-drift.md b/.changeset/amc-garage-live-drift.md new file mode 100644 index 000000000..63aad96d0 --- /dev/null +++ b/.changeset/amc-garage-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add GaRAGe RAG-grounding live-drift receipts with fail-closed source, dataset, annotation-schema, baseline/live, alert, signed-evidence, and row-hash proof. diff --git a/.changeset/amc-gdpr-art5-accountability.md b/.changeset/amc-gdpr-art5-accountability.md new file mode 100644 index 000000000..e01dc9c2b --- /dev/null +++ b/.changeset/amc-gdpr-art5-accountability.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add the GDPR Article 5(2) accountability mapping to built-in compliance mappings and cross-framework GDPR coverage. diff --git a/.changeset/amc-geospatial-provider-drift.md b/.changeset/amc-geospatial-provider-drift.md new file mode 100644 index 000000000..bf21bb14e --- /dev/null +++ b/.changeset/amc-geospatial-provider-drift.md @@ -0,0 +1,7 @@ +--- +"agent-maturity-compass": patch +--- + +Add GeoBenchX-style geospatial tool-calling proof to provider/model drift canaries. + +Provider-drift rows can now bind geospatial benchmark ids, task-set hashes, dataset snapshots, tool registries, reference solutions, trace exports, judge panels/configs, human calibration, result reports, token-cost reports, complexity groups, solvable/unsolvable task counts, tool counts, and max tool iterations. Missing geospatial proof emits a fail-closed Watch/API alert and is included in eval-pack row hashes, CI/lifecycle gates, public methodology, and docs. diff --git a/.changeset/amc-graph-eval-judge-calibration.md b/.changeset/amc-graph-eval-judge-calibration.md new file mode 100644 index 000000000..18c6cd1de --- /dev/null +++ b/.changeset/amc-graph-eval-judge-calibration.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add graph-eval judge calibration proof receipts with node-graph, metric-branch, cache, report, model-routing, parser, cost-estimate, signed-evidence, Watch alert, and Shield verification gates. diff --git a/.changeset/amc-gto-wizard-poker-replay.md b/.changeset/amc-gto-wizard-poker-replay.md new file mode 100644 index 000000000..f0786ff32 --- /dev/null +++ b/.changeset/amc-gto-wizard-poker-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add GTO Wizard-style poker-agent replay corpus proof with API-scope, no-solver policy, hand-history, legal-action trace, AIVAT metric, replay command, CI receipt, summary, and fail-closed evidence gates. diff --git a/.changeset/amc-guardbench-guardrail-metric-validity.md b/.changeset/amc-guardbench-guardrail-metric-validity.md new file mode 100644 index 000000000..77aeeae03 --- /dev/null +++ b/.changeset/amc-guardbench-guardrail-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add GuardBench-style guardrail metric-validity gates so dataset/access/format proof, moderation contracts, guardrail model and threshold configs, prediction scores, metric-suite and confusion-matrix reports, language coverage, export reports, owners, confidence intervals, signed evidence, and row hashes fail closed. diff --git a/.changeset/amc-hedrarag-live-drift.md b/.changeset/amc-hedrarag-live-drift.md new file mode 100644 index 000000000..5b3c7a7ca --- /dev/null +++ b/.changeset/amc-hedrarag-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add fail-closed HedraRAG artifact-eval live-drift proof with latency, throughput, memory, replay, evidence coverage, workflow/framework/runtime distributions, public methodology r138 docs, and source-safe documentation for Leo9660/HedraRAG_AE. diff --git a/.changeset/amc-hermes-bench-metric-validity.md b/.changeset/amc-hermes-bench-metric-validity.md new file mode 100644 index 000000000..14cf68558 --- /dev/null +++ b/.changeset/amc-hermes-bench-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Hermes Bench-style local LLM/agent benchmark metric-validity receipts across Score, Shield, Watch, public methodology, and docs. AMC now fails closed unless Hermes Bench claims bind source/license refs, default-branch snapshots, README/build specs, backend runners, judge calibration, task registries, model/server configs, adapter coverage, result schemas, frontend review surfaces, backend/frontend regressions, Docker runtimes, owners, confidence intervals, signed evidence, artifact hashes, and row hashes. diff --git a/.changeset/amc-hermes-turbo-question-explainability.md b/.changeset/amc-hermes-turbo-question-explainability.md new file mode 100644 index 000000000..f688e0182 --- /dev/null +++ b/.changeset/amc-hermes-turbo-question-explainability.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add Hermes Turbo-style question-explainability receipts for source-backed performance dashboard claims. Score, Shield, Watch, and docs now require source/license refs, live default-branch commit/tree, benchmark/perf-budget/daily-score workflows, turbo-score script, dashboard, benchmark report, baseline/candidate results, latency/throughput traces, score manifest, CI, thresholds, accepted/rejected evidence, repair hints, and row hashes before low-latency or turbo-score claims can pass. diff --git a/.changeset/amc-heurekabench-scientific-replay.md b/.changeset/amc-heurekabench-scientific-replay.md new file mode 100644 index 000000000..0ef9dbc45 --- /dev/null +++ b/.changeset/amc-heurekabench-scientific-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add HeurekaBench scientific co-scientist replay receipts to the public methodology manifest, API docs, benchmark boundaries, and source-review notes. diff --git a/.changeset/amc-humanstudybench-metric-validity.md b/.changeset/amc-humanstudybench-metric-validity.md new file mode 100644 index 000000000..533e8b129 --- /dev/null +++ b/.changeset/amc-humanstudybench-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add HumanStudy-Bench-style participant-simulation metric-validity receipts with fail-closed source, study, response, evaluator, reliability, validation-pipeline, CI, owner, confidence-interval, signed-evidence, and row-hash proof. diff --git a/.changeset/amc-improve-archetype-l3-examples.md b/.changeset/amc-improve-archetype-l3-examples.md new file mode 100644 index 000000000..69ff81c2a --- /dev/null +++ b/.changeset/amc-improve-archetype-l3-examples.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add product-specific L3 examples for common agent archetypes in `amc improve`. diff --git a/.changeset/amc-improve-product-language.md b/.changeset/amc-improve-product-language.md new file mode 100644 index 000000000..cfe9cf870 --- /dev/null +++ b/.changeset/amc-improve-product-language.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Explain L3 in product language in `amc improve` and add product outcomes to roadmap items. diff --git a/.changeset/amc-industry-adjust-current-score.md b/.changeset/amc-industry-adjust-current-score.md new file mode 100644 index 000000000..d1d8775a7 --- /dev/null +++ b/.changeset/amc-industry-adjust-current-score.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Let `amc score industry-adjust` reuse the latest scored run when `--score` is omitted. diff --git a/.changeset/amc-industry-adjust-drilldown.md b/.changeset/amc-industry-adjust-drilldown.md new file mode 100644 index 000000000..2b5b44ee0 --- /dev/null +++ b/.changeset/amc-industry-adjust-drilldown.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc score industry-adjust --drilldown` for per-dimension industry weighting inspection, with matching `--json --drilldown` rows for automation. diff --git a/.changeset/amc-industry-adjust-explanation.md b/.changeset/amc-industry-adjust-explanation.md new file mode 100644 index 000000000..b86be949b --- /dev/null +++ b/.changeset/amc-industry-adjust-explanation.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Explain raw versus adjusted score differences in `score industry-adjust` output. diff --git a/.changeset/amc-industry-adjust-history-report.md b/.changeset/amc-industry-adjust-history-report.md new file mode 100644 index 000000000..0da88e3bf --- /dev/null +++ b/.changeset/amc-industry-adjust-history-report.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add exportable industry-adjusted history reports across scored runs. diff --git a/.changeset/amc-inference-optimization-metric-validity.md b/.changeset/amc-inference-optimization-metric-validity.md new file mode 100644 index 000000000..642df9edf --- /dev/null +++ b/.changeset/amc-inference-optimization-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add InferenceBench-style inference optimization metric-validity proof. AMC now fails closed unless benchmark-backed inference-serving optimization claims bind scenario objectives, hardware budgets, server contracts, runtime backends, search spaces, baseline comparisons, quality/integrity gates, supervised relaunches, latency/throughput/tail metrics, exploration traces, owners, confidence intervals, signed evidence, and row hashes. diff --git a/.changeset/amc-innovatorbench-research-replay.md b/.changeset/amc-innovatorbench-research-replay.md new file mode 100644 index 000000000..7198dcfeb --- /dev/null +++ b/.changeset/amc-innovatorbench-research-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add InnovatorBench-style research replay receipts for source-backed LLM research-agent benchmarks with ResearchGym, tool, environment, checkpoint, score, replay, CI, and no-shortcut evidence boundaries. diff --git a/.changeset/amc-iot-firmware-question-explainability.md b/.changeset/amc-iot-firmware-question-explainability.md new file mode 100644 index 000000000..65ad1f0d4 --- /dev/null +++ b/.changeset/amc-iot-firmware-question-explainability.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Adsum IoT Coder-style IoT firmware question-explainability proof so Score, Shield, Watch, and API drilldowns fail closed unless firmware task claims bind platform, board/chip, hardware session, device logs, build/flash/test artifacts, privacy boundary, benchmark report, metric thresholds, accepted evidence, rejected reasons, repair hints, and row hashes. diff --git a/.changeset/amc-judge-calibration.md b/.changeset/amc-judge-calibration.md new file mode 100644 index 000000000..8291c9204 --- /dev/null +++ b/.changeset/amc-judge-calibration.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add judge calibration receipts with rubric hashes, calibration set hashes, disagreement/error/variance metrics, appeal outcomes, stability-aware checkpoint ranking checks, signed evidence, Watch alerts, Shield verification, and CI/lifecycle fail-closed gates. diff --git a/.changeset/amc-judgeit-llm-judge-replay.md b/.changeset/amc-judgeit-llm-judge-replay.md new file mode 100644 index 000000000..f0802f59d --- /dev/null +++ b/.changeset/amc-judgeit-llm-judge-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add JudgeIt-style LLM-as-judge replay receipts so dataset/golden/generated manifests, pipeline and judge configs, human-eval references, batch/export/metric/replay artifacts, precision/recall/F1, false-negative, blackbox/whitebox/negative-test metrics, signed evidence, and row hashes fail closed. diff --git a/.changeset/amc-k8s-ai-metric-validity.md b/.changeset/amc-k8s-ai-metric-validity.md new file mode 100644 index 000000000..c256925fd --- /dev/null +++ b/.changeset/amc-k8s-ai-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Kubernetes operational-agent metric-validity receipts with source/license, release, build workflow, agent module, MCP server, tool inventory, diagnostics, resource/log-analysis, CI, owner, confidence interval, artifact hash, and row-hash proof. diff --git a/.changeset/amc-kite-rag-live-drift.md b/.changeset/amc-kite-rag-live-drift.md new file mode 100644 index 000000000..b93d3ea75 --- /dev/null +++ b/.changeset/amc-kite-rag-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add KITE-style end-to-end RAG live-drift evidence fields, coverage gates, alerts, and methodology docs so corpus/query/rubric/judge/grade drift fails closed with signed receipts. diff --git a/.changeset/amc-knowlytics-ai-replay-corpus.md b/.changeset/amc-knowlytics-ai-replay-corpus.md new file mode 100644 index 000000000..2f8a9711b --- /dev/null +++ b/.changeset/amc-knowlytics-ai-replay-corpus.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Knowlytics-AI-style MCQ/RAG replay-corpus receipts with source/no-license, owned fixture, trace, feedback, CI, and row-hash proof. diff --git a/.changeset/amc-kubernetes-helm-deployment-guide.md b/.changeset/amc-kubernetes-helm-deployment-guide.md new file mode 100644 index 000000000..90ee2ec33 --- /dev/null +++ b/.changeset/amc-kubernetes-helm-deployment-guide.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add a Kubernetes/Helm deployment guide and Terraform Helm-release example covering bootstrap secrets, probes, rollback, and raw Kustomize manifests. diff --git a/.changeset/amc-leaf-scenario-simulation-replay.md b/.changeset/amc-leaf-scenario-simulation-replay.md new file mode 100644 index 000000000..b047032c0 --- /dev/null +++ b/.changeset/amc-leaf-scenario-simulation-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add scenario-simulation action-level replay receipts for benchmark replay corpus rows, including scenario project, action trace, evaluation report, visualization, persistence, checkpoint resume, summary, CI receipt, methodology, and docs coverage. diff --git a/.changeset/amc-legacybench-metric-validity.md b/.changeset/amc-legacybench-metric-validity.md new file mode 100644 index 000000000..6aa7d44af --- /dev/null +++ b/.changeset/amc-legacybench-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Legacy-Bench-style legacy-software metric-validity receipts so source/license, default-branch snapshots, README and task-corpus manifests, legacy-language coverage, environments, harness runners, agent tasks, patch submissions, test oracles, evaluator registries, CI, result artifacts, replay commands, owners, confidence intervals, signed evidence, and row hashes fail closed. diff --git a/.changeset/amc-legal-agent-live-drift.md b/.changeset/amc-legal-agent-live-drift.md new file mode 100644 index 000000000..fa20c1f8b --- /dev/null +++ b/.changeset/amc-legal-agent-live-drift.md @@ -0,0 +1,7 @@ +--- +"agent-maturity-compass": patch +--- + +Add LegalAgentBench-style legal-agent live drift receipts with final-success, +process-rate, tool-use, citation, evidence, token-cost, corpus, task-type, +difficulty, tool-context, signed-evidence, and row-hash fail-closed coverage. diff --git a/.changeset/amc-legal-code-rag-metric-validity.md b/.changeset/amc-legal-code-rag-metric-validity.md new file mode 100644 index 000000000..98268eb9a --- /dev/null +++ b/.changeset/amc-legal-code-rag-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Legal Code RAG metric-validity proof fields, fail-closed gates, methodology r110 docs, and tests for French legal-code RAG evidence. diff --git a/.changeset/amc-live-drift-alerts.md b/.changeset/amc-live-drift-alerts.md new file mode 100644 index 000000000..771069f60 --- /dev/null +++ b/.changeset/amc-live-drift-alerts.md @@ -0,0 +1,23 @@ +--- +"agent-maturity-compass": minor +--- + +Add live score, behavior, and optional data-science lifecycle-stage drift receipts with Score, Watch, and Shield API surfaces. + +Add persona-policy drift fields and alerts so live receipts can detect when realistic task-preserving persona populations collapse to cooperative defaults despite stable score and behavior signatures. + +Add live CTF evaluation controls for event/challenge context, flag acceptance, first-correct-flag forwarding, external-search contamination risk, competition-impact risk, and per-agent independence violations. + +Add partial-credit CTF drift controls for VM/sandbox/rubric context, execution trace coverage, checkpoint completion, partial-credit score, and isolation violations. + +Add survey-backed agent-evaluation dimension drift controls for planning, tool-use, self-reflection, memory, application-domain, generalist, framework, trend, and emergent-direction coverage. + +Add OmniEval-style RAG live drift controls for corpus/chunking context, retriever/generator/framework identity, retrieval top-k, generated-data finalization, model/rule/close-book evaluator mode, hallucination evaluator coverage, and RAG accuracy/completeness/utilization/numerical-accuracy/hallucination metrics. + +Add calculator-agent-style tool-use RL live drift controls for reward, answer verification, judge agreement, tool-call validity, rollout diversity, eval-improvement delta, and model/dataset/reward/verifier/environment/rollout/judge context drift. + +Add paper-trading agent live drift controls for win rate, risk/reward, drawdown, realized PnL, risk-limit violations, claim-validation failures, chart-vision agreement, memory retrieval, provider fallback, and market/strategy/risk/provider/memory/chart/indicator/validation/news/ledger context drift. + +Add GenoTEX-style genomics live drift controls for dataset-selection, preprocessing, and statistical-analysis task stages; reference/prediction/metadata hashes; expert-curation hashes; format conformance; stage metrics; and genomics stage/context drift. + +Add ALERT-style safety red-team live drift controls for unsafe-response rate, compliance, guard score, dataset/taxonomy/attack/guard evidence coverage, and risk-category/attack/subset/guard-label distribution drift. diff --git a/.changeset/amc-llm-evaluation-system-jury-replay.md b/.changeset/amc-llm-evaluation-system-jury-replay.md new file mode 100644 index 000000000..fd26940da --- /dev/null +++ b/.changeset/amc-llm-evaluation-system-jury-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Adds LLM Evaluation System-style jury replay receipts to the replay benchmark corpus, public methodology, benchmark docs, and API docs. AMC now fails closed unless source/repository/license/package/MCP, dataset generation, synthetic QA, document grounding, judge config, jury roster, binary scoring, execution/OpenTelemetry/Bedrock, result/analysis/PDF/S3, replay/CI, no-copy, no-config-only, no-report-only, threshold, signed-evidence, and row-hash proof are present. diff --git a/.changeset/amc-llm-fighter-live-drift.md b/.changeset/amc-llm-fighter-live-drift.md new file mode 100644 index 000000000..f9227767c --- /dev/null +++ b/.changeset/amc-llm-fighter-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add LLM Fighter live score and behavior drift receipts with signed evidence coverage, Watch alert metric ids, Shield-verifiable receipt hashing, methodology/docs boundaries, and fail-closed tests. diff --git a/.changeset/amc-llm-prompting-tests-public-methodology.md b/.changeset/amc-llm-prompting-tests-public-methodology.md new file mode 100644 index 000000000..efc504c82 --- /dev/null +++ b/.changeset/amc-llm-prompting-tests-public-methodology.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add llm-prompting-tests public methodology receipts with source/no-license/default-branch, README, prompt-catalog, prompt-taxonomy, rubric, self-check/no-external-assets, judge-calibration, regression-threshold, changelog, deprecation, migration, no-copy, signed-evidence, and row-hash requirements. diff --git a/.changeset/amc-llm-rag-eval-suite-live-drift.md b/.changeset/amc-llm-rag-eval-suite-live-drift.md new file mode 100644 index 000000000..71e48ab4e --- /dev/null +++ b/.changeset/amc-llm-rag-eval-suite-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add AIAnytime-style LLM/RAG multi-metric live-drift proof with signed eval-suite evidence, semantic similarity, bias risk, hallucination/faithfulness drift, fail-closed coverage/context alerts, public methodology docs, and tests. diff --git a/.changeset/amc-llmops-lifecycle-methodology.md b/.changeset/amc-llmops-lifecycle-methodology.md new file mode 100644 index 000000000..6f1746776 --- /dev/null +++ b/.changeset/amc-llmops-lifecycle-methodology.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add LLMOPS-style lifecycle public methodology versioning boundaries. diff --git a/.changeset/amc-local-system-live-drift.md b/.changeset/amc-local-system-live-drift.md new file mode 100644 index 000000000..784e3bffa --- /dev/null +++ b/.changeset/amc-local-system-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add local-system monitor live-drift receipts with monitor/device/hardware/process/sensor/alert proof, workload and hardware context drift, thermal-baseline deviation, voltage SPC anomalies, process identity, ghost-driver handling, proactive alert, local-only privacy, signed evidence, and fail-closed Watch alerts. diff --git a/.changeset/amc-logistics-sector-pack.md b/.changeset/amc-logistics-sector-pack.md new file mode 100644 index 000000000..1ba755fc1 --- /dev/null +++ b/.changeset/amc-logistics-sector-pack.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add the `freight-3pl-warehouse` industry sector pack under Mobility and expose logistics-contextual operational reliability scoring with `amc score operational-independence --domain logistics`. diff --git a/.changeset/amc-m2rag-multimodal-methodology.md b/.changeset/amc-m2rag-multimodal-methodology.md new file mode 100644 index 000000000..6a15589cf --- /dev/null +++ b/.changeset/amc-m2rag-multimodal-methodology.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add an M2RAG-style multimodal RAG methodology-versioning boundary for mixed text/image score claims. diff --git a/.changeset/amc-mcp-security-bench-nrp.md b/.changeset/amc-mcp-security-bench-nrp.md new file mode 100644 index 000000000..87f253ca3 --- /dev/null +++ b/.changeset/amc-mcp-security-bench-nrp.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add MCP Security Bench security-resilience scoring and Net Resilient Performance to MCP compliance and resilience analysis. diff --git a/.changeset/amc-medask-clinical-benchmark-replay.md b/.changeset/amc-medask-clinical-benchmark-replay.md new file mode 100644 index 000000000..0c8aacc36 --- /dev/null +++ b/.changeset/amc-medask-clinical-benchmark-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add MedAsk-style clinical benchmark replay receipts for SymptomCheck diagnostic and Triage urgency-classification evidence. diff --git a/.changeset/amc-memeval-memory-system-replay.md b/.changeset/amc-memeval-memory-system-replay.md new file mode 100644 index 000000000..3d3c9ec8d --- /dev/null +++ b/.changeset/amc-memeval-memory-system-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add MemEval-style long-term-memory replay receipts with fail-closed evidence checks for benchmark manifests, memory-system rosters, scoring configs, token-cost traces, metric coverage, replay pass rates, and CI receipt row IDs. diff --git a/.changeset/amc-methodology-reproducibility-packet.md b/.changeset/amc-methodology-reproducibility-packet.md new file mode 100644 index 000000000..19b5ad352 --- /dev/null +++ b/.changeset/amc-methodology-reproducibility-packet.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc methodology --reproducibility` to export a public methodology packet with question-bank hashes, full question metadata, formulas, source paths, commands, and limitations. diff --git a/.changeset/amc-methodology-sample-case-study-dataset.md b/.changeset/amc-methodology-sample-case-study-dataset.md new file mode 100644 index 000000000..369f07901 --- /dev/null +++ b/.changeset/amc-methodology-sample-case-study-dataset.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc methodology --sample-dataset` to export a source-generated public synthetic L0-L5 sample case-study dataset with dataset-card metadata, methodology hashes, evidence profiles, layer scores, privacy notes, and explicit non-empirical-validation boundaries. diff --git a/.changeset/amc-metric-validation-eval-pack.md b/.changeset/amc-metric-validation-eval-pack.md new file mode 100644 index 000000000..2bd6d5056 --- /dev/null +++ b/.changeset/amc-metric-validation-eval-pack.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add metric-validation eval-pack and CI/lifecycle gate artifacts with deterministic row hashes, signed evidence refs, replayable dataset hashes, and fail-closed metric IDs. diff --git a/.changeset/amc-metric-validity.md b/.changeset/amc-metric-validity.md new file mode 100644 index 000000000..eab7843a7 --- /dev/null +++ b/.changeset/amc-metric-validity.md @@ -0,0 +1,15 @@ +--- +"agent-maturity-compass": minor +--- + +Add diagnostic metric validity and reliability tables with construct validity, judge agreement, test-retest stability, confidence intervals, metric owners, lifecycle-aware coverage gates, ranking-stability coverage gates, dynamic tool-sandbox coverage gates, continual-learning coverage gates, strategic-interaction coverage gates, RAG-pipeline coverage gates, business-workflow coverage gates, and fail-closed status. + +Add a data-agent analytical coverage gate so heterogeneous analytical-query benchmark metrics fail closed without task-type, database/source-modality, difficulty, metric-computation, agent-workflow, expert-validation, cost/latency, and submission-schema evidence. + +Clarify RAG-pipeline coverage gates so custom-domain RAG metrics require document/test sets, solution roster/configs, selected metrics, query-level result records, metric-computation traces, report/export artifacts, and performance/cost evidence before they can support external claims. + +Clarify domain-specific legal RAG metric-validity evidence so corpus provenance, jurisdiction/language/task coverage, retriever/reranker configs, model and judge configs, logged samples, and agent-framework evidence are bound before legal RAG benchmark claims are accepted. + +Add an opt-in RAG evaluation-pipeline proof gate so Semantic Kernel-style RAG evaluation metrics fail closed without signed ground-truth question/answer sets, pipeline config, metric definitions, query/retrieval/generation traces, evaluation report, metric owner, sample size, confidence interval, and row-hash-bound eval-pack evidence. + +Add an opt-in architecture-reality proof gate so agent architecture metrics fail closed without signed wrapper-agent, marketing-agent, real-agent, planning, memory, recovery, stress, network, cost, ensemble, statistical-confidence, and row-hash-bound eval-pack evidence. diff --git a/.changeset/amc-metronous-methodology-versioning.md b/.changeset/amc-metronous-methodology-versioning.md new file mode 100644 index 000000000..c4cab50c9 --- /dev/null +++ b/.changeset/amc-metronous-methodology-versioning.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Metronous-style methodology-versioning assurance receipts for report and badge comparability. diff --git a/.changeset/amc-miniappbench-interactive-html-replay.md b/.changeset/amc-miniappbench-interactive-html-replay.md new file mode 100644 index 000000000..1462cf7f9 --- /dev/null +++ b/.changeset/amc-miniappbench-interactive-html-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add MiniAppBench-style interactive HTML replay receipts for replay-corpus runs, including browser-automation traces, generated MiniApp/source-code proof, live-instance evidence, withheld-reference and no-copy boundaries, MiniAppBench summary counters, CI receipt fields, and public methodology documentation. diff --git a/.changeset/amc-minimal-startup-path.md b/.changeset/amc-minimal-startup-path.md new file mode 100644 index 000000000..62922828d --- /dev/null +++ b/.changeset/amc-minimal-startup-path.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc init --minimal` and `amc quickstart --minimal` for startup-friendly workspace setup without vault prompting or immediate full-score prompting. diff --git a/.changeset/amc-mirage-drug-repositioning-validity.md b/.changeset/amc-mirage-drug-repositioning-validity.md new file mode 100644 index 000000000..05d0abc13 --- /dev/null +++ b/.changeset/amc-mirage-drug-repositioning-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add ARIASHA/MiRAGE-style drug-repositioning metric-validity proof. AMC metric-validation rows can now fail closed unless signed evidence binds dataset release, train/test split, drug-disease mapping, drug and disease features, similarity matrices, negative sampling, classifier config, feature selection, score calculation, evaluation report, case-study validation, owner, sample size, confidence interval, and row hashes. diff --git a/.changeset/amc-mirage-multimodal-rag-replay.md b/.changeset/amc-mirage-multimodal-rag-replay.md new file mode 100644 index 000000000..216b0e829 --- /dev/null +++ b/.changeset/amc-mirage-multimodal-rag-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add MiRAGE-style multimodal multihop RAG dataset-generation replay proof to replay benchmark corpus receipts, summaries, CI gates, Watch alerts, and docs. diff --git a/.changeset/amc-mirage-rag-metric-validity.md b/.changeset/amc-mirage-rag-metric-validity.md new file mode 100644 index 000000000..14608a1f4 --- /dev/null +++ b/.changeset/amc-mirage-rag-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add MIRAGE-style RAG metric-validity gates so benchmark identity, dataset, QA/context/retrieval pools, base/oracle/mixed protocol, retriever/model configs, LLM/retriever/MIRAGE metric reports, score formula, owner, sample-size, confidence-interval, signed-evidence, and row-hash proof fail closed. diff --git a/.changeset/amc-ml-dev-benchmark-replay.md b/.changeset/amc-ml-dev-benchmark-replay.md new file mode 100644 index 000000000..fa1e3f87d --- /dev/null +++ b/.changeset/amc-ml-dev-benchmark-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add ML-development workflow replay receipts to the replay benchmark corpus, including typed task/category/domain/harness evidence, deterministic replay artifacts, fail-closed metric thresholds, CI receipt fields, Watch alerts, methodology r58 docs, and source-use posture for ml-dev-bench/ml-dev-bench. diff --git a/.changeset/amc-mobile-agent-metric-validity.md b/.changeset/amc-mobile-agent-metric-validity.md new file mode 100644 index 000000000..d4ce93548 --- /dev/null +++ b/.changeset/amc-mobile-agent-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add MobileBench-style mobile-agent metric-validity gates requiring signed environment, app inventory, API catalog, UI trace, task dataset, task complexity, multi-app task, checkpoint, reset/device-state, license-boundary, owner, sample-size, confidence-interval, and row-hash proof before mobile-agent benchmark claims can be used externally. diff --git a/.changeset/amc-mobile-bridge-fetch.md b/.changeset/amc-mobile-bridge-fetch.md new file mode 100644 index 000000000..60219fdaf --- /dev/null +++ b/.changeset/amc-mobile-bridge-fetch.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add a mobile-safe AMC Bridge fetch wrapper for React Native-style integrations and document the mobile Bridge path for React Native and Flutter. The wrapper avoids Node-only imports, strips provider auth by default, injects AMC correlation headers, and keeps self-scoring guards active before mobile requests leave the app. diff --git a/.changeset/amc-multi-user-question-explainability.md b/.changeset/amc-multi-user-question-explainability.md new file mode 100644 index 000000000..c0b833783 --- /dev/null +++ b/.changeset/amc-multi-user-question-explainability.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Multi-User-LLM-Agent-style question explainability lenses with scenario, role, policy, trace, evaluator, metric-threshold, accepted/rejected evidence, repair-hint, and row-hash proof. diff --git a/.changeset/amc-navi-bench-web-agent-live-drift.md b/.changeset/amc-navi-bench-web-agent-live-drift.md new file mode 100644 index 000000000..23f003b53 --- /dev/null +++ b/.changeset/amc-navi-bench-web-agent-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Navi-Bench-style real-website web-agent live-drift receipts with source, dataset, task-config, evaluator, browser-provider, trajectory, visualization, crash-adjusted score, and evidence coverage gates. diff --git a/.changeset/amc-nika-network-troubleshooting-validity.md b/.changeset/amc-nika-network-troubleshooting-validity.md new file mode 100644 index 000000000..631ecfa50 --- /dev/null +++ b/.changeset/amc-nika-network-troubleshooting-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add NIKA-style network troubleshooting metric-validity gates with typed proof coverage, signed eval-pack fields, fail-closed warnings, public methodology r90 documentation, and API surface documentation. diff --git a/.changeset/amc-nomiracl-multilingual-rag-live-drift.md b/.changeset/amc-nomiracl-multilingual-rag-live-drift.md new file mode 100644 index 000000000..1eb6450c9 --- /dev/null +++ b/.changeset/amc-nomiracl-multilingual-rag-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add NoMIRACL-style multilingual RAG live-drift evidence gates, receipts, methodology boundaries, and docs. diff --git a/.changeset/amc-nuclia-rag-triad-replay.md b/.changeset/amc-nuclia-rag-triad-replay.md new file mode 100644 index 000000000..a72b52ddd --- /dev/null +++ b/.changeset/amc-nuclia-rag-triad-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Adds Nuclia-style RAG-triad replay receipts to the replay benchmark corpus. Rows now fail closed unless source/repository/license proof, package/model-cache/auth/evaluator/dataset/QA-context/metric traces, triad scores, replay proof, model-access boundary, no-raw-context-copy boundary, signed evidence, and row hashes are present. diff --git a/.changeset/amc-observability-live-drift.md b/.changeset/amc-observability-live-drift.md new file mode 100644 index 000000000..9e263167a --- /dev/null +++ b/.changeset/amc-observability-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add observability/SRE live-drift evidence receipts for o11y-bench-style monitoring, including task-spec and generated-task hashes, Grafana stack and Docker config proof, scenario-clock alignment, trajectory/stdout/grading/reward/result/report artifacts, deterministic-check, rubric, resolution, incident/task/data-source/tool-mode drift metrics, Watch alerts, public methodology, and docs coverage. diff --git a/.changeset/amc-observe-api-routes.md b/.changeset/amc-observe-api-routes.md new file mode 100644 index 000000000..f84e55b72 --- /dev/null +++ b/.changeset/amc-observe-api-routes.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Expose `amc observe` timeline and anomaly data through read-only `/api/v1/observe/*` API routes, document them in OpenAPI/API surfaces, and add regression coverage for route registration and CLI-parity payloads. diff --git a/.changeset/amc-occubench-professional-task-explainability.md b/.changeset/amc-occubench-professional-task-explainability.md new file mode 100644 index 000000000..868e5297e --- /dev/null +++ b/.changeset/amc-occubench-professional-task-explainability.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add OccuBench-style professional-task question explainability proof to Score, Shield, Watch, Studio drilldown, public methodology, and docs. diff --git a/.changeset/amc-ollama-metrics-live-drift.md b/.changeset/amc-ollama-metrics-live-drift.md new file mode 100644 index 000000000..13f77c2de --- /dev/null +++ b/.changeset/amc-ollama-metrics-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Ollama-metrics-style live drift receipts for local LLM proxy observability, including token, request-duration, time-per-token, loaded-model, RAM, error-rate, model/deployment/proxy-context drift, evidence coverage, methodology docs, and source-boundary notes. diff --git a/.changeset/amc-openapi-webhook-payload-schemas.md b/.changeset/amc-openapi-webhook-payload-schemas.md new file mode 100644 index 000000000..7d33a8f19 --- /dev/null +++ b/.changeset/amc-openapi-webhook-payload-schemas.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Document reusable OpenAPI webhook payload, event, receipt, and signature-header schemas and link product portal payload examples to those contracts. diff --git a/.changeset/amc-opencode-lab-metric-validity.md b/.changeset/amc-opencode-lab-metric-validity.md new file mode 100644 index 000000000..7a46beb78 --- /dev/null +++ b/.changeset/amc-opencode-lab-metric-validity.md @@ -0,0 +1,5 @@ +--- +"@agentmaturity/compass": patch +--- + +Add OpenCode-lab-style metric-validity gates that fail closed on missing source, lab, context, prompt, tool, AGENTS policy, repeated-run, fork-agreement, model-variance, ground-truth, metric-definition, CI, result, owner, sample-size, confidence-interval, signed-evidence, and row-hash proof. diff --git a/.changeset/amc-osuniverse-gui-navigation-live-drift.md b/.changeset/amc-osuniverse-gui-navigation-live-drift.md new file mode 100644 index 000000000..ea619c57d --- /dev/null +++ b/.changeset/amc-osuniverse-gui-navigation-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add OSUniverse-style GUI-navigation live-drift proof receipts with task category, complexity level, validator, trajectory, screenshot, step-limit, and runtime-context drift alerts. diff --git a/.changeset/amc-pack-entry-resolution.md b/.changeset/amc-pack-entry-resolution.md new file mode 100644 index 000000000..af9a83801 --- /dev/null +++ b/.changeset/amc-pack-entry-resolution.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Fix local pack test entry-point resolution for `index.mjs` scaffolds, retain legacy `index.js` fallback, document the pack authoring flow, and add regression coverage. diff --git a/.changeset/amc-pack-init-directory.md b/.changeset/amc-pack-init-directory.md new file mode 100644 index 000000000..562bd1248 --- /dev/null +++ b/.changeset/amc-pack-init-directory.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Make `amc pack init --name ` scaffold into `.//`, add `--dir`, and reject unnamed non-interactive scaffolding. diff --git a/.changeset/amc-pack-publish-destination.md b/.changeset/amc-pack-publish-destination.md new file mode 100644 index 000000000..0b4fddab9 --- /dev/null +++ b/.changeset/amc-pack-publish-destination.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Make `amc pack publish` create an explicit local bundle by default and upload to a registry only when `--registry` is provided. diff --git a/.changeset/amc-pack-test-path-ux.md b/.changeset/amc-pack-test-path-ux.md new file mode 100644 index 000000000..85e82268e --- /dev/null +++ b/.changeset/amc-pack-test-path-ux.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Make `amc pack test` validate AMC pack manifests, auto-detect one child pack, and print clearer path guidance. diff --git a/.changeset/amc-paper-read-skill-live-drift.md b/.changeset/amc-paper-read-skill-live-drift.md new file mode 100644 index 000000000..c7479fdf6 --- /dev/null +++ b/.changeset/amc-paper-read-skill-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add paper-read-skill-style live-drift receipts with source-boundary proof, row proof hashes, fail-closed evidence coverage alerts, public methodology r201 bindings, and docs. diff --git a/.changeset/amc-paperarena-replay-corpus.md b/.changeset/amc-paperarena-replay-corpus.md new file mode 100644 index 000000000..842770ed3 --- /dev/null +++ b/.changeset/amc-paperarena-replay-corpus.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Adds PaperArena replay-corpus receipts for scientific-literature tool-use claims, binding source/no-license proof, README/requirements, hub runner/scorer artifacts, dataset-builder/tool/RAG/reflector/run-script trees, Hugging Face dataset snapshots, paper/QA manifests, replay commands, CI receipts, tool surfaces, counts, evaluator agreement, trace/result coverage, signed evidence, and row hashes. diff --git a/.changeset/amc-parallel-research-skill-metric-validity.md b/.changeset/amc-parallel-research-skill-metric-validity.md new file mode 100644 index 000000000..bdeec164c --- /dev/null +++ b/.changeset/amc-parallel-research-skill-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Parallel/OpenClaw research-skill metric-validity receipts that fail closed unless source, license-boundary, skill/API/search/deep-research/chat/extract/citation/source-policy/batch/monitoring/security/dependency, benchmark-validation, owner, sample-size, confidence-interval, signed-evidence, and row-hash proof is present. diff --git a/.changeset/amc-pawbench-replay-corpus.md b/.changeset/amc-pawbench-replay-corpus.md new file mode 100644 index 000000000..b9d360ed7 --- /dev/null +++ b/.changeset/amc-pawbench-replay-corpus.md @@ -0,0 +1,7 @@ +--- +"agent-maturity-compass": patch +--- + +Add PawBench-style model-harness replay receipts for replay benchmark corpora. + +Replay rows can now bind model id, harness id, task id/source, task taxonomy, grading mode, prompt/workspace/timeout/task metadata hashes, grader or judge rubric proof, transcript and metrics artifacts, submission and slice payloads, replay command, result-version path, deterministic seed, task count, preservation artifacts, signed evidence, and row hashes. Missing model-harness replay evidence fails closed through the manifest, CI receipt, Shield verification, and Watch alerts. diff --git a/.changeset/amc-pbsai-governance.md b/.changeset/amc-pbsai-governance.md new file mode 100644 index 000000000..8c8d5acc4 --- /dev/null +++ b/.changeset/amc-pbsai-governance.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Implement PBSAI governance support with a twelve-domain framework mapping, structured context-envelope diagnostic, and signed attestation envelope metadata. diff --git a/.changeset/amc-pentest-metric-validity.md b/.changeset/amc-pentest-metric-validity.md new file mode 100644 index 000000000..f1eabb9b7 --- /dev/null +++ b/.changeset/amc-pentest-metric-validity.md @@ -0,0 +1,8 @@ +--- +"agent-maturity-compass": minor +--- + +Add Apex-style pentest and threat-model benchmark metric-validity coverage with +typed proof for manifests, coverage distributions, ground truth, traps, +security controls, execution/report artifacts, owners, sample size, confidence +intervals, eval-pack rows, public methodology, and API documentation. diff --git a/.changeset/amc-personagym-metric-validity.md b/.changeset/amc-personagym-metric-validity.md new file mode 100644 index 000000000..8cfb16bcf --- /dev/null +++ b/.changeset/amc-personagym-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add PersonaGym-style persona-agent metric-validity proof with typed persona, environment, benchmark question, model/provider, rubric, PersonaScore-style metric, calibration, evaluation-output, result, owner, sample-size, confidence-interval, signed-evidence, and row-hash gates. diff --git a/.changeset/amc-physicianbench-ehr-live-drift.md b/.changeset/amc-physicianbench-ehr-live-drift.md new file mode 100644 index 000000000..e3d1ff3a4 --- /dev/null +++ b/.changeset/amc-physicianbench-ehr-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add PhysicianBench-style clinical EHR live-drift receipts so FHIR, patient-record, checkpoint, trajectory, workspace, eval-log, clinical metric, signed-evidence, and row-hash proof fail closed. diff --git a/.changeset/amc-piarena-prompt-injection-live-drift.md b/.changeset/amc-piarena-prompt-injection-live-drift.md new file mode 100644 index 000000000..eeaaefad1 --- /dev/null +++ b/.changeset/amc-piarena-prompt-injection-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add PIArena-style prompt-injection live-drift receipts so attack success, defense block, false positives, agent task success, tool-call success, evidence coverage, attack/defense/dataset/agent-benchmark drift, signed evidence, and row hashes fail closed. diff --git a/.changeset/amc-playground-scenario-expansion.md b/.changeset/amc-playground-scenario-expansion.md new file mode 100644 index 000000000..72838fe4d --- /dev/null +++ b/.changeset/amc-playground-scenario-expansion.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Expand the CLI and browser playground scenario libraries with real-world alignment, supply-chain/logistics, healthcare, and finance cases, plus regression coverage for scenario breadth and static browser scenario checks. diff --git a/.changeset/amc-pokereval-live-drift.md b/.changeset/amc-pokereval-live-drift.md new file mode 100644 index 000000000..d3e28a550 --- /dev/null +++ b/.changeset/amc-pokereval-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add PokerEval-style live score and behavior drift receipts with package, citation, simulation, opponent-pool, run, hand-history, metric-report, BB/100, all-in adjusted, EV, VPIP, hand-count, context drift, methodology, docs, and synthetic tests. diff --git a/.changeset/amc-polymath-logic-replay.md b/.changeset/amc-polymath-logic-replay.md new file mode 100644 index 000000000..043042d24 --- /dev/null +++ b/.changeset/amc-polymath-logic-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Polymath-style logic benchmark replay receipts to the replay corpus so symbolic-reasoning benchmark claims bind source, dataset access, license, environment, inference boundary, tool, replay, evaluator, metric, signed evidence, and row-hash proof before Score/Shield/Watch accept them. diff --git a/.changeset/amc-promptware-kill-chain.md b/.changeset/amc-promptware-kill-chain.md new file mode 100644 index 000000000..81fa3fc66 --- /dev/null +++ b/.changeset/amc-promptware-kill-chain.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add promptware kill-chain scenarios to the injection assurance pack for multi-stage persistence, lateral movement, and exfiltration attacks. diff --git a/.changeset/amc-prospect-demo-share.md b/.changeset/amc-prospect-demo-share.md new file mode 100644 index 000000000..d3c2cd9c4 --- /dev/null +++ b/.changeset/amc-prospect-demo-share.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add guided prospect demo and static share bundle commands for sales walkthroughs, with DEMO_ONLY evidence boundaries and public URL manifests. diff --git a/.changeset/amc-provider-drift-benchmark.md b/.changeset/amc-provider-drift-benchmark.md new file mode 100644 index 000000000..6b45828a9 --- /dev/null +++ b/.changeset/amc-provider-drift-benchmark.md @@ -0,0 +1,9 @@ +--- +"agent-maturity-compass": minor +--- + +Add provider/model canary drift benchmarking with fail-closed score, refusal, latency, and cost thresholds plus watch alert and waiver outputs. + +Add user-aware agent-quality drift dimensions for progress AUC, progress per turn, pass@k, pass^k, subgoal completion, expected-tool-call coverage, persona coverage, and clustered error-rate analysis so stable headline scores cannot hide multi-turn or user-proxy regressions. + +Add standard evaluator-suite drift dimensions for evaluator coverage, guardrail pass rate, score-threshold pass rate, and retry stability so provider/model promotions fail closed when evaluator-library or pytest-style regression coverage weakens despite a stable headline score. diff --git a/.changeset/amc-provider-drift-eval-framework-proof.md b/.changeset/amc-provider-drift-eval-framework-proof.md new file mode 100644 index 000000000..9e0f3bcf1 --- /dev/null +++ b/.changeset/amc-provider-drift-eval-framework-proof.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Eval-ai-library-style provider-drift evaluator framework proof with framework/version, provider route, metric suite, evaluator config, generated test-data, verdict aggregation, dashboard artifact, signed evidence, row-hash, Watch alert, and CI fail-closed coverage. diff --git a/.changeset/amc-provider-drift-signed-evidence.md b/.changeset/amc-provider-drift-signed-evidence.md new file mode 100644 index 000000000..c243fea78 --- /dev/null +++ b/.changeset/amc-provider-drift-signed-evidence.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Fail provider-drift gates closed when canary rows lack signed evidence references, and document the signed-evidence requirement for replayable provider-drift eval packs. diff --git a/.changeset/amc-provider-observability-pipeline-drift.md b/.changeset/amc-provider-observability-pipeline-drift.md new file mode 100644 index 000000000..2e9986070 --- /dev/null +++ b/.changeset/amc-provider-observability-pipeline-drift.md @@ -0,0 +1,7 @@ +--- +"agent-maturity-compass": patch +--- + +Add Opik-style provider observability pipeline proof to provider/model drift canaries. + +Provider-drift rows can now bind pipeline orchestrator/run, experiment tracker/run, observability project, datastore, retrieval index, content dataset, summary artifact, QA dataset, trace export, metric report, and pipeline config proof. Missing observability-pipeline evidence emits a fail-closed Watch/API alert and is included in eval-pack row hashes, CI/lifecycle gates, public methodology, and docs. diff --git a/.changeset/amc-public-leaderboard-export.md b/.changeset/amc-public-leaderboard-export.md new file mode 100644 index 000000000..4ff96df75 --- /dev/null +++ b/.changeset/amc-public-leaderboard-export.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc leaderboard public-export` for anonymized public leaderboard dataset-card and JSONL exports with minimum-cohort and metadata opt-in privacy controls. diff --git a/.changeset/amc-public-methodology.md b/.changeset/amc-public-methodology.md new file mode 100644 index 000000000..d18a0f492 --- /dev/null +++ b/.changeset/amc-public-methodology.md @@ -0,0 +1,23 @@ +--- +"agent-maturity-compass": minor +--- + +Publish a public AMC scoring methodology manifest and bind its id, version, hash, benchmark methodology versioning rules, and modality/lifecycle-aware metric-validation gates into diagnostic reports, badges, the CLI, and the config version API. + +Add a persona-policy realism score-claim boundary so cooperative simulator success is not overclaimed as robust, human-like, task-preserving persona evaluation. + +Add a ranking-stability metric-validation gate so checkpoint, model, or candidate rankings cannot be treated as external proof without subsampling-confidence, tail-failure, data-quality, OCR/readability, and ordering evidence. + +Add a live CTF evaluation integrity score-claim boundary so cybersecurity benchmark or flag-solving claims require contamination, competition-impact, first-correct-flag forwarding, and per-agent independence evidence. + +Add a partial-credit CTF validity score-claim boundary so VM challenge or checkpoint-completion claims require environment snapshots, checkpoint rubrics, execution traces, labelling evidence, and isolation context. + +Add a business-workflow metric-validation gate so workflow automation benchmark metrics cannot be treated as external proof without domain/task coverage, simple-baseline evidence, public/private score caveats, toolset/config controls, programmatic end-state assertions, partial-credit and strict pass-rate semantics, export artifacts, and multi-run comparison evidence. + +Clarify the RAG-pipeline metric-validation methodology so custom-domain RAG benchmark claims disclose document/test sets, selected metrics, query-level computation records, solution comparisons, and performance/cost evidence. + +Clarify the RAG-pipeline metric-validation methodology for legal-domain RAG benchmark claims, including corpus provenance, jurisdiction/language/task coverage, retriever/reranker configs, model and judge configs, logged samples, and agent-framework evidence. + +Add an iterative tournament-learning score-claim boundary so leaderboard, tournament, peer-learning, and code-agent strategy-improvement claims require signed tournament protocol, opponent-pool, code-artifact, battle-log/replay, ranking, repeated-validation, uncertainty, learning-delta, access-policy, and contamination-boundary evidence. + +Add a data-agent analytical metric-validation methodology gate so heterogeneous data-agent benchmark claims require task-type, database/source-modality, difficulty, metric-computation, agent-workflow, expert-validation, cost/latency, and submission-schema evidence. diff --git a/.changeset/amc-public-question-count-drift.md b/.changeset/amc-public-question-count-drift.md new file mode 100644 index 000000000..81d0f77b1 --- /dev/null +++ b/.changeset/amc-public-question-count-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Reconcile public diagnostic-question and industry-pack count drift across product docs, website pages, station pages, and Batch 5 audit notes, with regression coverage for source catalog counts and current-facing copy. diff --git a/.changeset/amc-public-test-count-drift.md b/.changeset/amc-public-test-count-drift.md new file mode 100644 index 000000000..761e81ea9 --- /dev/null +++ b/.changeset/amc-public-test-count-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Normalize current-facing public test-count claims to a reproducible Vitest static inventory and add a regression test for stale count drift. diff --git a/.changeset/amc-pulumi-helm-release.md b/.changeset/amc-pulumi-helm-release.md new file mode 100644 index 000000000..4b42e302f --- /dev/null +++ b/.changeset/amc-pulumi-helm-release.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add a Pulumi Kubernetes Helm-release example for deploying the AMC Helm chart without storing bootstrap secret literals in Pulumi stack config. diff --git a/.changeset/amc-question-explainability.md b/.changeset/amc-question-explainability.md new file mode 100644 index 000000000..c9c021795 --- /dev/null +++ b/.changeset/amc-question-explainability.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add question-level score explainability receipts with accepted signed evidence, rejected evidence reasons, missing gates, repair hints, typed routing, red-team benchmark, off-policy optimization evaluation criteria, SkillLens-style rubric lens checks, row hashes, and Score/Shield/Watch/Passport bindings. Add Promptflow-style RAG flow diagnostics for flow DAG, parameter config, eval set, evaluator flow, ground-truth mapping, variant/deployment artifacts, accepted/rejected evidence, and fail-closed repair hints. diff --git a/.changeset/amc-question-set-count-drift.md b/.changeset/amc-question-set-count-drift.md new file mode 100644 index 000000000..7a328463b --- /dev/null +++ b/.changeset/amc-question-set-count-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Correct diagnostic question-set metadata after the four research-backed additions: the default compatibility set now reports 244 questions, lifecycle reports 264, user-facing copy no longer hardcodes stale 240/260 assumptions, and question-set tests derive expected counts from the bank. diff --git a/.changeset/amc-quickscore-answers-json.md b/.changeset/amc-quickscore-answers-json.md new file mode 100644 index 000000000..b74819ea2 --- /dev/null +++ b/.changeset/amc-quickscore-answers-json.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc quickscore --answers ` for non-interactive answer-based scoring in full and rapid quickscore modes. diff --git a/.changeset/amc-quickscore-auto-no-evidence.md b/.changeset/amc-quickscore-auto-no-evidence.md new file mode 100644 index 000000000..de7811a1f --- /dev/null +++ b/.changeset/amc-quickscore-auto-no-evidence.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Make `amc quickscore --auto --json` fail closed with structured `AUTO_NO_EVIDENCE` metadata when execution evidence is missing, avoiding normal-looking zero-score output for first-run users. diff --git a/.changeset/amc-quickscore-first-run-hint.md b/.changeset/amc-quickscore-first-run-hint.md new file mode 100644 index 000000000..40223acc5 --- /dev/null +++ b/.changeset/amc-quickscore-first-run-hint.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add a first-run interactive hint to non-interactive quickscore placeholder output, including JSON metadata, rapid/full CLI output, quickstart output, and docs/audit coverage. diff --git a/.changeset/amc-quickscore-json-placeholder.md b/.changeset/amc-quickscore-json-placeholder.md new file mode 100644 index 000000000..4928bf44a --- /dev/null +++ b/.changeset/amc-quickscore-json-placeholder.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Keep non-interactive quickscore JSON parseable by moving placeholder-score warnings into structured metadata and strengthening human quickstart warning copy. diff --git a/.changeset/amc-quickscore-nontty-fail-closed.md b/.changeset/amc-quickscore-nontty-fail-closed.md new file mode 100644 index 000000000..ef164cafd --- /dev/null +++ b/.changeset/amc-quickscore-nontty-fail-closed.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Make non-interactive `amc quickscore` fail closed before placeholder L0 scoring unless answers or evidence mode are provided. diff --git a/.changeset/amc-quickscore-share-validation.md b/.changeset/amc-quickscore-share-validation.md new file mode 100644 index 000000000..869b49e56 --- /dev/null +++ b/.changeset/amc-quickscore-share-validation.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Validate the compiled `quickscore --rapid --share` path as an offline badge and summary flow without requiring Studio. diff --git a/.changeset/amc-quickstart-noninteractive-guard.md b/.changeset/amc-quickstart-noninteractive-guard.md new file mode 100644 index 000000000..fb3959fe3 --- /dev/null +++ b/.changeset/amc-quickstart-noninteractive-guard.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Fail closed for non-interactive `amc quickstart` runs instead of producing placeholder L0 results. diff --git a/.changeset/amc-rag-chunking-technique-metric-validity.md b/.changeset/amc-rag-chunking-technique-metric-validity.md new file mode 100644 index 000000000..d9022254d --- /dev/null +++ b/.changeset/amc-rag-chunking-technique-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add IBM/rag-chunking-techniques-style RAG chunking technique metric-validity receipts with fail-closed Score/Shield/Watch surfaces, methodology/docs bindings, and eval-pack proof fields. diff --git a/.changeset/amc-rag-contradiction-detector-replay.md b/.changeset/amc-rag-contradiction-detector-replay.md new file mode 100644 index 000000000..cc4041448 --- /dev/null +++ b/.changeset/amc-rag-contradiction-detector-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add public methodology and source-boundary receipts for RAG_Contradiction_Detector-style biomedical RAG contradiction replay claims. diff --git a/.changeset/amc-rag-dataset-builder-live-drift.md b/.changeset/amc-rag-dataset-builder-live-drift.md new file mode 100644 index 000000000..14254c7c9 --- /dev/null +++ b/.changeset/amc-rag-dataset-builder-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add RAG QA dataset-builder live-drift receipts and alerts for source-document, license, QA-pair, passage, config, tier, question-type, build-stage, grounding, human-verification, citation, answer-support, cost, concurrency, count, and row-hash evidence. diff --git a/.changeset/amc-rag-eval-document-qa-replay.md b/.changeset/amc-rag-eval-document-qa-replay.md new file mode 100644 index 000000000..4b88aade6 --- /dev/null +++ b/.changeset/amc-rag-eval-document-qa-replay.md @@ -0,0 +1,5 @@ +--- +"@agent-maturity/compass": patch +--- + +Add rag-eval-style document QA dataset replay receipts. Replay-corpus rows can now fail closed on missing source/repository/license proof, input document manifests, processor/prompt/generator proof, generated QA dataset proof, endpoint response traces, ranking/evaluation reports, replay/CI receipts, question and endpoint counts, score delta, replay pass rate, endpoint response coverage, signed evidence, and row hashes. diff --git a/.changeset/amc-rag-eval-flow-replay.md b/.changeset/amc-rag-eval-flow-replay.md new file mode 100644 index 000000000..4f8b66468 --- /dev/null +++ b/.changeset/amc-rag-eval-flow-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Rag-Eval-flow-style local RAG replay proof to the replay benchmark corpus, including source/repository/license, pipeline, data-source, model, judge, metric, prompt-template, eval-pack, fixture, replay, result, score-delta, CI, sample-size, seed, replay-pass, metric-coverage, signed-evidence, and row-hash gates. diff --git a/.changeset/amc-ragas-notebook-metric-validity.md b/.changeset/amc-ragas-notebook-metric-validity.md new file mode 100644 index 000000000..6ccad8e10 --- /dev/null +++ b/.changeset/amc-ragas-notebook-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add fail-closed RAGAS notebook metric-validity receipts for Coding-Crashkurse/RAG-Evaluation-with-Ragas-style evidence, including source/no-license-boundary, notebook, dependency, document corpus, chunking, testset generation, evolution mix, generated testset, RAG chain, retriever/vectorstore, model/embedding, answer-context, RAGAS metric, LangFuse, visualization, sample/CI, signed-evidence, and row-hash proof. diff --git a/.changeset/amc-ragscore-rag-audit-methodology.md b/.changeset/amc-ragscore-rag-audit-methodology.md new file mode 100644 index 000000000..ee46a088c --- /dev/null +++ b/.changeset/amc-ragscore-rag-audit-methodology.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add a RagScore-style RAG audit methodology boundary for generated QA datasets, support-span grounding, detailed RAG metrics, failure diagnosis, privacy mode, and MCP/server telemetry claims. diff --git a/.changeset/amc-rail-score-live-drift.md b/.changeset/amc-rail-score-live-drift.md new file mode 100644 index 000000000..55dccfcca --- /dev/null +++ b/.changeset/amc-rail-score-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add RAIL Score live-drift receipts with fail-closed source, release, PyPI package, guardrail, telemetry, compliance, agent-tool, Watch alert, row-hash, and public methodology evidence. diff --git a/.changeset/amc-ravig-bench-metric-validity.md b/.changeset/amc-ravig-bench-metric-validity.md new file mode 100644 index 000000000..0ab5f6216 --- /dev/null +++ b/.changeset/amc-ravig-bench-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add RAViG-Bench metric-validity receipts with source/evaluation/dataset/evaluator proof, fail-closed Score/API/CI gates, public methodology r209 binding, and docs for the live verified source boundary. diff --git a/.changeset/amc-realign-simulation-metric-validity.md b/.changeset/amc-realign-simulation-metric-validity.md new file mode 100644 index 000000000..fd601b610 --- /dev/null +++ b/.changeset/amc-realign-simulation-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Realign-style simulation metric-validity proof fields, fail-closed judge/regression thresholds, public methodology boundaries, API surface documentation, and benchmark no-copy guidance for simulation-driven AI app evaluation evidence. diff --git a/.changeset/amc-realtalk-long-conversation-replay.md b/.changeset/amc-realtalk-long-conversation-replay.md new file mode 100644 index 000000000..086f480ca --- /dev/null +++ b/.changeset/amc-realtalk-long-conversation-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add REALTALK-style long-term conversation replay receipts so real-dialogue memory claims bind provenance, privacy/consent, temporal split, LoCoMo comparison, task-specific evaluator artifacts, metrics, signed evidence, and row-hash proof before Score/Shield/Watch accept them. diff --git a/.changeset/amc-recovery-bench-live-drift.md b/.changeset/amc-recovery-bench-live-drift.md new file mode 100644 index 000000000..406a50ea3 --- /dev/null +++ b/.changeset/amc-recovery-bench-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Recovery-Bench-style live-drift proof so failed-trajectory replay, corrupted-environment recovery, recovery success/reward, message-mode/harness/task drift, signed evidence, and row hashes fail closed across Score, Shield, and Watch. diff --git a/.changeset/amc-redteam-adversarial-regression.md b/.changeset/amc-redteam-adversarial-regression.md new file mode 100644 index 000000000..6443fa585 --- /dev/null +++ b/.changeset/amc-redteam-adversarial-regression.md @@ -0,0 +1,8 @@ +--- +"agent-maturity-compass": minor +--- + +Add RedTeam-style adversarial benchmark regression proof to replay-corpus rows, +including benchmark/question-set/reference-answer/scoring/backend/model/result +hashes, scoring modes, optional prompt optimization, release-gate binding, +summary fields, methodology, and documentation coverage. diff --git a/.changeset/amc-redteam-cvss-scoring.md b/.changeset/amc-redteam-cvss-scoring.md new file mode 100644 index 000000000..7b4f3d8f6 --- /dev/null +++ b/.changeset/amc-redteam-cvss-scoring.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add deterministic CVSS v4.0-style base scoring metadata to red-team vulnerability reports, including score, qualitative rating, vector string, metric values, and an explicit approximation note for AI-agent findings. diff --git a/.changeset/amc-redteam-evil-mcp-integration.md b/.changeset/amc-redteam-evil-mcp-integration.md new file mode 100644 index 000000000..4262ddca9 --- /dev/null +++ b/.changeset/amc-redteam-evil-mcp-integration.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Expose built-in Evil MCP agent-provider scenarios through `amc redteam run --evil-mcp`, add `--mcp-attacks` category selection with common aliases, embed linked MCP JSON/Markdown evidence paths in the primary red-team report, and fix MCP provider category filtering for `tool-poisoning`. diff --git a/.changeset/amc-redteam-guide-gaming-resistance.md b/.changeset/amc-redteam-guide-gaming-resistance.md new file mode 100644 index 000000000..4fabf8a48 --- /dev/null +++ b/.changeset/amc-redteam-guide-gaming-resistance.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Document `amc score gaming-resistance` in the red-team guide as the primary score-gaming resistance check, alongside `amc redteam run`, Evil MCP coverage, assurance packs, and shield analysis. diff --git a/.changeset/amc-redteam-no-sign-status.md b/.changeset/amc-redteam-no-sign-status.md new file mode 100644 index 000000000..31ff5d177 --- /dev/null +++ b/.changeset/amc-redteam-no-sign-status.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add explicit unsigned-valid status semantics to `amc redteam run`, expose `--no-sign` in the CLI, and verify that red-team reports run without creating or unlocking a vault while persisting JSON/Markdown evidence. Also refresh canon and diagnostic-bank schema counts from the live question bank so fresh workspace initialization works after the 244-question expansion. diff --git a/.changeset/amc-replay-benchmark-corpus.md b/.changeset/amc-replay-benchmark-corpus.md new file mode 100644 index 000000000..de0ea9269 --- /dev/null +++ b/.changeset/amc-replay-benchmark-corpus.md @@ -0,0 +1,19 @@ +--- +"agent-maturity-compass": minor +--- + +Add replayable benchmark corpus manifests with fixture hashes, score deltas, optional multi-turn tool-risk attack-success-rate checks, optional code-execution runtime artifact hashes, optional pairwise, paired-modality, interactive-episode, skill-lifecycle, RAG-evaluation, AI-research-task, scientific-evaluation-suite, dynamic tool-sandbox, platform-evaluation, binary-audit, and long-term-memory receipts, Watch alerts, and Shield-verifiable CI receipts. + +Add Arthur Engine-style adversarial regression evidence for continuous-eval trace/annotation ids, evaluator versions, transforms, eval/rerun statuses, criteria/variables/explanation hashes, guardrail rule results, prompt-injection detection, alert-rule query hashes, thresholds, and webhook refs. + +Add Level-Navi-style web-search benchmark replay evidence for dataset/source-link/config hashes, navigation/search/citation traces, result JSONL and metric-report hashes, domain/question-type/source-link coverage, final score, component metrics, and pass rate. + +Add StreamBench-style streaming continuous-improvement replay evidence for source dataset manifests, ordered sequence hashes, agent/benchmark configs, update/prediction/evaluation/sanity traces, result artifacts, online-update gates, improvement deltas, retention, and catastrophic forgetting. + +Add industrial multimodal RAG replay evidence for text/image corpus hashes, PDF and image-extraction traces, image summaries, separate or combined vector-store hashes, multimodal embedding config hashes, baseline and correct-context run hashes, judge model/rubric evidence, modality-specific context and faithfulness metrics, modality coverage, and fail-closed Watch alerts when those artifacts are missing. + +Add BioDSA-style biomedical agent benchmark replay evidence for biomedical task/workflow types, task/dataset/knowledge-base/tool/workflow/model/sandbox hashes, execution and code-execution traces, structured result/report/artifact/evaluator hashes, completion counts, score thresholds, and safe-code-execution gates. + +Add VAKRA-style enterprise multi-hop, multi-source tool-calling replay evidence for local API/database/document/MCP/tool-schema/policy hashes, trajectory replay, tool-call/tool-response/retrieved-evidence traces, output validation, deterministic replay, policy adherence, groundedness, exact tool-response matching, final-answer metrics, and leaderboard/repro metadata. + +Add AgenticVBench-style video post-production replay evidence for task families, media manifests, verifier reward/metric artifacts, agent trajectories, Harbor result summaries, trial logs, executors, sandbox image hashes, oracle/baseline solver hashes, leaderboard or human-review proof, judge modes, task/clip counts, reward thresholds, and fail-closed CI receipts. diff --git a/.changeset/amc-report-latest-alias.md b/.changeset/amc-report-latest-alias.md new file mode 100644 index 000000000..08974741d --- /dev/null +++ b/.changeset/amc-report-latest-alias.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc report latest` and unique run-id prefix resolution so customer-success and sales workflows can render the newest saved report without copying full UUIDs. diff --git a/.changeset/amc-report-share-url.md b/.changeset/amc-report-share-url.md new file mode 100644 index 000000000..a9e9c1ce4 --- /dev/null +++ b/.changeset/amc-report-share-url.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add diagnostic report share bundles. `amc report --share` now writes a static `index.html` plus `share-manifest.json`, prints a local file URL, and can print a public URL with `--public-base-url` for user-custodied static hosting. diff --git a/.changeset/amc-report-status-explanations.md b/.changeset/amc-report-status-explanations.md new file mode 100644 index 000000000..867b113a6 --- /dev/null +++ b/.changeset/amc-report-status-explanations.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Explain diagnostic report statuses in Markdown, executive, and HTML outputs with evidence status labels, claim boundaries, and next verification steps for valid, invalid, unsigned, and trust-boundary-blocked reports. diff --git a/.changeset/amc-researchgym-live-drift.md b/.changeset/amc-researchgym-live-drift.md new file mode 100644 index 000000000..4e661cf8e --- /dev/null +++ b/.changeset/amc-researchgym-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add ResearchGym-style autonomous research-run live drift proof so Watch/Score receipts fail closed on missing task, artifact, budget, inspection, violation, score-improvement, subtask-completion, task-domain, or runtime-context evidence. diff --git a/.changeset/amc-researchharness-agent-replay.md b/.changeset/amc-researchharness-agent-replay.md new file mode 100644 index 000000000..7dcd30944 --- /dev/null +++ b/.changeset/amc-researchharness-agent-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add ResearchHarness-style tool-using agent harness replay receipts to AgentBench replay rows, including runtime contract, tool surface, native tool-call, OpenAI-compatible API, workspace boundary, trace, adapter, provider matrix, baseline/meta-harness, policy, metric, summary, CI, methodology, and documentation bindings. diff --git a/.changeset/amc-resume-rag-evaluator-metric-validity.md b/.changeset/amc-resume-rag-evaluator-metric-validity.md new file mode 100644 index 000000000..d719466d8 --- /dev/null +++ b/.changeset/amc-resume-rag-evaluator-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add resume-RAG evaluator metric-validity receipts for local Ollama resume parser and candidate-evaluator claims, including fail-closed source/license, upload/parser, job-description, RAG strategy, retrieval, model, endpoint, rating, batch, privacy, dependency, owner, sample-size, confidence-interval, signed-evidence, and row-hash proof. diff --git a/.changeset/amc-retail-sales-question-explainability.md b/.changeset/amc-retail-sales-question-explainability.md new file mode 100644 index 000000000..768acf1bb --- /dev/null +++ b/.changeset/amc-retail-sales-question-explainability.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add ShampooSalesAgent-style retail sales question-explainability receipts with source/product/customer/order/provider/policy/privacy proof, metric thresholds, evidence drilldown previews, public methodology binding, and fail-closed validation. diff --git a/.changeset/amc-rss-market-impact-methodology.md b/.changeset/amc-rss-market-impact-methodology.md new file mode 100644 index 000000000..52b7e3bf7 --- /dev/null +++ b/.changeset/amc-rss-market-impact-methodology.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add RSS market-impact public methodology versioning receipts with fail-closed source/no-license, feed, model route, prompt/schema, dedupe, analysis, push, threshold, outcome/backtest, migration, signed-evidence, and row-hash requirements. diff --git a/.changeset/amc-sap-agent-eval-live-drift.md b/.changeset/amc-sap-agent-eval-live-drift.md new file mode 100644 index 000000000..f73882e99 --- /dev/null +++ b/.changeset/amc-sap-agent-eval-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add SAP agent-evaluation tutorial live-drift proof with objective, process, enterprise-context, notebook, dataset, baseline log, live sample, metric, tooling, policy, alert-receipt, signed-evidence, distribution-drift, and fail-closed Watch gates. diff --git a/.changeset/amc-scientific-literature-metric-validity.md b/.changeset/amc-scientific-literature-metric-validity.md new file mode 100644 index 000000000..b3b11bb28 --- /dev/null +++ b/.changeset/amc-scientific-literature-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add AutoResearchBench-style scientific literature discovery metric-validity gates with signed benchmark, task, dataset, search-tool, metric, owner, confidence-interval, eval-pack, and fail-closed CI proof. diff --git a/.changeset/amc-sconebench-smart-contract-validity.md b/.changeset/amc-sconebench-smart-contract-validity.md new file mode 100644 index 000000000..df91ea06d --- /dev/null +++ b/.changeset/amc-sconebench-smart-contract-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add SconeBench-style smart-contract exploit metric-validity proof to pentest benchmark checks, requiring dataset, historical fork, problem metadata, FlawVerifier, Forge grader, profit threshold, anti-cheat reset, cutoff split, signed evidence, confidence interval, and row-hash receipts. diff --git a/.changeset/amc-scorable-studio-drilldown.md b/.changeset/amc-scorable-studio-drilldown.md new file mode 100644 index 000000000..d6a0816dc --- /dev/null +++ b/.changeset/amc-scorable-studio-drilldown.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Scorable SDK Studio evidence drilldown receipts with source/package integrity, trace, receipt, policy, artifact, empty-state, and error-state proof. diff --git a/.changeset/amc-sdk-consolidation-inventory.md b/.changeset/amc-sdk-consolidation-inventory.md new file mode 100644 index 000000000..5feed5d9d --- /dev/null +++ b/.changeset/amc-sdk-consolidation-inventory.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Consolidate Node/TypeScript, Python, Go, and OpenAPI SDK surfaces in the SDK docs and update the Batch 5 integration audit evidence. diff --git a/.changeset/amc-securevibebench-metric-validity.md b/.changeset/amc-securevibebench-metric-validity.md new file mode 100644 index 000000000..83e364fb0 --- /dev/null +++ b/.changeset/amc-securevibebench-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add SecureVibeBench secure-coding metric-validity receipts with fail-closed source, dataset, runner, adapter, scenario, test-script, utility, CI, owner, confidence interval, signed-evidence, artifact-hash, and row-hash proof. diff --git a/.changeset/amc-securing-mcp-governance.md b/.changeset/amc-securing-mcp-governance.md new file mode 100644 index 000000000..bc26bce78 --- /dev/null +++ b/.changeset/amc-securing-mcp-governance.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Securing MCP supply-chain governance scoring and unintentional-adversary excessive-agency probes. diff --git a/.changeset/amc-shield-runtime-analysis.md b/.changeset/amc-shield-runtime-analysis.md new file mode 100644 index 000000000..faa38118c --- /dev/null +++ b/.changeset/amc-shield-runtime-analysis.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc shield analyze-runtime`, a runtime action-analysis surface that wraps the Shield trust pipeline with instruction-source, sensitive-field, credential freshness, confidence, risk, recommendation, and evidence-chain reporting. diff --git a/.changeset/amc-signed-ci-rollout-guidance.md b/.changeset/amc-signed-ci-rollout-guidance.md new file mode 100644 index 000000000..c18bdb5de --- /dev/null +++ b/.changeset/amc-signed-ci-rollout-guidance.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Document signed CI graduation after `ci init --no-sign`, including vault setup, CI secret handling, signed rerun command, and when to remove `--no-sign`. diff --git a/.changeset/amc-skill-forge-replay-corpus.md b/.changeset/amc-skill-forge-replay-corpus.md new file mode 100644 index 000000000..cbd405393 --- /dev/null +++ b/.changeset/amc-skill-forge-replay-corpus.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Skill Forge-style autoresearch replay receipts to SkillBench regression proof, requiring source, agent-role, mutation/revert policy, replay manifest, CI, score-delta, and row-hash evidence before autonomous skill-improvement claims pass. diff --git a/.changeset/amc-skillbench-regression-replay.md b/.changeset/amc-skillbench-regression-replay.md new file mode 100644 index 000000000..05db32570 --- /dev/null +++ b/.changeset/amc-skillbench-regression-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add SkillBench-style adversarial skill regression replay receipts with fail-closed with-skill/baseline agent config, eval-case, deterministic grader, static-analysis, security-scan, rerun, release-gate, score, and decision proof. diff --git a/.changeset/amc-skillmatch-resume-live-drift.md b/.changeset/amc-skillmatch-resume-live-drift.md new file mode 100644 index 000000000..bfeaae5bd --- /dev/null +++ b/.changeset/amc-skillmatch-resume-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Watch evidence receipts and public methodology boundaries for SkillMatch-style resume live-drift claims. diff --git a/.changeset/amc-sld-scaling-law-live-drift.md b/.changeset/amc-sld-scaling-law-live-drift.md new file mode 100644 index 000000000..2edb2d160 --- /dev/null +++ b/.changeset/amc-sld-scaling-law-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add SLDBench-style scaling-law discovery live-drift proof with signed benchmark/source, task, dataset split, source-experiment, config, model-route, program, checkpoint, result, formula, extrapolation, R2, NMSE, NMAE, evidence coverage, task-type, context, row-hash, and fail-closed Watch alerts. diff --git a/.changeset/amc-social-reasoning-bench-replay.md b/.changeset/amc-social-reasoning-bench-replay.md new file mode 100644 index 000000000..ff81bd383 --- /dev/null +++ b/.changeset/amc-social-reasoning-bench-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Microsoft Social Reasoning Bench-style replay-corpus receipts with fail-closed Score/Shield/Watch surfaces, methodology/docs bindings, source-boundary proof, and CI summary fields. diff --git a/.changeset/amc-sparkorbit-orbit-monitor-provider-drift.md b/.changeset/amc-sparkorbit-orbit-monitor-provider-drift.md new file mode 100644 index 000000000..695c456be --- /dev/null +++ b/.changeset/amc-sparkorbit-orbit-monitor-provider-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add SparkOrbit-style orbit-monitor proof to provider drift canaries with fail-closed checks for source catalogs, leaderboard/model/benchmark/news snapshots, reload runs, ranking policy, summary artifacts, source/category counts, daily reload verification, eval-pack row hashes, Watch alerts, and CI gates. diff --git a/.changeset/amc-spent-session-cost-replay.md b/.changeset/amc-spent-session-cost-replay.md new file mode 100644 index 000000000..0da45c154 --- /dev/null +++ b/.changeset/amc-spent-session-cost-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add spent-style session-cost replay receipts to the replay benchmark corpus, including fail-closed source/repository/license, hook config, JSONL manifest, pricing, classifier, dashboard, privacy, session/tool-event, efficiency, cost, replay, classification-coverage, CI, API, methodology, and Watch/Shield summary bindings. diff --git a/.changeset/amc-sre-incident-question-explainability.md b/.changeset/amc-sre-incident-question-explainability.md new file mode 100644 index 000000000..cee46ba1d --- /dev/null +++ b/.changeset/amc-sre-incident-question-explainability.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add SRE incident-triage question-explainability receipts with OpenEnv scenario proof, raw log and metric hashes, user reports, action payloads, deterministic grader and feedback hashes, reward/root-cause/red-herring/ordered-remediation thresholds, step bounds, evidence refs, repair hints, Studio drilldown previews, and r41 methodology docs. diff --git a/.changeset/amc-sso-scim-entrypoints.md b/.changeset/amc-sso-scim-entrypoints.md new file mode 100644 index 000000000..eef397b9d --- /dev/null +++ b/.changeset/amc-sso-scim-entrypoints.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add discoverable enterprise auth setup entrypoints. `amc sso configure ` now configures signed host identity providers, and `amc scim init` enables SCIM provisioning with optional first-token creation without resetting existing providers. diff --git a/.changeset/amc-startup-guidance.md b/.changeset/amc-startup-guidance.md new file mode 100644 index 000000000..a5bd8ac67 --- /dev/null +++ b/.changeset/amc-startup-guidance.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc quickstart --startup-plan` and `--what-broken` for role-aware startup guidance, framework detection, sample answer files, and blocker-only output. diff --git a/.changeset/amc-strands-benchmark-harness-live-drift.md b/.changeset/amc-strands-benchmark-harness-live-drift.md new file mode 100644 index 000000000..5d159be48 --- /dev/null +++ b/.changeset/amc-strands-benchmark-harness-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Adds Strands benchmark-harness live-drift receipt fields, thresholds, row hashing, Watch alerts, methodology docs, and tests for signed trajectory/evidence coverage, task-success, patch-apply, test-pass, latency, cost, suite, runtime, and task-family drift. diff --git a/.changeset/amc-studio-evidence-drilldown.md b/.changeset/amc-studio-evidence-drilldown.md new file mode 100644 index 000000000..1a2444bef --- /dev/null +++ b/.changeset/amc-studio-evidence-drilldown.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add a Studio evidence drilldown for question-level score findings, including a UI route, API view model, source artifact links, accepted/rejected evidence previews, missing gate reasons, repair hints, and fail-closed empty states. diff --git a/.changeset/amc-subtlememory-metric-validity.md b/.changeset/amc-subtlememory-metric-validity.md new file mode 100644 index 000000000..4a08860f6 --- /dev/null +++ b/.changeset/amc-subtlememory-metric-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add SubtleMemory-style relational-memory metric-validity receipts so Score, Shield, and Watch fail closed when source/license, default-branch, arXiv, Hugging Face dataset, persona split, bench/history manifest, relation taxonomy, construction pipeline, staged evaluation, adapter roster, judge/evaluator, score/diagnostic, CI validation, owner, sample-size, confidence-interval, signed-evidence, artifact-hash, and row-hash proof is missing. diff --git a/.changeset/amc-sutro-batch-methodology-versioning.md b/.changeset/amc-sutro-batch-methodology-versioning.md new file mode 100644 index 000000000..dfed1df98 --- /dev/null +++ b/.changeset/amc-sutro-batch-methodology-versioning.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Sutro-style batch methodology-versioning assurance for public AMC reports and badges, including source snapshot, license, function/schema, data-source, input-order, priority, dry-run cost, model-pool, observability, export, retention, multi-model, embedding, diagnostic receipt, and no-copy boundary requirements. diff --git a/.changeset/amc-systemic-sycophancy-probes.md b/.changeset/amc-systemic-sycophancy-probes.md new file mode 100644 index 000000000..85bf068c6 --- /dev/null +++ b/.changeset/amc-systemic-sycophancy-probes.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Extend the sycophancy assurance pack with systemic objective-decoupling probes for biased feedback loops, evaluator collusion, and unsafe majority-feedback updates. diff --git a/.changeset/amc-terminalworld-replay-corpus.md b/.changeset/amc-terminalworld-replay-corpus.md new file mode 100644 index 000000000..4b480747f --- /dev/null +++ b/.changeset/amc-terminalworld-replay-corpus.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add TerminalWorld-style replay-corpus receipts that fail closed unless public-recording provenance, privacy/quality filters, synthesized task proof, Docker environment reproduction, state-based tests, AllPassing/Nop/Partial trials, result/replay/CI proof, signed evidence, and row hashes are present. diff --git a/.changeset/amc-terrarium-living-environment-validity.md b/.changeset/amc-terrarium-living-environment-validity.md new file mode 100644 index 000000000..06ba4c774 --- /dev/null +++ b/.changeset/amc-terrarium-living-environment-validity.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Terrarium-style living-environment metric-validity gates so stateful multi-turn benchmark claims require task-program, mutable-environment, capability, sandbox, agent-adapter, checker, trial-result, aggregate-metric, pass@k, proactive-trigger, owner, confidence-interval, signed-evidence, and row-hash proof before Score/Shield/Watch accept them. diff --git a/.changeset/amc-test-suite-question-explainability.md b/.changeset/amc-test-suite-question-explainability.md new file mode 100644 index 000000000..7a57ff56e --- /dev/null +++ b/.changeset/amc-test-suite-question-explainability.md @@ -0,0 +1,7 @@ +--- +"agent-maturity-compass": patch +--- + +Add test-suite evaluation lenses to question-level score explainability. + +Question receipts can now bind suite/source identity, language/framework/adapter, dataset and test-case hashes, evaluator config, judge context, experiment results, export artifacts, CI proof, agent trace/tool-call validation, pass-rate and score thresholds, cost/latency/tokens, accepted and rejected evidence, repair hints, and row hashes. Missing test-suite evidence fails closed through question explainability replayability, generated reports, Shield inspection, API docs, and public methodology r69. diff --git a/.changeset/amc-text2sql-agent-replay.md b/.changeset/amc-text2sql-agent-replay.md new file mode 100644 index 000000000..48902c6f0 --- /dev/null +++ b/.changeset/amc-text2sql-agent-replay.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add Text2SQL business-database replay receipts to the replay benchmark corpus, including schema/database/governance/security/audit evidence, fail-closed SQL accuracy and safety thresholds, CI receipt fields, Watch alerts, methodology r59 docs, and source-use posture for Tangxihong0922/QueryMind. diff --git a/.changeset/amc-tiered-top-level-help.md b/.changeset/amc-tiered-top-level-help.md new file mode 100644 index 000000000..48285dc4e --- /dev/null +++ b/.changeset/amc-tiered-top-level-help.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Show compact grouped top-level help by default and keep the full list behind `amc --help --all`. diff --git a/.changeset/amc-toolsafe-diagnostic-question.md b/.changeset/amc-toolsafe-diagnostic-question.md new file mode 100644 index 000000000..85150903e --- /dev/null +++ b/.changeset/amc-toolsafe-diagnostic-question.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add a ToolSafe-backed diagnostic question for proactive pre-execution tool invocation guardrails and feedback evidence. diff --git a/.changeset/amc-trace-evaluation-metric-validity.md b/.changeset/amc-trace-evaluation-metric-validity.md new file mode 100644 index 000000000..e0f62f2de --- /dev/null +++ b/.changeset/amc-trace-evaluation-metric-validity.md @@ -0,0 +1,7 @@ +--- +"agent-maturity-compass": patch +--- + +Add trace-derived agent-evaluation metric-validity proof for Bedrock-style agent quality loops. + +Metric-validation reports now support optional `traceEvaluationChecks` plus `requireTraceEvaluationProof`. When required, AMC fails closed unless the row binds signed evidence for model config, agent parameters, tool registry, trace manifest, repeatable cases, dynamic validators, bulk runs, run permutations, mocked LLM controls, metric definitions, measurement exports, production monitor bindings, threshold alarms, owner, sample size, confidence interval, and row hashes. diff --git a/.changeset/amc-trust-report-nosign-public-key-fallback.md b/.changeset/amc-trust-report-nosign-public-key-fallback.md new file mode 100644 index 000000000..cb01d7bcf --- /dev/null +++ b/.changeset/amc-trust-report-nosign-public-key-fallback.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Treat missing monitor public keys as unsigned `fleet trust-report --no-sign` prerequisites instead of hard failures. diff --git a/.changeset/amc-up-demo-console-open.md b/.changeset/amc-up-demo-console-open.md new file mode 100644 index 000000000..3f8f39a6a --- /dev/null +++ b/.changeset/amc-up-demo-console-open.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Open the Compass Console by default after `amc up --demo` startup, with `--no-open` for headless runs. diff --git a/.changeset/amc-up-demo-readonly.md b/.changeset/amc-up-demo-readonly.md new file mode 100644 index 000000000..e0a2d1e10 --- /dev/null +++ b/.changeset/amc-up-demo-readonly.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc up --demo` / `--read-only` and `--dry-run` so users can explore Studio without a vault passphrase while keeping signed startup vault-gated. diff --git a/.changeset/amc-value-webhook-route-contract.md b/.changeset/amc-value-webhook-route-contract.md new file mode 100644 index 000000000..7896563f6 --- /dev/null +++ b/.changeset/amc-value-webhook-route-contract.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Publish the Studio value webhook route contract with explicit single-workspace and host-mode paths, `x-amc-webhook-token` auth semantics, and vault-token wording. diff --git a/.changeset/amc-vla-world-model-replay.md b/.changeset/amc-vla-world-model-replay.md new file mode 100644 index 000000000..dc2391017 --- /dev/null +++ b/.changeset/amc-vla-world-model-replay.md @@ -0,0 +1,8 @@ +--- +"agent-maturity-compass": minor +--- + +Add a VLA/world-model replay lane to the replay benchmark corpus with typed +taxonomy, manifest, trajectory, simulator/reward, policy, replay, seed, count, +coverage, task-success, score, summary, CI receipt, Watch alert, methodology, +and documentation coverage. diff --git a/.changeset/amc-watch-safety-test-category-verbose.md b/.changeset/amc-watch-safety-test-category-verbose.md new file mode 100644 index 000000000..ed861949d --- /dev/null +++ b/.changeset/amc-watch-safety-test-category-verbose.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add `amc watch safety-test --category` and `--verbose`, plus scenario-level safety test details and alignment-specific probes. diff --git a/.changeset/amc-web-agent-privacy-live-drift.md b/.changeset/amc-web-agent-privacy-live-drift.md new file mode 100644 index 000000000..0e3ab140c --- /dev/null +++ b/.changeset/amc-web-agent-privacy-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add typed web-agent privacy leakage live-drift receipts for AgentDAM-style data-minimization proof. Watch now hashes benchmark/task/browser/privacy evidence, checks minimization/leakage/task metrics, fails closed on missing proof, and alerts on environment, observation-mode, and context drift. diff --git a/.changeset/amc-web-eval-dataset-metric-validity.md b/.changeset/amc-web-eval-dataset-metric-validity.md new file mode 100644 index 000000000..66060dc7d --- /dev/null +++ b/.changeset/amc-web-eval-dataset-metric-validity.md @@ -0,0 +1,11 @@ +--- +"agent-maturity-compass": patch +--- + +Add web eval dataset metric-validity proof for web-search RAG evaluation dataset claims. + +AMC now fails closed unless web eval dataset metric rows bind source, subject, +generated-query, search-provider, retrieved-document, filter, QA-generation, +reference-answer, export-target, freshness, source-coverage, answer-grounding, +owner, confidence-interval, signed-evidence, and row-hash proof before using +Tavily-style generated web/RAG evaluation datasets as external evidence. diff --git a/.changeset/amc-web-operator-live-drift.md b/.changeset/amc-web-operator-live-drift.md new file mode 100644 index 000000000..bdf7df0a4 --- /dev/null +++ b/.changeset/amc-web-operator-live-drift.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": minor +--- + +Add web-operator live-drift receipts with self-report success, independent LLM-evaluation success, self-report overclaim and mismatch rates, task reliability, replay-artifact coverage, task timing, step-limit violations, provider/context drift alerts, row-hash binding, Watch projections, tests, and r44 methodology docs. diff --git a/.changeset/amc-website-404-relative-links.md b/.changeset/amc-website-404-relative-links.md new file mode 100644 index 000000000..b3579a7ca --- /dev/null +++ b/.changeset/amc-website-404-relative-links.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Fix the static 404 page navigation to use domain-agnostic relative links and add regression coverage against the old `/AgentMaturityCompass/` deployment path. diff --git a/.changeset/amc-website-backup-cleanup.md b/.changeset/amc-website-backup-cleanup.md new file mode 100644 index 000000000..fa7eba29a --- /dev/null +++ b/.changeset/amc-website-backup-cleanup.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Remove tracked public `website/script-backup*.js` files so stale backup assets are not shipped from the website root. diff --git a/.changeset/amc-website-content-validation.md b/.changeset/amc-website-content-validation.md new file mode 100644 index 000000000..0deb90b78 --- /dev/null +++ b/.changeset/amc-website-content-validation.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Validate website blog/changelog content and add a static changelog fallback so the public changelog is useful even when remote release-note loading fails. diff --git a/.changeset/amc-website-focus-visible.md b/.changeset/amc-website-focus-visible.md new file mode 100644 index 000000000..5125f0140 --- /dev/null +++ b/.changeset/amc-website-focus-visible.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Normalize static website keyboard focus-visible styling and add regression coverage for focus indicators and removed backup artifacts. diff --git a/.changeset/amc-website-skip-links.md b/.changeset/amc-website-skip-links.md new file mode 100644 index 000000000..164593568 --- /dev/null +++ b/.changeset/amc-website-skip-links.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Add skip-navigation links and `main-content` targets across static website/docs HTML pages, with shared docs skip-link styles and regression coverage. diff --git a/.changeset/amc-whitepaper-citation-placeholders.md b/.changeset/amc-whitepaper-citation-placeholders.md new file mode 100644 index 000000000..16f3ff7c9 --- /dev/null +++ b/.changeset/amc-whitepaper-citation-placeholders.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Resolve whitepaper placeholder citation markers by replacing `[CITATION: ...]` tags with numeric references, adding OpenAI's official agent tooling source, removing unsupported McKinsey/Gartner adoption claims, and adding regression coverage that blocks placeholder citations and unsupported industry-report claims from returning. diff --git a/.changeset/amc-whitepaper-homepage-discovery.md b/.changeset/amc-whitepaper-homepage-discovery.md new file mode 100644 index 000000000..9c0872f8f --- /dev/null +++ b/.changeset/amc-whitepaper-homepage-discovery.md @@ -0,0 +1,5 @@ +--- +"agent-maturity-compass": patch +--- + +Expose the AMC whitepaper from the public homepage research section, footer, and docs index, with regression coverage and Batch 5 audit closure for research-artifact discoverability. diff --git a/.gitignore b/.gitignore index 9f54cf719..a57d744f9 100644 --- a/.gitignore +++ b/.gitignore @@ -90,6 +90,9 @@ tmp/ .amc/blobs/ .amc/ledger/ .amc/reports/ +.amc/dashboard/ +.amc/gatePolicy.json +test-results/ swarm-report.json qa/node_modules/ qa/dist/ diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 6b40a9026..425466a48 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -26,7 +26,7 @@ git clone https://github.com/AgentMaturity/AgentMaturityCompass.git cd AgentMaturityCompass npm ci npm run build # must compile with 0 TypeScript errors -npm test # 3,980 tests, all must pass +npm test # 5,394 collected Vitest tests, all must pass in CI ``` **Python platform:** @@ -173,6 +173,17 @@ npx vitest run tests/assurance/myAttackPack.test.ts amc assurance run --pack my-attack-pack --verbose # manual test ``` +### Step 6: Review registry readiness + +Before a community pack is uploaded to any shared registry, complete the review gates in `docs/ASSURANCE_LAB.md#community-registry-review-gates`. + +At minimum, reviewers should confirm: + +- `amc pack test .` passes locally +- provenance, citations, and license terms are documented +- no secrets, malware, hidden network calls, unsafe prompts, or unlicensed copied content are present +- the manifest names a maintenance owner and compatible pack version + ### Guidelines for good packs - **Deterministic:** no randomness, no external dependencies @@ -379,7 +390,7 @@ export const myModule: ScoringModule = { 1. **Fork** the repo 2. **Branch:** `git checkout -b feat/my-contribution` 3. **Build:** `npm run build` — must compile with 0 errors -4. **Test:** `npm test` — all 3,980+ tests must pass +4. **Test:** `npm test` — the collected Vitest suite must pass 5. **Commit** with a descriptive message (`feat:`, `fix:`, `docs:`, `test:`) 6. **Push** and open a PR diff --git a/README.md b/README.md index 94c00e8e8..87321e481 100644 --- a/README.md +++ b/README.md @@ -14,7 +14,7 @@ npm version downloads CI - tests + tests MIT

@@ -91,6 +91,7 @@ Want a fast legacy pulse check instead of the full evidence score? ```bash amc quickscore --rapid # optional rapid check, not the full score +amc quickscore --answers answers.json --json # non-interactive answer-based score ```
@@ -108,9 +109,12 @@ brew tap AgentMaturity/amc && brew install agent-maturity-compass **Docker** ```bash -docker run -it --rm ghcr.io/agentmaturity/amc-quickstart amc +docker build -t amc-quickstart -f docker/Dockerfile.quickstart . +docker run -it --rm amc-quickstart amc ``` +Use the local build command unless a GHCR package has been verified public. + **From source** ```bash git clone https://github.com/AgentMaturity/AgentMaturityCompass.git @@ -159,19 +163,19 @@ AMC is not an observability tool and not an eval harness. It is a **trust scorec | Supply Chain | Dependency attacks, MCP server poisoning, SBOM integrity | | Behavioral | Sycophancy, self-preservation, sabotage, over-compliance | -### 40 Industry Domain Packs +### 41 Industry Domain Packs | Sector | Packs | Key Regulations | |--------|-------|-----------------| | đŸ„ Health | 9 | HIPAA, FDA 21 CFR Part 11, EU MDR, ICH E6(R3) | | 💰 Wealth | 5 | MiFID II, PSD2, EU DORA, MiCA, FATF | | 🎓 Education | 5 | FERPA, COPPA, IDEA, EU AI Act Annex III | -| 🚇 Mobility | 5 | UNECE WP.29, ETSI EN 303 645, EU NIS2 | +| 🚇 Mobility | 6 | UNECE WP.29, ETSI EN 303 645, EU NIS2, ISO 28000, GS1 EPCIS | | 💡 Technology | 5 | EU AI Act Art. 13, EU Data Act, DSA Art. 34 | | 🌿 Environment | 6 | EU Farm-to-Fork, REACH, IEC 61850 | | đŸ›ïž Governance | 5 | EU eIDAS 2.0, UNCAC, UNGPs | -Industry Packs are paid content: `$9.99/month` unlocks all 40 packs in the CLI and Studio. Run `amc domain pack checkout` to open the subscription flow, then paste the returned key into Studio or run `amc domain pack activate --key `. +Industry Packs are paid content: `$9.99/month` unlocks all 41 packs in the CLI and Studio. Run `amc domain pack checkout` to open the subscription flow, then paste the returned key into Studio or run `amc domain pack activate --key `. ### 🔼 Simulation & Forecast Evaluation Lane @@ -291,6 +295,12 @@ amc run --question-set lifecycle # opt-in 260-question lifecycle expansion Need a fast pulse check for a demo or README badge? Use `amc quickscore --rapid` explicitly. +Need a CI-safe score without terminal prompts? Use `amc quickscore --answers answers.json --json`, where `answers.json` maps question IDs to L0-L5 numbers. + +If `amc quickscore` prints a placeholder L0 because no terminal prompt was available, it now shows a first-run hint: "Did you mean to run the interactive score?" Run it in a terminal, or pass answers explicitly for CI. + +If `amc quickscore --auto --json` cannot find captured execution evidence, it fails closed with `scoreStatus: "AUTO_NO_EVIDENCE"` and does not emit a measured zero score. Capture evidence with `amc wrap -- `, or use `amc quickscore --answers answers.json --json` for CI-safe survey scoring. + Advanced proof check: ```bash @@ -410,8 +420,16 @@ amc lite-score # score a non-agent cha ```bash amc business kpi # correlate maturity to outcomes +amc business risk --maturity 3 --baseline-frequency 4 --incident-cost 50000 --json +amc business fair-scenario --scenario claims-ai-data-leak --maturity 3 --frequency-min 2 --frequency-most-likely 5 --frequency-max 9 --loss-min 20000 --loss-most-likely 75000 --loss-max 250000 --out fair-scenario.md +amc business roi --current-maturity 2 --target-maturity 3 --baseline-frequency 5 --incident-cost 20000 --annual-control-cost 15000 --implementation-cost 5000 +amc business heatmap --portfolio risk-portfolio.json --out risk-heatmap.md +amc business grc-export --portfolio risk-portfolio.json --out grc-treatment-plan.csv amc business report # stakeholder-ready business summary +amc executive brief --run latest --out board-brief.html # print-ready board one-pager amc leaderboard show # compare agents across a fleet +amc leaderboard public-export --output public-leaderboard # anonymized leaderboard dataset bundle +amc compare --output compare.json --badge # run diff plus compare-badge.svg amc inventory scan --deep # discover agents, frameworks, model files amc comms-check --text "Guaranteed 40% return" --domain wealth ``` @@ -461,6 +479,16 @@ jobs: ### Badge for your README +For a run-to-run or model-route comparison badge, generate it from the real comparison command: + +```bash +amc compare --output compare.json --badge +amc compare gpt-4o-mini claude-3-haiku --agent support-bot --output model-compare.json --badge +# writes compare-badge.svg or model-compare-badge.svg beside the report +``` + +Standalone maturity badges are also available when you only need a README trust marker: + ```markdown [![AMC Score](https://img.shields.io/badge/AMC-L3_(72.5)-green?logo=data:image/svg+xml;base64,PHN2ZyB4bWxucz0iaHR0cDovL3d3dy53My5vcmcvMjAwMC9zdmciIHZpZXdCb3g9IjAgMCAyNCAyNCI+PHBhdGggZmlsbD0iI2ZmZiIgZD0iTTEyIDJMMiA3bDEwIDUgMTAtNXptMCA5bC04LjUtNC4yNUwyIDEybDEwIDUgMTAtNXptMCA5bC04LjUtNC4yNUwyIDIxbDEwIDUgMTAtNXoiLz48L3N2Zz4=)](https://github.com/AgentMaturity/AgentMaturityCompass) @@ -535,9 +563,12 @@ curl -fsSL https://agentmaturity.co/install.sh | sh ### Docker ```bash -docker run -it --rm ghcr.io/agentmaturity/amc-quickstart amc +docker build -t amc-quickstart -f docker/Dockerfile.quickstart . +docker run -it --rm amc-quickstart amc ``` +Use the local build command unless a GHCR package has been verified public. + ### From source ```bash git clone https://github.com/AgentMaturity/AgentMaturityCompass.git @@ -562,11 +593,11 @@ The full trust stack is **free and MIT licensed**. The only paid surface is Indu | Tier | What you get | |---|---| -| **Free / Open Source** | Everything — Score, Shield, Enforce, Vault, Watch, Comply, Fleet, Passport, all 14 adapters, 1,084 registered CLI command paths, browser playground, CI gates | -| **Industry Packs** | Everything in Free + all 40 Industry Domain Packs for `$9.99/month` | +| **Free / Open Source** | Everything — Score, Shield, Enforce, Vault, Watch, Comply, Fleet, Passport, all 14 adapters, 1,140 registered CLI command paths, browser playground, CI gates | +| **Industry Packs** | Everything in Free + all 41 Industry Domain Packs for `$9.99/month` | | **Enterprise** | Everything in Industry Packs + priority support + custom pack development + deployment assistance | -> Industry Packs are 40 sector-specific domain packs (healthcare, finance, education, government, etc.) that require ongoing regulatory research and maintenance. The core trust stack stays free forever. +> Industry Packs are 41 sector-specific domain packs (healthcare, finance, education, logistics, government, etc.) that require ongoing regulatory research and maintenance. The core trust stack stays free forever. --- @@ -578,6 +609,7 @@ The full trust stack is **free and MIT licensed**. The only paid surface is Indu | **CLI** | Real agent scoring, evidence capture, shareable outputs | `npx agent-maturity-compass` | | **CI/CD** | Release gates, score thresholds, PR comments | [CI Templates](docs/CI_TEMPLATES.md) | | **Enterprise** | Self-hosted, managed deployment | [Deployment Options](docs/DEPLOYMENT_OPTIONS.md) | +| **Cloud self-hosting** | AWS, GCP, Azure reference architectures | [Cloud Reference Architectures](docs/CLOUD_REFERENCE_ARCHITECTURES.md) | ### Start by persona @@ -592,17 +624,20 @@ The full trust stack is **free and MIT licensed**. The only paid surface is Indu | | | |--|--| | [Getting Started (2 min)](docs/GETTING_STARTED.md) | [Agent Guide](docs/AGENT_GUIDE.md) | +| [Community Demo Kit](docs/COMMUNITY_DEMO_KIT.md) | [Why AMC One-Pager](docs/WHY_AMC_ONE_PAGER.md) | | [Solo Dev Quickstart](docs/SOLO_DEV_QUICKSTART.md) | [Platform Engineer Quickstart](docs/PLATFORM_ENGINEER_QUICKSTART.md) | | [Security & Compliance Quickstart](docs/SECURITY_COMPLIANCE_QUICKSTART.md) | [Troubleshooting](docs/TROUBLESHOOTING.md) | -| [CLI Reference (1,084 command paths)](docs/AMC_MASTER_REFERENCE.md) | [Architecture](docs/ARCHITECTURE_MAP.md) | +| [CLI Reference (1,140 command paths)](docs/CLI_COMMAND_INVENTORY.md) | [Architecture](docs/ARCHITECTURE_MAP.md) | | [Compatibility Matrix](docs/COMPATIBILITY_MATRIX.md) | [Starter Blueprints](docs/STARTER_BLUEPRINTS.md) | | [Install Packages](docs/INSTALL_PACKAGES.md) | [Support Policy](docs/SUPPORT_POLICY.md) | | [Release Cadence](docs/RELEASE_CADENCE.md) | [CI Templates](docs/CI_TEMPLATES.md) | | [Hardening Guide](docs/HARDENING.md) | [Community](docs/COMMUNITY.md) | +| [Cloud Reference Architectures](docs/CLOUD_REFERENCE_ARCHITECTURES.md) | [Deployment Options](docs/DEPLOYMENT_OPTIONS.md) | | [Assurance Lab](docs/ASSURANCE_LAB.md) | [Domain Packs](docs/SECTOR_PACKS.md) | | [EU AI Act Compliance](docs/EU_AI_ACT_COMPLIANCE.md) | [Multi-Agent Trust](docs/MULTI_AGENT_TRUST.md) | -| [Executive Overview](docs/EXECUTIVE_OVERVIEW.md) | [White Paper](whitepaper/AMC_WHITEPAPER_v1.md) | -| [Example Projects](examples/) | [Web Playground](https://agentmaturity.co/playground.html) | +| [Executive Overview](docs/EXECUTIVE_OVERVIEW.md) | [Board L3 Risk Memo](docs/BOARD_RISK_L3_MEMO.md) | +| [White Paper](whitepaper/AMC_WHITEPAPER_v1.md) | [Example Projects](examples/) | +| [Web Playground](https://agentmaturity.co/playground.html) | [Docs Index](docs/INDEX.md) |
More docs @@ -616,6 +651,7 @@ The full trust stack is **free and MIT licensed**. The only paid surface is Indu - [docs/EXAMPLES_INDEX.md](docs/EXAMPLES_INDEX.md) — example index - [docs/RECIPES.md](docs/RECIPES.md) — extended recipes - [docs/DEPLOYMENT_OPTIONS.md](docs/DEPLOYMENT_OPTIONS.md) — deployment options +- [docs/CLOUD_REFERENCE_ARCHITECTURES.md](docs/CLOUD_REFERENCE_ARCHITECTURES.md) — AWS, GCP, and Azure self-hosted reference architectures - [docs/PRODUCT_EDITIONS.md](docs/PRODUCT_EDITIONS.md) — product editions - [docs/PRICING.md](docs/PRICING.md) — pricing details - [docs/BUYER_PACKAGES.md](docs/BUYER_PACKAGES.md) — buyer packages @@ -652,7 +688,11 @@ AMC now includes a scheduled GitHub Actions workflow that validates packaged CLI AMC now supports lightweight workspace config presets for `.amc/amc.config.yaml`: ```bash +amc init --minimal amc init --profile dev +amc quickstart --minimal +amc quickstart --startup-plan --answers-out amc-startup-answers.json +amc quickstart --what-broken amc quickstart --profile ci amc config profile prod ``` @@ -662,6 +702,9 @@ Current MVP behavior: - `ci` → isolated trust boundary, proxy env enabled - `prod` → isolated trust boundary, proxy env disabled - explicit `--trust-boundary` still overrides the profile when you need it +- `--minimal` → startup-friendly setup without a vault prompt or immediate full-score prompt +- `--startup-plan` → role-aware 10-minute startup plan, framework detection, and optional sample answer file +- `--what-broken` → single-command startup blocker report without running the interactive score --- @@ -671,7 +714,7 @@ AMC is MIT licensed. We welcome contributions — especially new **assurance pac ```bash git clone https://github.com/AgentMaturity/AgentMaturityCompass.git -cd AgentMaturityCompass && npm ci && npm test # 5,098 tests +cd AgentMaturityCompass && npm ci && npm test # 5,394 collected Vitest tests ``` **→ [CONTRIBUTING.md](CONTRIBUTING.md)** — includes guides for writing packs, mapping research papers, and adding adapters. @@ -692,6 +735,6 @@ cd AgentMaturityCompass && npm ci && npm test # 5,098 tests ---

- 240 default diagnostic questions + 20 lifecycle expansion questions · 147 assurance packs · 40 domain packs · 14 adapters · 79 scoring modules · 5,098 tests
+ 244 default diagnostic questions + 20 lifecycle expansion questions · 147 assurance packs · 41 domain packs · 14 adapters · 79 scoring modules · 5,394 collected tests
Stop trusting. Start verifying.

diff --git a/deploy/helm/amc/README.md b/deploy/helm/amc/README.md index 21e74fcee..7f716039f 100644 --- a/deploy/helm/amc/README.md +++ b/deploy/helm/amc/README.md @@ -20,6 +20,8 @@ helm install amc ./deploy/helm/amc \ --set image.tag=latest ``` +Production guide: `docs/KUBERNETES_HELM_DEPLOYMENT.md`. + ## Example values Render internal-only deployment: @@ -50,3 +52,11 @@ helm template amc ./deploy/helm/amc -f ./deploy/helm/amc/examples/values-persist - TLS ingress support - NetworkPolicy with ingress-controller-only ingress and DNS/upstream egress controls - PDB + ServiceAccount templates + +## Terraform + +The example under `deploy/terraform/helm-release/` deploys this chart with Terraform's Helm provider while keeping bootstrap secrets outside Terraform state. + +## Pulumi + +The example under `deploy/pulumi/helm-release/` deploys this chart with Pulumi's Kubernetes Helm v3 Release resource while keeping bootstrap secrets outside Pulumi stack config/state. diff --git a/deploy/pulumi/helm-release/Pulumi.dev.yaml.example b/deploy/pulumi/helm-release/Pulumi.dev.yaml.example new file mode 100644 index 000000000..229310717 --- /dev/null +++ b/deploy/pulumi/helm-release/Pulumi.dev.yaml.example @@ -0,0 +1,12 @@ +config: + amc-helm-release:releaseName: amc + amc-helm-release:namespace: amc-system + amc-helm-release:chartPath: ../../helm/amc + amc-helm-release:imageRepository: ghcr.io/your-org/amc-studio + amc-helm-release:imageTag: latest + amc-helm-release:bootstrapSecretName: amc-bootstrap + amc-helm-release:ingressEnabled: false + amc-helm-release:ingressHost: amc.example.com + amc-helm-release:storageClassName: "" + amc-helm-release:workspaceStorageSize: 10Gi + amc-helm-release:timeoutSeconds: 600 diff --git a/deploy/pulumi/helm-release/Pulumi.yaml b/deploy/pulumi/helm-release/Pulumi.yaml new file mode 100644 index 000000000..7e8fc567f --- /dev/null +++ b/deploy/pulumi/helm-release/Pulumi.yaml @@ -0,0 +1,56 @@ +name: amc-helm-release +description: Deploy the AMC Helm chart to an existing Kubernetes cluster with Pulumi. +runtime: nodejs +config: + kubeconfig: + type: string + secret: true + description: Optional kubeconfig content. Omit to use the active kubeconfig environment. + kubeContext: + type: string + description: Optional kubeconfig context name. + releaseName: + type: string + default: amc + namespace: + type: string + default: amc-system + chartPath: + type: string + default: ../../helm/amc + imageRepository: + type: string + default: ghcr.io/your-org/amc-studio + imageTag: + type: string + default: latest + imagePullPolicy: + type: string + default: IfNotPresent + replicaCount: + type: integer + default: 1 + bootstrapSecretName: + type: string + default: amc-bootstrap + ingressEnabled: + type: boolean + default: false + ingressClassName: + type: string + default: "" + ingressHost: + type: string + default: amc.example.com + ingressTlsSecretName: + type: string + default: "" + storageClassName: + type: string + default: "" + workspaceStorageSize: + type: string + default: 10Gi + timeoutSeconds: + type: integer + default: 600 diff --git a/deploy/pulumi/helm-release/README.md b/deploy/pulumi/helm-release/README.md new file mode 100644 index 000000000..408acab7a --- /dev/null +++ b/deploy/pulumi/helm-release/README.md @@ -0,0 +1,88 @@ +# AMC Pulumi Helm Release Example + +This example deploys the local `deploy/helm/amc` chart with a Pulumi Kubernetes Helm v3 Release. It is a provider-agnostic root module for existing Kubernetes clusters, not an AMC-operated hosted service. + +Official references checked on 2026-06-16: + +- Pulumi Kubernetes Helm v3 Release: https://www.pulumi.com/registry/packages/kubernetes/api-docs/helm/v3/release/ +- Pulumi configuration and secret config: https://www.pulumi.com/docs/iac/concepts/config/ +- Pulumi Kubernetes provider: https://www.pulumi.com/registry/packages/kubernetes/api-docs/provider/ + +## Usage + +```bash +cd deploy/pulumi/helm-release +npm install +pulumi stack init dev +pulumi config set imageRepository ghcr.io/your-org/amc-studio +pulumi config set imageTag latest +pulumi preview +pulumi up +``` + +Optional kubeconfig handling: + +```bash +pulumi config set kubeContext your-context-name +pulumi config set --secret kubeconfig "$(cat ~/.kube/config)" +``` + +If `kubeconfig` is omitted, Pulumi uses the active Kubernetes environment available to the provider. + +## Secret Handling + +Create the `amc-bootstrap` Kubernetes secret outside Pulumi before applying this release. Do not put vault passphrases or owner passwords in Pulumi stack config. Pulumi state/config is not the right place for bootstrap secret literals, even when secret config encryption is enabled. + +```bash +kubectl create namespace amc-system +kubectl -n amc-system create secret generic amc-bootstrap \ + --from-literal=vaultPassphrase='' \ + --from-literal=ownerUsername='' \ + --from-literal=ownerPassword='' \ + --from-literal=notaryPassphrase='' \ + --from-literal=notaryAuthSecret='' +``` + +For production, source those values from your cloud secret manager, sealed-secrets workflow, External Secrets Operator, or a separate secure pipeline. + +## Configuration + +Common stack config keys: + +| Key | Default | Purpose | +|---|---|---| +| `releaseName` | `amc` | Helm release name | +| `namespace` | `amc-system` | Kubernetes namespace | +| `chartPath` | `../../helm/amc` | Local AMC Helm chart path | +| `imageRepository` | `ghcr.io/your-org/amc-studio` | AMC image repository | +| `imageTag` | `latest` | AMC image tag | +| `bootstrapSecretName` | `amc-bootstrap` | Existing Kubernetes Secret with bootstrap keys | +| `ingressEnabled` | `false` | Enable chart ingress | +| `ingressHost` | `amc.example.com` | Hostname for ingress | +| `ingressTlsSecretName` | empty | Existing TLS secret name | +| `storageClassName` | empty | Workspace PVC storage class | +| `workspaceStorageSize` | `10Gi` | Workspace PVC size | +| `valuesFiles` | empty | Additional Helm values files | + +Example values-file overlay: + +```bash +pulumi config set --path 'valuesFiles[0]' ../../helm/amc/examples/values-ingress-tls.yaml +``` + +## Verify + +```bash +kubectl -n amc-system rollout status deploy/amc +kubectl -n amc-system get pods,svc,ingress,pvc +kubectl -n amc-system port-forward svc/amc 3212:3212 +``` + +Then open `http://127.0.0.1:3212/console`. + +Health endpoints: + +- `GET /healthz` +- `GET /readyz` + +See `docs/KUBERNETES_HELM_DEPLOYMENT.md` for the full operator guide and `docs/CLOUD_REFERENCE_ARCHITECTURES.md` for AWS, GCP, and Azure architecture choices. diff --git a/deploy/pulumi/helm-release/index.ts b/deploy/pulumi/helm-release/index.ts new file mode 100644 index 000000000..303c8a797 --- /dev/null +++ b/deploy/pulumi/helm-release/index.ts @@ -0,0 +1,105 @@ +import * as k8s from "@pulumi/kubernetes"; +import * as pulumi from "@pulumi/pulumi"; + +const config = new pulumi.Config(); + +const kubeconfig = config.getSecret("kubeconfig"); +const kubeContext = config.get("kubeContext"); +const providerArgs: k8s.ProviderArgs = {}; + +if (kubeconfig) { + providerArgs.kubeconfig = kubeconfig; +} + +if (kubeContext) { + providerArgs.context = kubeContext; +} + +const provider = new k8s.Provider("amc-k8s", providerArgs); + +const releaseName = config.get("releaseName") ?? "amc"; +const namespace = config.get("namespace") ?? "amc-system"; +const chartPath = config.get("chartPath") ?? "../../helm/amc"; +const imageRepository = config.get("imageRepository") ?? "ghcr.io/your-org/amc-studio"; +const imageTag = config.get("imageTag") ?? "latest"; +const imagePullPolicy = config.get("imagePullPolicy") ?? "IfNotPresent"; +const replicaCount = config.getNumber("replicaCount") ?? 1; +const bootstrapSecretName = config.get("bootstrapSecretName") ?? "amc-bootstrap"; +const ingressEnabled = config.getBoolean("ingressEnabled") ?? false; +const ingressClassName = config.get("ingressClassName") ?? ""; +const ingressHost = config.get("ingressHost") ?? "amc.example.com"; +const ingressTlsSecretName = config.get("ingressTlsSecretName") ?? ""; +const storageClassName = config.get("storageClassName") ?? ""; +const workspaceStorageSize = config.get("workspaceStorageSize") ?? "10Gi"; +const timeoutSeconds = config.getNumber("timeoutSeconds") ?? 600; +const valuesFiles = config.getObject("valuesFiles") ?? []; + +const inlineValues: Record = { + replicaCount, + image: { + repository: imageRepository, + tag: imageTag, + pullPolicy: imagePullPolicy + }, + ingress: { + enabled: ingressEnabled, + className: ingressClassName, + hosts: [ + { + host: ingressHost, + paths: [{ path: "/", pathType: "Prefix" }] + } + ], + tls: ingressTlsSecretName + ? [ + { + secretName: ingressTlsSecretName, + hosts: [ingressHost] + } + ] + : [] + }, + workspace: { + persistence: { + enabled: true, + storageClassName, + size: workspaceStorageSize + } + }, + bootstrap: { + enabled: true, + ownerUsernameSecret: { name: bootstrapSecretName, key: "ownerUsername" }, + ownerPasswordSecret: { name: bootstrapSecretName, key: "ownerPassword" }, + vaultPassphraseSecret: { name: bootstrapSecretName, key: "vaultPassphrase" } + }, + notary: { + passphraseSecret: { name: bootstrapSecretName, key: "notaryPassphrase" }, + authSecret: { name: bootstrapSecretName, key: "notaryAuthSecret" } + } +}; + +const valueYamlFiles = valuesFiles.map((path) => new pulumi.asset.FileAsset(path)); + +const release = new k8s.helm.v3.Release( + "amc", + { + name: releaseName, + namespace, + chart: chartPath, + createNamespace: true, + atomic: true, + cleanupOnFail: true, + lint: true, + wait: true, + timeout: timeoutSeconds, + skipAwait: false, + values: inlineValues, + valueYamlFiles + }, + { provider } +); + +export const amcReleaseName = release.name; +export const amcNamespace = release.namespace; +export const amcChart = chartPath; +export const amcLocalPortForward = pulumi.interpolate`kubectl -n ${release.namespace} port-forward svc/${release.name} 3212:3212`; diff --git a/deploy/pulumi/helm-release/package.json b/deploy/pulumi/helm-release/package.json new file mode 100644 index 000000000..c3624e218 --- /dev/null +++ b/deploy/pulumi/helm-release/package.json @@ -0,0 +1,18 @@ +{ + "name": "amc-pulumi-helm-release", + "private": true, + "version": "0.0.0", + "type": "module", + "scripts": { + "preview": "pulumi preview", + "up": "pulumi up", + "destroy": "pulumi destroy" + }, + "dependencies": { + "@pulumi/kubernetes": "^4.32.0", + "@pulumi/pulumi": "^3.0.0" + }, + "devDependencies": { + "typescript": "^5.0.0" + } +} diff --git a/deploy/pulumi/helm-release/tsconfig.json b/deploy/pulumi/helm-release/tsconfig.json new file mode 100644 index 000000000..b1abe9b76 --- /dev/null +++ b/deploy/pulumi/helm-release/tsconfig.json @@ -0,0 +1,11 @@ +{ + "compilerOptions": { + "strict": true, + "target": "ES2020", + "module": "NodeNext", + "moduleResolution": "NodeNext", + "esModuleInterop": true, + "skipLibCheck": true, + "forceConsistentCasingInFileNames": true + } +} diff --git a/deploy/terraform/helm-release/README.md b/deploy/terraform/helm-release/README.md new file mode 100644 index 000000000..61d472aee --- /dev/null +++ b/deploy/terraform/helm-release/README.md @@ -0,0 +1,35 @@ +# AMC Terraform Helm Release Example + +This example deploys the local `deploy/helm/amc` chart with Terraform's Helm provider. It is an example root module, not a provider-specific cloud reference architecture. + +## Usage + +```bash +cp terraform.tfvars.example terraform.tfvars +terraform init +terraform plan +terraform apply +``` + +## Secret Handling + +Create the `amc-bootstrap` Kubernetes secret outside Terraform before applying this release. Do not put vault passphrases or owner passwords in `terraform.tfvars`; Terraform state is not the right place for those values. + +```bash +kubectl create namespace amc-system +kubectl -n amc-system create secret generic amc-bootstrap \ + --from-literal=vaultPassphrase='replace-with-long-random-passphrase' \ + --from-literal=ownerUsername='admin' \ + --from-literal=ownerPassword='replace-with-long-random-password' \ + --from-literal=notaryPassphrase='replace-with-long-random-passphrase' \ + --from-literal=notaryAuthSecret='replace-with-long-random-secret' +``` + +## Verify + +```bash +kubectl -n amc-system rollout status deploy/amc +kubectl -n amc-system get pods,svc,ingress,pvc +``` + +See `docs/KUBERNETES_HELM_DEPLOYMENT.md` for the full operator guide. diff --git a/deploy/terraform/helm-release/main.tf b/deploy/terraform/helm-release/main.tf new file mode 100644 index 000000000..c7f2f1fa0 --- /dev/null +++ b/deploy/terraform/helm-release/main.tf @@ -0,0 +1,62 @@ +provider "helm" { + kubernetes { + config_path = var.kubeconfig_path + config_context = var.kube_context + } +} + +locals { + inline_values = { + replicaCount = var.replica_count + image = { + repository = var.image_repository + tag = var.image_tag + pullPolicy = var.image_pull_policy + } + ingress = { + enabled = var.ingress_enabled + className = var.ingress_class_name + hosts = [ + { + host = var.ingress_host + paths = [ + { + path = "/" + pathType = "Prefix" + } + ] + } + ] + tls = var.ingress_tls_secret_name == "" ? [] : [ + { + secretName = var.ingress_tls_secret_name + hosts = [var.ingress_host] + } + ] + } + workspace = { + persistence = { + enabled = true + storageClassName = var.storage_class_name + size = var.workspace_storage_size + } + } + } +} + +resource "helm_release" "amc" { + name = var.release_name + namespace = var.namespace + chart = var.chart_path + create_namespace = true + atomic = true + cleanup_on_fail = true + lint = true + wait = true + timeout = var.timeout_seconds + + values = concat( + [yamlencode(local.inline_values)], + [for path in var.values_files : file(path)] + ) +} diff --git a/deploy/terraform/helm-release/outputs.tf b/deploy/terraform/helm-release/outputs.tf new file mode 100644 index 000000000..ecf65aae9 --- /dev/null +++ b/deploy/terraform/helm-release/outputs.tf @@ -0,0 +1,14 @@ +output "release_name" { + description = "Helm release name." + value = helm_release.amc.name +} + +output "namespace" { + description = "Kubernetes namespace." + value = helm_release.amc.namespace +} + +output "status" { + description = "Helm release status." + value = helm_release.amc.status +} diff --git a/deploy/terraform/helm-release/terraform.tfvars.example b/deploy/terraform/helm-release/terraform.tfvars.example new file mode 100644 index 000000000..30c4c76a9 --- /dev/null +++ b/deploy/terraform/helm-release/terraform.tfvars.example @@ -0,0 +1,10 @@ +image_repository = "ghcr.io/your-org/amc-studio" +image_tag = "latest" +replica_count = 2 + +# Optional production overrides: +# values_files = ["../../helm/amc/examples/values-ingress-tls.yaml"] +# ingress_enabled = true +# ingress_class_name = "nginx" +# ingress_host = "amc.example.com" +# ingress_tls_secret_name = "amc-tls" diff --git a/deploy/terraform/helm-release/variables.tf b/deploy/terraform/helm-release/variables.tf new file mode 100644 index 000000000..ae3672a86 --- /dev/null +++ b/deploy/terraform/helm-release/variables.tf @@ -0,0 +1,101 @@ +variable "kubeconfig_path" { + type = string + description = "Path to kubeconfig used by the Helm provider." + default = "~/.kube/config" +} + +variable "kube_context" { + type = string + description = "Optional kubeconfig context. Leave empty for the current context." + default = null +} + +variable "release_name" { + type = string + description = "Helm release name." + default = "amc" +} + +variable "namespace" { + type = string + description = "Kubernetes namespace for AMC." + default = "amc-system" +} + +variable "chart_path" { + type = string + description = "Path to the local AMC Helm chart." + default = "../../helm/amc" +} + +variable "values_files" { + type = list(string) + description = "Additional Helm values files, applied after inline Terraform values." + default = [] +} + +variable "image_repository" { + type = string + description = "AMC container image repository." + default = "ghcr.io/your-org/amc-studio" +} + +variable "image_tag" { + type = string + description = "AMC container image tag." + default = "latest" +} + +variable "image_pull_policy" { + type = string + description = "Kubernetes image pull policy." + default = "IfNotPresent" +} + +variable "replica_count" { + type = number + description = "Number of AMC Studio replicas." + default = 2 +} + +variable "workspace_storage_size" { + type = string + description = "Persistent volume size for the AMC workspace." + default = "10Gi" +} + +variable "storage_class_name" { + type = string + description = "Optional Kubernetes StorageClass name." + default = "" +} + +variable "ingress_enabled" { + type = bool + description = "Enable Kubernetes Ingress." + default = false +} + +variable "ingress_class_name" { + type = string + description = "IngressClass name when ingress is enabled." + default = "" +} + +variable "ingress_host" { + type = string + description = "Ingress host when ingress is enabled." + default = "amc.example.com" +} + +variable "ingress_tls_secret_name" { + type = string + description = "Existing TLS secret name. Empty disables chart TLS entries." + default = "" +} + +variable "timeout_seconds" { + type = number + description = "Helm operation timeout in seconds." + default = 600 +} diff --git a/deploy/terraform/helm-release/versions.tf b/deploy/terraform/helm-release/versions.tf new file mode 100644 index 000000000..2323e7d15 --- /dev/null +++ b/deploy/terraform/helm-release/versions.tf @@ -0,0 +1,10 @@ +terraform { + required_version = ">= 1.6.0" + + required_providers { + helm = { + source = "hashicorp/helm" + version = ">= 2.13.0" + } + } +} diff --git a/docs/ACCESSIBILITY.md b/docs/ACCESSIBILITY.md new file mode 100644 index 000000000..b840e12a4 --- /dev/null +++ b/docs/ACCESSIBILITY.md @@ -0,0 +1,48 @@ +# Accessibility + +Last reviewed: 2026-06-16 + +AMC aims to make the CLI, website, generated dashboard, and local Studio usable for keyboard-only users, screen-reader users, low-vision users, and users who disable terminal color. + +## Standards Target + +- Website and generated dashboard: WCAG 2.2 AA-aligned checks, including text contrast, visible focus, skip navigation, and accessible names for chart surfaces. +- CLI: color is supplemental, not the only signal. Use `NO_COLOR=1` or `--no-color` for plain output where supported. +- Generated dashboard: use Settings -> Theme -> High contrast, or cycle the top-right theme control until `HC` appears. +- Generated dashboard heatmaps: score, target, gap, and confidence are exposed as visible text plus ARIA grid, gridcell, selected-state, and meter values; color is supplemental. +- Evidence: accessibility fixes should be backed by repeatable tests, not only visual inspection. + +External standards used for this review: + +- W3C WCAG 2.2 Success Criterion 1.4.1 Use of Color: https://www.w3.org/TR/WCAG22/#use-of-color +- W3C WCAG 2.2 Success Criterion 1.4.3 Contrast (Minimum): https://www.w3.org/TR/WCAG22/#contrast-minimum +- W3C WCAG 2.2 Success Criterion 4.1.2 Name, Role, Value: https://www.w3.org/TR/WCAG22/#name-role-value +- W3C WAI Accessibility Statement guidance: https://www.w3.org/WAI/planning/statements/ +- WAI-ARIA accessible name guidance: https://www.w3.org/TR/accname-1.2/ + +## Current Support + +- Static website pages include skip links and keyboard focus-visible styling. +- Website hero canvas is decorative and marked `aria-hidden="true"`. +- Playground and generated console chart canvases expose `role="img"` plus descriptive `aria-label` text. +- Generated dashboard secondary text on dark backgrounds uses a higher-contrast token than the previously flagged `rgba(244,244,245,.55)`. +- Generated dashboard onboarding traps keyboard focus while open and restores focus when closed. +- Generated dashboard question heatmap rows include visible maturity level, gap, and confidence labels, plus ARIA grid semantics and confidence meters. +- The Playwright accessibility suite uses `@axe-core/playwright` for the core static pages. + +## Known Limitations + +- Interactive terminal prompts depend on the user's terminal and screen reader support. +- Some generated reports include dense tables and code blocks that may require additional screen-reader review with real data. +- Automated axe checks do not replace manual keyboard and assistive-technology testing. + +## Feedback + +Open an issue at https://github.com/AgentMaturity/AgentMaturityCompass/issues with: + +- Page, command, or generated artifact path. +- Assistive technology and browser or terminal used. +- Expected result and actual result. +- Screenshot, terminal output, or axe report when available. + +Accessibility findings should be treated as product defects and should include a regression test where practical. diff --git a/docs/ADAPTERS.md b/docs/ADAPTERS.md index 397df4eb4..200837008 100644 --- a/docs/ADAPTERS.md +++ b/docs/ADAPTERS.md @@ -4,6 +4,8 @@ Adapters are AMC's one-liner integration system. They wrap any AI agent CLI or S Adapters are part of AMC's current TypeScript integration path. They are not a separate runtime. For the architecture overview, read `docs/ARCHITECTURE_BRIEF.md`. For core-versus-wrapper clarity across SDKs, extensions, actions, and legacy paths, read `docs/IMPLEMENTATION_REALITY_MAP.md`. +Need an adapter for a framework AMC does not ship yet? Read `docs/CUSTOM_ADAPTER.md` for the declarative plugin adapter schema, SDK wrapper path, evidence contract, and acceptance checklist. + ## How Adapters Work When you run `amc adapters run`: @@ -84,6 +86,16 @@ const response = await fetchWithAmc("https://api.openai.com/v1/chat/completions" }); ``` +## Mobile Apps (React Native / Flutter) + +Mobile apps should use AMC Bridge over HTTPS, not the Node `wrapFetch` runtime: + +- React Native: use `createReactNativeAMCFetch` from `agent-maturity-compass/sdk/mobile-fetch` when your package manager can consume the mobile-safe subpath, or vendor `src/sdk/mobileFetch.ts`. +- Flutter: call `https:///bridge//...` directly with `authorization: Bearer `, `x-amc-agent-id`, and `x-amc-correlation-id`. +- Keep provider API keys on your backend or in AMC Bridge; do not embed provider keys in mobile apps. + +See `docs/SDK.md#mobile-react-native--flutter` for full examples. + ## Custom SDK Integration For programmatic evidence capture: @@ -175,6 +187,8 @@ Create a runnable local sample for library-based frameworks: amc adapters init-project --agent my-agent --adapter openai-agents-sdk ``` +For a framework not covered by a built-in sample, use `docs/CUSTOM_ADAPTER.md` to choose between a declarative plugin adapter and an SDK wrapper adapter. + ## Lease Compatibility AMC accepts leases via these headers: diff --git a/docs/ADAPTER_COMPATIBILITY.md b/docs/ADAPTER_COMPATIBILITY.md index 4a0995f55..4d01677ef 100644 --- a/docs/ADAPTER_COMPATIBILITY.md +++ b/docs/ADAPTER_COMPATIBILITY.md @@ -61,7 +61,7 @@ amc doctor ## Adding Custom Adapters -AMC supports plugin adapters for frameworks not covered by built-ins. See the [adapter development guide](ARCHITECTURE_MAP.md) for details. +AMC supports plugin adapters for frameworks not covered by built-ins. See the [custom adapter authoring guide](CUSTOM_ADAPTER.md) for the adapter schema, SDK wrapper path, evidence contract, and validation checklist. ## Reporting Issues diff --git a/docs/AGENT_COUNCIL_100_REPORT.md b/docs/AGENT_COUNCIL_100_REPORT.md index 425bf6685..62f2d7cde 100644 --- a/docs/AGENT_COUNCIL_100_REPORT.md +++ b/docs/AGENT_COUNCIL_100_REPORT.md @@ -146,7 +146,7 @@ All produce valid reports with `--json` support: | MITRE_ATLAS | MITRE_ATLAS, mitre | ✅ | | OWASP_API_TOP10 | OWASP_API_TOP10, owasp | ✅ | -### Industry Sector Packs (40 packs across 7 domains) +### Industry Sector Packs (41 packs across 7 domains) All accessible via `amc domain pack list/describe/run`: | Domain | Packs | Status | diff --git a/docs/AGENT_GUIDE.md b/docs/AGENT_GUIDE.md index e1c3142be..456d1c306 100644 --- a/docs/AGENT_GUIDE.md +++ b/docs/AGENT_GUIDE.md @@ -19,7 +19,7 @@ AMC Score → Gap Analysis → Severity Tagging → Guardrails + Agent Instructi Re-score → Diff → Repeat ``` -1. **Score** — AMC scores your agent from execution evidence (195 questions, 5 dimensions) +1. **Score** — AMC scores your agent from execution evidence (244 default questions, 5 dimensions) 2. **Analyze** — Guide identifies every gap between current and target level 3. **Tag** — Each gap gets a severity: 🔮 Critical (gap ≄ 3), 🟡 High (gap ≄ 2), đŸ”” Medium (gap = 1) 4. **Generate** — Produces guardrails (rules), agent instructions (what to do), and human guide (what to fix) diff --git a/docs/AMC_MASTER_REFERENCE.md b/docs/AMC_MASTER_REFERENCE.md index c13ccee22..12e735e08 100644 --- a/docs/AMC_MASTER_REFERENCE.md +++ b/docs/AMC_MASTER_REFERENCE.md @@ -116,8 +116,12 @@ Telemetry is **off by default**. When enabled, only sends: OS, Node version, AMC | Command | Description | |---------|-------------| | `amc run --agent --window ` | Run maturity diagnostic | -| `amc report ` | Render report for a run | -| `amc history` | List diagnostic run history | +| `amc report ` | Render report for a run, saved alias, or latest run | +| `amc report --share --public-base-url ` | Generate a static report share bundle and URL manifest | +| `amc demo prospect` | Run a guided 5-minute prospect demo flow | +| `amc demo share --public-base-url ` | Generate a static DEMO_ONLY prospect leave-behind bundle | +| `amc run-alias set ` | Name a diagnostic run for report and history workflows | +| `amc history` | List diagnostic run history with aliases | | `amc compare ` | Compare two runs | | `amc verify` | Verify integrity across AMC artifacts | | `amc verify all --json` | Full verification in one pass | @@ -425,7 +429,9 @@ Telemetry is **off by default**. When enabled, only sends: OS, Node version, AMC |---------|-------------| | `amc identity init` | Initialize identity config | | `amc identity provider add oidc\|saml` | Add SSO provider | +| `amc sso configure oidc\|saml` | Discoverable SSO setup shortcut | | `amc identity mapping add` | Group-to-role mapping | +| `amc scim init` | Enable SCIM provisioning and optionally create first token | | `amc scim token create` | Create SCIM provisioning token | ## Host Mode (Multi-Workspace) diff --git a/docs/AMC_STANDARD_RFC.md b/docs/AMC_STANDARD_RFC.md index 4d427d2ed..e5f9244fe 100644 --- a/docs/AMC_STANDARD_RFC.md +++ b/docs/AMC_STANDARD_RFC.md @@ -27,6 +27,7 @@ The Agent Maturity Compass (AMC) Standard defines an open, evidence-based framew 8. [Compliance Mapping](#8-compliance-mapping) 9. [Implementation Requirements](#9-implementation-requirements) 10. [Security Considerations](#10-security-considerations) +11. [Citation](#11-citation) - [Appendix A: Question Bank Schema](#appendix-a-question-bank-schema) - [Appendix B: Assurance Pack Schema](#appendix-b-assurance-pack-schema) - [Appendix C: Certificate Format](#appendix-c-certificate-format) @@ -812,6 +813,25 @@ The hash chain prevents retroactive tampering: modifying any historical event ch --- +## 11. Citation + +Use this citation when referring to the AMC Standard. No DOI or arXiv identifier is assigned to this RFC-style specification as of 2026-06-16; cite the repository document until an external identifier is issued. + +```bibtex +@misc{amcstandard2026, + title = {AMC Standard: Agent Maturity Compass, RFC-style Specification v1.0}, + author = {{AMC Labs}}, + year = {2026}, + month = mar, + url = {https://github.com/AgentMaturity/AgentMaturityCompass/blob/main/docs/AMC_STANDARD_RFC.md}, + note = {Repository specification; DOI and arXiv identifier not assigned as of 2026-06-16} +} +``` + +Related whitepaper citation: `whitepaper/AMC_WHITEPAPER_v1.md` includes the companion BibTeX entry for the AMC maturity framework paper. + +--- + ## Appendix A: Question Bank Schema ```typescript diff --git a/docs/API_REFERENCE.md b/docs/API_REFERENCE.md index 49236204c..78bd8d89a 100644 --- a/docs/API_REFERENCE.md +++ b/docs/API_REFERENCE.md @@ -1,6 +1,6 @@ # AMC API Reference -> Auto-generated from source on 2026-03-10 +> Auto-generated from source on 2026-06-16 ## Table of Contents @@ -13,4390 +13,9099 @@ ## CLI Commands -AMC provides 842 CLI commands organized into subcommand groups. +AMC provides 1,140 public CLI command paths in the live command inventory. | # | Command | Description | |---|---------|-------------| -| 1 | `help [commandPath...]` | Show help for a command (for example: amc help run) | -| 2 | `init` | Initialize .amc workspace | -| 3 | `doctor` | Check runtime availability and wrap readiness | -| 4 | `doctor-fix` | Auto-repair common setup issues | -| 5 | `improve` | Guided improvement — shows what to fix next based on your current score | -| 6 | `guide` | Generate personalized improvement guide with exportable agent instructions | -| 7 | `quickscore` | Zero-config rapid assessment — auto-scores from evidence, or interactive 5-question fallback | -| 8 | `explain ` | Plain-English explanation for a diagnostic question (example: AMC-2.1) | -| 9 | `bootstrap` | Bootstrap workspace for production deployment (non-interactive) | -| 10 | `up` | Start AMC control plane in one command (studio + gateway + bridge) | -| 11 | `host` | Multi-workspace host mode operations | -| 12 | `init` | Initialize host metadata database | -| 13 | `bootstrap` | Bootstrap host admin + default workspace from secret files | -| 14 | `user` | Host user management | -| 15 | `workspace` | Host workspace lifecycle | -| 16 | `migrate` | Migrate an existing single-workspace AMC directory into host mode | -| 17 | `membership` | Host membership management | -| 18 | `list` | List host users and workspaces | -| 19 | `down` | Stop AMC Studio local control plane | -| 20 | `status` | Show AMC Studio and vault status | -| 21 | `config` | Inspect resolved runtime configuration | -| 22 | `print` | Print resolved runtime config (secret-safe) | -| 23 | `explain` | Explain config source precedence and risky settings | -| 24 | `logs` | Print latest AMC Studio logs | -| 25 | `studio` | Studio API helpers | -| 26 | `ping` | Ping local Studio API /health endpoint | -| 27 | `start` | Start Studio in foreground (non-interactive, deployment-safe) | -| 28 | `healthcheck` | Health/readiness probe for deployment runtime | -| 29 | `lan` | LAN mode controls for Compass Console | -| 30 | `enable` | Enable LAN mode with pairing gate | -| 31 | `disable` | Disable LAN mode and revert to localhost-only | -| 32 | `connect` | Connect wizard for any agent/provider runtime | -| 33 | `adapters` | Built-in adapter system for one-line agent integration | -| 34 | `init` | Create signed adapters.yaml defaults | -| 35 | `verify` | Verify adapters.yaml signature | -| 36 | `list` | List built-in adapters and per-agent preferences | -| 37 | `detect` | Detect installed adapter runtimes and versions | -| 38 | `configure` | Set adapter profile for an agent (signed adapters.yaml) | -| 39 | `env` | Print adapter-compatible environment exports without lease token | -| 40 | `init-project` | Generate runnable local adapter sample for library-based frameworks | -| 41 | `run` | Run adapter with minted lease, routed through gateway, with observed evidence capture | -| 42 | `plugin` | Signed content-only extension marketplace | -| 43 | `keygen` | Generate plugin publisher keypair | -| 44 | `pack` | Create signed .amcplug package from a plugin folder | -| 45 | `verify` | Verify plugin package signature + artifact hashes | -| 46 | `print` | Print plugin manifest summary | -| 47 | `init` | Initialize signed plugin workspace files | -| 48 | `workspace-verify` | Verify workspace plugin signatures/integrity | -| 49 | `list` | List installed plugins and verification status | -| 50 | `registry` | Manage plugin registries | -| 51 | `init` | Initialize local signed plugin registry directory | -| 52 | `publish` | Publish plugin package into registry and re-sign index | -| 53 | `verify` | Verify registry signature and package hashes | -| 54 | `serve` | Serve plugin registry over local HTTP | -| 55 | `search` | Search a plugin registry by id/fingerprint | -| 56 | `registries` | List signed workspace registry configuration | -| 57 | `registries-apply` | Apply and sign workspace registries.yaml from JSON or YAML file | -| 58 | `install` | Request plugin install (requires SECURITY dual-control approval) | -| 59 | `upgrade` | Request plugin upgrade (requires SECURITY dual-control approval) | -| 60 | `remove` | Request plugin removal (requires SECURITY dual-control approval) | -| 61 | `execute` | Execute approved plugin install/upgrade/remove request | -| 62 | `registry-fingerprint` | Compute registry public key fingerprint | -| 63 | `wrap` | Wrap runtime and capture tamper-evident evidence | -| 64 | `supervise` | Supervise any agent process and inject gateway/proxy routing env vars | -| 65 | `monitor` | Continuous production monitoring — real-time scoring, drift detection, and alerting | -| 66 | `start` | Start continuous monitoring: scores agent at intervals, detects drift, sends alerts on degradation | -| 67 | `check` | One-shot trust drift analysis (check for degradation without running continuously) | -| 68 | `status` | Show monitoring status for all agents | -| 69 | `events` | Show recent monitoring events | -| 70 | `metrics` | Get metrics for a specific agent | -| 71 | `run` | Run maturity diagnostic | -| 72 | `report` | Render report for run ID | -| 73 | `history` | List diagnostic run history | -| 74 | `compare` | Compare two runs | -| 75 | `verify` | Verify integrity across AMC artifacts | -| 76 | `all` | Verify trust/policies/plugins/logs/ledger/artifacts in one pass | -| 77 | `target` | Target profile operations | -| 78 | `eval` | Eval interop import and coverage status | -| 79 | `evidence` | Evidence lifecycle workflows | -| 80 | `incidents` | Incident operations and dispatch workflows | -| 81 | `policy` | Policy-as-code operations | -| 82 | `governor` | Autonomy Governor checks | -| 83 | `tools` | ToolHub tools config | -| 84 | `workorder` | Signed work order operations | -| 85 | `ticket` | Execution ticket operations | -| 86 | `gateway` | AMC universal LLM proxy gateway | -| 87 | `bundle` | Portable evidence bundle operations | -| 88 | `ci` | CI/CD release gate helpers | -| 89 | `archetype` | Archetype packs | -| 90 | `export` | Export policy packs and badges | -| 91 | `assurance` | Assurance Lab red-team packs | -| 92 | `toctou` | Run TOCTOU assurance pack | -| 93 | `compound-threats` | Run compound threat assurance pack | -| 94 | `shutdown-compliance` | Run shutdown compliance pack | -| 95 | `advanced-threats` | Run advanced threats assurance pack | -| 96 | `cert` | Certificate operations | -| 97 | `dashboard` | Device-first Compass dashboard | -| 98 | `vault` | Encrypted key vault operations | -| 99 | `notary` | AMC Notary signing boundary operations | -| 100 | `trust` | Trust mode and Notary enforcement configuration | -| 101 | `canon` | Compass Canon signed content operations | -| 102 | `cgx` | Context Graph (CGX) build and verify operations | -| 103 | `diagnostic` | Diagnostic bank/render operations | -| 104 | `truthguard` | Deterministic output truth-constraint validator | -| 105 | `mode` | Switch CLI role mode | -| 106 | `loop` | Continuous self-serve maturity loop | -| 107 | `user` | Multi-user RBAC account management | -| 108 | `identity` | Enterprise identity (OIDC/SAML) configuration | -| 109 | `scim` | SCIM token management | -| 110 | `pair` | LAN pairing code operations | -| 111 | `transparency` | Append-only transparency log operations | -| 112 | `compliance` | Evidence-linked compliance map operations | -| 113 | `federate` | Offline federation sync operations | -| 114 | `integrations` | Integration hub operations | -| 115 | `outcomes` | Outcome contracts, value signals, and reports | -| 116 | `value` | Value realization engine (contracts, scoring, ROI) | -| 117 | `audit` | Audit binder and compliance maps | -| 118 | `admin` | Administrative controls, identity, and trust operations | -| 119 | `passport` | Agent Passport (shareable maturity credential) | -| 120 | `standard` | Open Compass Standard schema bundle and validation | -| 121 | `forecast` | Deterministic evidence-gated forecasting and planning | -| 122 | `advisory` | Forecast advisories (list/show/ack) | -| 123 | `casebook` | Signed casebook operations | -| 124 | `incident` | Incident tracking and response operations | -| 125 | `experiment` | Deterministic baseline vs candidate experiments | -| 126 | `release` | Deterministic release engineering and offline verification | -| 127 | `ops` | Operational hardening policy controls | -| 128 | `blobs` | Encrypted evidence blob operations | -| 129 | `retention` | Retention/archive payload lifecycle operations | -| 130 | `backup` | Signed encrypted backup/restore operations | -| 131 | `maintenance` | Operational maintenance operations | -| 132 | `metrics` | Prometheus metrics endpoint helpers | -| 133 | `lifecycle` | Agent lifecycle responsibility and governance mapping | -| 134 | `merkle` | Merkle transparency root/proof operations | -| 135 | `action` | Signed autonomy action policy | -| 136 | `approval` | Signed dual-control approval policy | -| 137 | `pack` | Policy packs by archetype and risk tier | -| 138 | `import` | Import eval outputs (LangSmith, DeepEval, Promptfoo, OpenAI Evals, W&B, Langfuse) into signed AMC evidence | -| 139 | `status` | Show imported eval coverage per AMC dimension | -| 140 | `run` | One-shot evaluation: read amcconfig.yaml, run all diagnostic tests, output results | -| 141 | `status` | Show lifecycle stage, accountability matrix, governance gates, and transition trail | -| 142 | `advance` | Advance lifecycle stage after governance gate confirmation | -| 143 | `help` | Show high-signal evidence command groups | -| 144 | `collect` | Guided wizard to connect your agent and capture evidence | -| 145 | `verify` | Run full workspace verification suite | -| 146 | `help` | Show incident-focused command groups | -| 147 | `alert` | Dispatch INCIDENT_CREATED to configured integration channels | -| 148 | `help` | Show admin-focused command groups | -| 149 | `status` | Show operational admin status for control-plane services | -| 150 | `init` | Create and sign .amc/action-policy.yaml | -| 151 | `verify` | Verify action policy signature | -| 152 | `init` | Create and sign .amc/approval-policy.yaml | -| 153 | `verify` | Verify approval-policy signature | -| 154 | `list` | List built-in policy packs | -| 155 | `describe` | Describe policy pack contents | -| 156 | `diff` | Show deterministic diff for applying a policy pack | -| 157 | `apply` | Apply policy pack and sign updated configs/targets | -| 158 | `list` | List incidents for an agent | -| 159 | `show ` | Show incident details | -| 160 | `create` | Create a manual incident | -| 161 | `link ` | Link evidence to an incident | -| 162 | `close ` | Close an incident with a resolution summary | -| 163 | `init` | Create and sign .amc/ops-policy.yaml | -| 164 | `verify` | Verify ops-policy signature | -| 165 | `print` | Print effective ops policy | -| 166 | `circuit-breaker-init` | Initialize circuit breaker policy | -| 167 | `circuit-breaker-status` | Show circuit breaker status | -| 168 | `circuit-breaker-reset` | Reset all circuit breakers | -| 169 | `dead-letters` | Show dead letter queue | -| 170 | `mode` | Show or set degradation mode | -| 171 | `backpressure` | Show backpressure pipeline health | -| 172 | `slo` | Show governance SLO dashboard | -| 173 | `latency` | Show latency accounting report | -| 174 | `init` | Create and sign .amc/canon/canon.yaml | -| 175 | `verify` | Verify canonical compass content signature | -| 176 | `print` | Print effective Compass Canon | -| 177 | `init` | Create and sign .amc/cgx/policy.yaml | -| 178 | `build` | Build deterministic signed context graph | -| 179 | `verify` | Verify CGX policy/graph/pack signatures | -| 180 | `show` | Show latest CGX graph or agent context pack | -| 181 | `simulate` | Simulate impact propagation when a node changes | -| 182 | `diff` | Diff two CGX graph snapshots | -| 183 | `delta-to-l5` | Generate L4→L5 delta report showing what separates current state from L5 | -| 184 | `control-classification` | Show control enforcement classification (ARCHITECTURAL/POLICY_ENFORCED/CONVENTION) | -| 185 | `prompt` | Northstar prompt policy + pack operations | -| 186 | `init` | Create and sign .amc/prompt/policy.yaml | -| 187 | `verify` | Verify prompt policy, pack, lint and scheduler signatures | -| 188 | `policy` | Prompt policy operations | -| 189 | `print` | Print prompt policy | -| 190 | `apply` | Apply prompt policy from YAML file and sign | -| 191 | `status` | List per-agent prompt pack status | -| 192 | `pack` | Prompt pack artifact operations | -| 193 | `build` | Build and sign .amcprompt for an agent | -| 194 | `verify` | Verify .amcprompt signature and lint signature | -| 195 | `show` | Show provider-specific enforced system prompt | -| 196 | `diff` | Diff latest prompt pack against previous snapshot | -| 197 | `scheduler` | Prompt pack recurrence scheduler | -| 198 | `status` | Show prompt scheduler status | -| 199 | `run-now` | Run prompt scheduler now for one agent or all | -| 200 | `enable` | Enable prompt scheduler | -| 201 | `disable` | Disable prompt scheduler | -| 202 | `init` | Create and sign .amc/passport/policy.yaml | -| 203 | `verify-policy` | Verify signed passport policy | -| 204 | `policy` | Passport policy operations | -| 205 | `print` | Print effective passport policy | -| 206 | `apply` | Apply passport policy from JSON/YAML file | -| 207 | `create` | Create deterministic signed .amcpass artifact | -| 208 | `verify` | Verify .amcpass artifact offline | -| 209 | `show` | Show .amcpass as JSON or single-line badge | -| 210 | `badge` | Print deterministic single-line badge from latest cache | -| 211 | `export-latest` | Export latest passport for a scope to .amcpass | -| 212 | `share` | Generate shareable passport material | -| 213 | `compare` | Compare two agents by passport maturity dimensions | -| 214 | `generate` | Generate signed Open Compass schema bundle under .amc/standard/ | -| 215 | `verify` | Verify schema bundle signatures and manifest digests | -| 216 | `print` | Print one generated schema | -| 217 | `validate` | Validate a JSON file or AMC artifact against a standard schema | -| 218 | `schemas` | List generated schemas with digests | -| 219 | `bank` | Signed diagnostic 126-question bank operations | -| 220 | `init` | Create and sign .amc/diagnostic/bank/bank.yaml | -| 221 | `verify` | Verify diagnostic bank signature | -| 222 | `render` | Render contextualized 126-question diagnostic for an agent | -| 223 | `validate` | Validate structured agent output claims against deterministic truth constraints | -| 224 | `verify` | Verify encrypted blob index and payload integrity | -| 225 | `key` | Blob key management | -| 226 | `init` | Initialize encrypted blob key material | -| 227 | `rotate` | Rotate encrypted blob key material | -| 228 | `reencrypt` | Re-encrypt blob batch from one key version to another | -| 229 | `status` | Show retention/archive status | -| 230 | `run` | Run archival + payload prune lifecycle | -| 231 | `verify` | Verify archive manifests/signatures and ledger continuity | -| 232 | `create` | Create signed encrypted backup bundle | -| 233 | `verify` | Verify signed backup bundle offline | -| 234 | `restore` | Restore a verified backup into target directory | -| 235 | `print` | Print backup manifest summary | -| 236 | `stats` | Show DB/blob/archive/cache operational stats | -| 237 | `vacuum` | Run SQLite VACUUM + ANALYZE | -| 238 | `reindex` | Ensure operational SQLite indexes | -| 239 | `rotate-logs` | Rotate Studio logs based on ops policy | -| 240 | `prune-cache` | Prune dashboard/console/transform cache artifacts | -| 241 | `status` | Show configured metrics endpoint bind/port | -| 242 | `check` | Evaluate whether an action is allowed now (simulate vs execute) | -| 243 | `explain` | Explain policy requirements for an action class | -| 244 | `report` | Render matrix of current SIMULATE/EXECUTE allowance per ActionClass | -| 245 | `init` | Create and sign .amc/tools.yaml | -| 246 | `verify` | Verify tools.yaml signature | -| 247 | `list` | List allowed ToolHub tools and action classes | -| 248 | `create` | Create and sign a work order | -| 249 | `list` | List work orders for agent | -| 250 | `show` | Show signed work order JSON | -| 251 | `verify` | Verify work order signature | -| 252 | `expire` | Expire/revoke a work order | -| 253 | `issue` | Issue short-lived signed execution ticket | -| 254 | `verify` | Verify signed execution ticket | -| 255 | `init` | Create and sign .amc/gateway.yaml | -| 256 | `start` | Start local reverse-proxy gateway and signed evidence capture | -| 257 | `status` | Check gateway reachability and route URLs | -| 258 | `verify-config` | Verify .amc/gateway.yaml signature | -| 259 | `bind-agent` | Bind a gateway route prefix to an agent ID for deterministic attribution | -| 260 | `export` | Export a portable, signed evidence bundle for a run | -| 261 | `verify` | Verify evidence bundle offline | -| 262 | `inspect` | Inspect bundle metadata | -| 263 | `diff` | Diff two bundles (maturity/integrity/targets) | -| 264 | `export` | Export verifier-ready evidence (json|csv|pdf) | -| 265 | `audit-packet` | Generate external-auditor packet with verifier-ready evidence | -| 266 | `gate` | Evaluate a run bundle against a signed gate policy | -| 267 | `init` | Generate GitHub workflow and signed gate policy | -| 268 | `print` | Print suggested CI pipeline steps | -| 269 | `check` | One-liner CI gate: quickscore + threshold check (exit 1 if below) | -| 270 | `list` | List built-in archetype packs | -| 271 | `describe` | Describe an archetype | -| 272 | `apply` | Apply archetype context/targets/guardrails/evals to an agent | -| 273 | `policy` | Export framework-agnostic North Star policy integration pack | -| 274 | `badge` | Export deterministic maturity badge SVG for a run | -| 275 | `build` | Build responsive offline dashboard for an agent | -| 276 | `serve` | Serve dashboard locally | -| 277 | `list` | List available assurance packs | -| 278 | `describe` | Describe assurance pack details | -| 279 | `init` | Initialize signed assurance policy | -| 280 | `verify-policy` | Verify assurance policy signature | -| 281 | `policy` | Print current assurance policy | -| 282 | `policy-apply` | Apply assurance policy from YAML/JSON file | -| 283 | `run` | Run assurance pack(s) with deterministic validation | -| 284 | `runs` | List assurance lab runs | -| 285 | `show` | Show assurance run artifacts | -| 286 | `cert-issue` | Issue signed assurance certificate for a run | -| 287 | `cert-verify` | Verify assurance certificate bundle offline | -| 288 | `scheduler` | Assurance scheduler controls | -| 289 | `status` | Show scheduler status | -| 290 | `run-now` | Run assurance scheduler immediately | -| 291 | `enable` | Enable assurance scheduler | -| 292 | `disable` | Disable assurance scheduler | -| 293 | `waiver` | Assurance threshold waiver controls | -| 294 | `request` | Request time-limited readiness waiver (dual-control approval required) | -| 295 | `status` | Show waiver status (activates approved pending waivers) | -| 296 | `revoke` | Revoke active or specific waiver | -| 297 | `history` | List assurance run history | -| 298 | `verify` | Verify assurance run determinism and signatures | -| 299 | `patch` | Apply deterministic patch kit for failed assurance findings | -| 300 | `certify` | Issue signed, offline-verifiable certificate bundle | -| 301 | `generate` | Generate execution-proof trust certificate (signed PDF or JSON) | -| 302 | `verify` | Verify certificate bundle offline | -| 303 | `inspect` | Inspect certificate bundle contents | -| 304 | `revoke` | Create signed revocation file for a certificate | -| 305 | `verify-revocation` | Verify revocation file signature | -| 306 | `init` | Initialize encrypted vault for signing keys | -| 307 | `unlock` | Unlock vault into memory for signing operations | -| 308 | `lock` | Lock vault and clear in-memory private keys | -| 309 | `status` | Show vault status | -| 310 | `rotate-keys` | Rotate monitor signing key and append to public key history | -| 311 | `init` | Initialize AMC Notary config and signing backend | -| 312 | `start` | Start AMC Notary service (foreground) | -| 313 | `status` | Show notary backend and log status | -| 314 | `pubkey` | Print notary public key and fingerprint | -| 315 | `attest` | Generate signed notary runtime attestation bundle (.amcattest) | -| 316 | `verify-attest` | Verify a .amcattest bundle offline | -| 317 | `sign` | Sign a payload file using Notary (admin utility) | -| 318 | `log-verify` | Verify notary append-only signing log + seal signature | -| 319 | `init` | Create and sign .amc/trust.yaml | -| 320 | `enable-notary` | Enable fail-closed NOTARY trust mode | -| 321 | `status` | Show trust mode, signature status, and notary health | -| 322 | `freshness` | Report temporal trust freshness and half-life decay | -| 323 | `owner` | Switch to owner mode (configuration + signing allowed) | -| 324 | `agent` | Switch to agent mode (read-only / self-check commands) | -| 325 | `init` | Initialize signed users.yaml with first OWNER user | -| 326 | `add` | Add a user with RBAC roles | -| 327 | `list` | List RBAC users | -| 328 | `revoke` | Revoke a user account | -| 329 | `role` | Set user roles | -| 330 | `set` | Replace roles for a user | -| 331 | `verify` | Verify users.yaml signature | -| 332 | `init` | Create and sign host-level identity.yaml | -| 333 | `verify` | Verify identity.yaml signature | -| 334 | `provider` | Identity provider management | -| 335 | `add` | Add an identity provider | -| 336 | `mapping` | Signed group-to-role mapping rules | -| 337 | `add` | Add a group mapping rule | -| 338 | `token` | SCIM bearer token operations | -| 339 | `create` | Create a SCIM bearer token and store hash in host vault | -| 340 | `create` | Create one-time pairing code (LAN login pairing or agent bridge pairing) | -| 341 | `redeem` | Redeem pairing code for a lease token file | -| 342 | `init` | Initialize append-only transparency log | -| 343 | `verify` | Verify transparency chain + seal signature | -| 344 | `tail` | Tail transparency entries | -| 345 | `export` | Export transparency bundle | -| 346 | `verify-bundle` | Verify exported transparency bundle | -| 347 | `rebuild` | Rebuild Merkle leaves/roots from transparency log | -| 348 | `root` | Show current Merkle root and history | -| 349 | `prove` | Export signed inclusion proof bundle for entry hash | -| 350 | `verify-proof` | Verify signed inclusion proof bundle | -| 351 | `init` | Create and sign compliance-maps.yaml | -| 352 | `verify` | Verify compliance maps signature | -| 353 | `report` | Generate evidence-linked compliance report | -| 354 | `fleet` | Generate fleet compliance summary | -| 355 | `diff` | Diff two compliance report JSON files | -| 356 | `init` | Initialize federation identity and signed config | -| 357 | `verify` | Verify federation config signature | -| 358 | `peer` | Federation peer trust anchors | -| 359 | `add` | Add a peer publisher public key | -| 360 | `list` | List federation peers | -| 361 | `export` | Export offline federation sync package (.amcfed) | -| 362 | `import` | Import and verify federation package | -| 363 | `verify-bundle` | Verify .amcfed package | -| 364 | `init` | Create and sign integrations.yaml with vault-backed secret refs | -| 365 | `verify` | Verify integrations config signature | -| 366 | `status` | Show integration channels and routing | -| 367 | `test` | Dispatch deterministic test event to an integration channel | -| 368 | `dispatch` | Dispatch a deterministic integration event | -| 369 | `export-journal` | Export integration delivery journal (receipts + dead letters) | -| 370 | `init` | Create and sign outcome contract | -| 371 | `verify` | Verify outcome contract signature | -| 372 | `report` | Generate outcomes report (agent) or fleet outcomes report | -| 373 | `diff` | Diff two outcome reports | -| 374 | `attest` | Record manual attested outcome signal | -| 375 | `init` | Initialize signed value policy, default contract, and scheduler | -| 376 | `verify-policy` | Verify signed value policy | -| 377 | `policy` | Value policy operations | -| 378 | `contract` | Value contract operations | -| 379 | `scheduler` | Value scheduler controls | -| 380 | `print` | Print effective value policy JSON | -| 381 | `default` | Print default value policy JSON | -| 382 | `apply` | Apply signed value policy from YAML/JSON file | -| 383 | `init` | Create and sign value contract template | -| 384 | `apply` | Apply value contract from YAML/JSON file | -| 385 | `verify` | Verify value contract signature | -| 386 | `print` | Print value contract and signature status | -| 387 | `ingest` | Ingest value webhook payload JSON | -| 388 | `import` | Import numeric KPI points from CSV (ts,value) | -| 389 | `snapshot` | Generate/load latest signed value snapshot | -| 390 | `report` | Generate signed value report | -| 391 | `status` | Show value scheduler status | -| 392 | `run-now` | Run value scheduler now | -| 393 | `enable` | Enable value scheduler | -| 394 | `disable` | Disable value scheduler | -| 395 | `verify` | Verify value workspace signatures/artifacts | -| 396 | `init` | Create and sign forecast policy | -| 397 | `verify` | Verify forecast policy signature | -| 398 | `print-policy` | Print effective forecast policy | -| 399 | `latest` | Render latest forecast for scope | -| 400 | `refresh` | Refresh forecast snapshot for scope | -| 401 | `scheduler` | Forecast renewal scheduler controls | -| 402 | `status` | Show scheduler status | -| 403 | `run-now` | Run scheduler refresh immediately | -| 404 | `enable` | Enable forecast scheduler | -| 405 | `disable` | Disable forecast scheduler | -| 406 | `policy` | Forecast policy operations | -| 407 | `apply` | Apply and sign forecast policy from file | -| 408 | `default` | Print default forecast policy JSON | -| 409 | `list` | List advisories for scope | -| 410 | `show` | Show one advisory by ID | -| 411 | `ack` | Acknowledge an advisory | -| 412 | `init` | Create a signed casebook | -| 413 | `add` | Add signed case from existing workorder | -| 414 | `list` | List casebooks | -| 415 | `verify` | Verify signed casebook and case files | -| 416 | `create` | Create an experiment | -| 417 | `set-baseline` | Set experiment baseline config | -| 418 | `set-candidate` | Set experiment candidate signed config overlay | -| 419 | `run` | Run deterministic experiment against signed casebook | -| 420 | `analyze` | Analyze latest experiment run | -| 421 | `gate` | Evaluate latest experiment run against gate policy | -| 422 | `gate-template` | Write an experiment gate policy template | -| 423 | `list` | List experiments | -| 424 | `fix-signatures` | Verify and re-sign gateway/fleet/agent configs | -| 425 | `init` | Initialize recurring loop config | -| 426 | `run` | Run recurring diagnostic + assurance + dashboard + snapshot | -| 427 | `plan` | Print recurring loop plan | -| 428 | `schedule` | Print OS scheduler config (no automatic installation) | -| 429 | `snapshot` | Generate Unified Clarity Snapshot markdown | -| 430 | `indices` | Compute deterministic failure-risk indices | -| 431 | `fleet` | Compute failure-risk indices across fleet | -| 432 | `fleet` | Fleet operations | -| 433 | `agent` | Agent registry operations | -| 434 | `provider` | Provider template operations | -| 435 | `sandbox` | Hardened sandbox execution | -| 436 | `init` | Create and sign .amc/fleet.yaml | -| 437 | `report` | Generate fleet maturity report (md) or fleet compliance report (pdf) | -| 438 | `health` | Show fleet health dashboard aggregates | -| 439 | `policy` | Fleet governance policy operations | -| 440 | `slo` | Fleet governance SLO operations | -| 441 | `apply` | Apply a governance policy to all fleet agents or one environment | -| 442 | `list` | List effective fleet governance policies | -| 443 | `tag` | Tag an agent with an environment | -| 444 | `define` | | -| 445 | `status` | Show fleet SLO compliance status | -| 446 | `list` | List fleet SLO definitions | -| 447 | `trust-init` | Initialize trust composition config | -| 448 | `trust-add-edge` | Add a delegation edge (orchestrator → worker) | -| 449 | `trust-remove-edge` | Remove a delegation edge | -| 450 | `trust-edges` | List all delegation edges | -| 451 | `trust-report` | Generate trust composition report across fleet | -| 452 | `trust-receipts` | Verify cross-agent receipt chains | -| 453 | `dag` | Visualize orchestration delegation graph | -| 454 | `trust-mode` | Set trust inheritance policy mode | -| 455 | `handoff` | Manage handoff packets | -| 456 | `contradictions` | Detect cross-agent contradictions | -| 457 | `add` | Interactively add an agent to the fleet | -| 458 | `list` | List fleet agents | -| 459 | `remove` | Remove an agent from the fleet | -| 460 | `use` | Set current agent | -| 461 | `diagnose` | Lease-auth self-run diagnostic (agent-triggered, evidence-scored server-side) | -| 462 | `list` | List provider templates | -| 463 | `add` | Assign or update provider template for an agent | -| 464 | `run` | Run agent command in hardened Docker sandbox | -| 465 | `ingest` | Ingest external logs/transcripts as SELF_REPORTED evidence | -| 466 | `attest` | Auditor-attest an ingest session to upgrade trust tier to ATTESTED | -| 467 | `set` | Interactive equalizer wizard | -| 468 | `verify` | Verify target profile signature | -| 469 | `diff` | Diff run against target profile | -| 470 | `learn` | Education flow for a specific maturity question | -| 471 | `lineage-init` | Initialize governance lineage tables | -| 472 | `lineage-report` | Generate governance lineage report | -| 473 | `lineage-claim` | Show full governance lineage for a specific claim | -| 474 | `lineage-policy-intents` | List all policy change intents for an agent | -| 475 | `claim-confidence` | Generate per-claim confidence report with citation-backed scoring | -| 476 | `claim-confidence-gate` | Check if claims for given questions pass confidence threshold | -| 477 | `overhead-report` | Generate per-feature overhead accounting report | -| 478 | `overhead-profile` | Set the overhead mode profile (STRICT, BALANCED, LEAN) | -| 479 | `micro-canary-run` | Run all micro-canary probes immediately | -| 480 | `micro-canary-report` | Generate micro-canary status report | -| 481 | `micro-canary-alerts` | Show active micro-canary alerts | -| 482 | `experiment-architecture` | Run a controlled architecture comparison experiment | -| 483 | `experiment-architecture-probes` | List the standard probe set for architecture experiments | -| 484 | `canary-start` | Start a policy canary with candidate vs stable policy | -| 485 | `canary-status` | Show current canary status and stats | -| 486 | `canary-stop` | Stop the active canary | -| 487 | `canary-report` | Generate full policy canary report | -| 488 | `rollback-create` | Create a rollback pack from the current policy file | -| 489 | `emergency-override` | Activate an emergency policy override with strict TTL | -| 490 | `policy-debt-add` | Register a temporary policy waiver (debt) | -| 491 | `policy-debt-list` | List active policy debt entries | -| 492 | `governance-drift` | Detect governance drift for an agent | -| 493 | `cgx-integrity` | Run graph integrity check on CGX with semantic overlay | -| 494 | `cgx-propagation` | Simulate risk propagation from a source node | -| 495 | `memory-extract` | Extract lessons from verified effective corrections | -| 496 | `memory-advisories` | Show advisories from correction memory for prompt injection | -| 497 | `memory-report` | Generate correction memory report | -| 498 | `memory-expire` | Expire stale lessons past their TTL | -| 499 | `own` | Ownership flow for top maturity gaps | -| 500 | `commit` | Commitment plan flow (7/14/30-day checklist) | -| 501 | `tune` | Mechanic mode tuning wizard | -| 502 | `upgrade` | Generate upgrade plan | -| 503 | `guard` | Guard check proposed output from stdin | -| 504 | `lease` | Issue/verify/revoke short-lived agent leases | -| 505 | `issue` | | -| 506 | `verify` | | -| 507 | `revoke` | | -| 508 | `budgets` | Signed autonomy and usage budgets | -| 509 | `init` | | -| 510 | `verify` | | -| 511 | `status` | | -| 512 | `reset` | | -| 513 | `drift` | Drift/regression detection and reporting | -| 514 | `check` | | -| 515 | `report` | | -| 516 | `freeze` | Execution freeze status and controls | -| 517 | `status` | | -| 518 | `lift` | | -| 519 | `alerts` | Signed drift alert configuration and dispatch | -| 520 | `init` | | -| 521 | `verify` | | -| 522 | `test` | | -| 523 | `bom` | Maturity Bill of Materials | -| 524 | `generate` | | -| 525 | `sign` | | -| 526 | `verify` | | -| 527 | `approvals` | Signed approval inbox operations | -| 528 | `list` | | -| 529 | `show` | | -| 530 | `approve` | | -| 531 | `deny` | | -| 532 | `whatif` | Equalizer what-if simulator | -| 533 | `targets` | | -| 534 | `equalizer` | | -| 535 | `transform` | Transformation OS (4C plans, tracking, attestations) | -| 536 | `init` | Initialize signed .amc/transform-map.yaml | -| 537 | `verify` | Verify signed transform map | -| 538 | `map` | Inspect or apply transform map | -| 539 | `show` | | -| 540 | `apply` | | -| 541 | `plan` | | -| 542 | `status` | | -| 543 | `track` | | -| 544 | `report` | | -| 545 | `attest` | | -| 546 | `attest-verify` | | -| 547 | `org` | Org graph and real-time comparative scorecards | -| 548 | `init` | | -| 549 | `verify` | Verify signed org.yaml | -| 550 | `add` | | -| 551 | `node` | | -| 552 | `assign` | | -| 553 | `unassign` | | -| 554 | `score` | | -| 555 | `report` | | -| 556 | `compare` | | -| 557 | `learn` | | -| 558 | `own` | | -| 559 | `commit` | | -| 560 | `init` | Initialize signed audit policy and compliance maps | -| 561 | `verify-policy` | Verify signed audit policy | -| 562 | `export` | Export enterprise audit logs for Splunk, Datadog, CloudTrail, or Azure Monitor | -| 563 | `policy` | Audit binder policy operations | -| 564 | `map` | Audit compliance map operations | -| 565 | `binder` | Audit binder artifact operations | -| 566 | `request` | Audit evidence request operations | -| 567 | `scheduler` | Audit binder cache scheduler | -| 568 | `print` | Print effective audit policy | -| 569 | `apply` | Apply and sign audit policy from file | -| 570 | `list` | List builtin/active audit maps | -| 571 | `show` | Show audit map | -| 572 | `apply` | Apply active audit map from file | -| 573 | `verify` | Verify builtin and active map signatures | -| 574 | `create` | Create deterministic signed .amcaudit artifact | -| 575 | `verify` | Verify .amcaudit file | -| 576 | `list` | List exported binders and cached workspace binder | -| 577 | `export-request` | Create dual-control approval request for external binder sharing | -| 578 | `export-execute` | Execute previously approved external binder export | -| 579 | `create` | Create auditor evidence request | -| 580 | `list` | List audit evidence requests | -| 581 | `approve` | Owner approves request (starts dual-control approval flow) | -| 582 | `reject` | Reject evidence request | -| 583 | `fulfill` | Fulfill approved evidence request by exporting restricted binder | -| 584 | `status` | Show audit scheduler status | -| 585 | `run-now` | Run audit binder cache refresh immediately | -| 586 | `enable` | Enable audit scheduler | -| 587 | `disable` | Disable audit scheduler | -| 588 | `verify` | Verify audit workspace signatures/artifacts | -| 589 | `bench` | Public benchmark registry + ecosystem comparative view | -| 590 | `init` | Initialize signed bench policy | -| 591 | `verify-policy` | Verify signed bench policy | -| 592 | `print-policy` | Print effective bench policy | -| 593 | `create` | Create deterministic signed .amcbench artifact | -| 594 | `verify` | Verify .amcbench artifact offline | -| 595 | `print` | Print bench manifest summary without modification | -| 596 | `registry` | Manage static bench registries | -| 597 | `init` | | -| 598 | `publish` | | -| 599 | `verify` | | -| 600 | `serve` | | -| 601 | `search` | Browse a bench registry index | -| 602 | `import` | Import one bench artifact from allowlisted registry | -| 603 | `list-imports` | List imported bench artifacts | -| 604 | `list-exports` | List locally exported bench artifacts | -| 605 | `compare` | Compute local vs imported ecosystem comparison | -| 606 | `comparison-latest` | Read latest bench comparison artifact | -| 607 | `registries` | Print signed bench registry allowlist | -| 608 | `registries-apply` | Apply bench registries config from JSON file | -| 609 | `publish` | Dual-control bench publish flow | -| 610 | `request` | | -| 611 | `execute` | | -| 612 | `benchmark` | Signed ecosystem benchmark snapshots | -| 613 | `export` | | -| 614 | `verify` | | -| 615 | `ingest` | | -| 616 | `list` | | -| 617 | `report` | | -| 618 | `stats` | | -| 619 | `mechanic` | Mechanic Workbench (targets, plans, simulation) | -| 620 | `init` | | -| 621 | `targets` | Manage signed equalizer targets | -| 622 | `init` | | -| 623 | `set` | | -| 624 | `apply` | | -| 625 | `print` | | -| 626 | `verify` | | -| 627 | `profile` | Apply one-click signed target profiles | -| 628 | `list` | | -| 629 | `apply` | | -| 630 | `verify` | | -| 631 | `tuning` | Manage signed mechanic tuning intent | -| 632 | `init` | | -| 633 | `set` | | -| 634 | `apply` | | -| 635 | `print` | | -| 636 | `verify` | | -| 637 | `gap` | | -| 638 | `plan` | Create, diff, approve, and execute upgrade plans | -| 639 | `create` | | -| 640 | `show` | | -| 641 | `diff` | | -| 642 | `request-approval` | | -| 643 | `execute` | | -| 644 | `simulate` | | -| 645 | `simulations` | Show latest signed simulation artifact | -| 646 | `verify` | Verify mechanic signatures and artifacts | -| 647 | `init` | Initialize AMC release signing keypair | -| 648 | `pack` | Build a signed deterministic .amcrelease bundle | -| 649 | `verify` | Verify a .amcrelease bundle offline | -| 650 | `sbom` | Generate deterministic CycloneDX SBOM | -| 651 | `licenses` | Generate dependency license inventory | -| 652 | `provenance` | Generate AMC provenance record | -| 653 | `scan` | Run strict secret scan on a .amcrelease bundle | -| 654 | `print` | Print release bundle manifest summary | -| 655 | `e2e` | End-to-end smoke verification | -| 656 | `smoke` | Run go-live smoke tests: local, docker, or helm-template | -| 657 | `_studio-daemon` | | -| 658 | `lab-templates` | List available experiment templates | -| 659 | `lab-create` | Create a new lab experiment | -| 660 | `lab-simulate` | Simulate running all probes for an experiment | -| 661 | `lab-report` | Generate a lab experiment report | -| 662 | `lab-compare` | Compare two lab experiments | -| 663 | `lab-list` | List all lab experiments | -| 664 | `insider-risk-report` | Generate insider risk analytics report | -| 665 | `insider-alerts` | Show insider risk alerts | -| 666 | `insider-risk-scores` | Show insider risk scores by actor | -| 667 | `attestation-export` | Export attestation bundle for external auditors | -| 668 | `fp-submit` | Submit a false positive report for an assurance scenario | -| 669 | `fp-resolve` | Resolve a false positive report | -| 670 | `fp-list` | List false positive reports | -| 671 | `fp-cost` | Show false positive cost summary | -| 672 | `fp-tuning-report` | Generate false positive tuning report with recommendations | -| 673 | `wiring-status` | Show production wiring status for all modules (Items 11-16) | -| 674 | `python-sdk` | Generate the Python SDK package for AMC Bridge API | -| 675 | `residency-policy` | Create or list data residency policies | -| 676 | `tenant-register` | Register a tenant boundary | -| 677 | `tenant-isolation-check` | Check tenant isolation between all registered tenants | -| 678 | `legal-hold` | Issue or manage legal holds | -| 679 | `redaction-test` | Run privacy redaction tests against built-in rules | -| 680 | `residency-report` | Generate data residency compliance report for a tenant | -| 681 | `key-custody-modes` | List available key custody modes and their configurations | -| 682 | `operator-dashboard` | Generate operator dashboard showing why questions are capped and how to unlock | -| 683 | `why-capped` | Show why each question is capped at its current level | -| 684 | `action-queue` | Show prioritized actions sorted by risk-reduction-per-effort | -| 685 | `confidence-heatmap` | Display confidence heatmap by question and layer | -| 686 | `role-presets` | List available dashboard role presets | -| 687 | `integrate` | Generate integration scaffold for a framework | -| 688 | `integrate-list` | List available integration frameworks | -| 689 | `contract-tests` | Generate and display contract test suite for bridge API | -| 690 | `simulate-bridge` | Run a simulated bridge request for local testing | -| 691 | `openapi-generate` | Generate live OpenAPI spec (Studio + Bridge + Gateway) | -| 692 | `limits` | Show current plugin sandbox resource limits | -| 693 | `code-scan` | Scan repository for semantic code edges | -| 694 | `claims-stale` | List stale claims for an agent | -| 695 | `claims-sweep` | Process all stale claims for an agent (auto-demote to PROVISIONAL) | -| 696 | `confidence-drift` | Track confidence drift per question across diagnostic runs | -| 697 | `lessons-list` | List lessons learned from corrections | -| 698 | `lessons-promote` | Promote a correction to a reusable lesson | -| 699 | `corrections-verify-closure` | Show open feedback loops that need closure | -| 700 | `receipts-chain` | Show full delegation chain for a receipt | -| 701 | `policy-canary-start` | Start policy canary mode (observation-only) | -| 702 | `policy-canary-report` | Generate canary mode report for an agent | -| 703 | `debt-add` | Add a policy debt entry (waiver/override/exception) | -| 704 | `debt-list` | List policy debt entries | -| 705 | `governor-override` | Activate an emergency governance override with TTL | -| 706 | `governor-override-alerts` | Show alerts for active/expired overrides | -| 707 | `community` | Community/platform governance scoring | -| 708 | `init` | | -| 709 | `score` | | -| 710 | `capabilities-add` | Add capability declaration to agent passport | -| 711 | `search` | Search agents by capability and minimum maturity level | -| 712 | `link` | Link agent passport to external platform identity | -| 713 | `unknowns` | | -| 714 | `meta-confidence` | Report confidence in the maturity score itself | -| 715 | `confidence-check` | Check if action is allowed given confidence-adjusted maturity | -| 716 | `confidence-components` | Show per-component confidence breakdown | -| 717 | `shield` | Threat detection and security scanning | -| 718 | `analyze ` | Run static code analyzer on a skill file | -| 719 | `sandbox ` | Check sandbox configuration for an agent | -| 720 | `sbom ` | Generate software bill of materials from package.json | -| 721 | `reputation ` | Check reputation score for a tool | -| 722 | `conversation-integrity ` | Check conversation integrity for an agent (demo) | -| 723 | `threat-intel ` | Check threat intelligence for an input | -| 724 | `detect-injection ` | Detect prompt injection attempts in text | -| 725 | `sanitize ` | Sanitize text — strip XSS, injection, and dangerous patterns | -| 726 | `enforce` | Policy enforcement and guardrails | -| 727 | `check ` | Check policy for an agent action | -| 728 | `exec-guard ` | Check if a command is safe to execute | -| 729 | `ato-detect ` | Detect account takeover attempts (demo) | -| 730 | `numeric-check ` | Validate a numeric value within bounds | -| 731 | `taint ` | Track tainted input through the system | -| 732 | `blind-secrets ` | Redact secrets from text | -| 733 | `watch` | Observability, attestation, and safety testing | -| 734 | `attest ` | Attest an agent output | -| 735 | `explain ` | Generate explainability packet for an agent run | -| 736 | `safety-test ` | Run safety tests for an agent | -| 737 | `host-hardening` | Check host hardening status for this AMC deployment | -| 738 | `product` | Product operations: routing, autonomy, metering, workflows | -| 739 | `route ` | Route a task to the best model/provider | -| 740 | `autonomy ` | Decide autonomy level for an agent | -| 741 | `loop-detect ` | Detect infinite loops in agent behavior | -| 742 | `metering ` | Show metering and billing for an agent | -| 743 | `retry ` | Execute a command with retry logic | -| 744 | `plan ` | Generate an execution plan for a goal | -| 745 | `workflow` | Workflow management | -| 746 | `create ` | Create a new workflow | -| 747 | `rag-guard ` | Guard RAG chunks against injection | -| 748 | `classify ` | Classify data sensitivity level | -| 749 | `scrub ` | Scrub metadata from a file | -| 750 | `dsar-status` | Show DSAR (Data Subject Access Request) status | -| 751 | `privacy-budget ` | Check privacy budget for an agent | -| 752 | `glossary` | Domain terminology management | -| 753 | `domain` | Domain-specific architecture and compliance operations | -| 754 | `features` | List product features | -| 755 | `features-recommended` | Show top recommended product features | -| 756 | `define ` | Define a glossary term | -| 757 | `lookup ` | Look up a glossary term | -| 758 | `list` | List all 7 domains with metadata | -| 759 | `assess` | Run full domain assessment | -| 760 | `modules` | Show module activation map for domain | -| 761 | `gaps` | Show compliance gaps for an agent and domain | -| 762 | `report` | Build full domain report and write it to a file | -| 763 | `assurance` | Run domain-specific assurance packs | -| 764 | `roadmap` | Generate 30/60/90-day roadmap for this domain | -| 765 | `score` | Maturity scoring, adversarial testing, and evidence collection | -| 766 | `formal-spec ` | Compute formal maturity score for an agent | -| 767 | `adversarial ` | Test gaming resistance of scoring | -| 768 | `collect-evidence ` | Collect evidence for scoring an agent | -| 769 | `production-ready ` | Run production readiness gate for an agent | -| 770 | `operational-independence ` | Calculate operational independence score | -| 771 | `evidence-coverage ` | Show automated vs manual evidence coverage | -| 772 | `lean-profile` | Show lean AMC profile | -| 773 | `behavioral-contract` | Score agent behavioral contract maturity (alignment card, permitted/forbidden actions) | -| 774 | `fail-secure` | Score fail-secure tool governance (deny-by-default, rate limiting, anomaly detection) | -| 775 | `output-integrity` | Score output integrity maturity (OWASP LLM02, confidence calibration, citation) | -| 776 | `state-portability` | Score agent state portability (vendor-neutral format, serialization, integrity on transfer) | -| 777 | `eu-ai-act` | Score EU AI Act compliance maturity (Art. 9-17, GPAI systemic risk) | -| 778 | `owasp-llm` | Score OWASP LLM Top 10 coverage (all 10 risks) | -| 779 | `regulatory-readiness` | Compute weighted regulatory readiness score (EU AI Act + ISO + OWASP) | -| 780 | `self-knowledge` | Score prior art self-knowledge maturity (typed attention, trace layer, confidence+citation) | -| 781 | `kernel-sandbox` | Score kernel-level sandbox maturity (OS isolation, filesystem/network restrictions) | -| 782 | `runtime-identity` | Score runtime execution identity maturity (JIT credentials, user propagation, revocation) | -| 783 | `calibration-gap` | Measure delta between agent self-reported confidence and observed behavior | -| 784 | `evidence-conflict` | Measure internal consistency of evidence — detect conflicting signals | -| 785 | `density-map` | Heatmap of evidence density per question per dimension — reveals blind spots | -| 786 | `evidence-ingest` | Ingest evidence from external systems (openai-evals, langsmith, mlflow, custom) | -| 787 | `level-transition` | Track formal promotion/demotion events with evidence gates | -| 788 | `gaming-resistance` | Test whether adversarial evidence injection can inflate scores | -| 789 | `sleeper-detection` | Detect context-dependent behavioral inconsistencies | -| 790 | `audit-depth` | Score audit trail depth and completeness | -| 791 | `policy-consistency` | Test policy enforcement consistency across repeated trials (pass^k) | -| 792 | `autonomy-duration` | Track time between human checkpoints with domain risk profiles | -| 793 | `pause-quality` | Score quality of agent-initiated pauses | -| 794 | `task-horizon` | Score task-completion time horizon (METR-inspired) | -| 795 | `factuality` | Score factuality across parametric, retrieval, and grounded dimensions | -| 796 | `alignment-index` | Compute composite alignment index | -| 797 | `interpretability` | Score structural transparency and explainability | -| 798 | `faithfulness` | Score how well LLM output is grounded in provided context | -| 799 | `memory-integrity` | Score memory correction persistence and poisoning resistance | -| 800 | `output-attestation` | Score output signing and trust metadata for receiving agents | -| 801 | `mutual-verification` | Score agent-to-agent trust verification (challenge-response) | -| 802 | `transparency-log` | Score network transparency log (Merkle tree, inclusion proofs) | -| 803 | `memory` | Memory maturity assessment and management | -| 804 | `assess ` | Full memory maturity assessment | -| 805 | `oversight` | Human oversight quality assessment | -| 806 | `assess ` | Assess human oversight quality | -| 807 | `classify` | Classify agent vs workflow | -| 808 | `agent ` | Classify whether system is workflow or agent | -| 809 | `claims` | Evidence claim expiry tracking | -| 810 | `list ` | List all evidence claims with TTL status | -| 811 | `dag` | Orchestration DAG capture and scoring | -| 812 | `capture ` | Capture orchestration DAG for agents | -| 813 | `score` | Score DAG governance | -| 814 | `confidence` | Confidence drift tracking | -| 815 | `calibration` | Show calibration report | -| 816 | `drift` | Show drift trend | -| 817 | `tier` | Run tiered maturity assessment (quick/standard/deep) | -| 818 | `scan` | Zero-integration agent assessment scanner | -| 819 | `guardrails` | Simple guardrail management | -| 820 | `list` | List all available guardrails with status | -| 821 | `enable ` | Enable a guardrail | -| 822 | `disable ` | Disable a guardrail | -| 823 | `profile ` | Apply a guardrail profile (minimal, standard, strict, healthcare, financial) | -| 824 | `playground` | Interactive scenario runner | -| 825 | `run` | Run all demo scenarios | -| 826 | `list` | List available scenarios | -| 827 | `open` | Build and serve dashboard at localhost:3210 | -| 828 | `vibe-audit` | Run static safety checks for AI-generated code | -| 829 | `quickstart` | 2-minute quickstart with Quick Score assessment | -| 830 | `debug` | Structured evidence debug stream for an agent | -| 831 | `api` | REST API management | -| 832 | `status` | Show API integration status | -| 833 | `run ` | Run an AMC-governed agent (content-moderation, data-pipeline, legal-contract) | -| 834 | `harness` | Run the autonomous improvement harness loop | -| 835 | `demo` | Run interactive demos of AMC capabilities | -| 836 | `gap` | The 84-point documentation inflation gap — keyword vs execution scoring | -| 837 | `run` | Run a simulated agent through the AMC gateway and produce a real score (~30s) | -| 838 | `fix` | Generate remediation patches for identified gaps (auto-fix mode) | -| 839 | `redteam` | Run red-team attack simulations against a target agent | -| 840 | `run [agentId]` | Execute red-team plugins with chosen attack strategies and generate a vulnerability report | -| 841 | `strategies` | List available attack strategies | -| 842 | `plugins` | List available attack plugins (assurance packs) | +| 1 | `amc action-queue` | Show prioritized actions sorted by risk-reduction-per-effort | +| 2 | `amc adapters` | Built-in adapter system for one-line agent integration | +| 3 | `amc adapters configure` | Set adapter profile for an agent (signed adapters.yaml) | +| 4 | `amc adapters detect` | Detect installed adapter runtimes and versions | +| 5 | `amc adapters env` | Print adapter-compatible environment exports without lease token | +| 6 | `amc adapters init` | Create signed adapters.yaml defaults | +| 7 | `amc adapters init-project` | Generate runnable local adapter sample for library-based frameworks | +| 8 | `amc adapters list` | List built-in adapters and per-agent preferences | +| 9 | `amc adapters run` | Run adapter with minted lease, routed through gateway, with observed evidence capture | +| 10 | `amc adapters verify` | Verify adapters.yaml signature | +| 11 | `amc admin` | Administrative controls, identity, and trust operations | +| 12 | `amc admin help` | Show admin-focused command groups | +| 13 | `amc admin status` | Show operational admin status for control-plane services | +| 14 | `amc advisory` | Forecast advisories (list/show/ack) | +| 15 | `amc advisory ack` | Acknowledge an advisory | +| 16 | `amc advisory list` | List advisories for scope | +| 17 | `amc advisory show` | Show one advisory by ID | +| 18 | `amc agent` | Agent registry operations | +| 19 | `amc agent add` | Interactively add an agent to the fleet | +| 20 | `amc agent diagnose` | Lease-auth self-run diagnostic (agent-triggered, evidence-scored server-side) | +| 21 | `amc agent harness` | Run the autonomous improvement harness loop | +| 22 | `amc agent list` | List fleet agents | +| 23 | `amc agent remove` | Remove an agent from the fleet | +| 24 | `amc agent run` | Run an AMC-governed agent (content-moderation, data-pipeline, legal-contract) | +| 25 | `amc agent use` | Set current agent | +| 26 | `amc alert` | SIEM/webhook alerting — configure and send alerts from anomalies | +| 27 | `amc alert config` | Configure alert destinations (webhooks, Slack, PagerDuty) | +| 28 | `amc alert send` | Send an alert to a webhook endpoint | +| 29 | `amc alert test` | Send a test alert to all configured destinations | +| 30 | `amc alert watch` | Watch for anomalies and auto-send alerts to configured destinations | +| 31 | `amc alerts` | Signed drift alert configuration and dispatch | +| 32 | `amc alerts init` | - | +| 33 | `amc alerts test` | - | +| 34 | `amc alerts verify` | - | +| 35 | `amc api` | REST API management | +| 36 | `amc api docs` | Show API reference documentation summary and link | +| 37 | `amc api key` | Manage programmatic API keys | +| 38 | `amc api key create` | Create a programmatic API key and show the secret once | +| 39 | `amc api key list` | List programmatic API keys without printing secrets | +| 40 | `amc api key revoke` | Revoke a programmatic API key | +| 41 | `amc api routes` | List all available REST API route families | +| 42 | `amc api start` | Start the AMC API server (alias for 'amc up') | +| 43 | `amc api status` | Show API integration status | +| 44 | `amc approvals` | Signed approval inbox operations | +| 45 | `amc approvals approve` | - | +| 46 | `amc approvals deny` | - | +| 47 | `amc approvals list` | - | +| 48 | `amc approvals show` | - | +| 49 | `amc archetype` | Archetype packs | +| 50 | `amc archetype apply` | Apply archetype context/targets/guardrails/evals to an agent | +| 51 | `amc archetype describe` | Describe an archetype | +| 52 | `amc archetype list` | List built-in archetype packs | +| 53 | `amc assurance` | Assurance Lab red-team packs | +| 54 | `amc assurance advanced-threats` | Run advanced threats assurance pack | +| 55 | `amc assurance cert-issue` | Issue signed assurance certificate for a run | +| 56 | `amc assurance cert-verify` | Verify assurance certificate bundle offline | +| 57 | `amc assurance compound-threats` | Run compound threat assurance pack | +| 58 | `amc assurance describe` | Describe assurance pack details | +| 59 | `amc assurance history` | List assurance run history | +| 60 | `amc assurance init` | Initialize signed assurance policy | +| 61 | `amc assurance list` | List available assurance packs | +| 62 | `amc assurance patch` | Apply deterministic patch kit for failed assurance findings | +| 63 | `amc assurance policy` | Print current assurance policy | +| 64 | `amc assurance policy-apply` | Apply assurance policy from YAML/JSON file | +| 65 | `amc assurance run` | Run assurance pack(s) with deterministic validation | +| 66 | `amc assurance runs` | List assurance lab runs | +| 67 | `amc assurance scheduler` | Assurance scheduler controls | +| 68 | `amc assurance scheduler disable` | Disable assurance scheduler | +| 69 | `amc assurance scheduler enable` | Enable assurance scheduler | +| 70 | `amc assurance scheduler run-now` | Run assurance scheduler immediately | +| 71 | `amc assurance scheduler status` | Show scheduler status | +| 72 | `amc assurance show` | Show assurance run artifacts | +| 73 | `amc assurance shutdown-compliance` | Run shutdown compliance pack | +| 74 | `amc assurance toctou` | Run TOCTOU assurance pack | +| 75 | `amc assurance verify` | Verify assurance run determinism and signatures | +| 76 | `amc assurance verify-policy` | Verify assurance policy signature | +| 77 | `amc assurance waiver` | Assurance threshold waiver controls | +| 78 | `amc assurance waiver request` | Request time-limited readiness waiver (dual-control approval required) | +| 79 | `amc assurance waiver revoke` | Revoke active or specific waiver | +| 80 | `amc assurance waiver status` | Show waiver status (activates approved pending waivers) | +| 81 | `amc attest` | Auditor-attest an ingest session to upgrade trust tier to ATTESTED | +| 82 | `amc attestation-export` | Export attestation bundle for external auditors | +| 83 | `amc audit` | Audit binder and compliance maps | +| 84 | `amc audit binder` | Audit binder artifact operations | +| 85 | `amc audit binder create` | Create deterministic signed .amcaudit artifact | +| 86 | `amc audit binder export-execute` | Execute previously approved external binder export | +| 87 | `amc audit binder export-request` | Create dual-control approval request for external binder sharing | +| 88 | `amc audit binder list` | List exported binders and cached workspace binder | +| 89 | `amc audit binder verify` | Verify .amcaudit file | +| 90 | `amc audit export` | Export enterprise audit logs for Splunk, Datadog, CloudTrail, or Azure Monitor | +| 91 | `amc audit init` | Initialize signed audit policy and compliance maps | +| 92 | `amc audit map` | Audit compliance map operations | +| 93 | `amc audit map apply` | Apply active audit map from file | +| 94 | `amc audit map list` | List builtin/active audit maps | +| 95 | `amc audit map show` | Show audit map | +| 96 | `amc audit map verify` | Verify builtin and active map signatures | +| 97 | `amc audit policy` | Audit binder policy operations | +| 98 | `amc audit policy apply` | Apply and sign audit policy from file | +| 99 | `amc audit policy print` | Print effective audit policy | +| 100 | `amc audit request` | Audit evidence request operations | +| 101 | `amc audit request approve` | Owner approves request (starts dual-control approval flow) | +| 102 | `amc audit request create` | Create auditor evidence request | +| 103 | `amc audit request fulfill` | Fulfill approved evidence request by exporting restricted binder | +| 104 | `amc audit request list` | List audit evidence requests | +| 105 | `amc audit request reject` | Reject evidence request | +| 106 | `amc audit scheduler` | Audit binder cache scheduler | +| 107 | `amc audit scheduler disable` | Disable audit scheduler | +| 108 | `amc audit scheduler enable` | Enable audit scheduler | +| 109 | `amc audit scheduler run-now` | Run audit binder cache refresh immediately | +| 110 | `amc audit scheduler status` | Show audit scheduler status | +| 111 | `amc audit verify` | Verify audit workspace signatures/artifacts | +| 112 | `amc audit verify-policy` | Verify signed audit policy | +| 113 | `amc audit-packet` | Generate external-auditor packet with verifier-ready evidence | +| 114 | `amc backup` | Signed encrypted backup/restore operations | +| 115 | `amc backup create` | Create signed encrypted backup bundle | +| 116 | `amc backup print` | Print backup manifest summary | +| 117 | `amc backup restore` | Restore a verified backup into target directory | +| 118 | `amc backup verify` | Verify signed backup bundle offline | +| 119 | `amc badge` | Generate maturity badge for README/docs (markdown, HTML, or URL) | +| 120 | `amc bench` | Public benchmark registry + ecosystem comparative view | +| 121 | `amc bench compare` | Compute local vs imported ecosystem comparison | +| 122 | `amc bench comparison-latest` | Read latest bench comparison artifact | +| 123 | `amc bench create` | Create deterministic signed .amcbench artifact | +| 124 | `amc bench import` | Import one bench artifact from allowlisted registry | +| 125 | `amc bench init` | Initialize signed bench policy | +| 126 | `amc bench list-exports` | List locally exported bench artifacts | +| 127 | `amc bench list-imports` | List imported bench artifacts | +| 128 | `amc bench print` | Print bench manifest summary without modification | +| 129 | `amc bench print-policy` | Print effective bench policy | +| 130 | `amc bench publish` | Dual-control bench publish flow | +| 131 | `amc bench publish execute` | - | +| 132 | `amc bench publish request` | - | +| 133 | `amc bench registries` | Print signed bench registry allowlist | +| 134 | `amc bench registries-apply` | Apply bench registries config from JSON file | +| 135 | `amc bench registry` | Manage static bench registries | +| 136 | `amc bench registry init` | - | +| 137 | `amc bench registry publish` | - | +| 138 | `amc bench registry serve` | - | +| 139 | `amc bench registry verify` | - | +| 140 | `amc bench search` | Browse a bench registry index | +| 141 | `amc bench verify` | Verify .amcbench artifact offline | +| 142 | `amc bench verify-policy` | Verify signed bench policy | +| 143 | `amc benchmark` | Signed ecosystem benchmark snapshots | +| 144 | `amc benchmark compare` | Compare benchmark results between two agents head-to-head | +| 145 | `amc benchmark export` | - | +| 146 | `amc benchmark ingest` | - | +| 147 | `amc benchmark list` | - | +| 148 | `amc benchmark provider-drift` | Run provider/model canary drift benchmark with score, refusal, latency, and cost thresholds | +| 149 | `amc benchmark replay-corpus` | Run a replayable benchmark corpus with optional multi-turn tool-risk ASR checks | +| 150 | `amc benchmark report` | - | +| 151 | `amc benchmark run` | Run standard benchmark suite (latency, accuracy, safety, cost-efficiency, reliability) against an agent | +| 152 | `amc benchmark stats` | - | +| 153 | `amc benchmark verify` | - | +| 154 | `amc blobs` | Encrypted evidence blob operations | +| 155 | `amc blobs key` | Blob key management | +| 156 | `amc blobs key init` | Initialize encrypted blob key material | +| 157 | `amc blobs key rotate` | Rotate encrypted blob key material | +| 158 | `amc blobs reencrypt` | Re-encrypt blob batch from one key version to another | +| 159 | `amc blobs verify` | Verify encrypted blob index and payload integrity | +| 160 | `amc bom` | Maturity Bill of Materials | +| 161 | `amc bom generate` | - | +| 162 | `amc bom sign` | - | +| 163 | `amc bom verify` | - | +| 164 | `amc bootstrap` | Bootstrap workspace for production deployment (non-interactive) | +| 165 | `amc budgets` | Signed autonomy and usage budgets | +| 166 | `amc budgets init` | - | +| 167 | `amc budgets reset` | - | +| 168 | `amc budgets status` | - | +| 169 | `amc budgets verify` | - | +| 170 | `amc bundle` | Portable evidence bundle operations | +| 171 | `amc bundle diff` | Diff two bundles (maturity/integrity/targets) | +| 172 | `amc bundle export` | Export a portable, signed evidence bundle for a run | +| 173 | `amc bundle inspect` | Inspect bundle metadata | +| 174 | `amc bundle verify` | Verify evidence bundle offline | +| 175 | `amc business` | Business impact — KPI correlation, ROI tracking, and maturity-to-outcome mapping | +| 176 | `amc business fair-scenario` | Run a FAIR-style calibrated loss-distribution scenario | +| 177 | `amc business grc-export` | Export a GRC treatment-plan register from portfolio maturity risk inputs | +| 178 | `amc business heatmap` | Build a portfolio financial risk heatmap from maturity, likelihood, impact, and appetite | +| 179 | `amc business kpi` | Show business KPIs correlated with maturity levels | +| 180 | `amc business report` | Generate business impact report with maturity correlation | +| 181 | `amc business risk` | Quantify maturity-linked incident frequency and expected annual loss | +| 182 | `amc business roi` | Estimate first-year ROI and cost of a trust gap from maturity improvement | +| 183 | `amc business track` | Record a business outcome event (incident, audit finding, cost) | +| 184 | `amc canary-report` | Generate full policy canary report | +| 185 | `amc canary-start` | Start a policy canary with candidate vs stable policy | +| 186 | `amc canary-status` | Show current canary status and stats | +| 187 | `amc canary-stop` | Stop the active canary | +| 188 | `amc canon` | Compass Canon signed content operations | +| 189 | `amc canon init` | Create and sign .amc/canon/canon.yaml | +| 190 | `amc canon print` | Print effective Compass Canon | +| 191 | `amc canon verify` | Verify canonical compass content signature | +| 192 | `amc casebook` | Signed casebook operations | +| 193 | `amc casebook add` | Add signed case from existing workorder | +| 194 | `amc casebook init` | Create a signed casebook | +| 195 | `amc casebook list` | List casebooks | +| 196 | `amc casebook verify` | Verify signed casebook and case files | +| 197 | `amc cert` | Certificate operations | +| 198 | `amc cert generate` | Generate execution-proof trust certificate (signed PDF or JSON) | +| 199 | `amc cert inspect` | Inspect certificate bundle contents | +| 200 | `amc cert revoke` | Create signed revocation file for a certificate | +| 201 | `amc cert verify` | Verify certificate bundle offline | +| 202 | `amc cert verify-revocation` | Verify revocation file signature | +| 203 | `amc certify` | Issue signed, offline-verifiable certificate bundle | +| 204 | `amc cgx` | Context Graph (CGX) build and verify operations | +| 205 | `amc cgx build` | Build deterministic signed context graph | +| 206 | `amc cgx code-scan` | Scan repository for semantic code edges | +| 207 | `amc cgx diff` | Diff two CGX graph snapshots | +| 208 | `amc cgx init` | Create and sign .amc/cgx/policy.yaml | +| 209 | `amc cgx show` | Show latest CGX graph or agent context pack | +| 210 | `amc cgx simulate` | Simulate impact propagation when a node changes | +| 211 | `amc cgx verify` | Verify CGX policy/graph/pack signatures | +| 212 | `amc cgx-integrity` | Run graph integrity check on CGX with semantic overlay | +| 213 | `amc cgx-propagation` | Simulate risk propagation from a source node | +| 214 | `amc ci` | CI/CD release gate helpers | +| 215 | `amc ci check` | One-liner CI gate: quickscore + threshold check (exit 1 if below) | +| 216 | `amc ci init` | Generate GitHub workflow and signed gate policy | +| 217 | `amc ci print` | Print suggested CI pipeline steps | +| 218 | `amc ci redteam` | CI gate: run red-team plugins, optional Evil MCP, and score-gaming resistance checks | +| 219 | `amc claim-confidence` | Generate per-claim confidence report with citation-backed scoring | +| 220 | `amc claim-confidence-gate` | Check if claims for given questions pass confidence threshold | +| 221 | `amc claims` | Evidence claim expiry tracking | +| 222 | `amc claims list` | List all evidence claims with TTL status | +| 223 | `amc claims-stale` | List stale claims for an agent | +| 224 | `amc claims-sweep` | Process all stale claims for an agent (auto-demote to PROVISIONAL) | +| 225 | `amc classify` | Classify agent vs workflow | +| 226 | `amc classify agent` | Classify whether system is workflow or agent | +| 227 | `amc commands` | Generate the live AMC CLI command inventory from the registered command map | +| 228 | `amc commit` | Commitment plan flow (7/14/30-day checklist) | +| 229 | `amc comms-check` | Check a message/communication against compliance policies (lightweight communications firewall) | +| 230 | `amc compare` | Compare two runs OR multiple models (side-by-side evaluation) | +| 231 | `amc compare-models` | Run the same agent evaluation across multiple models and show comparison matrix | +| 232 | `amc compliance` | Evidence-linked compliance map operations | +| 233 | `amc compliance diff` | Diff two compliance report JSON files | +| 234 | `amc compliance fleet` | Generate fleet compliance summary | +| 235 | `amc compliance init` | Create and sign compliance-maps.yaml | +| 236 | `amc compliance matrix` | Generate multi-framework compliance coverage matrix with gap analysis | +| 237 | `amc compliance regulatory-check` | Check for regulatory changes from configured feeds | +| 238 | `amc compliance regulatory-feeds` | List all configured regulatory feed sources | +| 239 | `amc compliance regulatory-gap` | Run gap analysis against current AMC configuration | +| 240 | `amc compliance report` | Generate evidence-linked compliance report | +| 241 | `amc compliance risk-classify` | Classify agent into EU AI Act risk tiers (UNACCEPTABLE / HIGH / LIMITED / MINIMAL) | +| 242 | `amc compliance roadmap` | Generate step-by-step compliance plan for a framework | +| 243 | `amc compliance verify` | Verify compliance maps signature | +| 244 | `amc confidence` | Confidence drift tracking | +| 245 | `amc confidence calibration` | Show calibration report | +| 246 | `amc confidence drift` | Show drift trend | +| 247 | `amc confidence-components` | Show per-component confidence breakdown | +| 248 | `amc confidence-drift` | Track confidence drift per question across diagnostic runs | +| 249 | `amc confidence-heatmap` | Display confidence heatmap by question and layer | +| 250 | `amc config` | Inspect resolved runtime configuration | +| 251 | `amc config explain` | Explain config source precedence and risky settings | +| 252 | `amc config print` | Print resolved runtime config (secret-safe) | +| 253 | `amc config profile` | Print or apply workspace config profile (dev|ci|prod) | +| 254 | `amc connect` | Connect wizard for any agent/provider runtime | +| 255 | `amc contract-tests` | Generate and display contract test suite for bridge API | +| 256 | `amc control-classification` | Show control enforcement classification (ARCHITECTURAL/POLICY_ENFORCED/CONVENTION) | +| 257 | `amc correction` | Human feedback, corrections, and feedback loop tracking | +| 258 | `amc correction add` | Add a human correction/feedback for an agent | +| 259 | `amc correction effectiveness` | Show correction effectiveness metrics | +| 260 | `amc correction list` | List corrections for an agent | +| 261 | `amc correction report` | Generate feedback closure report | +| 262 | `amc corrections-verify-closure` | Show open feedback loops that need closure | +| 263 | `amc costs` | Track and analyze actual agent costs from observability data | +| 264 | `amc costs show` | Show cost report for an agent | +| 265 | `amc dag` | Orchestration DAG capture and scoring | +| 266 | `amc dag capture` | Capture orchestration DAG for agents | +| 267 | `amc dag score` | Score DAG governance | +| 268 | `amc dashboard` | Device-first Compass dashboard | +| 269 | `amc dashboard build` | Build responsive offline dashboard for an agent | +| 270 | `amc dashboard open` | Build and serve dashboard at localhost:3210 | +| 271 | `amc dashboard serve` | Serve dashboard locally | +| 272 | `amc dashboard view` | Build and open web UI showing maturity scores, test results, and comparison matrix with shareable URLs | +| 273 | `amc dataset` | Manage evaluation datasets (golden sets) — curate business-specific test cases | +| 274 | `amc dataset add-case` | Add a test case to a dataset | +| 275 | `amc dataset create` | Create a new evaluation dataset | +| 276 | `amc dataset import` | Import test cases from CSV/JSON file | +| 277 | `amc dataset list` | List all evaluation datasets | +| 278 | `amc dataset run` | Run a dataset against an agent (via gateway proxy) | +| 279 | `amc debt-add` | Add a policy debt entry (waiver/override/exception) | +| 280 | `amc debt-list` | List policy debt entries | +| 281 | `amc debug` | Structured evidence debug stream for an agent | +| 282 | `amc delta-to-l5` | Generate L4→L5 delta report showing what separates current state from L5 | +| 283 | `amc demo` | Run interactive demos of AMC capabilities | +| 284 | `amc demo gap` | The 84-point documentation inflation gap — keyword vs execution scoring | +| 285 | `amc demo prospect` | Run a guided 5-minute prospect demo flow | +| 286 | `amc demo run` | Run a simulated agent through the AMC gateway and produce a real score (~30s) | +| 287 | `amc demo share` | Generate a static client-facing prospect demo bundle | +| 288 | `amc diagnostic` | Diagnostic bank/render operations | +| 289 | `amc diagnostic bank` | Signed diagnostic 126-question bank operations | +| 290 | `amc diagnostic bank init` | Create and sign .amc/diagnostic/bank/bank.yaml | +| 291 | `amc diagnostic bank verify` | Verify diagnostic bank signature | +| 292 | `amc diagnostic render` | Render contextualized 126-question diagnostic for an agent | +| 293 | `amc dlp` | DLP scanner for PII and secrets | +| 294 | `amc dlp scan` | Scan text for PII and secrets | +| 295 | `amc doctor` | Check runtime availability and wrap readiness | +| 296 | `amc doctor-fix` | Auto-repair common setup issues | +| 297 | `amc domain` | Domain-specific architecture and compliance operations | +| 298 | `amc domain apply` | Apply domain-specific guardrails and industry pack rules to an agent | +| 299 | `amc domain assess` | Run full domain assessment | +| 300 | `amc domain assurance` | Run domain-specific assurance packs | +| 301 | `amc domain gaps` | Show compliance gaps for an agent and domain | +| 302 | `amc domain list` | List all 7 domains with metadata | +| 303 | `amc domain modules` | Show module activation map for domain | +| 304 | `amc domain pack` | Industry sector packs — 41 packs across 7 domains | +| 305 | `amc domain pack access` | Show Industry Packs subscription status and unlock instructions | +| 306 | `amc domain pack activate` | Activate Industry Packs after purchase | +| 307 | `amc domain pack checkout` | Create an Industry Packs checkout link | +| 308 | `amc domain pack describe` | Show details of a specific industry sector pack | +| 309 | `amc domain pack list` | List all available industry sector packs | +| 310 | `amc domain pack run` | Run an industry sector pack — interactive assessment or baseline score | +| 311 | `amc domain pack verify` | Verify an Industry Packs license key | +| 312 | `amc domain report` | Build full domain report and write it to a file | +| 313 | `amc domain roadmap` | Generate 30/60/90-day roadmap for this domain | +| 314 | `amc down` | Stop AMC Studio local control plane | +| 315 | `amc drift` | Drift/regression detection and reporting | +| 316 | `amc drift check` | - | +| 317 | `amc drift report` | - | +| 318 | `amc e2e` | End-to-end smoke verification | +| 319 | `amc e2e smoke` | Run go-live smoke tests: local, docker, or helm-template | +| 320 | `amc emergency-override` | Activate an emergency policy override with strict TTL | +| 321 | `amc enforce` | Policy enforcement and guardrails | +| 322 | `amc enforce ato-detect` | Detect account takeover attempts (demo) | +| 323 | `amc enforce blind-secrets` | Redact secrets from text | +| 324 | `amc enforce check` | Check policy for an agent action | +| 325 | `amc enforce exec-guard` | Check if a command is safe to execute | +| 326 | `amc enforce formal-verify` | Formally verify safety properties using proof trees and certificates | +| 327 | `amc enforce numeric-check` | Validate a numeric value within bounds | +| 328 | `amc enforce resources` | Snapshot, diff, and verify agent resources governed by Enforce | +| 329 | `amc enforce resources apply` | Accept current resources as the new signed manifest; dry-run unless --yes is set | +| 330 | `amc enforce resources contract` | Show the AMC-native governed resource lifecycle contract | +| 331 | `amc enforce resources diff` | Diff two Enforce resource manifests, or a manifest against the current workspace | +| 332 | `amc enforce resources evaluate` | Evaluate a resource proposal against Enforce gates | +| 333 | `amc enforce resources get` | Alias for inspect: read one resource from an Enforce resource manifest | +| 334 | `amc enforce resources history` | Show signed Enforce resource manifests, snapshots, and receipts | +| 335 | `amc enforce resources inspect` | Inspect one resource in an Enforce resource manifest | +| 336 | `amc enforce resources list` | List resources in an Enforce resource manifest | +| 337 | `amc enforce resources propose` | Create a dry-run resource change proposal from the latest manifest to current workspace state | +| 338 | `amc enforce resources restore` | Restore resources from an Enforce snapshot; dry-run unless --apply is set | +| 339 | `amc enforce resources rollback` | Alias for restore: rollback resources from an Enforce snapshot | +| 340 | `amc enforce resources snapshot` | Write the current Enforce resource manifest | +| 341 | `amc enforce resources validate` | Validate governed resource changes before accepting them | +| 342 | `amc enforce resources verify` | Verify the current workspace resources against an Enforce resource manifest | +| 343 | `amc enforce taint` | Track tainted input through the system | +| 344 | `amc enforce tla-spec` | Generate a TLA+ specification for the AMC safety model | +| 345 | `amc enforce verify-certificate` | Verify the integrity of a proof certificate (pass JSON as string) | +| 346 | `amc enterprise` | Enterprise tier — licensing, audit export, SSO, fleet governance | +| 347 | `amc enterprise activate` | Activate an enterprise license key (format: AMC-ENT-XXXX-XXXX-XXXX) | +| 348 | `amc enterprise audit-export` | Export audit trail in SIEM format | +| 349 | `amc enterprise status` | Show current license status, tier, and enabled features | +| 350 | `amc enterprise usage` | Show multi-tenant usage metering and quota utilization | +| 351 | `amc eval` | Eval interop import and coverage status | +| 352 | `amc eval import` | Import eval outputs (LangSmith, DeepEval, Promptfoo, OpenAI Evals, W&B, Langfuse) into signed AMC evidence | +| 353 | `amc eval run` | One-shot evaluation: read amcconfig.yaml, run all diagnostic tests, output results | +| 354 | `amc eval status` | Show imported eval coverage per AMC dimension | +| 355 | `amc evidence` | Evidence lifecycle workflows | +| 356 | `amc evidence collect` | Guided wizard to connect your agent and capture evidence | +| 357 | `amc evidence decisions` | List and inspect decision receipts generated by full-score runs | +| 358 | `amc evidence decisions inspect` | Inspect one decision receipt by receipt id or run id | +| 359 | `amc evidence decisions list` | List persisted decision receipts | +| 360 | `amc evidence decisions observe` | Update open decision receipts with observed outcomes from a later full-score run | +| 361 | `amc evidence episodes` | List, inspect, and export lifecycle evidence episodes | +| 362 | `amc evidence episodes export` | Export one EpisodeRecord as JSON or Markdown | +| 363 | `amc evidence episodes inspect` | Inspect one EpisodeRecord by episode id, lifecycle id, or run id | +| 364 | `amc evidence episodes list` | List persisted EpisodeRecord evidence objects | +| 365 | `amc evidence export` | Export verifier-ready evidence (json|csv|pdf) | +| 366 | `amc evidence finding-proofs` | List, inspect, and export finding proof chains | +| 367 | `amc evidence finding-proofs export` | Export finding proofs as JSON | +| 368 | `amc evidence finding-proofs inspect` | Inspect one finding proof by proof id, finding id, run id, or question id | +| 369 | `amc evidence finding-proofs list` | List persisted finding proof chains | +| 370 | `amc evidence help` | Show high-signal evidence command groups | +| 371 | `amc evidence lifecycle` | List, inspect, and export full lifecycle run artifacts | +| 372 | `amc evidence lifecycle export` | Export one lifecycle artifact as JSON | +| 373 | `amc evidence lifecycle inspect` | Inspect one lifecycle artifact by lifecycle id or run id | +| 374 | `amc evidence lifecycle list` | List persisted lifecycle run artifacts | +| 375 | `amc evidence lifecycle-receipts` | List, inspect, and export lifecycle proposal, validation, commit, rollback, and monitor receipts | +| 376 | `amc evidence lifecycle-receipts export` | Export lifecycle receipts as JSON | +| 377 | `amc evidence lifecycle-receipts inspect` | Inspect one lifecycle receipt by receipt id, run id, or lifecycle id | +| 378 | `amc evidence lifecycle-receipts list` | List persisted lifecycle change receipts | +| 379 | `amc evidence observability` | List and inspect component, experience, and decision observability records | +| 380 | `amc evidence observability inspect` | Inspect one observability lane record by observability id, lifecycle id, or run id | +| 381 | `amc evidence observability list` | List persisted observability lane records | +| 382 | `amc evidence verify` | Run full workspace verification suite | +| 383 | `amc executive` | Executive and board-ready AMC artifacts | +| 384 | `amc executive brief` | Generate a board-ready one-page executive brief from a diagnostic run | +| 385 | `amc experiment` | Deterministic baseline vs candidate experiments | +| 386 | `amc experiment analyze` | Analyze latest experiment run | +| 387 | `amc experiment create` | Create an experiment | +| 388 | `amc experiment gate` | Evaluate latest experiment run against gate policy | +| 389 | `amc experiment gate-template` | Write an experiment gate policy template | +| 390 | `amc experiment list` | List experiments | +| 391 | `amc experiment optimize` | Create governed optimizer candidates from a Fixer RCA report | +| 392 | `amc experiment optimizer-list` | List governed optimizer runs | +| 393 | `amc experiment optimizer-show` | Show a governed optimizer run | +| 394 | `amc experiment run` | Run deterministic experiment against signed casebook | +| 395 | `amc experiment set-baseline` | Set experiment baseline config | +| 396 | `amc experiment set-candidate` | Set experiment candidate signed config overlay | +| 397 | `amc experiment-architecture` | Run a controlled architecture comparison experiment | +| 398 | `amc experiment-architecture-probes` | List the standard probe set for architecture experiments | +| 399 | `amc explain` | Plain-English explanation for a diagnostic question (example: AMC-2.1) | +| 400 | `amc export` | Export policy packs and badges | +| 401 | `amc export badge` | Export deterministic maturity badge SVG for a run | +| 402 | `amc export policy` | Export framework-agnostic North Star policy integration pack | +| 403 | `amc federate` | Offline federation sync operations | +| 404 | `amc federate export` | Export offline federation sync package (.amcfed) | +| 405 | `amc federate import` | Import and verify federation package | +| 406 | `amc federate init` | Initialize federation identity and signed config | +| 407 | `amc federate peer` | Federation peer trust anchors | +| 408 | `amc federate peer add` | Add a peer publisher public key | +| 409 | `amc federate peer list` | List federation peers | +| 410 | `amc federate verify` | Verify federation config signature | +| 411 | `amc federate verify-bundle` | Verify .amcfed package | +| 412 | `amc firewall` | Runtime protection for live agent traffic | +| 413 | `amc firewall check` | Evaluate a request or response payload against Runtime Firewall | +| 414 | `amc firewall disable` | Disable Runtime Firewall for this workspace | +| 415 | `amc firewall enable` | Enable Runtime Firewall in observe, warn, or block mode | +| 416 | `amc firewall events` | List Runtime Firewall decision events | +| 417 | `amc firewall export` | Export Runtime Firewall decisions for SIEM or audit review | +| 418 | `amc firewall status` | Show Runtime Firewall policy and event status | +| 419 | `amc fix` | Generate remediation patches for identified gaps (auto-fix mode) | +| 420 | `amc fix-signatures` | Verify and re-sign gateway/fleet/agent configs | +| 421 | `amc fleet` | Fleet operations | +| 422 | `amc fleet contradictions` | Detect cross-agent contradictions | +| 423 | `amc fleet dag` | Visualize orchestration delegation graph | +| 424 | `amc fleet graph` | Typed multi-agent graph operations | +| 425 | `amc fleet graph list` | List saved typed multi-agent graphs | +| 426 | `amc fleet graph show` | Inspect the latest typed multi-agent graph | +| 427 | `amc fleet graph validate` | Validate the latest typed multi-agent graph | +| 428 | `amc fleet graph write` | Write the latest typed multi-agent graph from a JSON file | +| 429 | `amc fleet handoff` | Manage handoff packets | +| 430 | `amc fleet health` | Show fleet health dashboard aggregates | +| 431 | `amc fleet init` | Create and sign .amc/fleet.yaml | +| 432 | `amc fleet lifecycle` | Fleet parent/child lifecycle evidence | +| 433 | `amc fleet lifecycle list` | List parent fleet lifecycle artifacts | +| 434 | `amc fleet lifecycle show` | Inspect one parent fleet lifecycle artifact | +| 435 | `amc fleet policy` | Fleet governance policy operations | +| 436 | `amc fleet policy apply` | Apply a governance policy to all fleet agents or one environment | +| 437 | `amc fleet policy list` | List effective fleet governance policies | +| 438 | `amc fleet report` | Generate fleet maturity report (md) or fleet compliance report (pdf) | +| 439 | `amc fleet score` | Score multiple agents in one run with fleet-wide aggregates, weak-link detection, and pairwise comparison | +| 440 | `amc fleet slo` | Fleet governance SLO operations | +| 441 | `amc fleet slo define` | Define a fleet SLO, e.g. "95% of production agents must score L3+ on dimension 2" | +| 442 | `amc fleet slo list` | List fleet SLO definitions | +| 443 | `amc fleet slo status` | Show fleet SLO compliance status | +| 444 | `amc fleet status` | Show fleet overview (agent count, average score, health) | +| 445 | `amc fleet tag` | Tag an agent with an environment | +| 446 | `amc fleet trust-add-edge` | Add a delegation edge (orchestrator → worker) | +| 447 | `amc fleet trust-edges` | List all delegation edges | +| 448 | `amc fleet trust-init` | Initialize trust composition config | +| 449 | `amc fleet trust-mode` | Set trust inheritance policy mode | +| 450 | `amc fleet trust-receipts` | Verify cross-agent receipt chains | +| 451 | `amc fleet trust-remove-edge` | Remove a delegation edge | +| 452 | `amc fleet trust-report` | Generate trust composition report across fleet | +| 453 | `amc forecast` | Deterministic evidence-gated forecasting and planning | +| 454 | `amc forecast init` | Create and sign forecast policy | +| 455 | `amc forecast latest` | Render latest forecast for scope | +| 456 | `amc forecast policy` | Forecast policy operations | +| 457 | `amc forecast policy apply` | Apply and sign forecast policy from file | +| 458 | `amc forecast policy default` | Print default forecast policy JSON | +| 459 | `amc forecast print-policy` | Print effective forecast policy | +| 460 | `amc forecast refresh` | Refresh forecast snapshot for scope | +| 461 | `amc forecast scheduler` | Forecast renewal scheduler controls | +| 462 | `amc forecast scheduler disable` | Disable forecast scheduler | +| 463 | `amc forecast scheduler enable` | Enable forecast scheduler | +| 464 | `amc forecast scheduler run-now` | Run scheduler refresh immediately | +| 465 | `amc forecast scheduler status` | Show scheduler status | +| 466 | `amc forecast verify` | Verify forecast policy signature | +| 467 | `amc fp-cost` | Show false positive cost summary | +| 468 | `amc fp-list` | List false positive reports | +| 469 | `amc fp-resolve` | Resolve a false positive report | +| 470 | `amc fp-submit` | Submit a false positive report for an assurance scenario | +| 471 | `amc fp-tuning-report` | Generate false positive tuning report with recommendations | +| 472 | `amc framework-guide` | Framework-specific governance guidance | +| 473 | `amc freeze` | Execution freeze status and controls | +| 474 | `amc freeze lift` | - | +| 475 | `amc freeze status` | - | +| 476 | `amc gate` | Evaluate a run bundle against a signed gate policy | +| 477 | `amc gateway` | AMC universal LLM proxy gateway | +| 478 | `amc gateway bind-agent` | Bind a gateway route prefix to an agent ID for deterministic attribution | +| 479 | `amc gateway init` | Create and sign .amc/gateway.yaml | +| 480 | `amc gateway start` | Start local reverse-proxy gateway and signed evidence capture | +| 481 | `amc gateway status` | Check gateway reachability and route URLs | +| 482 | `amc gateway verify-config` | Verify .amc/gateway.yaml signature | +| 483 | `amc glossary` | Domain terminology management | +| 484 | `amc glossary define` | Define a glossary term | +| 485 | `amc glossary lookup` | Look up a glossary term | +| 486 | `amc governance-drift` | Detect governance drift for an agent | +| 487 | `amc governor` | Autonomy Governor checks | +| 488 | `amc governor check` | Evaluate whether an action is allowed now (simulate vs execute) | +| 489 | `amc governor confidence-check` | Check if action is allowed given confidence-adjusted maturity | +| 490 | `amc governor explain` | Explain policy requirements for an action class | +| 491 | `amc governor report` | Render matrix of current SIMULATE/EXECUTE allowance per ActionClass | +| 492 | `amc governor-override` | Activate an emergency governance override with TTL | +| 493 | `amc governor-override-alerts` | Show alerts for active/expired overrides | +| 494 | `amc guard` | Guard check proposed output from stdin | +| 495 | `amc guardrails` | Simple guardrail management | +| 496 | `amc guardrails disable` | Disable a guardrail | +| 497 | `amc guardrails enable` | Enable a guardrail | +| 498 | `amc guardrails list` | List all available guardrails with status | +| 499 | `amc guardrails profile` | Apply a guardrail profile (minimal, standard, strict, healthcare, financial) | +| 500 | `amc guide` | Generate personalized improvement guide with exportable agent instructions | +| 501 | `amc help` | Show help for a command (for example: amc help run) | +| 502 | `amc history` | List diagnostic run history | +| 503 | `amc host` | Multi-workspace host mode operations | +| 504 | `amc host bootstrap` | Bootstrap host admin + default workspace from secret files | +| 505 | `amc host init` | Initialize host metadata database | +| 506 | `amc host list` | List host users and workspaces | +| 507 | `amc host membership` | Host membership management | +| 508 | `amc host membership grant` | - | +| 509 | `amc host membership revoke` | - | +| 510 | `amc host migrate` | Migrate an existing single-workspace AMC directory into host mode | +| 511 | `amc host user` | Host user management | +| 512 | `amc host user add` | - | +| 513 | `amc host user disable` | - | +| 514 | `amc host workspace` | Host workspace lifecycle | +| 515 | `amc host workspace create` | - | +| 516 | `amc host workspace delete` | - | +| 517 | `amc host workspace purge` | - | +| 518 | `amc identity` | Enterprise identity (OIDC/SAML) configuration | +| 519 | `amc identity init` | Create and sign host-level identity.yaml | +| 520 | `amc identity mapping` | Signed group-to-role mapping rules | +| 521 | `amc identity mapping add` | Add a group mapping rule | +| 522 | `amc identity provider` | Identity provider management | +| 523 | `amc identity provider add` | Add an identity provider | +| 524 | `amc identity verify` | Verify identity.yaml signature | +| 525 | `amc import` | Import neutral traces, runs, workflow graphs, configs, memory, evals, and benchmarks | +| 526 | `amc imports` | List, inspect, and roll back neutral import runs | +| 527 | `amc imports list` | List recent neutral import runs | +| 528 | `amc imports rollback` | Remove files written by a neutral import run | +| 529 | `amc imports show` | Inspect a neutral import manifest | +| 530 | `amc improve` | Guided improvement — shows what to fix next based on your current score | +| 531 | `amc incident` | Incident tracking and response operations | +| 532 | `amc incident close` | Close an incident with a resolution summary | +| 533 | `amc incident create` | Create a manual incident | +| 534 | `amc incident link` | Link evidence to an incident | +| 535 | `amc incident list` | List incidents for an agent | +| 536 | `amc incident show` | Show incident details | +| 537 | `amc incidents` | Incident operations and dispatch workflows | +| 538 | `amc incidents alert` | Dispatch INCIDENT_CREATED to configured integration channels | +| 539 | `amc incidents help` | Show incident-focused command groups | +| 540 | `amc indices` | Compute deterministic failure-risk indices | +| 541 | `amc indices fleet` | Compute failure-risk indices across fleet | +| 542 | `amc ingest` | Ingest external logs/transcripts as SELF_REPORTED evidence | +| 543 | `amc init` | Initialize .amc workspace | +| 544 | `amc insider-alerts` | Show insider risk alerts | +| 545 | `amc insider-risk-report` | Generate insider risk analytics report | +| 546 | `amc insider-risk-scores` | Show insider risk scores by actor | +| 547 | `amc integrate` | Generate integration scaffold for a framework | +| 548 | `amc integrate-list` | List available integration frameworks | +| 549 | `amc integrations` | Integration hub operations | +| 550 | `amc integrations catalog` | List available integrations | +| 551 | `amc integrations dispatch` | Dispatch a deterministic integration event | +| 552 | `amc integrations export-journal` | Export integration delivery journal (receipts + dead letters) | +| 553 | `amc integrations init` | Create and sign integrations.yaml with vault-backed secret refs | +| 554 | `amc integrations setup` | Generate integration config files | +| 555 | `amc integrations status` | Show integration channels and routing | +| 556 | `amc integrations test` | Dispatch deterministic test event to an integration channel | +| 557 | `amc integrations verify` | Verify integrations config signature | +| 558 | `amc inventory` | AI asset inventory — discover and catalog AI agents, models, and tools | +| 559 | `amc inventory list` | List AI assets (alias for 'inventory scan') | +| 560 | `amc inventory scan` | Scan workspace for AI assets (agents, models, configs, API keys) | +| 561 | `amc key-custody-modes` | List available key custody modes and their configurations | +| 562 | `amc lab-compare` | Compare two lab experiments | +| 563 | `amc lab-create` | Create a new lab experiment | +| 564 | `amc lab-list` | List all lab experiments | +| 565 | `amc lab-report` | Generate a lab experiment report | +| 566 | `amc lab-simulate` | Simulate running all probes for an experiment | +| 567 | `amc lab-templates` | List available experiment templates | +| 568 | `amc leaderboard` | Benchmark leaderboard — compare agent maturity scores | +| 569 | `amc leaderboard export` | Export leaderboard as JSON/HTML for public sharing | +| 570 | `amc leaderboard public-export` | Build an anonymized public leaderboard dataset bundle | +| 571 | `amc leaderboard show` | Show fleet-wide maturity leaderboard | +| 572 | `amc learn` | Education flow for a specific maturity question | +| 573 | `amc lease` | Issue/verify/revoke short-lived agent leases | +| 574 | `amc lease issue` | - | +| 575 | `amc lease revoke` | - | +| 576 | `amc lease verify` | - | +| 577 | `amc legal-hold` | Issue or manage legal holds | +| 578 | `amc lessons-list` | List lessons learned from corrections | +| 579 | `amc lessons-promote` | Promote a correction to a reusable lesson | +| 580 | `amc lifecycle` | Agent lifecycle responsibility and governance mapping | +| 581 | `amc lifecycle advance` | Advance lifecycle stage after governance gate confirmation | +| 582 | `amc lifecycle status` | Show lifecycle stage, accountability matrix, governance gates, and transition trail | +| 583 | `amc lineage-claim` | Show full governance lineage for a specific claim | +| 584 | `amc lineage-init` | Initialize governance lineage tables | +| 585 | `amc lineage-policy-intents` | List all policy change intents for an agent | +| 586 | `amc lineage-report` | Generate governance lineage report | +| 587 | `amc lint` | Lint agent configuration files for schema compliance, anti-patterns, and best practices | +| 588 | `amc lint rules` | List all available lint rules | +| 589 | `amc lite-score` | Lite scoring mode for non-agent LLMs / chatbots — simplified assessment without agentic features | +| 590 | `amc logs` | Print latest AMC Studio logs | +| 591 | `amc loop` | Continuous self-serve maturity loop | +| 592 | `amc loop init` | Initialize recurring loop config | +| 593 | `amc loop plan` | Print recurring loop plan | +| 594 | `amc loop run` | Run recurring diagnostic + assurance + dashboard + snapshot | +| 595 | `amc loop schedule` | Print OS scheduler config (no automatic installation) | +| 596 | `amc maintenance` | Operational maintenance operations | +| 597 | `amc maintenance prune-cache` | Prune dashboard/console/transform cache artifacts | +| 598 | `amc maintenance reindex` | Ensure operational SQLite indexes | +| 599 | `amc maintenance rotate-logs` | Rotate Studio logs based on ops policy | +| 600 | `amc maintenance stats` | Show DB/blob/archive/cache operational stats | +| 601 | `amc maintenance vacuum` | Run SQLite VACUUM + ANALYZE | +| 602 | `amc marketplace` | AMC Pack Marketplace — browse, install, rate community packs | +| 603 | `amc marketplace deprecate` | Deprecate a pack | +| 604 | `amc marketplace featured` | Show featured packs | +| 605 | `amc marketplace info` | Show details for a specific pack | +| 606 | `amc marketplace install` | Install a pack from the marketplace | +| 607 | `amc marketplace list` | List installed packs | +| 608 | `amc marketplace rate` | Rate a pack | +| 609 | `amc marketplace search` | Search marketplace for packs | +| 610 | `amc marketplace undeprecate` | Remove deprecation from a pack | +| 611 | `amc marketplace uninstall` | Uninstall a pack | +| 612 | `amc mcp` | AMC Model Context Protocol (MCP) server for AI coding assistants | +| 613 | `amc mcp config` | Print MCP configuration snippets for supported AI coding assistants | +| 614 | `amc mcp list-tools` | List all tools exposed by the AMC MCP server | +| 615 | `amc mcp serve` | Start the AMC MCP server (stdio transport for IDE integration) | +| 616 | `amc mechanic` | Mechanic Workbench (targets, plans, simulation) | +| 617 | `amc mechanic export` | Export latest gap analysis as reward functions, DSPy targets, or fine-tune recipes | +| 618 | `amc mechanic gap` | - | +| 619 | `amc mechanic init` | - | +| 620 | `amc mechanic plan` | Create, diff, approve, and execute upgrade plans | +| 621 | `amc mechanic plan create` | - | +| 622 | `amc mechanic plan diff` | - | +| 623 | `amc mechanic plan execute` | - | +| 624 | `amc mechanic plan request-approval` | - | +| 625 | `amc mechanic plan show` | - | +| 626 | `amc mechanic profile` | Apply one-click signed target profiles | +| 627 | `amc mechanic profile apply` | - | +| 628 | `amc mechanic profile list` | - | +| 629 | `amc mechanic profile verify` | - | +| 630 | `amc mechanic rca` | Generate fixer root-cause reports from trace failure indexes | +| 631 | `amc mechanic rca list` | List generated fixer RCA reports | +| 632 | `amc mechanic rca run` | Classify a failed run and create regression-preserving fix proposals | +| 633 | `amc mechanic rca show` | Inspect a fixer RCA report | +| 634 | `amc mechanic simulate` | - | +| 635 | `amc mechanic simulations` | Show latest signed simulation artifact | +| 636 | `amc mechanic targets` | Manage signed equalizer targets | +| 637 | `amc mechanic targets apply` | - | +| 638 | `amc mechanic targets init` | - | +| 639 | `amc mechanic targets print` | - | +| 640 | `amc mechanic targets set` | - | +| 641 | `amc mechanic targets verify` | - | +| 642 | `amc mechanic tuning` | Manage signed mechanic tuning intent | +| 643 | `amc mechanic tuning apply` | - | +| 644 | `amc mechanic tuning init` | - | +| 645 | `amc mechanic tuning print` | - | +| 646 | `amc mechanic tuning set` | - | +| 647 | `amc mechanic tuning verify` | - | +| 648 | `amc mechanic verify` | Verify mechanic signatures and artifacts | +| 649 | `amc memory` | Memory maturity assessment and management | +| 650 | `amc memory assess` | Full memory maturity assessment | +| 651 | `amc memory retrieve` | Retrieve active reasoning memory for a consumer | +| 652 | `amc memory show` | Show one reasoning memory item | +| 653 | `amc memory writeback` | Write governed reasoning memory from an EpisodeRecord | +| 654 | `amc memory-advisories` | Show advisories from correction memory for prompt injection | +| 655 | `amc memory-expire` | Expire stale lessons past their TTL | +| 656 | `amc memory-extract` | Extract lessons from verified effective corrections | +| 657 | `amc memory-report` | Generate correction memory report | +| 658 | `amc meta-confidence` | Report confidence in the maturity score itself | +| 659 | `amc methodology` | Print the public AMC scoring methodology manifest and hash | +| 660 | `amc metrics` | Prometheus metrics endpoint helpers | +| 661 | `amc metrics status` | Show configured metrics endpoint bind/port | +| 662 | `amc micro-canary-alerts` | Show active micro-canary alerts | +| 663 | `amc micro-canary-report` | Generate micro-canary status report | +| 664 | `amc micro-canary-run` | Run all micro-canary probes immediately | +| 665 | `amc mirofish` | Agent behavior simulation framework — flight simulator for AI agents | +| 666 | `amc mirofish compare` | Side-by-side comparison of two scenarios | +| 667 | `amc mirofish create` | Interactive scenario builder | +| 668 | `amc mirofish list` | List available built-in scenarios | +| 669 | `amc mirofish run` | Run a Monte Carlo simulation with a scenario | +| 670 | `amc mirofish stress` | Find governance breaking points for a scenario | +| 671 | `amc mode` | Switch CLI role mode | +| 672 | `amc mode agent` | Switch to agent mode (read-only / self-check commands) | +| 673 | `amc mode owner` | Switch to owner mode (configuration + signing allowed) | +| 674 | `amc monitor` | Continuous production monitoring — real-time scoring, drift detection, and alerting | +| 675 | `amc monitor check` | One-shot trust drift analysis (check for degradation without running continuously) | +| 676 | `amc monitor events` | Show recent monitoring events | +| 677 | `amc monitor live` | Start real-time monitoring with live assurance checks on incoming traces | +| 678 | `amc monitor metrics` | Get metrics for a specific agent | +| 679 | `amc monitor start` | Start continuous monitoring: scores agent at intervals, detects drift, sends alerts on degradation | +| 680 | `amc monitor status` | Show monitoring status for all agents | +| 681 | `amc notary` | AMC Notary signing boundary operations | +| 682 | `amc notary attest` | Generate signed notary runtime attestation bundle (.amcattest) | +| 683 | `amc notary init` | Initialize AMC Notary config and signing backend | +| 684 | `amc notary log-verify` | Verify notary append-only signing log + seal signature | +| 685 | `amc notary pubkey` | Print notary public key and fingerprint | +| 686 | `amc notary sign` | Sign a payload file using Notary (admin utility) | +| 687 | `amc notary start` | Start AMC Notary service (foreground) | +| 688 | `amc notary status` | Show notary backend and log status | +| 689 | `amc notary verify-attest` | Verify a .amcattest bundle offline | +| 690 | `amc observe` | Observability — timeline, anomaly detection, and tracing | +| 691 | `amc observe anomalies` | Detect observability anomalies (evidence rate drops, trust regressions, score volatility) | +| 692 | `amc observe timeline` | Show agent evidence timeline with score progression | +| 693 | `amc openapi-generate` | Generate live OpenAPI spec (Studio + Bridge + Gateway) | +| 694 | `amc operator-dashboard` | Generate operator dashboard showing why questions are capped and how to unlock | +| 695 | `amc ops` | Operational hardening policy controls | +| 696 | `amc ops backpressure` | Show backpressure pipeline health | +| 697 | `amc ops circuit-breaker-init` | Initialize circuit breaker policy | +| 698 | `amc ops circuit-breaker-reset` | Reset all circuit breakers | +| 699 | `amc ops circuit-breaker-status` | Show circuit breaker status | +| 700 | `amc ops dead-letters` | Show dead letter queue | +| 701 | `amc ops init` | Create and sign .amc/ops-policy.yaml | +| 702 | `amc ops latency` | Show latency accounting report | +| 703 | `amc ops mode` | Show or set degradation mode | +| 704 | `amc ops print` | Print effective ops policy | +| 705 | `amc ops slo` | Show governance SLO dashboard | +| 706 | `amc ops verify` | Verify ops-policy signature | +| 707 | `amc org` | Org graph and real-time comparative scorecards | +| 708 | `amc org add` | - | +| 709 | `amc org add node` | - | +| 710 | `amc org assign` | - | +| 711 | `amc org commit` | - | +| 712 | `amc org community` | Community/platform governance scoring | +| 713 | `amc org community init` | - | +| 714 | `amc org community score` | - | +| 715 | `amc org compare` | - | +| 716 | `amc org init` | - | +| 717 | `amc org inspect` | - | +| 718 | `amc org learn` | - | +| 719 | `amc org own` | - | +| 720 | `amc org report` | - | +| 721 | `amc org roles` | List the canonical 70 AMC org roles | +| 722 | `amc org run` | Run the advanced 70-role org lifecycle loop with isolated role workspaces | +| 723 | `amc org runs` | List org lifecycle runs | +| 724 | `amc org score` | - | +| 725 | `amc org unassign` | - | +| 726 | `amc org verify` | Verify signed org.yaml | +| 727 | `amc outcomes` | Outcome contracts, value signals, and reports | +| 728 | `amc outcomes attest` | Record manual attested outcome signal | +| 729 | `amc outcomes diff` | Diff two outcome reports | +| 730 | `amc outcomes init` | Create and sign outcome contract | +| 731 | `amc outcomes report` | Generate outcomes report (agent) or fleet outcomes report | +| 732 | `amc outcomes verify` | Verify outcome contract signature | +| 733 | `amc overhead-profile` | Set the overhead mode profile (STRICT, BALANCED, LEAN) | +| 734 | `amc overhead-report` | Generate per-feature overhead accounting report | +| 735 | `amc oversight` | Human oversight quality assessment | +| 736 | `amc oversight assess` | Assess human oversight quality | +| 737 | `amc own` | Ownership flow for top maturity gaps | +| 738 | `amc pack` | Community assurance pack registry — NPM-style package management | +| 739 | `amc pack info` | Show detailed information about a pack | +| 740 | `amc pack init` | Initialize a new pack in the current directory | +| 741 | `amc pack install` | Install a community assurance pack | +| 742 | `amc pack list` | List installed packs | +| 743 | `amc pack publish` | Publish a pack to the registry | +| 744 | `amc pack registry` | Pack registry management | +| 745 | `amc pack registry init` | Initialize local pack registry | +| 746 | `amc pack registry serve` | Start a local pack registry server | +| 747 | `amc pack search` | Search for packs in the registry | +| 748 | `amc pack test` | Test a local pack in sandbox mode | +| 749 | `amc pack uninstall` | Uninstall a pack | +| 750 | `amc pair` | LAN pairing code operations | +| 751 | `amc pair create` | Create one-time pairing code (LAN login pairing or agent bridge pairing) | +| 752 | `amc pair redeem` | Redeem pairing code for a lease token file | +| 753 | `amc passport` | Agent Passport (shareable maturity credential) | +| 754 | `amc passport badge` | Print deterministic single-line badge from latest cache | +| 755 | `amc passport capabilities-add` | Add capability declaration to agent passport | +| 756 | `amc passport compare` | Compare two agents by passport maturity dimensions | +| 757 | `amc passport create` | Create deterministic signed .amcpass artifact | +| 758 | `amc passport export-latest` | Export latest passport for a scope to .amcpass | +| 759 | `amc passport init` | Create and sign .amc/passport/policy.yaml | +| 760 | `amc passport issue-token` | Issue an AMC Trust Token for an agent | +| 761 | `amc passport link` | Link agent passport to external platform identity | +| 762 | `amc passport policy` | Passport policy operations | +| 763 | `amc passport policy apply` | Apply passport policy from JSON/YAML file | +| 764 | `amc passport policy print` | Print effective passport policy | +| 765 | `amc passport search` | Search agents by capability and minimum maturity level | +| 766 | `amc passport share` | Generate shareable passport material | +| 767 | `amc passport show` | Show .amcpass as JSON or single-line badge | +| 768 | `amc passport translate-score` | Translate trust scores between scoring systems | +| 769 | `amc passport verify` | Verify .amcpass artifact offline | +| 770 | `amc passport verify-policy` | Verify signed passport policy | +| 771 | `amc passport verify-token` | Verify an AMC Trust Token (pass JSON string) | +| 772 | `amc playground` | Interactive scenario runner | +| 773 | `amc playground list` | List available scenarios | +| 774 | `amc playground run` | Run all demo scenarios | +| 775 | `amc plugin` | Signed content-only extension marketplace | +| 776 | `amc plugin execute` | Execute approved plugin install/upgrade/remove request | +| 777 | `amc plugin init` | Initialize signed plugin workspace files | +| 778 | `amc plugin install` | Request plugin install (requires SECURITY dual-control approval) | +| 779 | `amc plugin keygen` | Generate plugin publisher keypair | +| 780 | `amc plugin limits` | Show current plugin sandbox resource limits | +| 781 | `amc plugin list` | List installed plugins and verification status | +| 782 | `amc plugin pack` | Create signed .amcplug package from a plugin folder | +| 783 | `amc plugin print` | Print plugin manifest summary | +| 784 | `amc plugin registries` | List signed workspace registry configuration | +| 785 | `amc plugin registries-apply` | Apply and sign workspace registries.yaml from JSON or YAML file | +| 786 | `amc plugin registry` | Manage plugin registries | +| 787 | `amc plugin registry init` | Initialize local signed plugin registry directory | +| 788 | `amc plugin registry publish` | Publish plugin package into registry and re-sign index | +| 789 | `amc plugin registry serve` | Serve plugin registry over local HTTP | +| 790 | `amc plugin registry verify` | Verify registry signature and package hashes | +| 791 | `amc plugin registry-fingerprint` | Compute registry public key fingerprint | +| 792 | `amc plugin remove` | Request plugin removal (requires SECURITY dual-control approval) | +| 793 | `amc plugin search` | Search a plugin registry by id/fingerprint | +| 794 | `amc plugin upgrade` | Request plugin upgrade (requires SECURITY dual-control approval) | +| 795 | `amc plugin verify` | Verify plugin package signature + artifact hashes | +| 796 | `amc plugin workspace-verify` | Verify workspace plugin signatures/integrity | +| 797 | `amc policy` | Policy-as-code operations | +| 798 | `amc policy action` | Signed autonomy action policy | +| 799 | `amc policy action init` | Create and sign .amc/action-policy.yaml | +| 800 | `amc policy action verify` | Verify action policy signature | +| 801 | `amc policy approval` | Signed dual-control approval policy | +| 802 | `amc policy approval init` | Create and sign .amc/approval-policy.yaml | +| 803 | `amc policy approval verify` | Verify approval-policy signature | +| 804 | `amc policy pack` | Policy packs by archetype and risk tier | +| 805 | `amc policy pack apply` | Apply policy pack and sign updated configs/targets | +| 806 | `amc policy pack describe` | Describe policy pack contents | +| 807 | `amc policy pack diff` | Show deterministic diff for applying a policy pack | +| 808 | `amc policy pack list` | List built-in policy packs | +| 809 | `amc policy-canary-report` | Generate canary mode report for an agent | +| 810 | `amc policy-canary-start` | Start policy canary mode (observation-only) | +| 811 | `amc policy-debt-add` | Register a temporary policy waiver (debt) | +| 812 | `amc policy-debt-list` | List active policy debt entries | +| 813 | `amc product` | Product operations: routing, autonomy, metering, workflows | +| 814 | `amc product autonomy` | Decide autonomy level for an agent | +| 815 | `amc product features` | List product features | +| 816 | `amc product features-recommended` | Show top recommended product features | +| 817 | `amc product loop-detect` | Detect infinite loops in agent behavior | +| 818 | `amc product metering` | Show metering and billing for an agent | +| 819 | `amc product plan` | Generate an execution plan for a goal | +| 820 | `amc product retry` | Execute a command with retry logic | +| 821 | `amc product route` | Route a task to the best model/provider | +| 822 | `amc product workflow` | Workflow management | +| 823 | `amc product workflow create` | Create a new workflow | +| 824 | `amc prompt` | Northstar prompt policy + pack operations | +| 825 | `amc prompt init` | Create and sign .amc/prompt/policy.yaml | +| 826 | `amc prompt pack` | Prompt pack artifact operations | +| 827 | `amc prompt pack build` | Build and sign .amcprompt for an agent | +| 828 | `amc prompt pack diff` | Diff latest prompt pack against previous snapshot | +| 829 | `amc prompt pack show` | Show provider-specific enforced system prompt | +| 830 | `amc prompt pack verify` | Verify .amcprompt signature and lint signature | +| 831 | `amc prompt policy` | Prompt policy operations | +| 832 | `amc prompt policy apply` | Apply prompt policy from YAML file and sign | +| 833 | `amc prompt policy print` | Print prompt policy | +| 834 | `amc prompt scheduler` | Prompt pack recurrence scheduler | +| 835 | `amc prompt scheduler disable` | Disable prompt scheduler | +| 836 | `amc prompt scheduler enable` | Enable prompt scheduler | +| 837 | `amc prompt scheduler run-now` | Run prompt scheduler now for one agent or all | +| 838 | `amc prompt scheduler status` | Show prompt scheduler status | +| 839 | `amc prompt status` | List per-agent prompt pack status | +| 840 | `amc prompt verify` | Verify prompt policy, pack, lint and scheduler signatures | +| 841 | `amc provider` | Provider template operations | +| 842 | `amc provider add` | Assign or update provider template for an agent | +| 843 | `amc provider list` | List provider templates | +| 844 | `amc python-sdk` | Generate the Python SDK package for AMC Bridge API | +| 845 | `amc quality-report` | Show quality report | +| 846 | `amc quickscore` | Full default interactive diagnostic — or use --rapid for 5-question express, --auto for ledger evidence | +| 847 | `amc quickstart` | 2-minute quickstart with Quick Score assessment | +| 848 | `amc rate` | Rate agent run quality (thumbs up/down) | +| 849 | `amc receipts-chain` | Show full delegation chain for a receipt | +| 850 | `amc redaction-test` | Run privacy redaction tests against built-in rules | +| 851 | `amc redteam` | Run red-team attack simulations against a target agent | +| 852 | `amc redteam attack` | Run attack plugins (prompt-injection, data-exfiltration, privilege-escalation, model-manipulation, denial-of-service) | +| 853 | `amc redteam attack-list` | List available attack plugins | +| 854 | `amc redteam plugins` | List available attack plugins (assurance packs) | +| 855 | `amc redteam run` | Execute red-team plugins with chosen attack strategies and generate a vulnerability report | +| 856 | `amc redteam strategies` | List available attack strategies | +| 857 | `amc release` | Deterministic release engineering and offline verification | +| 858 | `amc release init` | Initialize AMC release signing keypair | +| 859 | `amc release licenses` | Generate dependency license inventory | +| 860 | `amc release pack` | Build a signed deterministic .amcrelease bundle | +| 861 | `amc release print` | Print release bundle manifest summary | +| 862 | `amc release provenance` | Generate AMC provenance record | +| 863 | `amc release sbom` | Generate deterministic CycloneDX SBOM | +| 864 | `amc release scan` | Run strict secret scan on a .amcrelease bundle | +| 865 | `amc release verify` | Verify a .amcrelease bundle offline | +| 866 | `amc report` | Render report for run ID, saved alias, prefix, or 'latest' | +| 867 | `amc residency-policy` | Create or list data residency policies | +| 868 | `amc residency-report` | Generate data residency compliance report for a tenant | +| 869 | `amc resource` | Govern prompts, tools, memory, policies, routes, and other agent-defining resources | +| 870 | `amc resource apply` | Accept current resources as the new signed manifest; dry-run unless --yes is set | +| 871 | `amc resource contract` | Show the AMC-native governed resource lifecycle contract | +| 872 | `amc resource diff` | Diff an Enforce resource manifest against the current workspace | +| 873 | `amc resource evaluate` | Evaluate a resource proposal against Enforce gates | +| 874 | `amc resource get` | Inspect one resource in an Enforce resource manifest | +| 875 | `amc resource history` | Show signed Enforce resource manifests, snapshots, and receipts | +| 876 | `amc resource list` | List resources in an Enforce resource manifest | +| 877 | `amc resource propose` | Create a dry-run resource change proposal from the latest manifest to current workspace state | +| 878 | `amc resource restore` | Restore resources from an Enforce snapshot; dry-run unless --apply is set | +| 879 | `amc resource rollback` | Alias for restore: rollback resources from an Enforce snapshot | +| 880 | `amc resource snapshot` | Write the current Enforce resource manifest | +| 881 | `amc resource validate` | Validate governed resource changes before accepting them | +| 882 | `amc retention` | Retention/archive payload lifecycle operations | +| 883 | `amc retention run` | Run archival + payload prune lifecycle | +| 884 | `amc retention status` | Show retention/archive status | +| 885 | `amc retention verify` | Verify archive manifests/signatures and ledger continuity | +| 886 | `amc role-presets` | List available dashboard role presets | +| 887 | `amc rollback-create` | Create a rollback pack from the current policy file | +| 888 | `amc run` | Full assessment — Score + Shield + Enforce + Vault + Watch + Comply + Fleet + Passport in one command | +| 889 | `amc run-alias` | Name diagnostic runs for report and history workflows | +| 890 | `amc run-alias list` | List diagnostic run aliases for the active agent | +| 891 | `amc run-alias remove` | Remove a diagnostic run alias | +| 892 | `amc run-alias set` | Assign a reusable alias to a diagnostic run | +| 893 | `amc runtime` | Runtime run manager for connected agents | +| 894 | `amc runtime cancel` | Cancel a runtime run cleanly | +| 895 | `amc runtime complete` | Complete a runtime run | +| 896 | `amc runtime create` | Create a persisted connected-agent runtime run | +| 897 | `amc runtime degrade` | Mark a runtime run degraded | +| 898 | `amc runtime event` | Append an event to a persisted runtime run | +| 899 | `amc runtime export` | Export runtime run events as JSON or JSONL | +| 900 | `amc runtime inspect` | Inspect a runtime run and its event stream | +| 901 | `amc runtime list` | List persisted runtime runs | +| 902 | `amc runtime resume` | Resume a running or degraded runtime run from persisted state | +| 903 | `amc runtime status` | Show persisted runtime run-manager status | +| 904 | `amc sandbox` | Hardened sandbox execution | +| 905 | `amc sandbox run` | Run agent command in hardened Docker sandbox | +| 906 | `amc scan` | Zero-integration agent assessment scanner | +| 907 | `amc scan model-scan` | Scan ML model files for security threats (malicious code, backdoors, supply chain attacks) | +| 908 | `amc scim` | SCIM token management | +| 909 | `amc scim init` | Enable SCIM provisioning and optionally create an initial bearer token | +| 910 | `amc scim token` | SCIM bearer token operations | +| 911 | `amc scim token create` | Create a SCIM bearer token and store hash in host vault | +| 912 | `amc score` | Maturity scoring, adversarial testing, and evidence collection | +| 913 | `amc score a2a-protocol` | Score agent-to-agent protocol maturity: card completeness, lifecycle, auth, format, errors, discovery | +| 914 | `amc score adversarial` | Test gaming resistance of scoring | +| 915 | `amc score alignment-index` | Compute composite alignment index | +| 916 | `amc score audit-depth` | Score audit trail depth and completeness | +| 917 | `amc score autonomy-duration` | Track time between human checkpoints with domain risk profiles | +| 918 | `amc score behavioral-contract` | Score agent behavioral contract maturity (alignment card, permitted/forbidden actions) | +| 919 | `amc score calibration-gap` | Measure delta between agent self-reported confidence and observed behavior | +| 920 | `amc score collect-evidence` | Collect evidence for scoring an agent | +| 921 | `amc score density-map` | Heatmap of evidence density per question per dimension — reveals blind spots | +| 922 | `amc score distributed-agents` | Score distributed multi-agent execution: partitions, sync, failover, consensus, load, observability | +| 923 | `amc score eu-ai-act` | Score EU AI Act compliance maturity (Art. 9-17, GPAI systemic risk) | +| 924 | `amc score evidence-conflict` | Measure internal consistency of evidence — detect conflicting signals | +| 925 | `amc score evidence-coverage` | Show automated vs manual evidence coverage | +| 926 | `amc score evidence-ingest` | Ingest evidence from external systems (openai-evals, langsmith, mlflow, custom) | +| 927 | `amc score factuality` | Score factuality across parametric, retrieval, and grounded dimensions | +| 928 | `amc score fail-secure` | Score fail-secure tool governance (deny-by-default, rate limiting, anomaly detection) | +| 929 | `amc score faithfulness` | Score how well LLM output is grounded in provided context | +| 930 | `amc score formal-spec` | Compute formal maturity score for an agent | +| 931 | `amc score gaming-resistance` | Test whether adversarial evidence injection can inflate scores | +| 932 | `amc score industry-adjust` | Adjust a score using an industry-specific trust model | +| 933 | `amc score industry-benchmark` | Show industry benchmark percentiles | +| 934 | `amc score industry-list` | List all available industry trust models | +| 935 | `amc score interpretability` | Score structural transparency and explainability | +| 936 | `amc score kernel-sandbox` | Score kernel-level sandbox maturity (OS isolation, filesystem/network restrictions) | +| 937 | `amc score lean-profile` | Show lean AMC profile | +| 938 | `amc score level-transition` | Track formal promotion/demotion events with evidence gates | +| 939 | `amc score memory-depth` | Score deep memory infrastructure: backend resilience, compression fidelity, cross-session consistency, TTL, capacity | +| 940 | `amc score memory-integrity` | Score memory correction persistence and poisoning resistance | +| 941 | `amc score mutual-verification` | Score agent-to-agent trust verification (challenge-response) | +| 942 | `amc score operational-independence` | Calculate operational independence score | +| 943 | `amc score output-attestation` | Score output signing and trust metadata for receiving agents | +| 944 | `amc score output-integrity` | Score output integrity maturity (OWASP LLM02, confidence calibration, citation) | +| 945 | `amc score owasp-llm` | Score OWASP LLM Top 10 coverage (all 10 risks) | +| 946 | `amc score pause-quality` | Score quality of agent-initiated pauses | +| 947 | `amc score policy-consistency` | Test policy enforcement consistency across repeated trials (pass^k) | +| 948 | `amc score production-ready` | Run production readiness gate for an agent | +| 949 | `amc score regulatory-readiness` | Compute weighted regulatory readiness score (EU AI Act + ISO + OWASP) | +| 950 | `amc score runtime-identity` | Score runtime execution identity maturity (JIT credentials, user propagation, revocation) | +| 951 | `amc score safety-research` | Run the AI Safety Research evaluation lane — 4-dimension assessment based on frontier safety research | +| 952 | `amc score self-knowledge` | Score prior art self-knowledge maturity (typed attention, trace layer, confidence+citation) | +| 953 | `amc score simulation-lane` | Run the Simulation & Forecast evaluation lane — 5-dimension assessment for simulation/forecast systems | +| 954 | `amc score sleeper-detection` | Detect context-dependent behavioral inconsistencies | +| 955 | `amc score state-portability` | Score agent state portability (vendor-neutral format, serialization, integrity on transfer) | +| 956 | `amc score task-horizon` | Score task-completion time horizon (METR-inspired) | +| 957 | `amc score tier` | Run tiered maturity assessment (quick/standard/deep) | +| 958 | `amc score transparency-log` | Score network transparency log (Merkle tree, inclusion proofs) | +| 959 | `amc sessions` | View and analyze user sessions | +| 960 | `amc sessions list` | List tracked sessions | +| 961 | `amc setup` | Setup wizard for the full-score path and Studio gateway | +| 962 | `amc shell` | Interactive AMC session — natural language + commands | +| 963 | `amc shield` | Threat detection and security scanning | +| 964 | `amc shield analyze` | Run static code analyzer on a skill file | +| 965 | `amc shield analyze-mcp` | Scan an MCP server definition for security risks (score L0–L5) | +| 966 | `amc shield analyze-runtime` | Analyze a proposed runtime agent action through the Shield trust pipeline | +| 967 | `amc shield confirm` | Controlled exploit confirmation with strict authorization gates | +| 968 | `amc shield confirm export` | Export a redacted safe proof without exploit instructions | +| 969 | `amc shield confirm proofs` | List safe exploit-confirmation proof artifacts | +| 970 | `amc shield confirm run` | Run authorized safe exploit confirmation from a task JSON file | +| 971 | `amc shield confirm scope-write` | Write a signed exploit-confirmation authorization scope from JSON | +| 972 | `amc shield confirm scopes` | List exploit-confirmation authorization scopes | +| 973 | `amc shield conversation-integrity` | Check conversation integrity for an agent (demo) | +| 974 | `amc shield detect-injection` | Detect prompt injection attempts in text | +| 975 | `amc shield red-team` | Run a quick red team campaign (5 attacks on demo target). Tip: For full red-team suite with strategies, use `amc redteam run` | +| 976 | `amc shield red-team-status` | Show current red team capabilities and attack template count | +| 977 | `amc shield reputation` | Check reputation score for a tool | +| 978 | `amc shield sandbox` | Check sandbox configuration for an agent | +| 979 | `amc shield sanitize` | Sanitize text — strip LLM prompt injection and dangerous AI patterns (not SQL/XSS) | +| 980 | `amc shield sbom` | Generate software bill of materials from package.json | +| 981 | `amc shield threat-intel` | Check threat intelligence for an input | +| 982 | `amc shield trust-pipeline` | Run end-to-end trust pipeline for an agent action | +| 983 | `amc simulate-bridge` | Run a simulated bridge request for local testing | +| 984 | `amc snapshot` | Generate Unified Clarity Snapshot markdown | +| 985 | `amc sso` | SSO setup shortcuts for OIDC and SAML providers | +| 986 | `amc sso configure` | Configure an OIDC or SAML SSO provider | +| 987 | `amc standard` | Open Compass Standard schema bundle and validation | +| 988 | `amc standard generate` | Generate signed Open Compass schema bundle under .amc/standard/ | +| 989 | `amc standard print` | Print one generated schema | +| 990 | `amc standard schemas` | List generated schemas with digests | +| 991 | `amc standard validate` | Validate a JSON file or AMC artifact against a standard schema | +| 992 | `amc standard verify` | Verify schema bundle signatures and manifest digests | +| 993 | `amc status` | Show AMC Studio and vault status | +| 994 | `amc strategy` | Compare inference strategies and govern route changes | +| 995 | `amc strategy compare` | Compare model/provider strategies with score, cost, latency, risk, and evidence | +| 996 | `amc strategy list` | List inference strategy comparison runs | +| 997 | `amc strategy rollback` | Roll back an accepted inference route change | +| 998 | `amc strategy show` | Inspect an inference strategy comparison run | +| 999 | `amc studio` | Studio API helpers | +| 1000 | `amc studio healthcheck` | Health/readiness probe for deployment runtime | +| 1001 | `amc studio lan` | LAN mode controls for Compass Console | +| 1002 | `amc studio lan disable` | Disable LAN mode and revert to localhost-only | +| 1003 | `amc studio lan enable` | Enable LAN mode with pairing gate | +| 1004 | `amc studio ping` | Ping local Studio API /health endpoint | +| 1005 | `amc studio start` | Start Studio in foreground (non-interactive, deployment-safe) | +| 1006 | `amc supervise` | Supervise any agent process and inject gateway/proxy routing env vars | +| 1007 | `amc target` | Target profile operations | +| 1008 | `amc target diff` | Diff run against target profile | +| 1009 | `amc target set` | Interactive equalizer wizard | +| 1010 | `amc target verify` | Verify target profile signature | +| 1011 | `amc tenant-isolation-check` | Check tenant isolation between all registered tenants | +| 1012 | `amc tenant-register` | Register a tenant boundary | +| 1013 | `amc ticket` | Execution ticket operations | +| 1014 | `amc ticket issue` | Issue short-lived signed execution ticket | +| 1015 | `amc ticket verify` | Verify signed execution ticket | +| 1016 | `amc tools` | ToolHub tools config | +| 1017 | `amc tools init` | Create and sign .amc/tools.yaml | +| 1018 | `amc tools list` | List allowed ToolHub tools and action classes | +| 1019 | `amc tools verify` | Verify tools.yaml signature | +| 1020 | `amc trace` | Trace explorer — inspect agent execution traces, sessions, and tool calls | +| 1021 | `amc trace failures` | Show top recurring failure clusters mined from trace indexes | +| 1022 | `amc trace index` | List or inspect distilled trace failure indexes | +| 1023 | `amc trace inspect` | Inspect evidence events — show tool calls, decisions, and trust tiers | +| 1024 | `amc trace list` | List recent agent sessions with evidence summary | +| 1025 | `amc trace stats` | Show trace statistics — event counts by type, trust tier, tool usage | +| 1026 | `amc transform` | Transformation OS (4C plans, tracking, attestations) | +| 1027 | `amc transform attest` | - | +| 1028 | `amc transform attest-verify` | - | +| 1029 | `amc transform init` | Initialize signed .amc/transform-map.yaml | +| 1030 | `amc transform map` | Inspect or apply transform map | +| 1031 | `amc transform map apply` | - | +| 1032 | `amc transform map show` | - | +| 1033 | `amc transform plan` | - | +| 1034 | `amc transform report` | - | +| 1035 | `amc transform status` | - | +| 1036 | `amc transform track` | - | +| 1037 | `amc transform verify` | Verify signed transform map | +| 1038 | `amc transparency` | Append-only transparency log operations | +| 1039 | `amc transparency export` | Export transparency bundle | +| 1040 | `amc transparency init` | Initialize append-only transparency log | +| 1041 | `amc transparency merkle` | Merkle transparency root/proof operations | +| 1042 | `amc transparency merkle prove` | Export signed inclusion proof bundle for entry hash | +| 1043 | `amc transparency merkle rebuild` | Rebuild Merkle leaves/roots from transparency log | +| 1044 | `amc transparency merkle root` | Show current Merkle root and history | +| 1045 | `amc transparency merkle verify-proof` | Verify signed inclusion proof bundle | +| 1046 | `amc transparency report` | Generate an Agent Transparency Report — what the agent does, can access, and how trustworthy it is | +| 1047 | `amc transparency tail` | Tail transparency entries | +| 1048 | `amc transparency verify` | Verify transparency chain + seal signature | +| 1049 | `amc transparency verify-bundle` | Verify exported transparency bundle | +| 1050 | `amc trust` | Trust mode and Notary enforcement configuration | +| 1051 | `amc trust enable-notary` | Enable fail-closed NOTARY trust mode | +| 1052 | `amc trust freshness` | Report temporal trust freshness and half-life decay | +| 1053 | `amc trust init` | Create and sign .amc/trust.yaml — sets up the trust mode (SELF/NOTARY) that governs artifact signing | +| 1054 | `amc trust status` | Show trust mode, signature status, and notary health | +| 1055 | `amc truthguard` | Deterministic output truth-constraint validator | +| 1056 | `amc truthguard validate` | Validate structured agent output claims against deterministic truth constraints | +| 1057 | `amc tune` | Mechanic mode tuning wizard | +| 1058 | `amc unknowns` | List known unknowns for an agent's latest diagnostic run | +| 1059 | `amc up` | Start AMC control plane in one command (studio + gateway + bridge) | +| 1060 | `amc upgrade` | Generate upgrade plan | +| 1061 | `amc user` | Multi-user RBAC account management | +| 1062 | `amc user add` | Add a user with RBAC roles | +| 1063 | `amc user init` | Initialize signed users.yaml with first OWNER user | +| 1064 | `amc user list` | List RBAC users | +| 1065 | `amc user revoke` | Revoke a user account | +| 1066 | `amc user role` | Set user roles | +| 1067 | `amc user role set` | Replace roles for a user | +| 1068 | `amc user verify` | Verify users.yaml signature | +| 1069 | `amc value` | Value realization engine (contracts, scoring, ROI) | +| 1070 | `amc value contract` | Value contract operations | +| 1071 | `amc value contract apply` | Apply value contract from YAML/JSON file | +| 1072 | `amc value contract init` | Create and sign value contract template | +| 1073 | `amc value contract print` | Print value contract and signature status | +| 1074 | `amc value contract verify` | Verify value contract signature | +| 1075 | `amc value import` | Import numeric KPI points from CSV (ts,value) | +| 1076 | `amc value ingest` | Ingest value webhook payload JSON | +| 1077 | `amc value init` | Initialize signed value policy, default contract, and scheduler | +| 1078 | `amc value policy` | Value policy operations | +| 1079 | `amc value policy apply` | Apply signed value policy from YAML/JSON file | +| 1080 | `amc value policy default` | Print default value policy JSON | +| 1081 | `amc value policy print` | Print effective value policy JSON | +| 1082 | `amc value report` | Generate signed value report | +| 1083 | `amc value scheduler` | Value scheduler controls | +| 1084 | `amc value scheduler disable` | Disable value scheduler | +| 1085 | `amc value scheduler enable` | Enable value scheduler | +| 1086 | `amc value scheduler run-now` | Run value scheduler now | +| 1087 | `amc value scheduler status` | Show value scheduler status | +| 1088 | `amc value snapshot` | Generate/load latest signed value snapshot | +| 1089 | `amc value verify` | Verify value workspace signatures/artifacts | +| 1090 | `amc value verify-policy` | Verify signed value policy | +| 1091 | `amc vault` | Encrypted key vault operations | +| 1092 | `amc vault classify` | Classify data sensitivity level | +| 1093 | `amc vault dlp` | DLP scanner for PII and secrets | +| 1094 | `amc vault dlp scan` | Scan text for PII and secrets | +| 1095 | `amc vault dsar` | Persistent DSAR (Data Subject Access Request) workflow | +| 1096 | `amc vault dsar complete` | Mark a DSAR request complete and append an audit event | +| 1097 | `amc vault dsar list` | List persistent DSAR requests | +| 1098 | `amc vault dsar status` | Show a persistent DSAR request | +| 1099 | `amc vault dsar submit` | Submit a persistent DSAR request | +| 1100 | `amc vault dsar-status` | Show DSAR (Data Subject Access Request) status | +| 1101 | `amc vault init` | Initialize encrypted vault for signing keys | +| 1102 | `amc vault lock` | Lock vault and clear in-memory private keys | +| 1103 | `amc vault privacy-budget` | Check privacy budget for an agent | +| 1104 | `amc vault rag-guard` | Guard RAG chunks against injection | +| 1105 | `amc vault rotate-keys` | Rotate monitor signing key and append to public key history | +| 1106 | `amc vault scrub` | Scrub metadata from a file | +| 1107 | `amc vault secret-share` | Split a secret into shares using Shamir's Secret Sharing | +| 1108 | `amc vault status` | Show vault status | +| 1109 | `amc vault unlock` | Unlock vault into memory for signing operations | +| 1110 | `amc vault zk-commit` | Create a Pedersen commitment to a value | +| 1111 | `amc vault zk-range-proof` | Create a zero-knowledge range proof that an AMC score meets a threshold | +| 1112 | `amc vault zk-verify` | Verify a ZK range proof (pass JSON as string) | +| 1113 | `amc verify` | Verify integrity across AMC artifacts | +| 1114 | `amc verify all` | Verify trust/policies/plugins/logs/ledger/artifacts in one pass | +| 1115 | `amc vibe-audit` | Run static safety checks for AI-generated code | +| 1116 | `amc watch` | Observability, attestation, and safety testing | +| 1117 | `amc watch alerts` | Show recent alerts for a monitored agent | +| 1118 | `amc watch attest` | Attest an agent output | +| 1119 | `amc watch connect` | Connect to an observability provider (langfuse, helicone, otlp, datadog, webhook) | +| 1120 | `amc watch explain` | Generate explainability packet for an agent run | +| 1121 | `amc watch host-hardening` | Check host hardening status for this AMC deployment | +| 1122 | `amc watch profiler-anomalies` | List detected behavioral anomalies for an agent | +| 1123 | `amc watch profiler-start` | Start behavioral profiling for an agent | +| 1124 | `amc watch profiler-status` | Show behavioral profiler status and any recent anomalies | +| 1125 | `amc watch providers` | Show connected observability providers and trace stats | +| 1126 | `amc watch safety-test` | Run safety tests for an agent | +| 1127 | `amc watch start` | Start continuous production monitoring for an agent | +| 1128 | `amc watch status` | Show all monitored agents and their current state | +| 1129 | `amc whatif` | Equalizer what-if simulator | +| 1130 | `amc whatif equalizer` | - | +| 1131 | `amc whatif targets` | - | +| 1132 | `amc why-capped` | Show why each question is capped at its current level | +| 1133 | `amc wiring-status` | Show production wiring status for all modules (Items 11-16) | +| 1134 | `amc workorder` | Signed work order operations | +| 1135 | `amc workorder create` | Create and sign a work order | +| 1136 | `amc workorder expire` | Expire/revoke a work order | +| 1137 | `amc workorder list` | List work orders for agent | +| 1138 | `amc workorder show` | Show signed work order JSON | +| 1139 | `amc workorder verify` | Verify work order signature | +| 1140 | `amc wrap` | Wrap runtime and capture tamper-evident evidence | ### Command Details -#### `amc init` +#### `amc action-queue` + +Show prioritized actions sorted by risk-reduction-per-effort -Initialize .amc workspace | Option | Description | |--------|-------------| -| `--trust-boundary ` | isolated|shared | +| `--limit ` | - | -#### `amc doctor` +#### `amc adapters configure` + +Set adapter profile for an agent (signed adapters.yaml) -Check runtime availability and wrap readiness | Option | Description | |--------|-------------| -| `--json` | emit structured JSON output | +| `--agent ` | - | +| `--adapter ` | - | +| `--route ` | - | +| `--model ` | - | +| `--mode ` | - | -#### `amc doctor-fix` +#### `amc adapters env` + +Print adapter-compatible environment exports without lease token -Auto-repair common setup issues | Option | Description | |--------|-------------| -| `--dry-run` | Preview fixes without applying | -| `--json` | Emit structured JSON output | +| `--agent ` | - | +| `--adapter ` | - | -#### `amc improve` +#### `amc adapters init-project` + +Generate runnable local adapter sample for library-based frameworks -Guided improvement — shows what to fix next based on your current score | Option | Description | |--------|-------------| -| `--json` | emit JSON output | +| `--adapter ` | - | +| `--agent ` | - | +| `--route ` | - | -#### `amc guide` +#### `amc adapters run` + +Run adapter with minted lease, routed through gateway, with observed evidence capture -Generate personalized improvement guide with exportable agent instructions | Option | Description | |--------|-------------| -| `--target ` | target maturity level (1-5) | -| `--export` | export markdown files to .amc/guides/ | -| `--agent-instructions` | export agent-consumable instructions (for AGENTS.md / system prompts) | -| `--guardrails` | generate operational guardrails (rules, not suggestions) | -| `--apply [file]` | apply guardrails directly to agent config file (auto-detects or specify path) | -| `--interactive` | mechanic mode — choose which gaps to fix | -| `--watch` | continuous monitoring — re-generate guide on trust drift | -| `--watch-interval ` | interval for watch mode in seconds | -| `--diff` | show what changed since last guide generation | -| `--frameworks` | list all supported frameworks | -| `--ci` | CI gate mode — exit non-zero if below --target level | -| `--dry-run` | preview --apply changes without writing files | -| `--quick` | instant guide from defaults — no interactive questions | -| `--auto-detect` | auto-detect framework from project files | -| `--status` | one-line status: current level, gap count, severities | -| `--go` | all-in-one: quick + auto-detect + export + apply (zero friction) | -| `--compliance [frameworks]` | generate compliance guardrails (EU_AI_ACT,ISO_42001,NIST_AI_RMF,SOC2,ISO_27001 or | -| `--agent ` | agent ID | -| `--framework ` | framework name for tailored instructions | -| `--json` | emit JSON output | +| `--agent ` | - | +| `--adapter ` | - | +| `--workorder ` | - | +| `--mode ` | - | -#### `amc quickscore` +#### `amc advisory ack` + +Acknowledge an advisory -Zero-config rapid assessment — auto-scores from evidence, or interactive 5-question fallback | Option | Description | |--------|-------------| -| `--json` | emit JSON output | -| `--quiet` | suppress non-JSON output (use with --json for clean piping) | -| `--eu-ai-act` | show EU AI Act risk classification mapping | -| `--auto` | auto-score from ledger evidence (no questions asked) | -| `--agent ` | agent ID for auto mode | +| `--note ` | - | +| `--by ` | - | -#### `amc explain ` +#### `amc advisory list` + +List advisories for scope -Plain-English explanation for a diagnostic question (example: AMC-2.1) | Option | Description | |--------|-------------| -| `--json` | emit JSON output | +| `--scope ` | - | +| `--id ` | - | -#### `amc bootstrap` +#### `amc agent diagnose` + +Lease-auth self-run diagnostic (agent-triggered, evidence-scored server-side) -Bootstrap workspace for production deployment (non-interactive) | Option | Description | |--------|-------------| -| `--workspace ` | workspace directory (defaults to AMC_WORKSPACE_DIR or cwd) | +| `--token-file ` | - | +| `--studio ` | - | -#### `amc user` +#### `amc agent harness` + +Run the autonomous improvement harness loop -Host user management | Option | Description | |--------|-------------| -| `--host-admin` | grant host-admin privileges | +| `--type ` | - | +| `--iterations ` | - | +| `--target ` | - | -#### `amc migrate` +#### `amc agent run` + +Run an AMC-governed agent (content-moderation, data-pipeline, legal-contract) -Migrate an existing single-workspace AMC directory into host mode | Option | Description | |--------|-------------| -| `--move` | move source directory instead of copying | -| `--username ` | host username to grant OWNER+AUDITOR in migrated workspace | -| `--name ` | workspace display name | +| `--input ` | - | -#### `amc print` +#### `amc alert config` + +Configure alert destinations (webhooks, Slack, PagerDuty) -Print resolved runtime config (secret-safe) | Option | Description | |--------|-------------| -| `--json` | emit JSON | +| `--set-webhook ` | - | +| `--set-slack ` | - | +| `--set-pagerduty ` | - | +| `--show` | - | +| `--json` | - | -#### `amc explain` +#### `amc alert send` + +Send an alert to a webhook endpoint -Explain config source precedence and risky settings | Option | Description | |--------|-------------| -| `--json` | emit JSON | +| `--url ` | - | +| `--message ` | - | +| `--severity ` | - | +| `--agent ` | - | +| `--json` | - | -#### `amc logs` +#### `amc alert watch` + +Watch for anomalies and auto-send alerts to configured destinations -Print latest AMC Studio logs | Option | Description | |--------|-------------| -| `--lines ` | lines per log file | +| `--agent ` | - | +| `--interval ` | - | -#### `amc start` +#### `amc api key create` + +Create a programmatic API key and show the secret once -Start Studio in foreground (non-interactive, deployment-safe) | Option | Description | |--------|-------------| -| `--workspace ` | workspace directory (defaults to AMC_WORKSPACE_DIR) | -| `--bind ` | api bind host override | -| `--port ` | api port override | -| `--dashboard-port ` | dashboard port override | +| `--scope ` | - | +| `--label