From 1cb2b32afe2c5c8d3b9aee9ae7d4a2f719efcea6 Mon Sep 17 00:00:00 2001 From: Jesper Schulz-Wedde Date: Wed, 16 Sep 2026 13:20:57 +0200 Subject: [PATCH 1/2] Add AL implementation guidance skill Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .claude-plugin/marketplace.json | 4 +- .github/scripts/validate_frontmatter.py | 6 +- .github/workflows/development-guidance.yml | 9 +- README.md | 31 +- docs/agent-consumption.md | 120 ++++++- evaluation/README.md | 38 +++ .../implementation-guidance-fixtures.json | 208 ++++++++++++ .../development/al-implementation-guidance.md | 62 ++++ plugin.json | 5 +- skills/README.md | 4 + skills/al-implementation-guidance/SKILL.md | 22 ++ skills/do.md | 120 ++++++- skills/entry.md | 7 +- tools/DevelopmentGuidance.Evidence.ps1 | 295 +++++++++++++++++- tools/Test-DevelopmentGuidanceFixtures.ps1 | 14 +- .../Test-ImplementationGuidanceEvaluator.ps1 | 206 ++++++++++++ 16 files changed, 1120 insertions(+), 31 deletions(-) create mode 100644 evaluation/implementation-guidance-fixtures.json create mode 100644 microsoft/skills/development/al-implementation-guidance.md create mode 100644 skills/al-implementation-guidance/SKILL.md create mode 100644 tools/Test-ImplementationGuidanceEvaluator.ps1 diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 1fd35cd9..f328f1de 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -8,8 +8,8 @@ { "name": "bcquality", "source": "./", - "description": "Business Central AL quality knowledge base and skills, packaged as an installable plugin. Exposes read-only plan enrichment and review adapters through BCQuality's Entry protocol.", - "version": "0.3.0", + "description": "Business Central AL quality knowledge base and skills, packaged as an installable plugin. Exposes read-only plan, implementation-guidance, and review adapters through BCQuality's Entry protocol.", + "version": "0.4.0", "skills": [ "./skills/" ] diff --git a/.github/scripts/validate_frontmatter.py b/.github/scripts/validate_frontmatter.py index 6c17e93e..56a6690b 100644 --- a/.github/scripts/validate_frontmatter.py +++ b/.github/scripts/validate_frontmatter.py @@ -46,9 +46,11 @@ STANDARD_INPUTS = { "pr-diff", "object-list", "file-path", "folder-path", "repository", "telemetry-query", - "development-plan", + "development-plan", "implementation-diff", "decision-context", "consumed-guidance", +} +ALLOWED_OUTPUTS = { + "findings-report", "development-guidance-report", "implementation-guidance-report", } -ALLOWED_OUTPUTS = {"findings-report", "development-guidance-report"} VALID_SAMPLE_KINDS = {"good", "bad"} ACTION_SKILL_SECTIONS = ["Source", "Relevance", "Worklist", "Action", "Output"] diff --git a/.github/workflows/development-guidance.yml b/.github/workflows/development-guidance.yml index 51e24b3f..d95ac3cf 100644 --- a/.github/workflows/development-guidance.yml +++ b/.github/workflows/development-guidance.yml @@ -2,7 +2,6 @@ name: Validate read-only development guidance on: pull_request: - branches: [main] push: branches: [main] @@ -23,3 +22,11 @@ jobs: - name: Run credential-free guidance evaluator regressions shell: pwsh run: ./tools/Test-DevelopmentGuidanceEvaluator.ps1 + + - name: Validate and prepare implementation-guidance fixtures + shell: pwsh + run: ./tools/Test-DevelopmentGuidanceFixtures.ps1 -Root . -ManifestPath ./evaluation/implementation-guidance-fixtures.json -PrepareDirectory "$env:RUNNER_TEMP/bcquality-implementation-guidance-fixtures" + + - name: Run implementation-guidance evaluator regressions + shell: pwsh + run: ./tools/Test-ImplementationGuidanceEvaluator.ps1 diff --git a/README.md b/README.md index c479736c..0c80480a 100644 --- a/README.md +++ b/README.md @@ -29,11 +29,13 @@ copilot plugin list The list should include `bcquality`. The plugin exposes [`al-code-review`](skills/al-code-review/SKILL.md) and the read-only [`al-development-plan`](skills/al-development-plan/SKILL.md) plan-enrichment -skill. Installation and skill discovery are the general pattern; reviewing an -app is one example of using it. +and [`al-implementation-guidance`](skills/al-implementation-guidance/SKILL.md) +consultation skills. Installation and skill discovery are the general pattern; +reviewing an app is one example of using it. -Plugin version `0.3.0` adds `al-development-plan`, a read-only adapter for -enriching an **existing** plan. It does not generate a plan or implement code. +Plugin version `0.4.0` adds `al-implementation-guidance`, a read-only, +just-in-time consultation over the current diff and decision context. It does +not edit code or run an implementation workflow. The adapters are intentionally not second implementations: @@ -47,6 +49,11 @@ standalone host skill: skills/al-development-plan/SKILL.md -> routing contract: skills/entry.md -> enrichment skill: microsoft/skills/development/al-development-plan.md -> referenced constraints for the consumer's existing workflow (read-only) + +standalone host skill: skills/al-implementation-guidance/SKILL.md + -> routing contract: skills/entry.md + -> consultation skill: microsoft/skills/development/al-implementation-guidance.md + -> focused constraints for the consumer's next decision (read-only) ``` Only the files under `skills/*/SKILL.md` follow the host's packaging format. @@ -113,13 +120,16 @@ intentionally left to those deterministic tools rather than duplicated here. The read-only `al-development-plan` interface selects relevant constraints before the consumer implements its own existing plan. +The distinct `al-implementation-guidance` interface consults the same corpus +after implementation begins and selects only constraints capable of changing +the next bounded implementation or validation decision. Repository-specific orchestrators retain planning, implementation, approvals, tests, environment, propagation, and delivery ownership. The intended flow is -consumer analysis and normalized plan -> read-only BCQuality guidance -> -existing implementation phases -> independent final BCQuality review -> -delivery. Consumer uptake and a real runtime pilot are follow-up work, not -implemented integrations or demonstrated authoring improvements. +consumer analysis and normalized plan -> read-only plan guidance -> consumer +implementation -> explicit read-only implementation checkpoints -> consumer +edits and tests -> independent final BCQuality review -> delivery. BCQuality +does not own checkpoint state or invoke itself automatically. `no-knowledge` means no additional applicable BCQuality constraints, with empty `knowledge`; it does not make a plan unsafe or prevent the consumer from using @@ -151,8 +161,9 @@ and apply the relevant knowledge. Both live in three layers: | [Community](community/) | Community-owned skills and their knowledge. | | [Custom](custom/) | Organization-specific additions and overrides in your own fork. | -Review skills emit a `findings-report`; plan enrichment emits a read-only -`development-guidance-report`. Both contracts are defined in +Review skills emit a `findings-report`; plan enrichment emits +`development-guidance-report`; implementation consultation emits +`implementation-guidance-report`. All contracts are defined in [`skills/do.md`](skills/do.md). See [how agents consume BCQuality](docs/agent-consumption.md) for the integration flow. diff --git a/docs/agent-consumption.md b/docs/agent-consumption.md index b8f2d584..02803665 100644 --- a/docs/agent-consumption.md +++ b/docs/agent-consumption.md @@ -79,7 +79,8 @@ Add scheduling, retries, and rendering only when needed, using the When BCQuality is installed as a standalone plugin, it additionally exposes `skills/al-code-review/SKILL.md` and -`skills/al-development-plan/SKILL.md`. These are host-format adapters, not +`skills/al-development-plan/SKILL.md`, and +`skills/al-implementation-guidance/SKILL.md`. These are host-format adapters, not additional action skills: each creates the task context and enters the same flow at Entry. @@ -175,6 +176,7 @@ The output contracts are defined in the DO meta-skill: - A **findings report** carries review findings, domain labels, references, confidence, and suppressions. - A **development guidance report** carries read-only knowledge constraints and validation considerations for an existing plan. +- An **implementation guidance report** carries focused additional constraints for a current diff, phase, and next implementation or validation decision. The orchestrator parses this **without skill-specific logic**. This is the point of the contract: orchestrators and action skills evolve independently. @@ -183,6 +185,11 @@ selects applicable knowledge, and returns constraints without changing the target. It does not generate a replacement plan, run tests, implement code, or drive a review/fix loop. Implementation stays in the consuming workflow. +For implementation consultation, the skill additionally reads the current +implementation diff and explicit decision context. It does not broaden the +plan again: it returns only knowledge that can change the next bounded +implementation or validation decision. + ### 7. Orchestrator integrates The orchestrator turns findings into PR comments, build gates, or IDE diagnostics. It can feed read-only guidance into its own implementation phases, preserving all existing approvals and delivery gates. @@ -205,11 +212,21 @@ integration boundary, not a shipped consumer integration: 3. Execute the dispatched `al-development-plan` skill. Persist the unchanged report and provenance **after** any consumer state initialization or cleanup that could erase them. BCQuality does not own the state directory or lifecycle. -4. Inject relevant constraints and validation considerations into the existing - Baseline, Implement, propagation (such as MiApp), and Critique phases, or - equivalents. Re-enrich on material plan or applicability changes; preserve - the relationship between plan version, guidance, and implementation attempt. -5. Run an independent final BCQuality review against the completed diff using +4. Implement in the consumer workflow. At explicit checkpoints, invoke + `al-implementation-guidance` with the existing plan, readable repository, + current implementation diff, and decision context. Useful checkpoints + include schema/data upgrade, public APIs/events/interfaces, permissions, + external effects/job queues/`HttpClient`, telemetry/privacy, UI/page + background tasks, and tests. The coding agent may also initiate a bounded + consultation. This hybrid recommendation is visible orchestration, not + automatic hidden behavior. +5. Apply returned constraints, edit, compile, test, and retry in the consumer. + Persist consumed article path, decision key, and evidence fingerprint in + consumer-owned state. Requery when the diff, affected files/symbols, + changed AL tokens, next decision, acceptance criteria, validation result, or + applicability context materially changes. Exact unchanged triples are + omitted deterministically; BCQuality stores no state. +6. Run an independent final BCQuality review against the completed diff using the **same recorded immutable checkout** used for enrichment. Review the actual changes, not the guidance report as proof of correctness, then apply the consumer's ordinary delivery gates. @@ -230,6 +247,90 @@ outcomes and recorded unknowns: seek missing context, re-enrich, escalate, or apply their existing risk controls without relabeling the report as successful. Do not fill gaps with generic articles simply to unlock implementation. +For implementation consultation, `no-knowledge` means no additional applicable +constraints for the exact current decision and evidence fingerprint. It does +not establish functional correctness, safe deployment, adequate tests, or +release readiness. + +### Minimal implementation consultation + +The consumer supplies the actual values separately from Entry's type list: + +```text +BCQuality root: C:\Knowledge\BCQuality +development-plan: +repository: C:\Repos\MyBusinessCentralApp +implementation-diff: +decision-context: + phase: public-contract-checkpoint + decision: Choose a compatible shape for the new provider capability. + decision-key: provider-capability-contract + evidence-fingerprint: sha256: + affected-files: [src/Provider/INotificationProvider.Interface.al] + affected-symbols: [interface Notification Provider] + changed-tokens: [interface, procedure] + tests: [Compile existing and new provider implementations.] + acceptance-criteria: [Existing implementations remain compatible.] + development-plan-pin: plan:v1 + implementation-evidence-pin: diff:v2 +consumed-guidance: [] + +Invoke skills/entry.md with inputs-available: +[development-plan, repository, implementation-diff, decision-context, +consumed-guidance]. Execute only the dispatched implementation-guidance skill +and return its implementation-guidance-report unchanged. +``` + +An abbreviated successful report is: + +```json +{ + "skill": { "id": "al-implementation-guidance", "version": 1 }, + "outcome": "completed", + "summary": { + "request": "Expose delivery status through the provider contract.", + "phase": "public-contract-checkpoint", + "decision": "Choose a compatible shape for the new provider capability.", + "decision-key": "provider-capability-contract", + "evidence-fingerprint": "sha256:example", + "candidates": 1, + "selected": 1, + "omitted-consumed": 0 + }, + "pins": { + "knowledge-checkout": "0123456789abcdef0123456789abcdef01234567", + "development-plan": "plan:v1", + "implementation-evidence": "diff:v2" + }, + "context": { + "bc-version": "28", + "technologies": ["al"], + "countries": ["w1"], + "application-area": ["all"], + "affected-files": ["src/Provider/INotificationProvider.Interface.al"], + "affected-symbols": ["interface Notification Provider"], + "changed-tokens": ["interface", "procedure"], + "unknown": [] + }, + "knowledge": [ + { + "path": "microsoft/knowledge/interfaces/extend-published-interfaces-dont-edit-them.md", + "used-for": "Choose a compatible public contract shape.", + "constraints": ["Preserve the shipped interface and use the article's compatible extension shape."], + "sample-paths": ["microsoft/knowledge/interfaces/extend-published-interfaces-dont-edit-them.good.al"] + } + ], + "validation-considerations": [], + "deduplication": { "strategy": "omit-exact-consumed-match", "omitted": [] }, + "suppressed": [], + "unresolved": [] +} +``` + +The example illustrates shape and provenance, not a functional-correctness +claim. Consumers must still open the cited knowledge through the skill, +compile, test, and independently review the resulting diff. + ### Pinning and pilot evidence A configured tag or ref alone does not establish runtime pinning. Record the @@ -248,6 +349,13 @@ final reviews, including failures and unresolved results. Credential-free fixture preparation and scorer regressions do not demonstrate improved repair quality, compilation, runtime success, or production integration. +A focused authoring experiment should use four matched arms: baseline without +guidance, plan guidance only, implementation guidance only, and combined plan +plus implementation guidance. Hold starting task/code, model, tools, runtime, +checkpoints, and delivery gates constant; evaluate resulting diffs, compile and +test evidence, independent final review, guidance precision, duplication, and +cost. This is a recommended design, not a reported experiment or result. + ## Knowledge-backed and agent findings BCQuality is an **additive** knowledge layer. The agent surfaces two kinds of findings, both shaped to the same DO output contract: diff --git a/evaluation/README.md b/evaluation/README.md index 83cf6050..109f90db 100644 --- a/evaluation/README.md +++ b/evaluation/README.md @@ -168,3 +168,41 @@ article-read traces, resulting diffs, compile/test outcomes, and independent review evidence, including failures, no-knowledge, partial, and unresolved results. No compile/run or authoring-quality claim follows from the credential-free checks above. + +## Read-only implementation guidance + +`implementation-guidance-fixtures.json` covers a distinct just-in-time +consultation contract. Its production-shaped synthetic AL contexts exercise a +new privacy constraint after the implementation surface expands beyond the +plan, exact consumed-guidance omission, an honest no-additional-guidance +control, missing current diff/decision context, and a public-interface +checkpoint. They are neutral fixtures, not copies of knowledge samples or +evidence of a production integration. + +Validate and prepare the implementation manifest with the shared scorer: + +```powershell +$run = Join-Path ([IO.Path]::GetTempPath()) 'bcquality-implementation-guidance-run' +pwsh ./tools/Test-DevelopmentGuidanceFixtures.ps1 -Root . ` + -ManifestPath ./evaluation/implementation-guidance-fixtures.json ` + -PrepareDirectory $run +pwsh ./tools/Test-ImplementationGuidanceEvaluator.ps1 +``` + +The same runner-owned baseline and workspace-map flow used for plan guidance +proves that evaluated target repositories remain byte-for-byte and +Git-identity stable across scoring. The implementation regression adds strict +phase/decision/pin preservation and exact path + decision key + evidence +fingerprint deduplication checks. This remains before/after evidence, not an OS +sandbox; it cannot detect a transient reverted write. + +BCQuality does not keep consumed state or run checkpoints automatically. The +consumer chooses explicit checkpoints, persists consumed triples, recomputes +evidence fingerprints, owns edits/build/tests/retries/delivery, and runs an +independent final review. `no-knowledge` is only an additive retrieval outcome, +never a functional-correctness or release-readiness claim. + +A future experiment should compare four matched arms: baseline, plan-only, +implementation-only, and combined. Hold task, starting code, model, tools, +runtime, checkpoints, and gates constant. This is a recommended evaluation +design, not a claimed result. diff --git a/evaluation/implementation-guidance-fixtures.json b/evaluation/implementation-guidance-fixtures.json new file mode 100644 index 00000000..e638bf37 --- /dev/null +++ b/evaluation/implementation-guidance-fixtures.json @@ -0,0 +1,208 @@ +{ + "version": 1, + "skill": "microsoft/skills/development/al-implementation-guidance.md", + "minimumKnowledgeRecall": 1.0, + "minimumKnowledgePrecision": 0.67, + "cases": [ + { + "id": "expanded-surface-adds-pii-field", + "title": "Current implementation expands beyond the plan into a privacy-sensitive schema surface", + "evidenceType": "production-shaped-synthetic", + "expectedOutcome": "completed", + "development-plan": { + "request": "Add an opt-in notification channel selected by an internal setup routine.", + "affected-files": ["src/Notifications/NotificationSetup.Codeunit.al"], + "test-strategy": "Verify the selected channel is persisted and used.", + "acceptance-criteria": ["A tenant can select and use the notification channel."] + }, + "implementation-diff": "diff --git a/src/Notifications/NotificationSetup.TableExt.al b/src/Notifications/NotificationSetup.TableExt.al\nnew file mode 100644\n--- /dev/null\n+++ b/src/Notifications/NotificationSetup.TableExt.al\n@@\n+tableextension 71000 \"Notification Setup Ext.\" extends \"Notification Setup\"\n+{\n+ fields\n+ {\n+ field(71000; \"Recipient E-Mail\"; Text[250])\n+ {\n+ DataClassification = ToBeClassified;\n+ }\n+ }\n+}", + "decision-context": { + "phase": "implement-schema", + "decision": "Finalize the newly introduced recipient field before adding persistence tests.", + "decision-key": "notification-recipient-schema", + "evidence-fingerprint": "sha256:expanded-surface-v1", + "affected-files": ["src/Notifications/NotificationSetup.TableExt.al"], + "affected-symbols": ["tableextension 71000 Notification Setup Ext.", "field 71000 Recipient E-Mail"], + "changed-tokens": ["tableextension", "field", "DataClassification", "ToBeClassified"], + "tests": ["Persist and retrieve a recipient email in the setup extension."], + "acceptance-criteria": ["The new recipient field is suitable for release."], + "development-plan-pin": "plan:notification-channel:v1", + "implementation-evidence-pin": "diff:expanded-surface-v1" + }, + "consumed-guidance": [ + { + "path": "microsoft/knowledge/privacy/data-classification-required-on-pii-fields.md", + "decision-key": "notification-recipient-schema", + "evidence-fingerprint": "sha256:plan-only-v0", + "prior-decision": "No recipient field existed in the implementation surface." + } + ], + "context": { + "bc-version": "28", + "technologies": ["al"], + "countries": ["w1"], + "application-area": ["all"], + "unknown": [] + }, + "requiredKnowledge": [ + "microsoft/knowledge/privacy/data-classification-required-on-pii-fields.md" + ], + "optionalKnowledge": [ + "microsoft/knowledge/privacy/resolve-tobeclassified-before-release.md" + ], + "expectedOmitted": [], + "requiresUnresolved": false + }, + { + "id": "exact-consumed-guidance-is-omitted", + "title": "Exact previously consumed guidance is not reissued", + "evidenceType": "production-shaped-synthetic", + "expectedOutcome": "no-knowledge", + "development-plan": { + "request": "Add a recipient email field to notification setup.", + "affected-files": ["src/Notifications/NotificationSetup.TableExt.al"], + "test-strategy": "Persist and retrieve the recipient.", + "acceptance-criteria": ["The recipient is stored with the intended schema."] + }, + "implementation-diff": "diff --git a/src/Notifications/NotificationSetup.TableExt.al b/src/Notifications/NotificationSetup.TableExt.al\n@@\n+ field(71000; \"Recipient E-Mail\"; Text[250])\n+ {\n+ DataClassification = ToBeClassified;\n+ }", + "decision-context": { + "phase": "implement-schema", + "decision": "Finalize the recipient field classification.", + "decision-key": "notification-recipient-schema", + "evidence-fingerprint": "sha256:recipient-schema-v1", + "affected-files": ["src/Notifications/NotificationSetup.TableExt.al"], + "affected-symbols": ["field 71000 Recipient E-Mail"], + "changed-tokens": ["DataClassification", "ToBeClassified"], + "tests": ["Persist and retrieve a recipient email."], + "acceptance-criteria": ["The field classification is finalized."], + "development-plan-pin": "plan:notification-recipient:v1", + "implementation-evidence-pin": "diff:recipient-schema-v1" + }, + "consumed-guidance": [ + { + "path": "microsoft/knowledge/privacy/data-classification-required-on-pii-fields.md", + "decision-key": "notification-recipient-schema", + "evidence-fingerprint": "sha256:recipient-schema-v1", + "prior-decision": "Classify the recipient field according to the data it stores." + } + ], + "context": { + "bc-version": "28", + "technologies": ["al"], + "countries": ["w1"], + "application-area": ["all"], + "unknown": [] + }, + "requiredKnowledge": [], + "optionalKnowledge": [], + "expectedOmitted": [ + "microsoft/knowledge/privacy/data-classification-required-on-pii-fields.md" + ], + "requiresUnresolved": false + }, + { + "id": "no-additional-guidance-control", + "title": "Comment-only implementation checkpoint has no additional guidance", + "evidenceType": "production-shaped-synthetic", + "expectedOutcome": "no-knowledge", + "development-plan": { + "request": "Correct a spelling error in an internal implementation comment without changing behavior.", + "affected-files": ["src/Notifications/NotificationDispatcher.Codeunit.al"], + "test-strategy": "Inspect the diff for comment-only changes.", + "acceptance-criteria": ["Executable AL is unchanged."] + }, + "implementation-diff": "diff --git a/src/Notifications/NotificationDispatcher.Codeunit.al b/src/Notifications/NotificationDispatcher.Codeunit.al\n@@\n- // Retreive the configured channel.\n+ // Retrieve the configured channel.", + "decision-context": { + "phase": "verify-diff", + "decision": "Confirm that the change remains comment-only.", + "decision-key": "notification-comment-spelling", + "evidence-fingerprint": "sha256:comment-only-v1", + "affected-files": ["src/Notifications/NotificationDispatcher.Codeunit.al"], + "affected-symbols": ["procedure DispatchNotification"], + "changed-tokens": ["comment"], + "tests": ["Inspect the patch for executable changes."], + "acceptance-criteria": ["Only comment text changes."], + "development-plan-pin": "plan:comment-spelling:v1", + "implementation-evidence-pin": "diff:comment-only-v1" + }, + "consumed-guidance": [], + "context": { + "bc-version": "28", + "technologies": ["al"], + "countries": ["w1"], + "application-area": ["all"], + "unknown": [] + }, + "requiredKnowledge": [], + "optionalKnowledge": [], + "expectedOmitted": [], + "requiresUnresolved": false + }, + { + "id": "missing-current-decision-and-diff", + "title": "Plan-only input is not an implementation consultation", + "evidenceType": "production-shaped-synthetic", + "expectedOutcome": "not-applicable", + "development-plan": { + "request": "Add notification delivery.", + "affected-files": ["src/Notifications/NotificationDispatcher.Codeunit.al"], + "test-strategy": "Verify delivery.", + "acceptance-criteria": ["Notifications are delivered."] + }, + "implementation-diff": "", + "decision-context": {}, + "consumed-guidance": [], + "context": { + "bc-version": "28", + "technologies": ["al"], + "countries": ["w1"], + "application-area": ["all"], + "unknown": [] + }, + "requiredKnowledge": [], + "optionalKnowledge": [], + "expectedOmitted": [], + "requiresUnresolved": false + }, + { + "id": "read-only-public-interface-checkpoint", + "title": "Public interface checkpoint returns constraints without repository mutation", + "evidenceType": "production-shaped-synthetic", + "expectedOutcome": "completed", + "development-plan": { + "request": "Expose delivery status through the existing notification provider contract.", + "affected-files": ["src/Notifications/INotificationProvider.Interface.al"], + "test-strategy": "Compile existing and new provider implementations.", + "acceptance-criteria": ["Existing dependent provider implementations remain compatible."] + }, + "implementation-diff": "diff --git a/src/Notifications/INotificationProvider.Interface.al b/src/Notifications/INotificationProvider.Interface.al\n@@\n interface \"Notification Provider\"\n {\n procedure Send(NotificationId: Guid);\n+ procedure GetDeliveryStatus(NotificationId: Guid): Enum \"Delivery Status\";\n }", + "decision-context": { + "phase": "public-contract-checkpoint", + "decision": "Choose a compatible shape for the new delivery-status capability.", + "decision-key": "notification-provider-status-contract", + "evidence-fingerprint": "sha256:provider-interface-v2", + "affected-files": ["src/Notifications/INotificationProvider.Interface.al"], + "affected-symbols": ["interface Notification Provider", "procedure GetDeliveryStatus"], + "changed-tokens": ["interface", "procedure", "Enum"], + "tests": ["Compile unchanged dependent provider implementations.", "Compile a provider opting into delivery status."], + "acceptance-criteria": ["Existing implementations remain compatible."], + "development-plan-pin": "plan:delivery-status:v1", + "implementation-evidence-pin": "diff:provider-interface-v2" + }, + "consumed-guidance": [], + "context": { + "bc-version": "28", + "technologies": ["al"], + "countries": ["w1"], + "application-area": ["all"], + "unknown": [] + }, + "requiredKnowledge": [ + "microsoft/knowledge/interfaces/extend-published-interfaces-dont-edit-them.md" + ], + "optionalKnowledge": [], + "expectedOmitted": [], + "requiresUnresolved": false + } + ] +} diff --git a/microsoft/skills/development/al-implementation-guidance.md b/microsoft/skills/development/al-implementation-guidance.md new file mode 100644 index 00000000..307b8600 --- /dev/null +++ b/microsoft/skills/development/al-implementation-guidance.md @@ -0,0 +1,62 @@ +--- +kind: action-skill +id: al-implementation-guidance +version: 1 +title: AL implementation guidance +description: Produces focused, read-only BCQuality guidance for the next Business Central AL implementation or validation decision. +inputs: [development-plan, repository, implementation-diff, decision-context, consumed-guidance] +outputs: [implementation-guidance-report] +bc-version: [all] +technologies: [al] +countries: [w1] +application-area: [all] +--- + +# AL implementation guidance + +Selects BCQuality knowledge that can change the consumer's next implementation or validation decision. It is a just-in-time consultation after implementation has begun, not a second plan-enrichment pass. + +A readable `repository`, non-empty `development-plan`, current `implementation-diff`, and current `decision-context` are required. The diff may be a patch or structured changed-file/source context, but it must identify exact affected files and current code. Decision context must identify the current phase, next decision, affected symbols, changed AL properties or tokens, tests or acceptance criteria, and stable `decision-key` and `evidence-fingerprint` values. Return `not-applicable` when any required focus input is absent or empty, when the repository is not an AL project, or when current implementation evidence cannot be distinguished from plan intent. Do not invent missing workflow state. + +`consumed-guidance` is optional consumer-owned state. Each entry contains an exact article `path`, `decision-key`, `evidence-fingerprint`, and `prior-decision` summary. BCQuality stores no state. An exact path/key/fingerprint match is omitted from `knowledge` and recorded in `deduplication.omitted`; a different decision key or evidence fingerprint is a new consultation surface and the article may be selected again. + +This skill never edits source, owns workflow state, runs an implementation or review/fix loop, compiles, deploys, stages, commits, or publishes. The consuming coding agent owns all implementation, tests, retries, review, and delivery. + +## Source + +Read the BCQuality knowledge index once, using the external path supplied by Entry when present. If no index is available, use READ's path-based discovery across enabled layers; inability to read the corpus is `failed`, not `no-knowledge`. The index is discovery metadata only. + +Inspect the supplied repository and implementation evidence read-only. Confirm exact changed and affected AL files, symbols, newly introduced properties and tokens, current tests and acceptance obligations, plan intent, and already-consumed guidance. Do not create scratch or generated files inside the target repository. + +## Relevance + +Apply READ's matching semantics using resolved target and implementation context. For upgrades, distinguish source and target versions. Retain conditionally applicable candidates only when resolving them could change the next decision. Record unknown dimensions in `context.unknown` and explain their materiality in `unresolved`. + +## Worklist + +1. Preserve the plan's intent, but derive retrieval vocabulary primarily from the current diff, exact affected files and symbols, changed AL properties and tokens, current phase and decision, tests, acceptance criteria, and validation failures. Do not regenerate or broadly enrich the plan. +2. Check the current surface at deterministic consumer-visible checkpoints: schema or data upgrade; public API, events, or interfaces; permissions; external effects, job queues, or `HttpClient`; telemetry or privacy; UI or page background tasks; and tests. Also consult when the consumer explicitly requests BC-specific guidance for a bounded decision. These are orchestration recommendations, not hidden automatic behavior. +3. Search only domains connected to concrete current evidence. Add an article only when its normative content can change the next implementation or validation decision. Applicability, a broad business noun, or plan membership alone is insufficient. +4. Open every selected article in full. Open only sibling `.good.*` or `.bad.*` samples needed to make the current constraint concrete. Never cite an unopened index row. +5. Resolve contradictory normative guidance with READ's layer precedence and record losing candidates in `suppressed`. +6. Apply stateless deduplication after selection. Omit an article only when a consumed entry's path, decision key, and evidence fingerprint all exactly match this invocation. Record the omitted entry. A changed affected surface or diff fingerprint permits selection again. + +Keep the worklist focused on the next decision. Do not return entire domains, generic engineering advice, previously consumed exact matches, or constraints that cannot affect implementation or validation. + +## Action + +For each worklist article: + +1. Copy its exact path and optional checkout SHA. +2. State `used-for` as the concrete current decision or affected surface. +3. Translate only its normative Best Practice and Anti Pattern into faithful constraints. +4. Include only opened, existing sibling samples. +5. Describe validation evidence the consumer should obtain without claiming it has run or passed. + +Preserve consumer-supplied plan and implementation evidence pins exactly. Use `unpinned` when the consumer supplied none; do not synthesize hashes. Before emitting, verify every knowledge and sample path exists in the live BCQuality checkout and was opened during this run. Reference-integrity failure is `failed`. + +## Output + +Return one `implementation-guidance-report` conforming to DO. `completed` requires at least one newly applicable or materially re-applicable article and no material unresolved applicability. `no-knowledge` means no additional applicable constraints for this decision and evidence fingerprint; it does not mean the implementation is safe, correct, complete, or ready to ship. + +Return `partial` for incomplete evaluation or materially unresolved applicability, and `failed` for retrieval or integrity failure. Keep every unresolved condition explicit. The consumer decides how to respond and owns edits, tests, retries, final independent review, commits, and delivery. diff --git a/plugin.json b/plugin.json index edb78c9d..24deda01 100644 --- a/plugin.json +++ b/plugin.json @@ -1,7 +1,7 @@ { "name": "bcquality", - "description": "Quality skills and knowledge for Business Central. Exposes read-only AL plan enrichment and code-review adapters backed by BCQuality's Entry protocol.", - "version": "0.3.0", + "description": "Quality skills and knowledge for Business Central. Exposes read-only AL plan, implementation-guidance, and code-review adapters backed by BCQuality's Entry protocol.", + "version": "0.4.0", "author": { "name": "microsoft/BCQuality", "url": "https://github.com/microsoft/BCQuality" @@ -13,6 +13,7 @@ "al", "business-central", "code-review", + "implementation-guidance", "plan-guidance", "quality" ], diff --git a/skills/README.md b/skills/README.md index b1d6391a..b444754d 100644 --- a/skills/README.md +++ b/skills/README.md @@ -32,6 +32,7 @@ READ and DO are read on demand — typically by the first action skill the agent |---|---| | [`al-code-review/SKILL.md`](al-code-review/SKILL.md) | Exposes BCQuality through the standard `SKILL.md` format when this repository is installed as a plugin. | | [`al-development-plan/SKILL.md`](al-development-plan/SKILL.md) | Enriches an existing AL plan read-only through the standard `SKILL.md` format; does not generate a plan or implement code. | +| [`al-implementation-guidance/SKILL.md`](al-implementation-guidance/SKILL.md) | Consults BCQuality read-only at bounded implementation checkpoints; the consuming agent still owns edits, tests, and delivery. | Each adapter is deliberately thin. It translates the caller's request into an Entry task context, then follows Entry's dispatch without owning routing, @@ -49,6 +50,9 @@ This gives the two skill formats distinct roles: - `microsoft/skills/development/al-development-plan.md` is the read-only knowledge-enrichment interface for existing plans. Consumers own format normalization, planning, implementation, and delivery. +- `microsoft/skills/development/al-implementation-guidance.md` is the focused + implementation-time consultation interface. It uses the current diff and + decision context, and never becomes an implementation workflow. Each host adapter deliberately shares its name with the internal action skill for the same operation. Their locations distinguish the host integration from diff --git a/skills/al-implementation-guidance/SKILL.md b/skills/al-implementation-guidance/SKILL.md new file mode 100644 index 00000000..3436def9 --- /dev/null +++ b/skills/al-implementation-guidance/SKILL.md @@ -0,0 +1,22 @@ +--- +name: al-implementation-guidance +description: Consult BCQuality read-only for focused Business Central AL constraints affecting the current implementation or validation decision. Does not edit or run the workflow. +--- + +# AL implementation guidance + +This host-native adapter translates current implementation evidence into Entry's task context. It does not generate or replace a plan, edit the target, own consumed-guidance state, run an implementation or review/fix loop, compile, deploy, stage, commit, or publish. + +## Execute + +1. Resolve `PLUGIN_ROOT` to the directory containing this plugin's root `plugin.json`, two levels above this file. +2. Preserve the caller's `development-plan`, readable `repository`, `implementation-diff`, and `decision-context` verbatim. Preserve optional `consumed-guidance` verbatim. The consumer must provide stable decision and evidence identifiers and normalize any workflow-specific payload. +3. Build the task context for `PLUGIN_ROOT/skills/entry.md`: + - Set `goal` to focused, read-only BCQuality consultation for the supplied current implementation decision. + - List only actually supplied inputs from `[development-plan, repository, implementation-diff, decision-context, consumed-guidance]` in `inputs-available`. + - Set `technologies: [al]` only when established, and pass other applicability dimensions only when supplied or reliably determined. + - Apply `BCQUALITY_ENABLED_LAYERS` and `BCQUALITY_DISABLED_SKILLS` as described in the `al-code-review` adapter. +4. Read and execute Entry, including Preparation. Resolve BCQuality paths against `PLUGIN_ROOT`, never the target repository. Keep index, report, and scratch artifacts outside the target. An unreadable corpus is a failure, not empty knowledge. +5. Follow Entry's dispatch and verify its output metadata before invocation. This operation accepts only `implementation-guidance-report`; return `failed` rather than execute another output kind. Pass the supplied inputs unchanged and return the report unchanged. Return Entry's `no-match` or `failed` record unchanged when nothing is dispatched. + +The dispatched skill returns `not-applicable` when required focus context is missing. Exact previously consumed path/decision/evidence matches are omitted deterministically, while changed evidence can produce new guidance. `no-knowledge` is not a correctness claim. The consumer retains all implementation, validation, state, review, and delivery ownership. diff --git a/skills/do.md b/skills/do.md index 1dd77097..1eb064a3 100644 --- a/skills/do.md +++ b/skills/do.md @@ -63,15 +63,24 @@ application-area: [all] `inputs` is a list of abstract input types the skill **accepts**. Standard values: `pr-diff`, `object-list`, `file-path`, `folder-path`, `repository`, -`telemetry-query`, and `development-plan`. Semantics are any-of: the +`telemetry-query`, `development-plan`, `implementation-diff`, +`decision-context`, and `consumed-guidance`. Semantics are any-of: the orchestrator supplies whichever listed input types it has, and the skill is invoked with a non-empty subset of its declared `inputs`. A skill that cannot proceed with the supplied subset MUST return `outcome: "not-applicable"`. +`implementation-diff` is the current patch or structured changed-file/source +context, including exact affected files. `decision-context` identifies the +current phase, bounded next decision, affected symbols, changed AL properties +or tokens, tests or acceptance obligations, and stable decision/evidence +identifiers. `consumed-guidance` is optional consumer-owned deduplication input; +BCQuality does not persist it. + `outputs` is always a single-element list naming the output kind: - `findings-report` — evaluates an input and reports defects or observations. - `development-guidance-report` — selects and summarizes applicable BCQuality knowledge for an existing development plan without changing the target repository. +- `implementation-guidance-report` — selects focused, additional BCQuality knowledge for a current implementation or validation decision without changing the target repository. `file-path` is one file. `folder-path` is a directory whose recursively contained files form the complete current-state input, such as a Business @@ -340,6 +349,108 @@ The skill is read-only with respect to the target repository: no edits, generate Reference SHAs, when present, identify the files read; they do not prove runtime pinning on their own. The consumer records and verifies the actual immutable BCQuality checkout used for both enrichment and final review, plus its filtering policy and run provenance outside the target repository. See [agent-consumption.md](../agent-consumption.md). +## Implementation-guidance-report contract + +An action skill with `outputs: [implementation-guidance-report]` emits one JSON document: + +```json +{ + "skill": { "id": "al-implementation-guidance", "version": 1 }, + "outcome": "completed | not-applicable | no-knowledge | partial | failed", + "outcome-reason": "string", + "summary": { + "request": "string", + "phase": "string", + "decision": "string", + "decision-key": "string", + "evidence-fingerprint": "string", + "candidates": 0, + "selected": 0, + "omitted-consumed": 0 + }, + "pins": { + "knowledge-checkout": "full commit SHA | unpinned", + "development-plan": "consumer-supplied id/version/digest | unpinned", + "implementation-evidence": "consumer-supplied id/digest | unpinned" + }, + "context": { + "bc-version": "string", + "technologies": ["string"], + "countries": ["string"], + "application-area": ["string"], + "affected-files": ["string"], + "affected-symbols": ["string"], + "changed-tokens": ["string"], + "unknown": ["bc-version | technologies | countries | application-area"] + }, + "knowledge": [ + { + "path": "string", + "sha": "string", + "used-for": "string", + "constraints": ["string"], + "sample-paths": ["string"] + } + ], + "validation-considerations": [ + { "id": "string", "reason": "string", "evidence": "string" } + ], + "deduplication": { + "strategy": "omit-exact-consumed-match", + "omitted": [ + { + "path": "string", + "decision-key": "string", + "evidence-fingerprint": "string", + "prior-decision": "string" + } + ] + }, + "suppressed": [ + { + "reference": { "path": "string", "sha": "string" }, + "reason": "layer-precedence | configuration" + } + ], + "unresolved": ["string"] +} +``` + +The report is strict JSON with no surrounding commentary and is read-only with +respect to both the target repository and consumer workflow state. Index, +report, and scratch artifacts stay outside the target. + +`summary` is the compact current phase/decision record. Its decision key and +evidence fingerprint are copied from `decision-context`, not generated by the +skill. `pins` preserves consumer-supplied plan and implementation identifiers; +`knowledge-checkout` is the actual checkout SHA only when established, otherwise +`unpinned`. A pin records identity, not safety or correctness. +`candidates` is the number of unique relevant articles before consumed-guidance +omission, so `selected + omitted-consumed` cannot exceed it. + +`context.affected-files`, `affected-symbols`, and `changed-tokens` contain exact +current implementation evidence, not broad planned domains. Paths use forward +slashes and are repository-relative. The knowledge, validation, suppression, +unknown-context, and reference-integrity rules from +`development-guidance-report` apply unchanged. + +`deduplication.strategy` is always `omit-exact-consumed-match`. An omitted entry +must exactly match a supplied consumed article path, decision key, and evidence +fingerprint. Its path is absent from `knowledge`. `summary.omitted-consumed` +equals `deduplication.omitted.length`. A changed decision key or evidence +fingerprint permits the article to be selected again. The consumer owns and +persists consumed state; BCQuality remains stateless. + +`completed` requires at least one selected article, complete evaluation, and no +material unresolved applicability. `not-applicable` means a required readable +repository, plan, implementation diff, or decision context is missing or the +task is outside applicability. `no-knowledge` requires empty `knowledge` and +means no additional applicable constraints for this exact decision/evidence; +it is not a functional-correctness, safety, completeness, or release claim. +`partial` keeps incomplete evaluation or materially unresolved applicability +explicit. `failed` covers retrieval, reference-integrity, or other errors. +`outcome-reason` is required for `partial` and `failed`. + ## Composition (super-skills) A **super-skill** is an action skill whose frontmatter declares a non-empty `sub-skills: [...]`. A super-skill does not evaluate knowledge files directly; it invokes other action skills and composes their output. @@ -440,4 +551,9 @@ Conforms to the DO output contract. ## How orchestrators consume output -An orchestrator invokes an action skill with an input appropriate to the skill's declared `inputs` and uses the single output kind declared in frontmatter. It maps a `findings-report` to PR comments, build gates, or IDE diagnostics, and a `development-guidance-report` to additional constraints for its existing implementation workflow. These are the two output schemas defined by this contract; the consumer retains ownership of implementation and delivery. +An orchestrator invokes an action skill with an input appropriate to the skill's +declared `inputs` and uses the single output kind declared in frontmatter. It +maps a `findings-report` to PR comments, build gates, or IDE diagnostics; a +`development-guidance-report` to constraints for an existing plan; and an +`implementation-guidance-report` to focused constraints for the next current +decision. The consumer retains ownership of implementation and delivery. diff --git a/skills/entry.md b/skills/entry.md index a1f13a98..6116b606 100644 --- a/skills/entry.md +++ b/skills/entry.md @@ -23,6 +23,9 @@ task-context: inputs-available: # values the orchestrator has ready to pass to a chosen skill - development-plan - repository + - implementation-diff + - decision-context + - consumed-guidance - pr-diff - file-path - folder-path @@ -48,7 +51,7 @@ Before routing, ensure the knowledge index is current for the **live** clone. Th It defaults to indexing this checkout and writes `knowledge-index.json` at the root in well under a second. When in doubt, rebuild: a sub-second rebuild is always cheaper than a stale or over-listing index, which is a correctness risk. - The paths above assume the checkout root is the current directory. A caller that enters Entry from elsewhere — a plugin host, whose working directory is the user's own project — MUST resolve them against the BCQuality root it already knows instead. The generator resolves its own root, so invoking it by absolute path indexes and writes the right tree. -- For read-only plan enrichment, generated artifacts MUST remain outside the target repository. When the target contains the BCQuality checkout, or that checkout is immutable, pass the generator's `-IndexPath` to an external runner-owned artifact location and supply that resolved index path to the dispatched skill. Do not regenerate inside the target or modify the immutable checkout. If generation is unavailable, use READ's path-based discovery; retrieval failure is not an empty corpus. +- For read-only development guidance, generated artifacts MUST remain outside the target repository. When the target contains the BCQuality checkout, or that checkout is immutable, pass the generator's `-IndexPath` to an external runner-owned artifact location and supply that resolved index path to the dispatched skill. Do not regenerate inside the target or modify the immutable checkout. If generation is unavailable, use READ's path-based discovery; retrieval failure is not an empty corpus. - Pruning is the consumer's job, not Entry's, and not every consumer does it: an installation that ships the whole tree gets no deny guarantee from this step. There, `enabled-layers` narrows discovery only, and the unlisted layers' files remain on disk. - This is a side step. It MUST NOT change Entry's output — the dispatch record below is the only thing Entry emits, and build logs are never part of the dispatch JSON. @@ -128,7 +131,7 @@ Emit a single JSON document conforming to the output contract below. Entry does - `skill.version` — copied from the dispatched skill's frontmatter so the orchestrator can detect drift between dispatch time and execution. - `rationale` — short human-readable string, for logs and traceability. - `inputs` — the intersection of `task-context.inputs-available` and the skill's declared `inputs`. The agent MUST pass exactly this subset when invoking the skill. Sending a strict intersection avoids accidental information leakage between skills. -- `outputs` — the dispatched skill's complete, single-element `outputs` value copied from frontmatter. This lets an orchestrator distinguish `findings-report` from read-only `development-guidance-report` before invocation. Check the declared contract and actual skill; output metadata is not a sandbox or proof of side effects. Unknown output kinds must not be silently treated as a supported report. +- `outputs` — the dispatched skill's complete, single-element `outputs` value copied from frontmatter. This lets an orchestrator distinguish `findings-report`, read-only `development-guidance-report`, and focused read-only `implementation-guidance-report` before invocation. Check the declared contract and actual skill; output metadata is not a sandbox or proof of side effects. Unknown output kinds must not be silently treated as a supported report. Ordering of `dispatch[]` is not significant. diff --git a/tools/DevelopmentGuidance.Evidence.ps1 b/tools/DevelopmentGuidance.Evidence.ps1 index 8c92a8fe..f23af61d 100644 --- a/tools/DevelopmentGuidance.Evidence.ps1 +++ b/tools/DevelopmentGuidance.Evidence.ps1 @@ -318,6 +318,87 @@ function Get-GuidancePlanRequest { return $Plan.request } +function Test-ImplementationGuidanceManifest { + param($Manifest) + return $Manifest.skill -ceq 'microsoft/skills/development/al-implementation-guidance.md' +} + +function Assert-ImplementationDecisionContext { + param($DecisionContext, [switch] $AllowEmpty) + Assert-GuidanceObject $DecisionContext 'decision-context' + if ($AllowEmpty -and -not $DecisionContext.Count) { return } + foreach ($key in @('phase', 'decision', 'decision-key', 'evidence-fingerprint', + 'affected-files', 'affected-symbols', 'changed-tokens', 'tests', + 'acceptance-criteria', 'development-plan-pin', 'implementation-evidence-pin')) { + if (-not $DecisionContext.Contains($key)) { throw "decision-context is missing required field '$key'." } + } + foreach ($key in @('phase', 'decision', 'decision-key', 'evidence-fingerprint', + 'development-plan-pin', 'implementation-evidence-pin')) { + Assert-GuidanceString $DecisionContext[$key] "decision-context.$key" + } + foreach ($key in @('affected-files', 'affected-symbols', 'changed-tokens', 'tests', 'acceptance-criteria')) { + Assert-GuidanceArray $DecisionContext[$key] "decision-context.$key" -Strings + } +} + +function Assert-ConsumedGuidance { + param($Consumed) + Assert-GuidanceArray $Consumed 'consumed-guidance' + $keys = [Collections.Generic.HashSet[string]]::new([StringComparer]::Ordinal) + foreach ($entry in $Consumed) { + Assert-GuidanceObject $entry 'consumed-guidance entry' @('path', 'decision-key', 'evidence-fingerprint', 'prior-decision') + foreach ($key in @('path', 'decision-key', 'evidence-fingerprint', 'prior-decision')) { + Assert-GuidanceString $entry[$key] "consumed-guidance.$key" + } + $identity = "$($entry.path)`n$($entry.'decision-key')`n$($entry.'evidence-fingerprint')" + if (-not $keys.Add($identity)) { throw 'consumed-guidance contains duplicate identities.' } + } +} + +function Assert-ImplementationGuidanceManifestCases { + param($Manifest, [string] $Root) + Assert-GuidanceArray $Manifest.cases 'manifest.cases' -NonEmpty + $ids = [Collections.Generic.HashSet[string]]::new([StringComparer]::Ordinal) + $modelIds = [Collections.Generic.HashSet[string]]::new([StringComparer]::Ordinal) + foreach ($case in $Manifest.cases) { + Assert-GuidanceObject $case 'manifest case' @( + 'id', 'expectedOutcome', 'development-plan', 'implementation-diff', + 'decision-context', 'consumed-guidance', 'context', 'requiredKnowledge', + 'optionalKnowledge', 'expectedOmitted', 'requiresUnresolved') + if ($case.id -isnot [string] -or $case.id -cnotmatch '^[a-z0-9]+(?:-[a-z0-9]+)*$') { + throw 'Fixture id must be kebab-case.' + } + if (-not $ids.Add($case.id) -or -not $modelIds.Add((Get-GuidanceCaseId $case.id))) { + throw 'Duplicate fixture or model case identity.' + } + if ($case.expectedOutcome -cnotin @('completed', 'not-applicable', 'no-knowledge', 'partial', 'failed')) { + throw 'Fixture expectedOutcome is invalid.' + } + Assert-GuidanceString $case.expectedOutcome 'fixture.expectedOutcome' + $null = Get-GuidancePlanRequest $case.'development-plan' + if ($case.'implementation-diff' -isnot [string]) { throw 'implementation-diff must be a string.' } + Assert-ImplementationDecisionContext $case.'decision-context' -AllowEmpty + Assert-ConsumedGuidance $case.'consumed-guidance' + Assert-GuidanceContext $case.context + foreach ($name in @('requiredKnowledge', 'optionalKnowledge', 'expectedOmitted')) { + Assert-GuidanceArray $case[$name] "fixture.$name" -Strings + } + if ($case.requiresUnresolved -isnot [bool]) { throw 'Fixture requiresUnresolved must be boolean.' } + $references = @($case.requiredKnowledge) + @($case.optionalKnowledge) + @($case.expectedOmitted) + foreach ($reference in @($references | Sort-Object -Unique)) { + $null = Resolve-GuidanceReference $Root $reference -Knowledge + } + if ($case.expectedOutcome -in @('no-knowledge', 'not-applicable') -and + (@($case.requiredKnowledge) + @($case.optionalKnowledge)).Count) { + throw 'Empty-knowledge outcomes cannot require or accept selected knowledge.' + } + if ($case.expectedOutcome -eq 'not-applicable' -and + -not ([string]::IsNullOrWhiteSpace($case.'implementation-diff') -or -not $case.'decision-context'.Count)) { + throw 'The not-applicable implementation fixture must omit current diff or decision context.' + } + } +} + function Assert-GuidanceContext { param($Context) Assert-GuidanceObject $Context 'context' @('bc-version', 'technologies', 'countries', 'application-area', 'unknown') @@ -349,6 +430,10 @@ function Assert-GuidanceManifest { $value -lt 0 -or $value -gt 1) { throw 'Manifest thresholds must be numbers between zero and one.' } } $null = Resolve-GuidanceReference $Root $Manifest.skill + if (Test-ImplementationGuidanceManifest $Manifest) { + Assert-ImplementationGuidanceManifestCases $Manifest $Root + return + } Assert-GuidanceArray $Manifest.cases 'manifest.cases' -NonEmpty $ids = [Collections.Generic.HashSet[string]]::new([StringComparer]::Ordinal) $modelIds = [Collections.Generic.HashSet[string]]::new([StringComparer]::Ordinal) @@ -382,8 +467,174 @@ function Assert-GuidanceManifest { } } +function Assert-ImplementationGuidanceResult { + param($Result, $Case, $Manifest, [string] $Root, [string] $Workspace) + Assert-GuidanceObject $Result 'result' @('caseId', 'guidanceReport') + Assert-GuidanceString $Result.caseId 'result.caseId' + if ($Result.caseId -cne (Get-GuidanceCaseId $Case.id)) { throw 'Result caseId mismatch.' } + if ($Result.Contains('workspaceRoot')) { + Assert-GuidanceString $Result.workspaceRoot 'result.workspaceRoot' + if (-not [IO.Path]::IsPathFullyQualified($Result.workspaceRoot) -or + (Get-GuidanceSafePath $Result.workspaceRoot -Directory) -cne $Workspace) { + throw 'Result workspaceRoot disagrees with the runner binding.' + } + } + $report = $Result.guidanceReport + Assert-GuidanceObject $report 'guidanceReport' @( + 'skill', 'outcome', 'summary', 'pins', 'context', 'knowledge', + 'validation-considerations', 'deduplication', 'suppressed', 'unresolved') + Assert-GuidanceObject $report.skill 'skill' @('id', 'version') + if ($report.skill.id -cne 'al-implementation-guidance' -or + $report.skill.version -isnot [long] -or $report.skill.version -ne 1) { + throw 'Report skill identity/version is invalid.' + } + Assert-GuidanceString $report.outcome 'outcome' + if ($report.outcome -cnotin @('completed', 'not-applicable', 'no-knowledge', 'partial', 'failed')) { + throw 'Report outcome enum is invalid.' + } + if ($report.outcome -cne $Case.expectedOutcome) { throw 'Report outcome does not match fixture expectedOutcome.' } + if ($report.outcome -in @('partial', 'failed') -or $report.Contains('outcome-reason')) { + Assert-GuidanceString $report['outcome-reason'] 'outcome-reason' + } + Assert-GuidanceObject $report.summary 'summary' @( + 'request', 'phase', 'decision', 'decision-key', 'evidence-fingerprint', + 'candidates', 'selected', 'omitted-consumed') + foreach ($key in @('request', 'phase', 'decision', 'decision-key', 'evidence-fingerprint')) { + Assert-GuidanceString $report.summary[$key] "summary.$key" + } + foreach ($key in @('candidates', 'selected', 'omitted-consumed')) { + Assert-GuidanceInteger $report.summary[$key] "summary.$key" + } + Assert-GuidanceObject $report.pins 'pins' @('knowledge-checkout', 'development-plan', 'implementation-evidence') + foreach ($key in @('knowledge-checkout', 'development-plan', 'implementation-evidence')) { + Assert-GuidanceString $report.pins[$key] "pins.$key" + } + if ($report.pins.'knowledge-checkout' -cne 'unpinned') { + if ($report.pins.'knowledge-checkout' -cnotmatch '^([0-9A-Fa-f]{40}|[0-9A-Fa-f]{64})$' -or + $report.pins.'knowledge-checkout' -cne (Invoke-GuidanceGit $Root @('rev-parse', '--verify', 'HEAD'))) { + throw 'Knowledge checkout pin does not identify the evaluated checkout.' + } + } + if ($Case.expectedOutcome -ne 'not-applicable') { + if ($report.summary.'decision-key' -cne $Case.'decision-context'.'decision-key' -or + $report.summary.'evidence-fingerprint' -cne $Case.'decision-context'.'evidence-fingerprint' -or + $report.summary.phase -cne $Case.'decision-context'.phase -or + $report.summary.decision -cne $Case.'decision-context'.decision) { + throw 'Report summary does not preserve current decision context.' + } + if ($report.pins.'development-plan' -cne $Case.'decision-context'.'development-plan-pin' -or + $report.pins.'implementation-evidence' -cne $Case.'decision-context'.'implementation-evidence-pin') { + throw 'Report does not preserve consumer-supplied pins.' + } + } + Assert-GuidanceContext $report.context + foreach ($key in @('affected-files', 'affected-symbols', 'changed-tokens')) { + if (-not $report.context.Contains($key)) { throw "context is missing required field '$key'." } + Assert-GuidanceArray $report.context[$key] "context.$key" -Strings + if ($Case.expectedOutcome -ne 'not-applicable' -and + ($report.context[$key] | ConvertTo-Json -Compress) -cne + ($Case.'decision-context'[$key] | ConvertTo-Json -Compress)) { + throw "context.$key does not preserve current implementation evidence." + } + } + foreach ($name in @('knowledge', 'validation-considerations', 'suppressed', 'unresolved')) { + Assert-GuidanceArray $report[$name] $name + } + Assert-GuidanceArray $report.unresolved 'unresolved' -Strings + Assert-GuidanceObject $report.deduplication 'deduplication' @('strategy', 'omitted') + if ($report.deduplication.strategy -cne 'omit-exact-consumed-match') { + throw 'Deduplication strategy is invalid.' + } + Assert-GuidanceArray $report.deduplication.omitted 'deduplication.omitted' + if ($report.summary.selected -ne $report.knowledge.Count -or + $report.summary.'omitted-consumed' -ne $report.deduplication.omitted.Count -or + ($report.summary.selected + $report.summary.'omitted-consumed') -gt $report.summary.candidates) { + throw 'Summary counts disagree with selected or omitted guidance.' + } + if ($report.outcome -in @('no-knowledge', 'not-applicable') -and $report.knowledge.Count) { + throw 'This outcome requires empty knowledge.' + } + if ($report.outcome -eq 'completed' -and -not $report.knowledge.Count) { + throw 'Completed requires selected knowledge; empty evaluation is no-knowledge.' + } + if (($report.outcome -eq 'partial' -or $Case.requiresUnresolved) -and -not $report.unresolved.Count) { + throw 'Partial/incomplete evaluation must explain unresolved gaps.' + } + foreach ($dimension in $report.context.unknown) { + if (-not @($report.unresolved | Where-Object { $_ -match [regex]::Escape($dimension) }).Count) { + throw 'Unknown dimensions require a corresponding unresolved explanation.' + } + } + $used = [Collections.Generic.HashSet[string]]::new([StringComparer]::Ordinal) + foreach ($entry in $report.knowledge) { + Assert-GuidanceObject $entry 'knowledge entry' @('path', 'used-for', 'constraints', 'sample-paths') + $null = Resolve-GuidanceReference $Root $entry.path -Knowledge + if (-not $used.Add($entry.path)) { throw 'Duplicate knowledge reference.' } + Assert-GuidanceString $entry.'used-for' 'knowledge.used-for' + Assert-GuidanceArray $entry.constraints 'knowledge.constraints' -Strings -NonEmpty + Assert-GuidanceArray $entry.'sample-paths' 'knowledge.sample-paths' -Strings + foreach ($sample in $entry.'sample-paths') { + $null = Resolve-GuidanceReference $Root $sample -Article $entry.path + } + Assert-GuidanceReferenceSha $entry $Root + } + $omitted = [Collections.Generic.HashSet[string]]::new([StringComparer]::Ordinal) + foreach ($entry in $report.deduplication.omitted) { + Assert-GuidanceObject $entry 'deduplication omitted entry' @( + 'path', 'decision-key', 'evidence-fingerprint', 'prior-decision') + foreach ($key in @('path', 'decision-key', 'evidence-fingerprint', 'prior-decision')) { + Assert-GuidanceString $entry[$key] "deduplication.omitted.$key" + } + $null = Resolve-GuidanceReference $Root $entry.path -Knowledge + $match = @($Case.'consumed-guidance' | Where-Object { + $_.path -ceq $entry.path -and + $_.'decision-key' -ceq $entry.'decision-key' -and + $_.'evidence-fingerprint' -ceq $entry.'evidence-fingerprint' + }) + if ($match.Count -ne 1 -or $entry.'decision-key' -cne $report.summary.'decision-key' -or + $entry.'evidence-fingerprint' -cne $report.summary.'evidence-fingerprint') { + throw 'Omitted guidance is not an exact consumed match for this decision evidence.' + } + if (-not $omitted.Add($entry.path) -or $used.Contains($entry.path)) { + throw 'Omitted guidance is duplicated or reissued.' + } + } + if (@($Case.expectedOmitted | Where-Object { -not $omitted.Contains($_) }).Count -or + @($omitted | Where-Object { $_ -cnotin $Case.expectedOmitted }).Count) { + throw 'Deduplication omitted set does not match fixture expectation.' + } + $validationIds = [Collections.Generic.HashSet[string]]::new([StringComparer]::Ordinal) + foreach ($entry in $report.'validation-considerations') { + Assert-GuidanceObject $entry 'validation consideration' @('id', 'reason', 'evidence') + foreach ($key in @('id', 'reason', 'evidence')) { + Assert-GuidanceString $entry[$key] "validation-considerations.$key" + } + if (-not $validationIds.Add($entry.id)) { throw 'Duplicate validation consideration id.' } + } + foreach ($entry in $report.suppressed) { + Assert-GuidanceObject $entry 'suppressed entry' @('reference', 'reason') + Assert-GuidanceObject $entry.reference 'suppressed.reference' @('path') + $null = Resolve-GuidanceReference $Root $entry.reference.path -Knowledge + Assert-GuidanceReferenceSha $entry.reference $Root + if ($entry.reason -cnotin @('layer-precedence', 'configuration')) { + throw 'Suppression reason is invalid.' + } + } + $matched = @($Case.requiredKnowledge | Where-Object { $used.Contains($_) }).Count + $recall = if ($Case.requiredKnowledge.Count) { $matched / $Case.requiredKnowledge.Count } else { 1.0 } + $accepted = @($Case.requiredKnowledge) + @($Case.optionalKnowledge) + $acceptedCount = @($used | Where-Object { $_ -cin $accepted }).Count + $precision = if ($used.Count) { $acceptedCount / $used.Count } elseif (-not $Case.requiredKnowledge.Count) { 1.0 } else { 0.0 } + if ($recall -lt $Manifest.minimumKnowledgeRecall) { throw 'Knowledge recall is below the manifest threshold.' } + if ($precision -lt $Manifest.minimumKnowledgePrecision) { throw 'Knowledge precision is below the manifest threshold.' } +} + function Assert-GuidanceResult { param($Result, $Case, $Manifest, [string] $Root, [string] $Workspace) + if (Test-ImplementationGuidanceManifest $Manifest) { + Assert-ImplementationGuidanceResult $Result $Case $Manifest $Root $Workspace + return + } Assert-GuidanceObject $Result 'result' @('caseId', 'guidanceReport') Assert-GuidanceString $Result.caseId 'result.caseId' if ($Result.caseId -cne (Get-GuidanceCaseId $Case.id)) { throw 'Result caseId mismatch.' } @@ -491,7 +742,49 @@ function Assert-GuidanceReferenceSha { } function Get-GuidanceResultSchema { - param([string] $CaseId) + param([string] $CaseId, [switch] $Implementation) + if ($Implementation) { + return [ordered]@{ + caseId = $CaseId + guidanceReport = [ordered]@{ + skill = [ordered]@{ id = 'al-implementation-guidance'; version = 1 } + outcome = 'completed | not-applicable | no-knowledge | partial | failed' + 'outcome-reason' = 'required for partial or failed' + summary = [ordered]@{ + request = 'preserved plan intent'; phase = 'current phase'; decision = 'next decision' + 'decision-key' = 'consumer key'; 'evidence-fingerprint' = 'consumer fingerprint' + candidates = 0; selected = 0; 'omitted-consumed' = 0 + } + pins = [ordered]@{ + 'knowledge-checkout' = 'full checkout SHA or unpinned' + 'development-plan' = 'consumer plan pin or unpinned' + 'implementation-evidence' = 'consumer evidence pin or unpinned' + } + context = [ordered]@{ + 'bc-version' = 'resolved target or unknown'; technologies = @('al') + countries = @('w1'); 'application-area' = @('all') + 'affected-files' = @(); 'affected-symbols' = @(); 'changed-tokens' = @(); unknown = @() + } + knowledge = @([ordered]@{ + path = 'repo-relative knowledge article'; sha = 'optional full checkout commit id' + 'used-for' = 'current decision'; constraints = @('faithful normative constraint') + 'sample-paths' = @() + }) + 'validation-considerations' = @([ordered]@{ + id = 'stable id'; reason = 'why needed'; evidence = 'evidence consumer should obtain' + }) + deduplication = [ordered]@{ + strategy = 'omit-exact-consumed-match' + omitted = @([ordered]@{ + path = 'consumed article path'; 'decision-key' = 'exact current key' + 'evidence-fingerprint' = 'exact current fingerprint'; 'prior-decision' = 'prior decision' + }) + } + suppressed = @() + unresolved = @() + } + } + } return [ordered]@{ caseId = $CaseId guidanceReport = [ordered]@{ diff --git a/tools/Test-DevelopmentGuidanceFixtures.ps1 b/tools/Test-DevelopmentGuidanceFixtures.ps1 index 466f1e32..10368bf1 100644 --- a/tools/Test-DevelopmentGuidanceFixtures.ps1 +++ b/tools/Test-DevelopmentGuidanceFixtures.ps1 @@ -157,10 +157,11 @@ try { & (Join-Path $Root 'tools\Build-KnowledgeIndex.ps1') -BCQualityRoot $Root ` -IndexPath (Join-Path $PrepareDirectory 'knowledge-index.json') | Out-Null $skillInstructions = [IO.File]::ReadAllText((Resolve-GuidanceReference $Root $manifest.skill)) + $implementation = Test-ImplementationGuidanceManifest $manifest foreach ($case in $manifest.cases) { $modelId = Get-GuidanceCaseId $case.id $request = [ordered]@{ - protocol = 'Run the supplied read-only skill on the runner-assigned repository and existing plan. Return only caseId and guidanceReport. The runner captures evidence before invocation; do not capture or modify it. Do not create artifacts in the target or knowledge checkout.' + protocol = 'Run the supplied read-only skill on the runner-assigned repository and supplied development context. Return only caseId and guidanceReport. The runner captures evidence before invocation; do not capture or modify it. Do not create artifacts in the target or knowledge checkout.' caseId = $modelId skill = $manifest.skill skillInstructions = $skillInstructions @@ -168,14 +169,21 @@ try { knowledgeRoot = $Root 'task-context' = [ordered]@{ goal = Get-GuidancePlanRequest $case.'development-plan' - 'inputs-available' = @('development-plan', 'repository') + 'inputs-available' = $(if ($implementation) { + @('development-plan', 'repository', 'implementation-diff', 'decision-context', 'consumed-guidance') + } else { @('development-plan', 'repository') }) 'bc-version' = $case.context.'bc-version' technologies = $case.context.technologies countries = $case.context.countries 'application-area' = $case.context.'application-area' } 'development-plan' = $case.'development-plan' - resultSchema = Get-GuidanceResultSchema $modelId + resultSchema = Get-GuidanceResultSchema $modelId -Implementation:$implementation + } + if ($implementation) { + $request.'implementation-diff' = $case.'implementation-diff' + $request.'decision-context' = $case.'decision-context' + $request.'consumed-guidance' = $case.'consumed-guidance' } if ($workspaces.Count) { $request.repository = $workspaces[$case.id] } Write-GuidanceNewJson (Join-Path $PrepareDirectory "request-$modelId.json") $request diff --git a/tools/Test-ImplementationGuidanceEvaluator.ps1 b/tools/Test-ImplementationGuidanceEvaluator.ps1 new file mode 100644 index 00000000..52f9ef74 --- /dev/null +++ b/tools/Test-ImplementationGuidanceEvaluator.ps1 @@ -0,0 +1,206 @@ +<# +.SYNOPSIS + Deterministic, offline regressions for implementation-guidance fixtures. +.DESCRIPTION + Exercises the shared guidance scorer with controlled reports and standalone + synthetic AL repositories. No model, compiler, deployment, or network access + is used. The uniquely named regression directory is removed in finally. +#> +[CmdletBinding()] +param() + +Set-StrictMode -Version Latest +$ErrorActionPreference = 'Stop' +. (Join-Path $PSScriptRoot 'DevelopmentGuidance.Evidence.ps1') + +$sourceRoot = (Get-Item -LiteralPath (Join-Path $PSScriptRoot '..')).FullName +$evaluator = Join-Path $PSScriptRoot 'Test-DevelopmentGuidanceFixtures.ps1' +$sourceManifestPath = Join-Path $sourceRoot 'evaluation\implementation-guidance-fixtures.json' +$scratch = Join-Path $sourceRoot ".implementation-guidance-regression-$([guid]::NewGuid().ToString('N'))" +$knowledgeRoot = Join-Path $scratch 'knowledge-checkout' +$runnerRoot = Join-Path $scratch 'runner' +$results = Join-Path $runnerRoot 'results' +$mapPath = Join-Path $runnerRoot 'workspace-map.json' +$baselinePath = Join-Path $runnerRoot 'baseline.json' +$manifestPath = Join-Path $knowledgeRoot 'evaluation\implementation-guidance-fixtures.json' +$tests = [Collections.Generic.List[string]]::new() + +function Set-TestJson([string] $Path, $Value) { + [IO.Directory]::CreateDirectory((Split-Path $Path -Parent)) | Out-Null + [IO.File]::WriteAllText($Path, ($Value | ConvertTo-Json -Depth 64)) +} + +function Invoke-TestGit([string] $Directory, [string[]] $Arguments) { + $null = @(& git --no-optional-locks -C $Directory @Arguments 2>&1) + if ($LASTEXITCODE -ne 0) { throw 'Regression Git setup failed.' } +} + +function Initialize-TestRepository([string] $Directory, [string] $CaseId) { + [IO.Directory]::CreateDirectory($Directory) | Out-Null + Invoke-TestGit $Directory @('init', '--quiet') + Invoke-TestGit $Directory @('config', 'user.email', 'implementation-guidance@example.invalid') + Invoke-TestGit $Directory @('config', 'user.name', 'Implementation guidance fixture') + Invoke-TestGit $Directory @('config', 'commit.gpgSign', 'false') + Invoke-TestGit $Directory @('config', 'core.autocrlf', 'false') + [IO.File]::WriteAllText((Join-Path $Directory 'app.json'), '{"name":"Synthetic implementation fixture","application":"28.0.0.0"}') + [IO.File]::WriteAllText((Join-Path $Directory 'CurrentImplementation.al'), "codeunit 71000 `"$CaseId`" { }") + Invoke-TestGit $Directory @('add', '.') + Invoke-TestGit $Directory @('commit', '--quiet', '-m', 'Synthetic implementation baseline') +} + +function Invoke-Evaluator([string[]] $Arguments, [bool] $ShouldPass, [string] $Name, [string] $Diagnostic = '') { + $output = @(& pwsh -NoProfile -File $evaluator @Arguments 2>&1) -join "`n" + if (($LASTEXITCODE -eq 0) -ne $ShouldPass -or + $output -notmatch $(if ($ShouldPass) { 'PASSED|captured' } else { 'FAILED' })) { + throw "Regression '$Name' failed unexpectedly. $output" + } + if ($Diagnostic -and $output -notmatch [regex]::Escape($Diagnostic)) { + throw "Regression '$Name' missed diagnostic '$Diagnostic'. $output" + } + $tests.Add($Name) +} + +function New-ControlledReport($Case, [string] $KnowledgeSha) { + $notApplicable = $Case.expectedOutcome -eq 'not-applicable' + $decision = if ($notApplicable) { + [ordered]@{ + phase = 'unavailable'; decision = 'Current implementation decision was not supplied.' + 'decision-key' = 'unavailable'; 'evidence-fingerprint' = 'unavailable' + 'affected-files' = @(); 'affected-symbols' = @(); 'changed-tokens' = @() + 'development-plan-pin' = 'unpinned'; 'implementation-evidence-pin' = 'unpinned' + } + } else { $Case.'decision-context' } + $knowledge = @( + foreach ($path in $Case.requiredKnowledge) { + [ordered]@{ + path = $path + sha = $KnowledgeSha + 'used-for' = $decision.decision + constraints = @('Apply the opened article normative constraint to this current decision.') + 'sample-paths' = @() + } + } + ) + $omitted = @( + foreach ($path in $Case.expectedOmitted) { + $consumed = @($Case.'consumed-guidance' | Where-Object path -CEQ $path)[0] + [ordered]@{ + path = $path + 'decision-key' = $consumed.'decision-key' + 'evidence-fingerprint' = $consumed.'evidence-fingerprint' + 'prior-decision' = $consumed.'prior-decision' + } + } + ) + return [ordered]@{ + skill = [ordered]@{ id = 'al-implementation-guidance'; version = 1 } + outcome = $Case.expectedOutcome + summary = [ordered]@{ + request = Get-GuidancePlanRequest $Case.'development-plan' + phase = $decision.phase + decision = $decision.decision + 'decision-key' = $decision.'decision-key' + 'evidence-fingerprint' = $decision.'evidence-fingerprint' + candidates = $knowledge.Count + $omitted.Count + selected = $knowledge.Count + 'omitted-consumed' = $omitted.Count + } + pins = [ordered]@{ + 'knowledge-checkout' = $KnowledgeSha + 'development-plan' = $decision.'development-plan-pin' + 'implementation-evidence' = $decision.'implementation-evidence-pin' + } + context = [ordered]@{ + 'bc-version' = $Case.context.'bc-version' + technologies = @($Case.context.technologies) + countries = @($Case.context.countries) + 'application-area' = @($Case.context.'application-area') + 'affected-files' = @($decision.'affected-files') + 'affected-symbols' = @($decision.'affected-symbols') + 'changed-tokens' = @($decision.'changed-tokens') + unknown = @($Case.context.unknown) + } + knowledge = $knowledge + 'validation-considerations' = @() + deduplication = [ordered]@{ + strategy = 'omit-exact-consumed-match' + omitted = $omitted + } + suppressed = @() + unresolved = @() + } +} + +try { + $manifest = Read-GuidanceJson $sourceManifestPath + $references = @($manifest.skill, 'tools/Build-KnowledgeIndex.ps1', + 'evaluation/implementation-guidance-fixtures.json') + + @($manifest.cases | ForEach-Object { + $_.requiredKnowledge + $_.optionalKnowledge + $_.expectedOmitted + }) + foreach ($reference in @($references | Sort-Object -Unique)) { + $source = Join-Path $sourceRoot $reference.Replace('/', [IO.Path]::DirectorySeparatorChar) + $destination = Join-Path $knowledgeRoot $reference.Replace('/', [IO.Path]::DirectorySeparatorChar) + [IO.Directory]::CreateDirectory((Split-Path $destination -Parent)) | Out-Null + [IO.File]::Copy($source, $destination) + } + Initialize-TestRepository $knowledgeRoot 'Knowledge Checkout' + [IO.Directory]::CreateDirectory($results) | Out-Null + $workspaces = [ordered]@{} + foreach ($case in $manifest.cases) { + $workspace = Join-Path $scratch "target-$($case.id)" + Initialize-TestRepository $workspace $case.id + $workspaces[$case.id] = $workspace + } + Set-TestJson $mapPath $workspaces + + Invoke-Evaluator @('-Root', $knowledgeRoot, '-ManifestPath', $manifestPath) $true 'public implementation manifest validates' + $prepared = Join-Path $runnerRoot 'prepared' + Invoke-Evaluator @('-Root', $knowledgeRoot, '-ManifestPath', $manifestPath, + '-PrepareDirectory', $prepared, '-WorkspaceMapPath', $mapPath) $true 'all implementation requests prepare' + $missingCase = @($manifest.cases | Where-Object id -CEQ 'missing-current-decision-and-diff')[0] + $missingRequest = Read-GuidanceJson (Join-Path $prepared "request-$(Get-GuidanceCaseId $missingCase.id).json") + if ($missingRequest.'implementation-diff' -cne '' -or $missingRequest.'decision-context'.Count) { + throw 'Preparation invented missing current implementation context.' + } + $tests.Add('missing context remains missing during preparation') + + Invoke-Evaluator @('-Root', $knowledgeRoot, '-ManifestPath', $manifestPath, + '-CaptureBaseline', '-WorkspaceMapPath', $mapPath, '-BaselinePath', $baselinePath) $true 'read-only baselines capture' + $knowledgeSha = Invoke-GuidanceGit $knowledgeRoot @('rev-parse', 'HEAD') + foreach ($case in $manifest.cases) { + Set-TestJson (Join-Path $results "result-$(Get-GuidanceCaseId $case.id).json") ([ordered]@{ + caseId = Get-GuidanceCaseId $case.id + guidanceReport = New-ControlledReport $case $knowledgeSha + }) + } + $scoreArgs = @('-Root', $knowledgeRoot, '-ManifestPath', $manifestPath, + '-ResultsDirectory', $results, '-BaselinePath', $baselinePath, + '-BaselineSha256', (Get-GuidanceHash $baselinePath)) + Invoke-Evaluator $scoreArgs $true 'focused, deduplicated, clean, missing, and boundary reports score' + + $consumedCase = @($manifest.cases | Where-Object id -CEQ 'exact-consumed-guidance-is-omitted')[0] + $consumedPath = Join-Path $results "result-$(Get-GuidanceCaseId $consumedCase.id).json" + $consumedResult = Read-GuidanceJson $consumedPath + $consumedResult.guidanceReport.deduplication.omitted = @() + $consumedResult.guidanceReport.summary.'omitted-consumed' = 0 + Set-TestJson $consumedPath $consumedResult + Invoke-Evaluator $scoreArgs $false 'exact consumed guidance cannot disappear from deduplication' 'omitted set' + Set-TestJson $consumedPath ([ordered]@{ + caseId = Get-GuidanceCaseId $consumedCase.id + guidanceReport = New-ControlledReport $consumedCase $knowledgeSha + }) + + [IO.File]::AppendAllText((Join-Path $workspaces['read-only-public-interface-checkpoint'] 'CurrentImplementation.al'), "`n// forbidden mutation") + Invoke-Evaluator $scoreArgs $false 'target repository mutation is detected' 'identity/content changed' + + Write-Host "Implementation guidance evaluator regressions PASSED: $($tests.Count) checks." +} finally { + if (Test-Path -LiteralPath $scratch) { + Remove-Item -LiteralPath $scratch -Recurse -Force + } +} + +exit 0 From 4ff0fe0160a65658ead8e01f11cf545643382d3d Mon Sep 17 00:00:00 2001 From: Jesper Schulz-Wedde Date: Wed, 16 Sep 2026 13:29:21 +0200 Subject: [PATCH 2/2] Run implementation guidance in existing CI Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .github/workflows/development-guidance.yml | 9 +-------- tools/Test-DevelopmentGuidanceEvaluator.ps1 | 2 ++ 2 files changed, 3 insertions(+), 8 deletions(-) diff --git a/.github/workflows/development-guidance.yml b/.github/workflows/development-guidance.yml index d95ac3cf..51e24b3f 100644 --- a/.github/workflows/development-guidance.yml +++ b/.github/workflows/development-guidance.yml @@ -2,6 +2,7 @@ name: Validate read-only development guidance on: pull_request: + branches: [main] push: branches: [main] @@ -22,11 +23,3 @@ jobs: - name: Run credential-free guidance evaluator regressions shell: pwsh run: ./tools/Test-DevelopmentGuidanceEvaluator.ps1 - - - name: Validate and prepare implementation-guidance fixtures - shell: pwsh - run: ./tools/Test-DevelopmentGuidanceFixtures.ps1 -Root . -ManifestPath ./evaluation/implementation-guidance-fixtures.json -PrepareDirectory "$env:RUNNER_TEMP/bcquality-implementation-guidance-fixtures" - - - name: Run implementation-guidance evaluator regressions - shell: pwsh - run: ./tools/Test-ImplementationGuidanceEvaluator.ps1 diff --git a/tools/Test-DevelopmentGuidanceEvaluator.ps1 b/tools/Test-DevelopmentGuidanceEvaluator.ps1 index 1bce91d6..ad18d77f 100644 --- a/tools/Test-DevelopmentGuidanceEvaluator.ps1 +++ b/tools/Test-DevelopmentGuidanceEvaluator.ps1 @@ -421,4 +421,6 @@ try { } # Intentional negative native-command probes leave LASTEXITCODE nonzero. +& (Join-Path $PSScriptRoot 'Test-ImplementationGuidanceEvaluator.ps1') +if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } exit 0