Compare commits
139 commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 740e461f4d | |||
| 440ff50df8 | |||
| 8ab40c95d0 | |||
| 878292f467 | |||
| b57114746a | |||
| bae9c70d50 | |||
| b290022c59 | |||
| 526927403e | |||
| 931425a002 | |||
| 35ac933590 | |||
| 97812d3263 | |||
| 2ba3cee060 | |||
| 67a34975fd | |||
| 0a3ce4f414 | |||
| 39c947ab26 | |||
| 14284e3895 | |||
| 79bb54412b | |||
| 50c38a72ea | |||
| 8ea54ec00e | |||
| 1c07afd056 | |||
| 41f769a166 | |||
| 9820f6c945 | |||
| 30d4340e94 | |||
| da16608c60 | |||
| 088cf06aa3 | |||
| 795d494781 | |||
| 96639f7e52 | |||
| 0470674801 | |||
| 99675d576c | |||
| 4032dfcd43 | |||
| 08f31250c4 | |||
| 80e17ec462 | |||
| 9013d250bd | |||
| 66fb567ec6 | |||
| beddba8dd3 | |||
| 0675d9a6dd | |||
| a9d472488d | |||
| 527fb0303f | |||
| 63644c1791 | |||
| 065a55f3c5 | |||
| 915fd6e4da | |||
| 8cde10f09a | |||
| b1307ad42b | |||
| 80a174b73a | |||
| 59c6c280b1 | |||
| c569bdc10e | |||
| 57a491ab66 | |||
| a1295a97a2 | |||
| 39eb2fd084 | |||
| cf23e90afc | |||
| 957ebef6da | |||
| 4042d0b94a | |||
| 43f0a5e4e9 | |||
| 28461cf7f7 | |||
| e1d344307c | |||
| 03caa72d58 | |||
| cbbff91208 | |||
| d8ce788709 | |||
| 4a36fd1853 | |||
| 94c99c46dd | |||
| b174db0936 | |||
| e4925c6b28 | |||
| e86948a71a | |||
| 064078b8c2 | |||
| b0b5890703 | |||
| 4fae3c46b2 | |||
| 8b270fbe99 | |||
| 9e7638cfa1 | |||
| 353b976425 | |||
| 8ac64e407a | |||
| 330d8199bb | |||
| 845d8fbdc9 | |||
| f52a486e3a | |||
| 533d451cd2 | |||
| e017f8bcc2 | |||
| 3043ff74d4 | |||
| 8e4c8b59c7 | |||
| a7cb4b2bb3 | |||
| 6b91c81935 | |||
| 1b29e9673a | |||
| e625645739 | |||
| 6bfe24c7e2 | |||
| 7011898bd0 | |||
| c4554708ca | |||
| 80bfd02726 | |||
| 2935039f44 | |||
| 6b457c02e2 | |||
| f9cc32982c | |||
| b4fc1d8289 | |||
| 9d3959ad85 | |||
| 3c017ac7d9 | |||
| 99156428b6 | |||
| 2eb9c8d840 | |||
| f7efc8ed21 | |||
| 706f44fb34 | |||
| 83ef8eaf7a | |||
| 312c3c0369 | |||
| 31e647f67a | |||
| ccd6b8875a | |||
| 372a92258a | |||
| af6c31c4c1 | |||
| 135f34eb9b | |||
| 76df39670d | |||
| 722a131eb7 | |||
| 6125f933be | |||
| f094e242d1 | |||
| 7b6025a731 | |||
| 712a143e58 | |||
| b3011da017 | |||
| b68514487c | |||
| 5b05009b44 | |||
| 9e5e4a338a | |||
| 3c70206dbd | |||
| 6224487987 | |||
| cef5ab6e80 | |||
| 5b75eefe5c | |||
| ddbe4954e0 | |||
| 1f80574d0d | |||
| 3ea3608a8d | |||
| c6c0987ba5 | |||
| fd01c65213 | |||
| a8ae1fd276 | |||
| 4794de0b79 | |||
| da8c682622 | |||
| de0d94cbc1 | |||
| 1ab4fc34cf | |||
| e999b74eda | |||
| 5a0e8d774a | |||
| a583599ab1 | |||
| 655f60a40d | |||
| 6db508c670 | |||
| afc041a6a2 | |||
| 841292ff36 | |||
| a0283efe84 | |||
| cb31fb9abd | |||
| b227c278eb | |||
| b5c44e6c6e | |||
| 6f82572dba | |||
| 36fd4cfe7b |
345 changed files with 33005 additions and 1124 deletions
|
|
@ -6,6 +6,6 @@
|
|||
"name": "Kjell Tore Guttormsen"
|
||||
},
|
||||
"license": "MIT",
|
||||
"repository": "https://git.fromaitochitta.com/open/ktg-plugin-marketplace",
|
||||
"repository": "https://git.fromaitochitta.com/open/ms-ai-architect",
|
||||
"keywords": ["microsoft", "azure", "ai-architect", "governance", "security", "norwegian-public-sector", "eu-ai-act"]
|
||||
}
|
||||
|
|
|
|||
7
.gitignore
vendored
7
.gitignore
vendored
|
|
@ -23,10 +23,15 @@ node_modules/
|
|||
.work/
|
||||
org/
|
||||
# Generated KB-update artifacts (registry, reports) are ignored, but the
|
||||
# hand-authored taxonomy (lag 0) and the decision ledger (lag 2) are tracked.
|
||||
# hand-authored taxonomy (lag 0), the decision ledger (lag 2), the curated
|
||||
# AI Act deadline source (RX-REG, sync-tested consumer contract) and the
|
||||
# adjudicated Layer B allowlist (Enhet A2 — the gate's strictness contract,
|
||||
# human-reviewed per entry, must survive a fresh clone) are tracked.
|
||||
scripts/kb-update/data/*
|
||||
!scripts/kb-update/data/domain-taxonomy.json
|
||||
!scripts/kb-update/data/decisions.json
|
||||
!scripts/kb-update/data/ai-act-deadlines.json
|
||||
!scripts/kb-update/data/layerb-allowlist.json
|
||||
# Generated skill-lifecycle detection report (Spor B / B1) — regenerated on demand,
|
||||
# like the kb-update reports above. The detector script + curated inputs are tracked.
|
||||
scripts/kb-eval/data/skill-lifecycle-report.json
|
||||
|
|
|
|||
13
CLAUDE.md
13
CLAUDE.md
|
|
@ -102,25 +102,26 @@ Agenter leser navngitte kjernefiler, ikke hele kataloger. «3 kjernefiler» er n
|
|||
|
||||
## Reference docs (read on demand)
|
||||
|
||||
- **Styrende sesjonsplan (R0–R18):** `docs/plugin-roadmap-2026-07.md` — sekvenserer alt gjenstående arbeid; les FØR ny arbeidssesjon startes
|
||||
- **Utvikling, testing, KB-refresh-workflow:** `docs/development.md`
|
||||
- **Playground v3 (decision-builder + rapport-viewer):** `docs/playground.md`
|
||||
- **Recommended MCP servers (detail):** `skills/ms-ai-advisor/references/architecture/recommended-mcp-servers.md`
|
||||
|
||||
## Viktige frister (EU AI Act)
|
||||
|
||||
> **NB — Digital Omnibus (provisorisk):** EU-rådet og Parlamentet ble 7. mai 2026 enige om å utsette høyrisiko-fristene (Digital Omnibus). Endringene trer i kraft først ved formell vedtakelse + publisering i Official Journal (ventet før 2026-08-02), så datoene under er **foreløpige**. Des. 2027 er en **ytre grense** — Kommisjonen kan fremskynde til 6 mnd etter at standarder/spesifikasjoner/veiledning er på plass. Kjør `/architect:classify` for systemspesifikk vurdering.
|
||||
> **NB — Digital Omnibus (vedtatt, avventer OJ):** Digital Omnibus utsetter høyrisiko-fristene i AI Act. Endringen er **formelt vedtatt** (Europaparlamentet 16. juni, Rådet 29. juni 2026) og trer i kraft tredje dag etter publisering i Official Journal — senest 2026-07-30 for anvendelse før 2026-08-02. Datoene under følger vedtatt tekst og bekreftes mot OJ ved publisering. Kjør `/architect:classify` for systemspesifikk vurdering.
|
||||
|
||||
| Frist | Krav | Status |
|
||||
|-------|------|--------|
|
||||
| 2025-02-02 | Forbudte AI-praksiser (Art. 5) | Gjeldende |
|
||||
| 2025-08-02 | GPAI-krav + governance/sanksjoner (Art. 99) | Gjeldende |
|
||||
| 2026-08-02 | Transparens (Art. 50): merking av syntetisk innhold gjelder | Gjeldende |
|
||||
| 2026-08-02 | Transparens (Art. 50): merking av syntetisk innhold | Fra 2026-08-02 |
|
||||
| 2026-12-02 | Art. 50(2): frist for maskinlesbar merking i eksisterende generative systemer | Overgang (Omnibus) |
|
||||
| 2027-12-02 | Annex III høyrisiko (frittstående) — utsatt fra 2026-08-02 | Provisorisk (Omnibus) |
|
||||
| 2028-08-02 | Annex I høyrisiko (innebygd i regulerte produkter) | Provisorisk (Omnibus) |
|
||||
| 2027-12-02 | Annex III høyrisiko (frittstående) — utsatt fra 2026-08-02 | Vedtatt (Omnibus, avventer OJ) |
|
||||
| 2028-08-02 | Annex I høyrisiko (innebygd i regulerte produkter) | Vedtatt (Omnibus, avventer OJ) |
|
||||
|
||||
**Tilsynsmyndigheter:** Datatilsynet (personvern), nasjonal AI-tilsynsmyndighet (under etablering), sektortilsyn.
|
||||
> Maskinlesbar kilde: `scripts/kb-update/data/ai-act-deadlines.json` — synk-testet mot denne tabellen, assessor-malen og hookene (`tests/kb-update/test-ai-act-deadlines-sync.test.mjs`). Oppdater kilden først.
|
||||
|
||||
**Tilsynsmyndigheter:** Nkom (koordinerende markedstilsynsmyndighet og nasjonalt kontaktpunkt for AI-forordningen), Datatilsynet (personvern), sektortilsyn kan utpekes i tillegg.
|
||||
|
||||
## Relaterte plugins (fremtidig)
|
||||
|
||||
|
|
|
|||
131
GOVERNANCE.md
131
GOVERNANCE.md
|
|
@ -1,131 +0,0 @@
|
|||
# Governance
|
||||
|
||||
How this marketplace is maintained, what you can expect from upstream, and how it's meant to be used.
|
||||
|
||||
## TL;DR
|
||||
|
||||
- Solo-maintained, AI-assisted development, MIT licensed.
|
||||
- **Fork-and-own is the default model.** Upstream is a starting point, not a vendor.
|
||||
- Issues welcome as signals. Pull requests are not accepted — see [Why no PRs](#pull-requests--no).
|
||||
- No SLA. Best-effort bug fixes and security advisories. Breaking changes happen and are noted in each plugin's CHANGELOG.
|
||||
|
||||
---
|
||||
|
||||
## Can I trust this?
|
||||
|
||||
Be honest with yourself about what you're adopting:
|
||||
|
||||
- **One maintainer.** If I get hit by a bus, the bus wins. The repos stay up under MIT, but no one owes you a fix.
|
||||
- **AI-generated code with human review.** Every plugin is built through dialog-driven development with Claude Code. I read, test, and judge the output before it ships, but I'm not auditing every line the way a security firm would. Treat it accordingly.
|
||||
- **No commercial interests.** I'm not selling a SaaS, not steering you toward a paid tier, not collecting telemetry. The plugins run locally in your Claude Code installation.
|
||||
- **MIT licensed.** Fork it, modify it, ship it under your own name.
|
||||
|
||||
If you work somewhere that needs vendor accountability, support contracts, or signed assurances — **this isn't that.** Use it as a reference implementation, fork it into your own organization, and own the result.
|
||||
|
||||
---
|
||||
|
||||
## How this is meant to be used
|
||||
|
||||
### Fork-and-own
|
||||
|
||||
The intended workflow:
|
||||
|
||||
1. **Fork** the marketplace (or a single plugin) into your own organization or namespace.
|
||||
2. **Tailor** it to your context — terminology, integrations, cycle lengths, regulatory framing, whatever doesn't fit out of the box.
|
||||
3. **Maintain it yourself.** Treat your fork as the canonical version for your team.
|
||||
4. **Watch upstream selectively.** Cherry-pick changes that help, ignore changes that don't. There's no obligation to stay in sync.
|
||||
|
||||
This isn't a workaround for not accepting PRs. It's the actual recommended adoption pattern, especially for plugins like `okr` and `ms-ai-architect` where every Norwegian public sector organization will need its own tildelingsbrev mappings, terminology, and integrations. A central "one true plugin" would be wrong for everyone.
|
||||
|
||||
### What to change first when you fork
|
||||
|
||||
Each plugin differs, but the common edits are:
|
||||
|
||||
- **Identity** — rename the plugin, replace authorship, update README.
|
||||
- **External integrations** — issue trackers, knowledge bases, dashboards, observability backends. The plugins ship as starting points, not pre-wired. Every organization must configure its own integrations.
|
||||
- **Norwegian-specific framing** — relevant for `okr` and `ms-ai-architect`. Other plugins are jurisdiction-neutral. Rewrite for your jurisdiction if you're outside Norway.
|
||||
- **Reference docs** — the knowledge base in each plugin reflects my reading. Replace with your organization's authoritative sources.
|
||||
- **Hooks and policies** — security thresholds, blocked commands, and audit gates are tuned to my taste. Tune them to yours.
|
||||
|
||||
### Staying current with upstream
|
||||
|
||||
If you want to pull in upstream changes later:
|
||||
|
||||
- **Cherry-pick, don't merge.** Each plugin moves independently and breaking changes land without ceremony.
|
||||
- **Read the CHANGELOG first.** Every plugin has one.
|
||||
- **Keep your customizations in clearly-named files.** The harder upstream is to merge cleanly, the more painful staying current becomes. A `local/` directory or `*.local.md` convention helps.
|
||||
|
||||
---
|
||||
|
||||
## What upstream provides
|
||||
|
||||
| | What I do | What I don't |
|
||||
|---|---|---|
|
||||
| **Bug fixes** | Best-effort when I notice or get a clear report | No SLA, no triage commitment |
|
||||
| **Security issues** | Investigate within reasonable time, document in CHANGELOG | No CVE process, no embargo coordination |
|
||||
| **New features** | When they fit my own usage | Not on request |
|
||||
| **Norwegian public sector context** | Kept current as long as the project lives | If I lose interest or change jobs, the framing freezes |
|
||||
| **Breaking changes** | Documented in CHANGELOG | They happen — version pin if you need stability |
|
||||
| **Compatibility** | Tracked against current Claude Code releases | No long-term support branches |
|
||||
|
||||
If any of this is a dealbreaker — fork now, version-pin, and stop reading upstream.
|
||||
|
||||
---
|
||||
|
||||
## How to contribute
|
||||
|
||||
### Issues — yes, please
|
||||
|
||||
Issues are the most valuable thing you can send me:
|
||||
|
||||
- **Bug reports** with reproduction steps. Even a screenshot helps.
|
||||
- **Use-case feedback.** "I tried to use this in my organization and X didn't fit" is genuinely useful, even if I can't fix it for you.
|
||||
- **Pointers to better sources.** If you know a DFØ veileder, an NSM guideline, or an academic paper that contradicts what's in a knowledge base, tell me.
|
||||
- **Security findings.** See each plugin's `SECURITY.md` for disclosure preference where one exists; otherwise email rather than open a public issue.
|
||||
|
||||
### Pull requests — no
|
||||
|
||||
This is deliberate, not laziness:
|
||||
|
||||
- **Solo review is a bottleneck.** Honest PR review takes me longer than rewriting from scratch. The math doesn't work.
|
||||
- **Forks are where the value is.** The fork-and-own model means upstream consolidation isn't the point. Your organization's adaptations belong in your fork, not mine.
|
||||
- **AI-generated code complicates provenance.** Every line here is produced through dialog with Claude Code, with me as the judge. Mixing in PRs from contributors with different processes and licensing assumptions creates a mess I'd rather not untangle.
|
||||
|
||||
If you've built something useful on top of a fork, **publish it under your own name and link back.** I'll happily list notable forks here once they exist.
|
||||
|
||||
### Notable forks
|
||||
|
||||
*(To be populated as forks emerge. If you've forked one of these plugins for production use, open an issue and I'll add a link.)*
|
||||
|
||||
---
|
||||
|
||||
## Relationship between plugins
|
||||
|
||||
These plugins are **independent**. Install one without the others, fork one without the others. They share conventions (slash command naming, hook patterns, AI-generated disclosure) but no runtime dependencies.
|
||||
|
||||
The marketplace is a **catalog**, not a suite. Don't fork the whole repo unless you actually want to maintain everything.
|
||||
|
||||
---
|
||||
|
||||
## Versioning and stability
|
||||
|
||||
- **Semantic versioning per plugin.** Each plugin has its own `CHANGELOG.md` and version number.
|
||||
- **Breaking changes happen.** I bump the major version when they do, but I don't run an LTS branch.
|
||||
- **Pin your version.** If stability matters more than features, install a specific version and stay there until you choose to upgrade.
|
||||
|
||||
---
|
||||
|
||||
## Public sector adoption notes
|
||||
|
||||
For Norwegian etater specifically:
|
||||
|
||||
- **DPIA-relevant data flows are documented in the relevant plugin README where applicable.** Read them before installation.
|
||||
- **No data leaves your machine** beyond what Claude Code itself sends to Anthropic. The plugins themselves do not call external services unless you configure an integration.
|
||||
- **Drøftingsplikt and ledelsesansvar** are not replaced by these tools. The `okr` plugin coaches; it does not decide. The `ms-ai-architect` plugin advises; it does not approve.
|
||||
- **Choose your Claude deployment carefully.** claude.ai vs. API direct vs. Bedrock in EU region have different data residency profiles. The plugins don't choose for you.
|
||||
|
||||
---
|
||||
|
||||
## License
|
||||
|
||||
MIT for all plugins in this marketplace. See each plugin's `LICENSE` file.
|
||||
79
README.md
79
README.md
|
|
@ -1,10 +1,12 @@
|
|||
# AI Architect Plugin for Claude Code
|
||||
|
||||
Microsoft AI Solution Architect — structured architecture guidance for the full Microsoft AI stack.
|
||||
|
||||
> Your virtual Microsoft AI solution architect — meet **Cosmo Skyberg**.
|
||||
|
||||
> **Solo-maintained, fork-and-own.** This plugin is a starting point, not a vendor product. Issues are welcome as signals; pull requests are not accepted. See [GOVERNANCE.md](GOVERNANCE.md) for the full model and what upstream provides.
|
||||
> **Solo-maintained, fork-and-own.** This plugin is a starting point, not a vendor product. Issues are welcome as signals; pull requests are not accepted. See [GOVERNANCE.md](https://git.fromaitochitta.com/open/repo-standard/src/branch/main/GOVERNANCE.md) for the full model and what upstream provides.
|
||||
|
||||
*AI-generated: all code produced by Claude Code through dialog-driven development. [Full disclosure →](../../README.md#ai-generated-code-disclosure)*
|
||||
*AI-generated: all code produced by Claude Code through dialog-driven development. Every change is human-directed, reviewed, and validated before commit.*
|
||||
|
||||

|
||||

|
||||
|
|
@ -14,12 +16,40 @@
|
|||
|
||||
A Claude Code plugin that provides structured architecture guidance across the full Microsoft AI stack. Cosmo Skyberg is a methodical, opinionated architect persona who understands the problem before recommending technology, verifies claims against live Microsoft Learn documentation via MCP, and delivers assessments calibrated for Norwegian public sector governance — while remaining useful for any enterprise context.
|
||||
|
||||
## Install
|
||||
|
||||
Both lines are required — the first adds the marketplace, the second installs this plugin from it:
|
||||
|
||||
```bash
|
||||
claude plugin marketplace add https://git.fromaitochitta.com/open/ktg-plugin-marketplace.git
|
||||
claude plugin install ms-ai-architect@ktg-plugin-marketplace
|
||||
```
|
||||
|
||||
To browse the other plugins in the marketplace instead, run `/plugin`.
|
||||
|
||||
Or enable directly in `~/.claude/settings.json`:
|
||||
|
||||
```json
|
||||
{
|
||||
"enabledPlugins": {
|
||||
"ms-ai-architect@ktg-plugin-marketplace": true
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Requirements
|
||||
|
||||
- [Claude Code](https://docs.anthropic.com/en/docs/claude-code) installed
|
||||
- Python with [uv](https://github.com/astral-sh/uv) (for the microsoft-learn MCP server)
|
||||
- Network access to `learn.microsoft.com`
|
||||
|
||||
---
|
||||
|
||||
## Table of Contents
|
||||
|
||||
- [What Is This?](#what-is-this)
|
||||
- [Quick Start](#quick-start)
|
||||
- [Non-goals](#non-goals)
|
||||
- [First Conversation](#first-conversation)
|
||||
- [Commands](#commands)
|
||||
- [Agent Architecture](#agent-architecture)
|
||||
- [Knowledge Base](#knowledge-base)
|
||||
|
|
@ -30,7 +60,7 @@ A Claude Code plugin that provides structured architecture guidance across the f
|
|||
- [Technology Coverage](#technology-coverage)
|
||||
- [Enterprise Onboarding](#enterprise-onboarding)
|
||||
- [Related Plugins](#related-plugins)
|
||||
- [Version History](#version-history)
|
||||
- [Changelog](#changelog)
|
||||
- [License & Attribution](#license--attribution)
|
||||
|
||||
---
|
||||
|
|
@ -58,33 +88,20 @@ Key capabilities:
|
|||
|
||||
---
|
||||
|
||||
## Quick Start
|
||||
## Non-goals
|
||||
|
||||
### Prerequisites
|
||||
What this plugin deliberately does not do — worth knowing before you adopt it:
|
||||
|
||||
- [Claude Code](https://docs.anthropic.com/en/docs/claude-code) installed
|
||||
- Python with [uv](https://github.com/astral-sh/uv) (for the microsoft-learn MCP server)
|
||||
- Network access to `learn.microsoft.com`
|
||||
- **It is not legal advice.** Classification, DPIA/PVK and ROS output is decision support for architects. It is drafted to be reviewed by your data protection officer, legal counsel and approving authority — not to replace them.
|
||||
- **It does not deploy or provision anything.** Every command produces a document or an assessment. No command creates, modifies or deletes Azure resources, and the plugin writes no infrastructure-as-code.
|
||||
- **It is not a live pricing source.** Cost estimates use published list prices with explicit disclaimers and P10/P50/P90 ranges. Verify against the Azure pricing calculator before committing a budget.
|
||||
- **It covers the Microsoft stack only.** AWS Bedrock, Google Vertex AI and direct-to-provider LLM APIs are out of scope, including comparisons against them.
|
||||
- **It is not a vendor product.** Solo-maintained, no SLA, pull requests are not accepted — see [GOVERNANCE.md](https://git.fromaitochitta.com/open/repo-standard/src/branch/main/GOVERNANCE.md).
|
||||
- **It is not affiliated with Microsoft.** Product names are trademarks of Microsoft Corporation, and nothing here is endorsed by them.
|
||||
|
||||
### Installation
|
||||
---
|
||||
|
||||
Add the marketplace and browse plugins with `/plugin`:
|
||||
|
||||
```bash
|
||||
claude plugin marketplace add https://git.fromaitochitta.com/open/ktg-plugin-marketplace.git
|
||||
```
|
||||
|
||||
Or enable directly in `~/.claude/settings.json`:
|
||||
|
||||
```json
|
||||
{
|
||||
"enabledPlugins": {
|
||||
"ms-ai-architect@ktg-plugin-marketplace": true
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### First Conversation
|
||||
## First Conversation
|
||||
|
||||
```
|
||||
> /architect
|
||||
|
|
@ -349,7 +366,7 @@ These MCP servers enhance the plugin's capabilities but are not required:
|
|||
|
||||
| Server | Purpose |
|
||||
|--------|---------|
|
||||
| [azure-mcp-server](https://github.com/microsoft/azure-mcp-server) | Live Azure infrastructure inspection (Storage, Key Vault, AI Search, RBAC) |
|
||||
| [azure-mcp](https://github.com/Azure/azure-mcp) | Live Azure infrastructure inspection (Storage, Key Vault, AI Search, RBAC) |
|
||||
| bicep-mcp-server | Infrastructure-as-Code generation for Azure resources |
|
||||
| [azure-devops-mcp](https://github.com/microsoft/azure-devops-mcp) | Work items, pipelines, repos integration |
|
||||
|
||||
|
|
@ -551,7 +568,7 @@ For organizations that need deeper customization beyond what onboarding provides
|
|||
|
||||
### LLM Security Plugin
|
||||
|
||||
The **[LLM Security Plugin](../llm-security)** is a companion plugin that covers the agentic AI attack surface — the runtime security dimension that complements this plugin's architecture-level assessments.
|
||||
The **[LLM Security Plugin](https://git.fromaitochitta.com/open/llm-security)** is a companion plugin that covers the agentic AI attack surface — the runtime security dimension that complements this plugin's architecture-level assessments.
|
||||
|
||||
While **ms-ai-architect** evaluates *what to build* (platform selection, compliance, cost, risk), the LLM Security Plugin evaluates *whether what you built is safe to deploy* by scanning Claude Code plugins, MCP servers, and AI agent configurations against the OWASP LLM Top 10.
|
||||
|
||||
|
|
@ -655,7 +672,9 @@ Category-to-skill routing is defined in `scripts/kb-update/data/domain-taxonomy.
|
|||
|
||||
---
|
||||
|
||||
## Version History
|
||||
## Changelog
|
||||
|
||||
Full history, including patch releases: [CHANGELOG.md](CHANGELOG.md). The table below summarizes the minor releases.
|
||||
|
||||
| Version | Date | Highlights |
|
||||
|---------|------|-----------|
|
||||
|
|
|
|||
32
SECURITY.md
Normal file
32
SECURITY.md
Normal file
|
|
@ -0,0 +1,32 @@
|
|||
# Security policy
|
||||
|
||||
## Reporting a vulnerability
|
||||
|
||||
Report privately to <security@fromaitochitta.com> - do not open a
|
||||
public issue.
|
||||
Canonical repository: https://git.fromaitochitta.com/open/ms-ai-architect
|
||||
|
||||
Please include the affected version or commit, a minimal reproduction,
|
||||
and the impact you see. We acknowledge every report within 5 working
|
||||
days, agree a fix and disclosure timeline with the reporter, and aim to
|
||||
disclose within 90 days of the initial report.
|
||||
|
||||
## Response process
|
||||
|
||||
1. Acknowledge within 5 working days.
|
||||
2. Triage and confirm severity within 10 working days.
|
||||
3. Develop and test a fix.
|
||||
4. Publish an advisory and credit the reporter unless they prefer
|
||||
to remain anonymous.
|
||||
|
||||
## Supported versions
|
||||
|
||||
The latest tagged release (currently 1.17.0) is the only supported
|
||||
version. Security fixes land on `main` and are released as a new tag;
|
||||
we do not backport fixes to older releases. See `CHANGELOG.md` for the
|
||||
release history.
|
||||
|
||||
## Advisories
|
||||
|
||||
We publish advisories for confirmed vulnerabilities once a fix is
|
||||
available, crediting the reporter unless they request anonymity.
|
||||
|
|
@ -2,12 +2,12 @@
|
|||
name: adr-writer-agent
|
||||
description: |
|
||||
Generates Architecture Decision Records (ADR) in MADR v3.0 format from structured input.
|
||||
Reads adr-template.md, fills in from session context, and writes to file.
|
||||
Reads adr-template.md, fills in from session context, and returns the ADR markdown to the main context.
|
||||
Use when architect:adr needs to generate a complete ADR document.
|
||||
Triggers on: ADR generation, decision documentation, architect:adr delegation.
|
||||
model: opus
|
||||
color: orange
|
||||
tools: ["Read", "Write", "Glob"]
|
||||
tools: ["Read", "Glob"]
|
||||
---
|
||||
|
||||
# ADR Writer Agent
|
||||
|
|
@ -40,7 +40,7 @@ Et kompakt sammendrag av virksomhetskonteksten injiseres ambient i hovedøkten v
|
|||
|
||||
### 1. Read Template
|
||||
|
||||
Read `skills/ms-ai-advisor/references/architecture/adr-template.md` for the MADR v3.0 format.
|
||||
Read `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/adr-template.md` for the MADR v3.0 format.
|
||||
|
||||
### 2. Parse Input
|
||||
|
||||
|
|
@ -87,9 +87,9 @@ Fill in every section of the MADR template:
|
|||
|
||||
**Validering og oppfølging**: Concrete next steps with responsible party.
|
||||
|
||||
### 4. Write to File
|
||||
### 4. Return to Main Context
|
||||
|
||||
Write the ADR to the location specified in the input. Default: `docs/adr/ADR-NNN-[slug].md`
|
||||
Return the complete ADR markdown as your final message. The main context (the `/architect:adr` command) writes it to file — do not write files yourself (you run as a subagent).
|
||||
|
||||
## Output Format
|
||||
|
||||
|
|
@ -101,7 +101,7 @@ The generated ADR should be:
|
|||
|
||||
## Quality Checklist
|
||||
|
||||
Before writing:
|
||||
Before returning:
|
||||
- [ ] All template sections filled (no placeholders)
|
||||
- [ ] Compliance section included (even if "Not assessed")
|
||||
- [ ] Confidence level reflects actual analysis quality
|
||||
|
|
|
|||
|
|
@ -23,17 +23,17 @@ You are a Norwegian regulatory compliance specialist focused on EU AI Act assess
|
|||
## Knowledge Base References
|
||||
|
||||
Read relevant files from:
|
||||
- `skills/ms-ai-governance/references/responsible-ai/ai-act-classification-methodology.md` — **OBLIGATORISK:** 4-stegs klassifiseringsmetodikk
|
||||
- `skills/ms-ai-governance/references/responsible-ai/ai-act-provider-obligations.md` — Provider-forpliktelser Art. 9-27
|
||||
- `skills/ms-ai-governance/references/responsible-ai/ai-act-deployer-obligations.md` — Deployer-forpliktelser Art. 26-27
|
||||
- `skills/ms-ai-governance/references/responsible-ai/ai-act-fria-template.md` — FRIA-mal Art. 27
|
||||
- `skills/ms-ai-governance/references/responsible-ai/ai-act-conformity-assessment.md` — Samsvarsvurdering Annex IV/VI/VII
|
||||
- `skills/ms-ai-governance/references/responsible-ai/ai-act-transparency-notices.md` — Art. 13/50 transparensnotiser
|
||||
- `skills/ms-ai-governance/references/responsible-ai/ai-act-microsoft-tools-mapping.md` — Artikkel-til-verktøy-mapping
|
||||
- `skills/ms-ai-governance/references/responsible-ai/ai-act-compliance-guide.md` — Generell compliance-veileder
|
||||
- `skills/ms-ai-governance/references/responsible-ai/ai-act-annex-iii-checklist.md` — Annex III sjekkliste med beslutningstre
|
||||
- `skills/ms-ai-governance/references/norwegian-public-sector-governance/norge-ai-strategy-government.md` — Norsk AI-strategi
|
||||
- `skills/ms-ai-governance/references/norwegian-public-sector-governance/forvaltningsloven-ai-decisions.md` — Forvaltningsloven og AI
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-classification-methodology.md` — **OBLIGATORISK:** 4-stegs klassifiseringsmetodikk
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-provider-obligations.md` — Provider-forpliktelser Art. 9-27
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-deployer-obligations.md` — Deployer-forpliktelser Art. 26-27
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-fria-template.md` — FRIA-mal Art. 27
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-conformity-assessment.md` — Samsvarsvurdering Annex IV/VI/VII
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-transparency-notices.md` — Art. 13/50 transparensnotiser
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-microsoft-tools-mapping.md` — Artikkel-til-verktøy-mapping
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-compliance-guide.md` — Generell compliance-veileder
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-annex-iii-checklist.md` — Annex III sjekkliste med beslutningstre
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/norwegian-public-sector-governance/norge-ai-strategy-government.md` — Norsk AI-strategi
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/norwegian-public-sector-governance/forvaltningsloven-ai-decisions.md` — Forvaltningsloven og AI
|
||||
|
||||
## Virksomhetskontekst (automatisk)
|
||||
|
||||
|
|
@ -59,7 +59,7 @@ Ekstraher fra brukerens input:
|
|||
|
||||
### Fase 2: Klassifisering (4-stegs)
|
||||
Les `ai-act-classification-methodology.md` og utfør:
|
||||
1. **Forbudt-sjekk (Art. 5):** Er noen av de 8 forbudte praksisene relevante?
|
||||
1. **Forbudt-sjekk (Art. 5):** Er noen av de forbudte praksisene relevante? (8 i kraft; i tillegg nudifiers/NCII/CSAM-forbudet — Digital Omnibus, anvendelse 2026-12-02, avventer OJ)
|
||||
2. **Annex III høyrisiko-sjekk:** Treffer systemet noen av de 8 kategoriene?
|
||||
3. **GPAI-sjekk:** Er systemet basert på generell AI-modell? Systemisk risiko?
|
||||
4. **Begrenset/Minimal:** Transparenskrav eller frivillig Code of Conduct?
|
||||
|
|
@ -142,8 +142,9 @@ Anbefal oppfølgingsaktiviteter:
|
|||
| 2025-02-02 | Forbudte AI-praksiser (Art. 5) | [Gjelder/Gjelder ikke] |
|
||||
| 2025-08-02 | GPAI-krav + governance/sanksjoner (Art. 99) | [Gjelder/Gjelder ikke] |
|
||||
| 2026-08-02 | Transparens (Art. 50, syntetisk innhold) | [Gjelder/Gjelder ikke] |
|
||||
| 2027-12-02 | Annex III høyrisiko — provisorisk (utsatt fra 2026-08-02 via Omnibus, avventer OJ) | [Gjelder/Gjelder ikke] |
|
||||
| 2028-08-02 | Annex I høyrisiko innebygd (provisorisk) | [Gjelder/Gjelder ikke] |
|
||||
| 2026-12-02 | Art. 50(2): maskinlesbar merking i eksisterende generative systemer | [Gjelder/Gjelder ikke] |
|
||||
| 2027-12-02 | Annex III høyrisiko — utsatt fra 2026-08-02 (Omnibus vedtatt, avventer OJ) | [Gjelder/Gjelder ikke] |
|
||||
| 2028-08-02 | Annex I høyrisiko innebygd (Omnibus vedtatt, avventer OJ) | [Gjelder/Gjelder ikke] |
|
||||
|
||||
### Referanser
|
||||
- [Liste over KB-filer og MCP-kilder brukt]
|
||||
|
|
@ -176,8 +177,8 @@ Bruk `microsoft_docs_search` for:
|
|||
## Norwegian Public Sector Context
|
||||
|
||||
- Alle vurderinger gjøres i norsk kontekst (EØS-implementering)
|
||||
- Datatilsynet er sannsynlig tilsynsmyndighet (personverndimensjon)
|
||||
- Nasjonal AI-tilsynsmyndighet er under etablering
|
||||
- Nkom er koordinerende markedstilsynsmyndighet og nasjonalt kontaktpunkt for AI-forordningen i Norge
|
||||
- Datatilsynet er tilsynsmyndighet for personverndimensjonen; sektortilsyn kan utpekes i tillegg
|
||||
- Forvaltningsloven gjelder i tillegg til AI Act for vedtakssystemer
|
||||
- Offentlig sektor er nesten alltid deployer, sjelden provider
|
||||
|
||||
|
|
|
|||
|
|
@ -63,7 +63,7 @@ For høyrisiko-systemer, verifiser:
|
|||
- [ ] **FRIA gjennomført (Art. 27):** Obligatorisk for offentlig sektor-deployers
|
||||
|
||||
**Ekstra KB-referanse:**
|
||||
- `skills/ms-ai-governance/references/responsible-ai/ai-act-conformity-assessment.md`
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-conformity-assessment.md`
|
||||
|
||||
### 3. Utredningsinstruksen (Analysis Requirements)
|
||||
- **Problem description**: Clear problem statement, affected parties identified
|
||||
|
|
@ -156,21 +156,21 @@ Read the architecture proposal. Extract:
|
|||
|
||||
### 2. Load Reference Knowledge
|
||||
Read relevant knowledge base files:
|
||||
- `skills/ms-ai-advisor/references/architecture/decision-trees.md` — Platform selection validation
|
||||
- `skills/ms-ai-advisor/references/architecture/security.md` — Security best practices
|
||||
- `skills/ms-ai-advisor/references/architecture/public-sector-checklist.md` — Norwegian compliance checklist
|
||||
- `skills/ms-ai-advisor/references/architecture/ai-utredning-template.md` — Utredningsinstruksen template
|
||||
- `skills/ms-ai-advisor/references/architecture/cost-models.md` — Cost estimation patterns
|
||||
- `skills/ms-ai-advisor/references/architecture/licensing-matrix.md` — License requirements
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/decision-trees.md` — Platform selection validation
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/security.md` — Security best practices
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/public-sector-checklist.md` — Norwegian compliance checklist
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/ai-utredning-template.md` — Utredningsinstruksen template
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/cost-models.md` — Cost estimation patterns
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/licensing-matrix.md` — License requirements
|
||||
|
||||
Load domain-specific references only when dimension requires depth (max 2-3 additional):
|
||||
- AI Act: `responsible-ai/ai-act-compliance-guide.md`, `responsible-ai/ai-act-annex-iii-checklist.md`
|
||||
- Governance: `responsible-ai/ai-governance-structure-framework.md`
|
||||
- Norwegian: `norwegian-public-sector-governance/utredningsinstruksen-ai-methodology.md`
|
||||
- Security: `ai-security-engineering/ai-threat-modeling-stride.md`
|
||||
- Cost: `cost-optimization/azure-ai-foundry-cost-governance.md`, `cost-optimization/deterministic-cost-calculation-model.md`
|
||||
- RAG-arkitektur (når løsningen er RAG-/gjenfinningsbasert): `skills/ms-ai-engineering/references/rag-architecture/rag-core-patterns.md`, `rag-architecture/agentic-rag-patterns.md`, `rag-architecture/rag-evaluation-frameworks.md`
|
||||
- MLOps/GenAIOps (når løsningen har produksjons-/livssyklusfokus): `skills/ms-ai-engineering/references/mlops-genaiops/genaiops-llm-specific-practices.md`, `mlops-genaiops/monitoring-observability-ml-systems.md`, `mlops-genaiops/model-deployment-strategies-azure.md`
|
||||
- AI Act: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-compliance-guide.md`, `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-annex-iii-checklist.md`
|
||||
- Governance: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-governance-structure-framework.md`
|
||||
- Norwegian: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/norwegian-public-sector-governance/utredningsinstruksen-ai-methodology.md`
|
||||
- Security: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/ai-security-engineering/ai-threat-modeling-stride.md`
|
||||
- Cost: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/cost-optimization/azure-ai-foundry-cost-governance.md`, `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/cost-optimization/deterministic-cost-calculation-model.md`
|
||||
- RAG-arkitektur (når løsningen er RAG-/gjenfinningsbasert): `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-engineering/references/rag-architecture/rag-core-patterns.md`, `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-engineering/references/rag-architecture/agentic-rag-patterns.md`, `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-engineering/references/rag-architecture/rag-evaluation-frameworks.md`
|
||||
- MLOps/GenAIOps (når løsningen har produksjons-/livssyklusfokus): `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-engineering/references/mlops-genaiops/genaiops-llm-specific-practices.md`, `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-engineering/references/mlops-genaiops/monitoring-observability-ml-systems.md`, `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-engineering/references/mlops-genaiops/model-deployment-strategies-azure.md`
|
||||
|
||||
## Virksomhetskontekst (automatisk)
|
||||
|
||||
|
|
|
|||
|
|
@ -47,7 +47,7 @@ Provide accurate, comprehensive cost estimates for Microsoft AI solutions includ
|
|||
|
||||
**ALWAYS start by reading:**
|
||||
```bash
|
||||
Read skills/ms-ai-advisor/references/architecture/cost-models.md
|
||||
Read ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/cost-models.md
|
||||
```
|
||||
|
||||
This file contains verified pricing data and calculation formulas.
|
||||
|
|
@ -55,14 +55,14 @@ This file contains verified pricing data and calculation formulas.
|
|||
## Knowledge Base References (max 3 per invokasjon)
|
||||
|
||||
Read these core files:
|
||||
- `skills/ms-ai-security/references/cost-optimization/deterministic-cost-calculation-model.md` — **OBLIGATORISK:** Enhetspriser, beregningsformler, P10/P50/P90 konfidensintervaller
|
||||
- `skills/ms-ai-security/references/cost-optimization/azure-ai-foundry-cost-governance.md` — FinOps-rammeverk
|
||||
- `skills/ms-ai-advisor/references/architecture/cost-models.md` — Cost model templates
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/cost-optimization/deterministic-cost-calculation-model.md` — **OBLIGATORISK:** Enhetspriser, beregningsformler, P10/P50/P90 konfidensintervaller
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/cost-optimization/azure-ai-foundry-cost-governance.md` — FinOps-rammeverk
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/cost-models.md` — Cost model templates
|
||||
|
||||
Load additional files only when estimate requires specific depth:
|
||||
- PTU: `cost-optimization/ptu-vs-paygo-economics.md`
|
||||
- Caching: `cost-optimization/semantic-caching-patterns.md`
|
||||
- Model selection: `cost-optimization/model-selection-price-performance.md`
|
||||
- PTU: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/cost-optimization/ptu-vs-paygo-economics.md`
|
||||
- Caching: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/cost-optimization/semantic-caching-patterns.md`
|
||||
- Model selection: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/cost-optimization/model-selection-price-performance.md`
|
||||
|
||||
## Virksomhetskontekst (automatisk)
|
||||
|
||||
|
|
|
|||
|
|
@ -45,7 +45,7 @@ Et kompakt sammendrag av virksomhetskonteksten injiseres ambient i hovedøkten v
|
|||
|
||||
Les prompt-maler fra:
|
||||
```
|
||||
skills/ms-ai-advisor/references/architecture/diagram-prompt-templates.md
|
||||
${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/diagram-prompt-templates.md
|
||||
```
|
||||
|
||||
## Azure-stilguide
|
||||
|
|
|
|||
|
|
@ -17,15 +17,15 @@ You are a Norwegian data protection specialist conducting structured DPIAs for A
|
|||
## Knowledge Base References (3 kjernefiler + betinget)
|
||||
|
||||
Read these core files:
|
||||
- `skills/ms-ai-governance/references/norwegian-public-sector-governance/dpia-norwegian-methodology-ai.md` — DPIA-metodikk
|
||||
- `skills/ms-ai-governance/references/responsible-ai/gdpr-compliance-ai-systems.md` — GDPR for AI
|
||||
- `skills/ms-ai-governance/references/responsible-ai/ai-impact-assessment-framework.md` — Konsekvensvurdering
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/norwegian-public-sector-governance/dpia-norwegian-methodology-ai.md` — DPIA-metodikk
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/gdpr-compliance-ai-systems.md` — GDPR for AI
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-impact-assessment-framework.md` — Konsekvensvurdering
|
||||
|
||||
Load additional files only when assessment requires specific depth:
|
||||
- Bias: `responsible-ai/bias-detection-mitigation-strategies.md`
|
||||
- PII: `ai-security-engineering/pii-detection-norwegian-context.md`
|
||||
- Data leakage: `ai-security-engineering/data-leakage-prevention-ai.md`
|
||||
- **Cross-border / Schrems II (OBLIGATORISK når data kan nås fra tredjeland — se Fase 3, risiko 7):** `monitoring-observability/data-residency-audit-monitoring.md` — EDPB seks-stegs-TIA, CLOUD Act/FISA 702/EO 12333-restanalyse, EO 14086/DPF-status, tekniske tilleggstiltak
|
||||
- Bias: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/bias-detection-mitigation-strategies.md`
|
||||
- PII: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/ai-security-engineering/pii-detection-norwegian-context.md`
|
||||
- Data leakage: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/ai-security-engineering/data-leakage-prevention-ai.md`
|
||||
- **Cross-border / Schrems II (OBLIGATORISK når data kan nås fra tredjeland — se Fase 3, risiko 7):** `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/monitoring-observability/data-residency-audit-monitoring.md` — EDPB seks-stegs-TIA, CLOUD Act/FISA 702/EO 12333-restanalyse, EO 14086/DPF-status, tekniske tilleggstiltak
|
||||
|
||||
## Virksomhetskontekst (automatisk)
|
||||
|
||||
|
|
@ -47,12 +47,12 @@ Før DPIA-vurderingen, sjekk om AI Act-klassifisering er utført:
|
|||
- Integrer deployer-forpliktelser fra `ai-act-deployer-obligations.md` som tiltak i Fase 4
|
||||
|
||||
### Hvis ikke klassifisert
|
||||
- Spør om det bør gjøres: "Er det gjennomført AI Act-klassifisering for dette systemet? Hvis nei, anbefaler vi `/architect:classify` — men DPIA fortsetter uansett."
|
||||
- Marker i rapporten at AI Act-klassifisering ikke er dokumentert, og anbefal `/architect:classify` som neste steg (du kjører som subagent uten brukertur — still ingen spørsmål)
|
||||
- Fortsett DPIA som normalt — klassifisering er ikke forutsetning
|
||||
|
||||
### Ekstra KB-referanser for AI Act
|
||||
- `skills/ms-ai-governance/references/responsible-ai/ai-act-deployer-obligations.md` — Deployer-krav inkl. FRIA og logging
|
||||
- `skills/ms-ai-governance/references/responsible-ai/ai-act-transparency-notices.md` — Art. 13/50 maler for transparenstiltak
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-deployer-obligations.md` — Deployer-krav inkl. FRIA og logging
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-transparency-notices.md` — Art. 13/50 maler for transparenstiltak
|
||||
|
||||
## DPIA Framework (5 Phases)
|
||||
|
||||
|
|
@ -93,7 +93,7 @@ Risk categories for AI systems:
|
|||
|
||||
#### Cross-border / Schrems II — obligatorisk TIA (risiko 7)
|
||||
|
||||
Når systemet bruker en amerikansk-eid skyleverandør (Azure/Microsoft 365/Foundry) eller data på annen måte kan nås fra tredjeland, **er det ikke nok å navngi risikoen** — load `monitoring-observability/data-residency-audit-monitoring.md` og gjennomfør EDPB seks-stegs Transfer Impact Assessment:
|
||||
Når systemet bruker en amerikansk-eid skyleverandør (Azure/Microsoft 365/Foundry) eller data på annen måte kan nås fra tredjeland, **er det ikke nok å navngi risikoen** — load `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/monitoring-observability/data-residency-audit-monitoring.md` og gjennomfør EDPB seks-stegs Transfer Impact Assessment:
|
||||
|
||||
1. Kartlegg overføringene (inkl. residual: support, troubleshooting, telemetri)
|
||||
2. Identifiser overføringsverktøyet (adekvansvedtak / SCCs / unntak)
|
||||
|
|
@ -141,9 +141,9 @@ Read the AI system description or architecture proposal. Extract:
|
|||
|
||||
### 2. Load Reference Knowledge
|
||||
Core files are loaded via Knowledge Base References above. For deeper analysis:
|
||||
- Fairness: `responsible-ai/fairness-testing-measurement.md`
|
||||
- Transparency: `responsible-ai/transparency-documentation-standards.md`
|
||||
- Human oversight: `responsible-ai/human-in-the-loop-oversight.md`
|
||||
- Fairness: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/fairness-testing-measurement.md`
|
||||
- Transparency: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/transparency-documentation-standards.md`
|
||||
- Human oversight: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/human-in-the-loop-oversight.md`
|
||||
|
||||
### 3. Validate Latest Guidance
|
||||
Use `microsoft_docs_search` for:
|
||||
|
|
@ -225,7 +225,7 @@ Follow the output format below with all sections completed.
|
|||
|
||||
If missing information:
|
||||
- State assumptions clearly
|
||||
- Request specific details needed
|
||||
- Note which specific inputs are missing rather than requesting them (you run as a non-interactive subagent with no user turn)
|
||||
- Provide conditional assessments
|
||||
- Note "Kan ikke vurdere [area] uten [info]"
|
||||
|
||||
|
|
|
|||
|
|
@ -41,11 +41,11 @@ Given a set of Microsoft license types, produce a complete capability map showin
|
|||
### 1. Read Reference Data
|
||||
|
||||
Read these files:
|
||||
- `skills/ms-ai-advisor/references/architecture/licensing-matrix.md` — master matrix
|
||||
- `skills/ms-ai-advisor/references/platforms/azure-ai-foundry.md` — Foundry capabilities
|
||||
- `skills/ms-ai-advisor/references/platforms/copilot-studio.md` — Copilot Studio capabilities
|
||||
- `skills/ms-ai-advisor/references/platforms/m365-copilot.md` — M365 Copilot capabilities
|
||||
- `skills/ms-ai-advisor/references/platforms/power-platform.md` — Power Platform capabilities
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/licensing-matrix.md` — master matrix
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/platforms/azure-ai-foundry.md` — Foundry capabilities
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/platforms/copilot-studio.md` — Copilot Studio capabilities
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/platforms/m365-copilot.md` — M365 Copilot capabilities
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/platforms/power-platform.md` — Power Platform capabilities
|
||||
|
||||
### 2. Map Licenses to Capabilities
|
||||
|
||||
|
|
|
|||
|
|
@ -34,8 +34,8 @@ Et kompakt sammendrag av virksomhetskonteksten injiseres ambient i hovedøkten v
|
|||
## Lokal KB-baseline (betinget — RAG / MLOps / engineering-temaer)
|
||||
|
||||
Når forskningstemaet er RAG, gjenfinning, MLOps eller GenAIOps, les den relevante engineering-kjernefilen **først** som hypotese-baseline — verifiser den deretter mot live Microsoft Learn. KB-en kan være utdatert; **MCP-resultatet er fasit**.
|
||||
- RAG/gjenfinning: `skills/ms-ai-engineering/references/rag-architecture/rag-core-patterns.md`, `rag-architecture/agentic-rag-patterns.md`
|
||||
- MLOps/GenAIOps: `skills/ms-ai-engineering/references/mlops-genaiops/genaiops-llm-specific-practices.md`, `mlops-genaiops/llm-evaluation-production.md`
|
||||
- RAG/gjenfinning: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-engineering/references/rag-architecture/rag-core-patterns.md`, `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-engineering/references/rag-architecture/agentic-rag-patterns.md`
|
||||
- MLOps/GenAIOps: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-engineering/references/mlops-genaiops/genaiops-llm-specific-practices.md`, `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-engineering/references/mlops-genaiops/llm-evaluation-production.md`
|
||||
|
||||
Les maks 2 baseline-filer. **Flagg eksplisitt** hvis live docs avviker fra KB-baselinen (samme avviks-flagging som Fase 4).
|
||||
|
||||
|
|
|
|||
|
|
@ -22,7 +22,7 @@ You are a Norwegian risk management specialist conducting structured ROS analyse
|
|||
|
||||
ROS er en bevisst KB-tung agent. En deterministisk analyse krever et fast **kjernesett** pluss et **betinget sett** lastet på definerte triggere. Dette er større enn det generelle «3 kjernefiler»-mønsteret i `CLAUDE.md` (security/cost/review) — det er en dokumentert, håndhevet last-rekkefølge, ikke fri lesing. **To analytikere som kjører samme system skal laste de samme filene i samme rekkefølge.**
|
||||
|
||||
Alle stier under `skills/ms-ai-governance/references/norwegian-public-sector-governance/` med mindre annet er angitt.
|
||||
Alle stier under `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/norwegian-public-sector-governance/` med mindre annet er angitt.
|
||||
|
||||
### Obligatorisk kjerne (last ALLTID, i denne rekkefølgen)
|
||||
1. `ros-ai-threat-library.md` — AI-trusselbibliotek (kilde for T-xxx-IDer)
|
||||
|
|
@ -36,12 +36,12 @@ Alle stier under `skills/ms-ai-governance/references/norwegian-public-sector-gov
|
|||
| Sektor oppdaget (helse/transport/finans/justis/utdanning) | `ros-sector-checklists.md` |
|
||||
| Multi-agent / agent-orkestrering | `ros-maestro-multiagent.md` (MAESTRO 7-lag) |
|
||||
| DPIA eller sikkerhetsvurdering skal integreres | `ros-dpia-security-integration.md` |
|
||||
| AI Act-dybde i dimensjon 6 | `responsible-ai/ai-act-classification-methodology.md` + `responsible-ai/ai-act-provider-obligations.md` (maks 2) |
|
||||
| AI Act-dybde i dimensjon 6 | `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-classification-methodology.md` + `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-provider-obligations.md` (maks 2) |
|
||||
|
||||
### Referanse (last kun ved eksplisitt behov, ikke default)
|
||||
- `skills/ms-ai-security/references/ai-security-engineering/security-scoring-rubrics-6x5.md` — scoringsmønster-referanse
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/ai-security-engineering/security-scoring-rubrics-6x5.md` — scoringsmønster-referanse
|
||||
- `ros-analyse-ai-systems.md` — generell ROS-bakgrunn
|
||||
- `responsible-ai/ai-risk-taxonomy-classification.md` — risikotaksonomi
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-risk-taxonomy-classification.md` — risikotaksonomi
|
||||
|
||||
**Budsjett:** kjerne (4) + betinget (maks 2-3 på trigger) = typisk 5-7 filer. Aldri last hele katalogen; last ikke en betinget fil hvis triggeren ikke utløses.
|
||||
|
||||
|
|
@ -118,8 +118,8 @@ I tillegg til eksisterende trusler i dimensjon 6, vurder følgende:
|
|||
- Art. 50 (transparens): Opptil 7,5 MEUR eller 1,5 % av global omsetning
|
||||
|
||||
**KB-referanser for AI Act-dybde i dimensjon 6** (betinget — last per last-kontrakten øverst, AI Act-trigger):
|
||||
- `skills/ms-ai-governance/references/responsible-ai/ai-act-classification-methodology.md`
|
||||
- `skills/ms-ai-governance/references/responsible-ai/ai-act-provider-obligations.md`
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-classification-methodology.md`
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-provider-obligations.md`
|
||||
|
||||
## 8-fase metodikk (NS 5814-compliant)
|
||||
|
||||
|
|
@ -273,7 +273,7 @@ Risk Levels: Low (1-6), Medium (7-12), High (13-19), Critical (20-25)
|
|||
|
||||
If missing information:
|
||||
- State assumptions clearly
|
||||
- Request specific details needed
|
||||
- Note which specific inputs are missing rather than requesting them (you run as a non-interactive subagent with no user turn)
|
||||
- Provide conditional assessments
|
||||
- Note "Kan ikke vurdere [area] uten [info]"
|
||||
|
||||
|
|
|
|||
|
|
@ -21,14 +21,14 @@ You are a Microsoft AI security specialist. You assess AI architectures against
|
|||
## Knowledge Base References (max 3 per invokasjon)
|
||||
|
||||
Read these core files:
|
||||
- `skills/ms-ai-security/references/ai-security-engineering/security-scoring-rubrics-6x5.md` — **OBLIGATORISK:** Deterministiske scoringsrubrikker
|
||||
- `skills/ms-ai-security/references/ai-security-engineering/ai-security-scoring-framework.md` — Scoring-rammeverk
|
||||
- `skills/ms-ai-security/references/ai-security-engineering/ai-threat-modeling-stride.md` — STRIDE trusselmodellering
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/ai-security-engineering/security-scoring-rubrics-6x5.md` — **OBLIGATORISK:** Deterministiske scoringsrubrikker
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/ai-security-engineering/ai-security-scoring-framework.md` — Scoring-rammeverk
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/ai-security-engineering/ai-threat-modeling-stride.md` — STRIDE trusselmodellering
|
||||
|
||||
Load additional files only when assessment requires specific depth:
|
||||
- Prompt injection: `ai-security-engineering/prompt-injection-defense-patterns.md`
|
||||
- Governance: `responsible-ai/ai-act-compliance-guide.md`
|
||||
- Norwegian context: `norwegian-public-sector-governance/nsm-grunnprinsipper-ai-mapping.md`
|
||||
- Prompt injection: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/ai-security-engineering/prompt-injection-defense-patterns.md`
|
||||
- Governance: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-compliance-guide.md`
|
||||
- Norwegian context: `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/norwegian-public-sector-governance/nsm-grunnprinsipper-ai-mapping.md`
|
||||
|
||||
## Virksomhetskontekst (automatisk)
|
||||
|
||||
|
|
@ -147,8 +147,8 @@ Read the architecture proposal or solution description. Look for:
|
|||
|
||||
### 2. Load Reference Knowledge
|
||||
Read these knowledge base files:
|
||||
- `skills/ms-ai-advisor/references/architecture/security.md` — Security best practices
|
||||
- `skills/ms-ai-advisor/references/architecture/public-sector-checklist.md` — Norwegian compliance (if exists)
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/security.md` — Security best practices
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/public-sector-checklist.md` — Norwegian compliance (if exists)
|
||||
|
||||
### 3. Validate Latest Guidance
|
||||
Use `microsoft_docs_search` for:
|
||||
|
|
|
|||
|
|
@ -42,13 +42,13 @@ Hvis `/architect:cost` ble brukt, inkluder kostnadsestimatet.
|
|||
Bruk Task-verktøyet til å delegere ADR-generering:
|
||||
|
||||
```
|
||||
Task(general-purpose): "Read agents/adr-writer-agent.md for your role and instructions.
|
||||
Task(ms-ai-architect:adr-writer-agent): "
|
||||
Generate an ADR based on the current session context.
|
||||
Beslutning: [beslutningstittel]
|
||||
Bakgrunn: [forretningskontekst]
|
||||
Alternativer: [vurderte alternativer]
|
||||
Valgt løsning: [beslutning med begrunnelse]
|
||||
Les også: skills/ms-ai-advisor/references/architecture/adr-template.md"
|
||||
Les også: ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/adr-template.md"
|
||||
```
|
||||
|
||||
### 4. Skriv til fil
|
||||
|
|
|
|||
|
|
@ -30,7 +30,7 @@ Spør om nøkkelinformasjon hvis ikke kjent:
|
|||
|
||||
### 3. Les kunnskapsbasen
|
||||
|
||||
- `skills/ms-ai-governance/references/norwegian-public-sector-governance/anskaffelser-ai-procurement-framework.md` — lovgrunnlag (anskaffelsesloven/-forskriften), EØS-regelverk, AI-spesifikk kravspesifikasjon, leverandørevaluering, etiske krav, DFØs IT-anskaffelsesveiledning
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/norwegian-public-sector-governance/anskaffelser-ai-procurement-framework.md` — lovgrunnlag (anskaffelsesloven/-forskriften), EØS-regelverk, AI-spesifikk kravspesifikasjon, leverandørevaluering, etiske krav, DFØs IT-anskaffelsesveiledning
|
||||
|
||||
For AI Act-deployer-/transparenskrav som skal inn i kravspec: koble til `/architect:requirements` og `/architect:classify`.
|
||||
|
||||
|
|
|
|||
|
|
@ -12,9 +12,9 @@ Du aktiverer nå **Cosmo Skyberg**, en erfaren Microsoft AI Solution Architect.
|
|||
|
||||
## Instruksjoner
|
||||
|
||||
1. Les og aktiver skillen `ms-ai-advisor/SKILL.md`
|
||||
1. Les og aktiver skillen `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/SKILL.md`
|
||||
2. Følg arbeidsprosessen definert i skillen
|
||||
3. Bruk kunnskapsbasene i `references/` for verifisering
|
||||
3. Bruk kunnskapsbasene i `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/` for verifisering
|
||||
4. Bruk `microsoft-learn` MCP-verktøy for oppdatert informasjon
|
||||
|
||||
## Oppstart
|
||||
|
|
|
|||
|
|
@ -33,8 +33,8 @@ Spør om nøkkeltall hvis ikke allerede kjent:
|
|||
|
||||
### 3. Les kunnskapsbasene
|
||||
|
||||
- `skills/ms-ai-governance/references/norwegian-public-sector-governance/samfunnsokonomisk-analyse-nnv.md` — NNV-formel, kalkulasjonsrente, diskonteringsfaktorer, skattefinansieringskostnad, prissatte vs. ikke-prissatte virkninger
|
||||
- `skills/ms-ai-governance/references/norwegian-public-sector-governance/gevinstrealisering-dfo-methodology.md` — DFØs 5-stegs modell + gevinstregister-mal
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/norwegian-public-sector-governance/samfunnsokonomisk-analyse-nnv.md` — NNV-formel, kalkulasjonsrente, diskonteringsfaktorer, skattefinansieringskostnad, prissatte vs. ikke-prissatte virkninger
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/norwegian-public-sector-governance/gevinstrealisering-dfo-methodology.md` — DFØs 5-stegs modell + gevinstregister-mal
|
||||
|
||||
For selve kostnadsestimatet: deleger til `/architect:cost` eller `cost-estimation-agent` og bruk resultatet som input til NNV-en.
|
||||
|
||||
|
|
|
|||
|
|
@ -33,7 +33,7 @@ Bruk samtalehistorikk hvis denne informasjonen allerede er gitt.
|
|||
Kjør AI Act-agenten via Task for klassifiseringen:
|
||||
|
||||
```
|
||||
Task(ai-act-assessor): "Read agents/ai-act-assessor.md for your role and instructions.
|
||||
Task(ms-ai-architect:ai-act-assessor): "
|
||||
Gjennomfør en EU AI Act-klassifisering (Fase 1-3) for følgende AI-system:
|
||||
|
||||
**System:** [systemnavn]
|
||||
|
|
@ -48,9 +48,9 @@ Gjennomfør en EU AI Act-klassifisering (Fase 1-3) for følgende AI-system:
|
|||
Modus: Klassifisering — fokus på risikonivå og rolle.
|
||||
|
||||
Les kunnskapsbasene:
|
||||
- skills/ms-ai-governance/references/responsible-ai/ai-act-classification-methodology.md
|
||||
- skills/ms-ai-governance/references/responsible-ai/ai-act-annex-iii-checklist.md
|
||||
- skills/ms-ai-governance/references/responsible-ai/ai-act-compliance-guide.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-classification-methodology.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-annex-iii-checklist.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-compliance-guide.md
|
||||
|
||||
Lever klassifiseringsresultat med risikonivå, Annex III-kategori, GPAI-status, rolle og begrunnelse."
|
||||
```
|
||||
|
|
|
|||
|
|
@ -36,16 +36,16 @@ Hvis bare én plattform er angitt, foreslå den mest relevante motparten basert
|
|||
Deleger research til `research-agent` via Task-verktøyet:
|
||||
|
||||
```
|
||||
Task(general-purpose): "Les agents/research-agent.md og utfør research.
|
||||
Task(ms-ai-architect:research-agent): "Utfør research.
|
||||
Sammenlign [Plattform A] og [Plattform B] for [use case].
|
||||
Fokusér på: kapabiliteter, begrensninger, prising, regional tilgjengelighet.
|
||||
Bruk microsoft_docs_search for begge plattformer."
|
||||
```
|
||||
|
||||
Les også relevant kunnskapsbase:
|
||||
- `skills/ms-ai-advisor/references/architecture/decision-trees.md` — beslutningsrammeverk
|
||||
- Les plattformfil(er) relevant for sammenligningen fra `skills/ms-ai-advisor/references/platforms/` (max 2-3 filer)
|
||||
- **Ved 3+ alternativer eller `--weighted`:** `skills/ms-ai-advisor/references/architecture/alternativanalyse-methodology.md` — vektet multi-kriterie-analyse (scoringsskala, standardkriterier, vekting, begrunnelsestabell)
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/decision-trees.md` — beslutningsrammeverk
|
||||
- Les plattformfil(er) relevant for sammenligningen fra `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/platforms/` (max 2-3 filer)
|
||||
- **Ved 3+ alternativer eller `--weighted`:** `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/alternativanalyse-methodology.md` — vektet multi-kriterie-analyse (scoringsskala, standardkriterier, vekting, begrunnelsestabell)
|
||||
|
||||
### 3. Bygg sammenligning
|
||||
|
||||
|
|
|
|||
|
|
@ -27,7 +27,7 @@ Avklar:
|
|||
### 2. Deleger til AI Act-agent
|
||||
|
||||
```
|
||||
Task(ai-act-assessor): "Read agents/ai-act-assessor.md for your role and instructions.
|
||||
Task(ms-ai-architect:ai-act-assessor): "
|
||||
Gjennomfør samsvarsvurdering for følgende AI-system:
|
||||
|
||||
**System:** [systemnavn]
|
||||
|
|
@ -40,8 +40,8 @@ Gjennomfør samsvarsvurdering for følgende AI-system:
|
|||
Modus: Conformity — Annex IV sjekkliste og samsvarserklæring.
|
||||
|
||||
Les kunnskapsbasene:
|
||||
- skills/ms-ai-governance/references/responsible-ai/ai-act-conformity-assessment.md
|
||||
- skills/ms-ai-governance/references/responsible-ai/ai-act-provider-obligations.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-conformity-assessment.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-provider-obligations.md
|
||||
|
||||
Lever:
|
||||
1. Annex IV 9-element sjekkliste med status per element
|
||||
|
|
|
|||
|
|
@ -24,25 +24,25 @@ Hvis informasjon mangler, spør brukeren om nøkkeltall.
|
|||
|
||||
### 2. Les kostnadsreferanse
|
||||
|
||||
Les `skills/ms-ai-advisor/references/architecture/cost-models.md` for baseline-priser per plattform.
|
||||
Les `skills/ms-ai-security/references/cost-optimization/deterministic-cost-calculation-model.md` for enhetspriser, beregningsformler og P10/P50/P90 konfidensintervaller.
|
||||
Les `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/cost-models.md` for baseline-priser per plattform.
|
||||
Les `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/cost-optimization/deterministic-cost-calculation-model.md` for enhetspriser, beregningsformler og P10/P50/P90 konfidensintervaller.
|
||||
|
||||
**Ved selvhostede modeller / GPU-inferens eller `--capacity`:** Les også
|
||||
- `skills/ms-ai-security/references/performance-scalability/gpu-compute-sizing.md` — GPU VM-serier, modellstørrelse→GPU-krav, minnebudsjett, batch/throughput
|
||||
- `skills/ms-ai-advisor/references/architecture/capacity-feasibility-benchmarks.md` — kompetanse-gap-matrise + tidsplan-validering mot bransjebenchmarks
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/performance-scalability/gpu-compute-sizing.md` — GPU VM-serier, modellstørrelse→GPU-krav, minnebudsjett, batch/throughput
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/capacity-feasibility-benchmarks.md` — kompetanse-gap-matrise + tidsplan-validering mot bransjebenchmarks
|
||||
|
||||
### 3. Deleger estimering
|
||||
|
||||
Bruk Task-verktøyet til å lansere `cost-estimation-agent`:
|
||||
|
||||
```
|
||||
Task(general-purpose): "Les agents/cost-estimation-agent.md og utfør kostnadsestimering.
|
||||
Task(ms-ai-architect:cost-estimation-agent): "Utfør kostnadsestimering.
|
||||
Plattform: [plattform]
|
||||
Brukere: [antall]
|
||||
Volum: [volum]
|
||||
Region: [region]
|
||||
Les også: skills/ms-ai-advisor/references/architecture/cost-models.md
|
||||
og skills/ms-ai-advisor/references/architecture/licensing-matrix.md
|
||||
Les også: ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/cost-models.md
|
||||
og ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/licensing-matrix.md
|
||||
Verifiser priser via microsoft_docs_search."
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -36,9 +36,9 @@ Avklar hvis ikke kjent (gjenbruk samtalehistorikk og `org/`-filer hvis onboardet
|
|||
|
||||
### 3. Les kunnskapsbasene
|
||||
|
||||
- `skills/ms-ai-advisor/references/architecture/decision-trees.md` — plattformvalg (Foundry / Copilot Studio / Power Platform / Agent Framework)
|
||||
- `skills/ms-ai-advisor/references/architecture/security.md` — sikkerhetsarkitektur og soneinndeling
|
||||
- `skills/ms-ai-advisor/references/architecture/cost-models.md` — kostnadsdimensjonering
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/decision-trees.md` — plattformvalg (Foundry / Copilot Studio / Power Platform / Agent Framework)
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/security.md` — sikkerhetsarkitektur og soneinndeling
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/cost-models.md` — kostnadsdimensjonering
|
||||
|
||||
For dybde, deleger til eksisterende kommandoer/agenter og bruk resultatene som input:
|
||||
- `/architect:compare` — strukturert alternativanalyse (bruk `--weighted` ved 3+ alternativer)
|
||||
|
|
|
|||
|
|
@ -46,11 +46,11 @@ Hvis kontekst mangler, still korte spørsmål:
|
|||
Kjør `diagram-generation-agent` via Task:
|
||||
|
||||
```
|
||||
Task(general-purpose): "Read agents/diagram-generation-agent.md for your role and instructions.
|
||||
Task(ms-ai-architect:diagram-generation-agent): "
|
||||
Generer [type]-diagram for [scenario].
|
||||
Komponenter: [liste over tjenester].
|
||||
Kontekst: [ekstra detaljer].
|
||||
Les: skills/ms-ai-advisor/references/architecture/diagram-prompt-templates.md"
|
||||
Les: ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/diagram-prompt-templates.md"
|
||||
```
|
||||
|
||||
## Format Parameter
|
||||
|
|
|
|||
|
|
@ -32,7 +32,7 @@ Bruk samtalehistorikk hvis denne informasjonen allerede er gitt.
|
|||
Kjør DPIA-agenten via Task for selve vurderingen:
|
||||
|
||||
```
|
||||
Task(architect:dpia-agent): "Read agents/dpia-agent.md for your role and instructions.
|
||||
Task(ms-ai-architect:dpia-agent): "
|
||||
Gjennomfør en komplett DPIA for følgende AI-system:
|
||||
|
||||
**System:** [systemnavnet]
|
||||
|
|
@ -41,14 +41,15 @@ Gjennomfør en komplett DPIA for følgende AI-system:
|
|||
**Registrerte:** [hvem som berøres]
|
||||
**Behandlingsgrunnlag:** [GDPR art. 6/9]
|
||||
**Kontekst:** [sektor/kontekst — offentlig, privat, finans, helse, etc.]
|
||||
**AI Act-klassifisering (fra /architect:classify, hvis utført):** [risikonivå + rolle + Annex III-kategori — ellers "ikke klassifisert"]
|
||||
|
||||
Les kunnskapsbasene (kjerne):
|
||||
- skills/ms-ai-governance/references/norwegian-public-sector-governance/dpia-norwegian-methodology-ai.md
|
||||
- skills/ms-ai-governance/references/responsible-ai/gdpr-compliance-ai-systems.md
|
||||
- skills/ms-ai-governance/references/responsible-ai/ai-impact-assessment-framework.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/norwegian-public-sector-governance/dpia-norwegian-methodology-ai.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/gdpr-compliance-ai-systems.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-impact-assessment-framework.md
|
||||
|
||||
Betinget (OBLIGATORISK hvis amerikansk-eid skyleverandør eller data nåbar fra tredjeland):
|
||||
- skills/ms-ai-governance/references/monitoring-observability/data-residency-audit-monitoring.md (EDPB seks-stegs-TIA + CLOUD Act/FISA 702/EO 14086-restanalyse for risiko 7)
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/monitoring-observability/data-residency-audit-monitoring.md (EDPB seks-stegs-TIA + CLOUD Act/FISA 702/EO 14086-restanalyse for risiko 7)
|
||||
|
||||
Lever en komplett DPIA-rapport med alle 5 faser, risikomatrise og anbefaling."
|
||||
```
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ Avklar:
|
|||
### 2. Deleger til AI Act-agent
|
||||
|
||||
```
|
||||
Task(ai-act-assessor): "Read agents/ai-act-assessor.md for your role and instructions.
|
||||
Task(ms-ai-architect:ai-act-assessor): "
|
||||
Gjennomfør en FRIA (Art. 27) for følgende AI-system:
|
||||
|
||||
**System:** [systemnavn]
|
||||
|
|
@ -41,8 +41,8 @@ Gjennomfør en FRIA (Art. 27) for følgende AI-system:
|
|||
Modus: FRIA — utfyll Art. 27-malen.
|
||||
|
||||
Les kunnskapsbasene:
|
||||
- skills/ms-ai-governance/references/responsible-ai/ai-act-fria-template.md
|
||||
- skills/ms-ai-governance/references/responsible-ai/ai-act-deployer-obligations.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-fria-template.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-deployer-obligations.md
|
||||
|
||||
Lever en komplett FRIA med alle 7 seksjoner: systembeskrivelse, berørte grupper, rettighetsmatrise (12 rettigheter), konsekvensanalyse, tilsynsnotifikasjon, godkjenning, vedlegg."
|
||||
```
|
||||
|
|
|
|||
|
|
@ -42,7 +42,7 @@ Dette gir ~15-20 skills per sesjon istedenfor ~5.
|
|||
|
||||
### Strategi: Én agent per skill
|
||||
|
||||
Hver skill delegeres til én `general-purpose` Task-agent (sonnet) som utfører:
|
||||
Hver skill delegeres til én `general-purpose` Task-agent (opus) som utfører:
|
||||
1. MCP-research (5-8 kall)
|
||||
2. Filskriving (Write-verktøyet)
|
||||
3. Returnerer kort kvittering
|
||||
|
|
@ -55,7 +55,7 @@ Kjør **5 agenter parallelt** i én melding. Vent på resultat, oppdater state,
|
|||
|
||||
### Agent-prompt (bruk denne malen)
|
||||
|
||||
For HVER skill, send denne prompten til en `general-purpose` Task-agent med `model: sonnet`:
|
||||
For HVER skill, send denne prompten til en `general-purpose` Task-agent med `model: opus`:
|
||||
|
||||
```
|
||||
Du er Cosmo Skyberg, senior Microsoft AI Solution Architect. Generer en kunnskapsreferanse.
|
||||
|
|
@ -64,7 +64,7 @@ Du er Cosmo Skyberg, senior Microsoft AI Solution Architect. Generer en kunnskap
|
|||
|
||||
Skriv kunnskapsreferanse: **{SKILL_TITLE}**
|
||||
Kategori: {CATEGORY_NAME}
|
||||
Fil: skills/{TARGET_SKILL}/references/{CATEGORY_DIR}/{SKILL_ID}.md
|
||||
Fil: ${CLAUDE_PLUGIN_ROOT}/skills/{TARGET_SKILL}/references/{CATEGORY_DIR}/{SKILL_ID}.md
|
||||
|
||||
## Steg 1: Research (OBLIGATORISK)
|
||||
|
||||
|
|
@ -80,7 +80,7 @@ Bruk MCP-verktøy for oppdatert informasjon:
|
|||
## Steg 2: Skriv filen
|
||||
|
||||
Bruk Write-verktøyet til å skrive filen til:
|
||||
{PLUGIN_ROOT}/skills/{TARGET_SKILL}/references/{CATEGORY_DIR}/{SKILL_ID}.md
|
||||
${CLAUDE_PLUGIN_ROOT}/skills/{TARGET_SKILL}/references/{CATEGORY_DIR}/{SKILL_ID}.md
|
||||
|
||||
Format (STRENGT — alle seksjoner påkrevd):
|
||||
|
||||
|
|
@ -167,11 +167,11 @@ error: {only if failed}
|
|||
|
||||
```
|
||||
# Batch 1: 5 parallelle agenter
|
||||
Task(general-purpose, sonnet): "Research + write skill: Hybrid Search..."
|
||||
Task(general-purpose, sonnet): "Research + write skill: Semantic Ranker..."
|
||||
Task(general-purpose, sonnet): "Research + write skill: Citation Tracking..."
|
||||
Task(general-purpose, sonnet): "Research + write skill: RAG Evaluation..."
|
||||
Task(general-purpose, sonnet): "Research + write skill: Multi-Index..."
|
||||
Task(general-purpose, opus): "Research + write skill: Hybrid Search..."
|
||||
Task(general-purpose, opus): "Research + write skill: Semantic Ranker..."
|
||||
Task(general-purpose, opus): "Research + write skill: Citation Tracking..."
|
||||
Task(general-purpose, opus): "Research + write skill: RAG Evaluation..."
|
||||
Task(general-purpose, opus): "Research + write skill: Multi-Index..."
|
||||
|
||||
# Vent på alle 5 → oppdater state.json → neste batch
|
||||
```
|
||||
|
|
|
|||
|
|
@ -23,17 +23,17 @@ Ekstraher lisenstype(r) fra argumentet. Vanlige kombinasjoner:
|
|||
|
||||
### 2. Les referanse
|
||||
|
||||
Les `skills/ms-ai-advisor/references/architecture/licensing-matrix.md` for komplett lisensmatrise.
|
||||
Les `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/licensing-matrix.md` for komplett lisensmatrise.
|
||||
|
||||
### 3. Deleger kartlegging
|
||||
|
||||
Bruk Task-verktøyet til å lansere `license-mapper-agent`:
|
||||
|
||||
```
|
||||
Task(general-purpose): "Les agents/license-mapper-agent.md og kartlegg lisenser.
|
||||
Task(ms-ai-architect:license-mapper-agent): "Kartlegg lisenser.
|
||||
Lisenser: [lisenstype(r)]
|
||||
Les: skills/ms-ai-advisor/references/architecture/licensing-matrix.md
|
||||
og skills/ms-ai-advisor/references/platforms/ (alle plattformfiler).
|
||||
Les: ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/licensing-matrix.md
|
||||
og ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/platforms/ (alle plattformfiler).
|
||||
Verifiser kritiske punkter via microsoft_docs_search."
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -23,7 +23,7 @@ Ekstraher:
|
|||
|
||||
### 2. Les migrasjonsreferanse
|
||||
|
||||
Les `skills/ms-ai-advisor/references/architecture/migration-patterns.md` for:
|
||||
Les `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/migration-patterns.md` for:
|
||||
- Migrasjonsmatrise (innsats, risiko, tidslinje)
|
||||
- Detaljerte migrasjonsmønstre med steg-for-steg
|
||||
- Kodeeksempler for vanlige migrasjoner
|
||||
|
|
|
|||
|
|
@ -86,7 +86,7 @@ Deretter start onboarding-agenten (se under).
|
|||
Sjekk eksisterende `$ORG_DIR/*.md`-filer for å avgjøre resume-punkt:
|
||||
|
||||
```
|
||||
Task(architect:onboarding-agent): "Read agents/onboarding-agent.md for your role and instructions.
|
||||
Task(ms-ai-architect:onboarding-agent): "
|
||||
|
||||
Gjennomfør onboarding-intervju for å samle virksomhetsspesifikk kontekst.
|
||||
|
||||
|
|
|
|||
|
|
@ -29,13 +29,13 @@ Spør brukeren om nøkkelinformasjon (hvis ikke allerede kjent):
|
|||
|
||||
### 3. Les template
|
||||
|
||||
Les `skills/ms-ai-advisor/references/architecture/poc-template.md` for komplett POC-rammeverk.
|
||||
Les `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/poc-template.md` for komplett POC-rammeverk.
|
||||
|
||||
### 3b. Les domene-spesifikke mønstre (betinget)
|
||||
|
||||
Hvis use-caset treffer et engineering-domene, les 1-2 kjernefiler for å forankre scope og suksesskriterier (ikke hele katalogen):
|
||||
- **RAG / gjenfinning:** `skills/ms-ai-engineering/references/rag-architecture/rag-core-patterns.md`, `rag-architecture/rag-evaluation-frameworks.md` — sett målbare gjenfinnings-/grounding-kriterier
|
||||
- **MLOps / produksjonssetting:** `skills/ms-ai-engineering/references/mlops-genaiops/genaiops-llm-specific-practices.md`, `mlops-genaiops/llm-evaluation-production.md` — POC-evaluering + driftskriterier
|
||||
- **RAG / gjenfinning:** `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-engineering/references/rag-architecture/rag-core-patterns.md`, `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-engineering/references/rag-architecture/rag-evaluation-frameworks.md` — sett målbare gjenfinnings-/grounding-kriterier
|
||||
- **MLOps / produksjonssetting:** `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-engineering/references/mlops-genaiops/genaiops-llm-specific-practices.md`, `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-engineering/references/mlops-genaiops/llm-evaluation-production.md` — POC-evaluering + driftskriterier
|
||||
|
||||
### 4. Generer POC-plan
|
||||
|
||||
|
|
|
|||
|
|
@ -27,7 +27,7 @@ Avklar:
|
|||
### 2. Deleger til AI Act-agent
|
||||
|
||||
```
|
||||
Task(ai-act-assessor): "Read agents/ai-act-assessor.md for your role and instructions.
|
||||
Task(ms-ai-architect:ai-act-assessor): "
|
||||
Kartlegg konkrete AI Act-forpliktelser (Fase 4-5) for følgende system:
|
||||
|
||||
**System:** [systemnavn]
|
||||
|
|
@ -40,12 +40,12 @@ Kartlegg konkrete AI Act-forpliktelser (Fase 4-5) for følgende system:
|
|||
Modus: Requirements — fokus på forpliktelser og tiltaksplan.
|
||||
|
||||
Les kunnskapsbasene:
|
||||
- skills/ms-ai-governance/references/responsible-ai/ai-act-provider-obligations.md
|
||||
- skills/ms-ai-governance/references/responsible-ai/ai-act-deployer-obligations.md
|
||||
- skills/ms-ai-governance/references/responsible-ai/ai-act-microsoft-tools-mapping.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-provider-obligations.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-deployer-obligations.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-microsoft-tools-mapping.md
|
||||
|
||||
Betinget (kun ved regulert privat sektor):
|
||||
- Hvis sektor = finans: skills/ms-ai-governance/references/norwegian-public-sector-governance/ros-sector-checklists.md (§3 Finans — 17-punkts sjekkliste med DORA, Finanstilsynets IKT-forskrift, EBA/GL/2023/06). Kartlegg DORA-forpliktelser i tillegg til AI Act-kravene.
|
||||
- Hvis sektor = finans: ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/norwegian-public-sector-governance/ros-sector-checklists.md (§3 Finans — 17-punkts sjekkliste med DORA, Finanstilsynets IKT-forskrift, EBA/GL/2023/06). Kartlegg DORA-forpliktelser i tillegg til AI Act-kravene.
|
||||
|
||||
Lever detaljert forpliktelsesliste med gap-analyse og tiltaksplan."
|
||||
```
|
||||
|
|
|
|||
|
|
@ -36,7 +36,7 @@ Ekstraher:
|
|||
Bruk Task-verktøyet til å lansere `research-agent`:
|
||||
|
||||
```
|
||||
Task(general-purpose): "Les agents/research-agent.md og utfør research.
|
||||
Task(ms-ai-architect:research-agent): "Utfør research.
|
||||
Plattform: [full plattformnavn]
|
||||
Tidsperiode: [periode]
|
||||
Fokusområder:
|
||||
|
|
|
|||
|
|
@ -44,16 +44,16 @@ Identifiser hvilke dimensjoner som er mest kritiske for scenarioet:
|
|||
Bruk Task-verktøyet til å lansere `architecture-review-agent`:
|
||||
|
||||
```
|
||||
Task(general-purpose): "Les agents/architecture-review-agent.md og utfør en
|
||||
Task(ms-ai-architect:architecture-review-agent): "Utfør en
|
||||
arkitekturgjennomgang for [løsningsnavn].
|
||||
Arkitekturbeskrivelse: [beskrivelse fra bruker]
|
||||
Kontekst: [offentlig sektor / sektor / stadium]
|
||||
Vurder alle 6 dimensjoner med 1-5 score.
|
||||
Les også:
|
||||
- skills/ms-ai-advisor/references/architecture/decision-trees.md
|
||||
- skills/ms-ai-advisor/references/architecture/public-sector-checklist.md
|
||||
- skills/ms-ai-advisor/references/architecture/security.md
|
||||
- skills/ms-ai-advisor/references/architecture/ai-utredning-template.md"
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/decision-trees.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/public-sector-checklist.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/security.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/ai-utredning-template.md"
|
||||
```
|
||||
|
||||
### 4. Berik med arkitekturperspektiv
|
||||
|
|
|
|||
|
|
@ -33,7 +33,7 @@ Sjekk om --quick er angitt. Bruk samtalehistorikk hvis info allerede er gitt.
|
|||
Kjør ROS-agenten via Task for selve vurderingen:
|
||||
|
||||
```
|
||||
Task(ros-analysis-agent): "Read agents/ros-analysis-agent.md for your role and instructions.
|
||||
Task(ms-ai-architect:ros-analysis-agent): "
|
||||
Gjennomfør en [komplett / quick] ROS-analyse for følgende AI-system:
|
||||
|
||||
**System:** [systemnavn]
|
||||
|
|
@ -43,21 +43,23 @@ Gjennomfør en [komplett / quick] ROS-analyse for følgende AI-system:
|
|||
**Sektor:** [sektor]
|
||||
**Borgermøtende:** [ja/nei]
|
||||
**Kontekst:** [ytterligere kontekst]
|
||||
**AI Act-klassifisering (dimensjon 6, fra /architect:classify):** [risikonivå + rolle — ellers "ikke klassifisert"]
|
||||
**DPIA-funn (fra /architect:dpia, hvis utført):** [sentrale personvernrisikoer — ellers "ikke utført"]
|
||||
[**Modus:** Quick (top-10 risikoer, trafikklys) — if --quick]
|
||||
|
||||
Les kunnskapsbasene per last-kontrakten i agentfilen — kjerne (alltid) + betinget (kun på trigger):
|
||||
|
||||
Kjerne (alltid, i rekkefølge):
|
||||
- skills/ms-ai-governance/references/norwegian-public-sector-governance/ros-ai-threat-library.md
|
||||
- skills/ms-ai-governance/references/norwegian-public-sector-governance/ros-scoring-rubrics-7x5.md
|
||||
- skills/ms-ai-governance/references/norwegian-public-sector-governance/ros-methodology-ns5814-iso31000.md
|
||||
- skills/ms-ai-governance/references/norwegian-public-sector-governance/ros-report-templates.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/norwegian-public-sector-governance/ros-ai-threat-library.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/norwegian-public-sector-governance/ros-scoring-rubrics-7x5.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/norwegian-public-sector-governance/ros-methodology-ns5814-iso31000.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/norwegian-public-sector-governance/ros-report-templates.md
|
||||
|
||||
Betinget (kun når triggeren utløses):
|
||||
- ros-sector-checklists.md (hvis relevant sektor oppdaget)
|
||||
- ros-maestro-multiagent.md (hvis multi-agent / agent-orkestrering)
|
||||
- ros-dpia-security-integration.md (hvis DPIA/sikkerhet skal integreres)
|
||||
- responsible-ai/ai-act-classification-methodology.md + ai-act-provider-obligations.md (hvis AI Act-dybde i dimensjon 6)
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-classification-methodology.md + ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-provider-obligations.md (hvis AI Act-dybde i dimensjon 6)
|
||||
|
||||
Lever en [komplett ROS-rapport med alle 8 faser / Quick ROS med top-10 og trafikklys]."
|
||||
```
|
||||
|
|
|
|||
|
|
@ -34,13 +34,13 @@ Identifiser hvilke sikkerhetsdimensjoner som er mest kritiske for scenarioet:
|
|||
Bruk Task-verktøyet til å lansere `security-assessment-agent`:
|
||||
|
||||
```
|
||||
Task(general-purpose): "Les agents/security-assessment-agent.md og utfør en
|
||||
Task(ms-ai-architect:security-assessment-agent): "Utfør en
|
||||
sikkerhetsassessment for [plattform] brukt til [scenario].
|
||||
Kontekst: [offentlig sektor / privat / etc.]
|
||||
Vurder alle 6 dimensjoner med 1-5 score.
|
||||
Les også: skills/ms-ai-advisor/references/architecture/security.md
|
||||
og skills/ms-ai-advisor/references/architecture/public-sector-checklist.md
|
||||
og skills/ms-ai-security/references/ai-security-engineering/security-scoring-rubrics-6x5.md"
|
||||
Les også: ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/security.md
|
||||
og ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/public-sector-checklist.md
|
||||
og ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/ai-security-engineering/security-scoring-rubrics-6x5.md"
|
||||
```
|
||||
|
||||
### 4. Berik med arkitekturperspektiv
|
||||
|
|
|
|||
|
|
@ -33,7 +33,7 @@ Hvis ingen vurderinger er gjennomført, informer brukeren om at summary krever m
|
|||
Kjør summary-agenten via Task:
|
||||
|
||||
```
|
||||
Task(general-purpose): "Read agents/summary-agent.md for your role and instructions.
|
||||
Task(ms-ai-architect:summary-agent): "
|
||||
Generer teknisk sammendrag og executive summary for:
|
||||
|
||||
**Løsning:** [navn]
|
||||
|
|
|
|||
|
|
@ -27,7 +27,7 @@ Avklar:
|
|||
### 2. Deleger til AI Act-agent
|
||||
|
||||
```
|
||||
Task(ai-act-assessor): "Read agents/ai-act-assessor.md for your role and instructions.
|
||||
Task(ms-ai-architect:ai-act-assessor): "
|
||||
Generer transparensnotiser for følgende AI-system:
|
||||
|
||||
**System:** [systemnavn]
|
||||
|
|
@ -39,7 +39,7 @@ Generer transparensnotiser for følgende AI-system:
|
|||
Modus: Transparens — generer Art. 13/50 notiser.
|
||||
|
||||
Les kunnskapsbasene:
|
||||
- skills/ms-ai-governance/references/responsible-ai/ai-act-transparency-notices.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-transparency-notices.md
|
||||
|
||||
Lever:
|
||||
1. Art. 50(1) AI-interaksjonsnotis (norsk)
|
||||
|
|
|
|||
|
|
@ -23,9 +23,9 @@ Hvis kommandoen kjøres etter `/architect` (Fase 1-3), gjenbruk innsamlet kontek
|
|||
### 1. Last kontekst
|
||||
|
||||
Les malen som styrer utredningen:
|
||||
- `skills/ms-ai-advisor/references/architecture/ai-utredning-template.md`
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/ai-utredning-template.md`
|
||||
|
||||
Aktiver Cosmo Skyberg-personaen fra `skills/ms-ai-advisor/SKILL.md`.
|
||||
Aktiver Cosmo Skyberg-personaen fra `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/SKILL.md`.
|
||||
|
||||
### 2. Parse input og bestem kompleksitet
|
||||
|
||||
|
|
@ -115,7 +115,7 @@ Orkestratoren gjør alt selv. Ingen TeamCreate.
|
|||
4. Fullfør J, L→M, skriv til fil
|
||||
5. Kjør diagram-generation-agent for S8.2 (arkitekturoversikt):
|
||||
```
|
||||
Task(architect:diagram-generation-agent): "Generer arkitekturoversikt-diagram for {scenario}.
|
||||
Task(ms-ai-architect:diagram-generation-agent): "Generer arkitekturoversikt-diagram for {scenario}.
|
||||
Komponenter: {fra S8.1}. Skriv til {output_dir}/.work/diagrams/architecture-overview.md"
|
||||
```
|
||||
6. Kjør summary-agent (steg N) — les worker-mal nedenfor
|
||||
|
|
@ -206,15 +206,15 @@ Alle arbeidere spawnes med `Task` og skriver output til `.work/`-filer. Bruk `te
|
|||
|
||||
#### Security Worker
|
||||
```
|
||||
Task(architect:security-assessment-agent, name="security-worker", team_name="{team}"):
|
||||
Task(ms-ai-architect:security-assessment-agent, name="security-worker", team_name="{team}"):
|
||||
"Utfør sikkerhetsvurdering for: {scenario}
|
||||
Plattform: {plattform}
|
||||
Kontekst: {sektor/virksomhet fra Fase 1}
|
||||
|
||||
Les relevante KB-filer (max 3):
|
||||
- skills/ms-ai-security/references/ai-security-engineering/security-scoring-rubrics-6x5.md
|
||||
- skills/ms-ai-security/references/ai-security-engineering/ai-security-scoring-framework.md
|
||||
- skills/ms-ai-advisor/references/architecture/security.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/ai-security-engineering/security-scoring-rubrics-6x5.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/ai-security-engineering/ai-security-scoring-framework.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/security.md
|
||||
|
||||
VIKTIG: Skriv KOMPLETT output til {output_dir}/.work/security.md med Write-verktøyet.
|
||||
Inkluder: Score-matrise (6 dimensjoner), P0/P1-funn, anbefalinger."
|
||||
|
|
@ -222,14 +222,14 @@ Inkluder: Score-matrise (6 dimensjoner), P0/P1-funn, anbefalinger."
|
|||
|
||||
#### Cost Worker
|
||||
```
|
||||
Task(architect:cost-estimation-agent, name="cost-worker", team_name="{team}"):
|
||||
Task(ms-ai-architect:cost-estimation-agent, name="cost-worker", team_name="{team}"):
|
||||
"Estimer kostnader for: {scenario}
|
||||
Plattform: {plattform}, Brukere: {antall}, Volum: {volum}
|
||||
|
||||
Les relevante KB-filer (max 3):
|
||||
- skills/ms-ai-security/references/cost-optimization/deterministic-cost-calculation-model.md
|
||||
- skills/ms-ai-security/references/cost-optimization/azure-ai-foundry-cost-governance.md
|
||||
- skills/ms-ai-advisor/references/architecture/cost-models.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/cost-optimization/deterministic-cost-calculation-model.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-security/references/cost-optimization/azure-ai-foundry-cost-governance.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/cost-models.md
|
||||
|
||||
VIKTIG: Skriv KOMPLETT output til {output_dir}/.work/cost.md med Write-verktøyet.
|
||||
Inkluder: Månedskostnad, TCO 3 år, alle alternativer, konfidensgradering."
|
||||
|
|
@ -237,14 +237,14 @@ Inkluder: Månedskostnad, TCO 3 år, alle alternativer, konfidensgradering."
|
|||
|
||||
#### DPIA Worker
|
||||
```
|
||||
Task(architect:dpia-agent, name="dpia-worker", team_name="{team}"):
|
||||
Task(ms-ai-architect:dpia-agent, name="dpia-worker", team_name="{team}"):
|
||||
"Gjennomfør DPIA/PVK for: {scenario}
|
||||
Datatype: {datatype}, Behandlingsgrunnlag: {grunnlag}
|
||||
|
||||
Les relevante KB-filer (max 3):
|
||||
- skills/ms-ai-governance/references/norwegian-public-sector-governance/dpia-norwegian-methodology-ai.md
|
||||
- skills/ms-ai-governance/references/responsible-ai/gdpr-compliance-ai-systems.md
|
||||
- skills/ms-ai-governance/references/responsible-ai/ai-impact-assessment-framework.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/norwegian-public-sector-governance/dpia-norwegian-methodology-ai.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/gdpr-compliance-ai-systems.md
|
||||
- ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-impact-assessment-framework.md
|
||||
|
||||
VIKTIG: Skriv KOMPLETT output til {output_dir}/.work/dpia.md med Write-verktøyet.
|
||||
Inkluder: Risikomatrise, tiltakstabell, bias/forklarbarhet/HITL-vurdering."
|
||||
|
|
@ -252,7 +252,7 @@ Inkluder: Risikomatrise, tiltakstabell, bias/forklarbarhet/HITL-vurdering."
|
|||
|
||||
#### Diagram Worker
|
||||
```
|
||||
Task(architect:diagram-generation-agent, name="diagram-worker", team_name="{team}"):
|
||||
Task(ms-ai-architect:diagram-generation-agent, name="diagram-worker", team_name="{team}"):
|
||||
"Generer diagrammer for: {scenario}
|
||||
Komponenter: {fra S8.1}
|
||||
|
||||
|
|
@ -263,7 +263,7 @@ Diagrammer å generere:
|
|||
- Sikkerhetssoner (S5.1) — hvis sikkerhet er kritisk
|
||||
- Implementeringstidslinje (S9.1) — hvis faseplan er definert
|
||||
|
||||
Les: skills/ms-ai-advisor/references/architecture/diagram-prompt-templates.md
|
||||
Les: ${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/diagram-prompt-templates.md
|
||||
|
||||
VIKTIG: Skriv output til {output_dir}/.work/diagrams/ (én fil per diagram).
|
||||
Hvis mcp-image er utilgjengelig: generer Mermaid-syntaks som fallback."
|
||||
|
|
@ -271,7 +271,7 @@ Hvis mcp-image er utilgjengelig: generer Mermaid-syntaks som fallback."
|
|||
|
||||
#### Summary Worker (kjøres ALLTID som siste agent)
|
||||
```
|
||||
Task(architect:summary-agent, name="summary-worker"):
|
||||
Task(ms-ai-architect:summary-agent, name="summary-worker"):
|
||||
"Generer sammendrag for utredningen.
|
||||
|
||||
Les utredningen: {output_dir}/utredning.md
|
||||
|
|
|
|||
|
|
@ -35,9 +35,9 @@ Avklar hvis ikke kjent:
|
|||
|
||||
### 3. Les kunnskapsbasene
|
||||
|
||||
- `skills/ms-ai-governance/references/monitoring-observability/data-residency-audit-monitoring.md` — Schrems II, EDPB seks-stegs-TIA, CLOUD Act/FISA 702-restanalyse for tredjelandsoverføring
|
||||
- `skills/ms-ai-governance/references/responsible-ai/ai-act-deployer-obligations.md` — deployerforpliktelser hvis leverandøren leverer et AI-system
|
||||
- `skills/ms-ai-advisor/references/architecture/security.md` — sikkerhetskrav til eksterne tjenester
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/monitoring-observability/data-residency-audit-monitoring.md` — Schrems II, EDPB seks-stegs-TIA, CLOUD Act/FISA 702-restanalyse for tredjelandsoverføring
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-governance/references/responsible-ai/ai-act-deployer-obligations.md` — deployerforpliktelser hvis leverandøren leverer et AI-system
|
||||
- `${CLAUDE_PLUGIN_ROOT}/skills/ms-ai-advisor/references/architecture/security.md` — sikkerhetskrav til eksterne tjenester
|
||||
|
||||
For personvern-/cross-border-dybde: deleger til `/architect:dpia` (full TIA). For anskaffelseskrav: `/architect:anskaffelse`.
|
||||
|
||||
|
|
|
|||
|
|
@ -38,7 +38,8 @@ Fase 1a A+B (sitemap-prefiks + skjemaløs URL-ekstraksjon) er levert (`e74646d`)
|
|||
|
||||
Fase 1b (innholdsfortegnelse i de **20 filene >800 linjer**) ble opprinnelig skopet som «trygt håndverk, uavhengig av Cosmo». Ground truth motbeviste premisset: **11 av 12 ikke-advisor-storfiler** (og alle advisor-storfilene) har en `## …Cosmo…`-seksjon på nivå 2. En TOC bygget fra `##`-overskriftene ville da emittere en **ny Cosmo-anker** (`- [For arkitekten (Cosmo)](#…)`) → bryter «aldri ny Cosmo-innhold», og å nøytralisere overskriften er nettopp denne persona-fjerningen (ikke en sidehandling). Store-fil-TOC folder derfor inn her.
|
||||
|
||||
- **Når du nøytraliserer `## …Cosmo…`-overskriften i en storfil:** kjør `node scripts/kb-update/backfill-toc.mjs --write <fil>` i samme diff. Den setter `## Innhold` rett før første ikke-fence `## `-seksjon (samme slot som `composeKbFile`), er idempotent og fence-aware, og lister da de **nøytraliserte** overskriftene. Dry-run uten `--write` for å se diff først.
|
||||
- **Når du nøytraliserer `## …Cosmo…`-overskriften i en storfil (R13/R14):** bygg `## Innhold` på nytt via `transform.insertToc` (primitiven overlever; det tidligere TOC-backfill-CLI-scriptet er **retired i R6 Step 6** — ikke-atomisk `writeFileSync`, 0 programmatiske kallere, superseded av `insertToc` i `transform.mjs` / `migrate-corpus`). NB: `migrate-corpus` fencer ut advisor OG mangler single-fil-modus, så den er **ikke** en drop-in for den advisor-rettede cosmo-use-casen — R13/R14 håndterer TOC via et advisor-kapabelt single-fil-steg, ELLER heading-fjerningen løser det selv.
|
||||
- **§8-residual (gap-discipline, [[gap-discipline-must-close]]):** `insertToc` er en no-op på en fil som allerede har `## Innhold`, så en heading-nøytralisering som endrer en **oppført** overskrift etterlater en stale TOC-entry (latent link-rot). Lukkes i R13/R14 ved å regenerere TOC etter nøytralisering (eller nøytralisere før TOC-en bygges).
|
||||
- **Mekanismen** er `transform.insertToc` (testet, `tests/kb-update/test-transform.test.mjs`). Nye filer fødes allerede med TOC via `composeKbFile` (Fase 1c, `2240f1e`).
|
||||
- **Verifisering:** `node scripts/kb-eval/eval.mjs` → `checkN4 hasToc` = true for de berørte filene; ingen diff-churn på små filer (<100 linjer røres ikke).
|
||||
- Eneste storfil uten `##`-Cosmo-overskrift: `zero-trust-ai-services.md` (eneste Cosmo-token er inline `**For Cosmo:**`) — kan TOC-es uten å skape Cosmo-anker, men tas naturlig i samme pass.
|
||||
|
|
|
|||
|
|
@ -147,6 +147,37 @@ at two existing chokepoints so it runs regardless of any hook:
|
|||
quarantine, operator adjudicates, never auto-committed. No new mechanism — this
|
||||
reuses "never auto-fix KB — flag → human → fix."
|
||||
|
||||
## 5b. Layer A-aktiveringsprotokoll (R7-gate)
|
||||
|
||||
R7 (og enhver senere fetch-økt) starter **ALDRI** før denne sjekklisten passerer. Den
|
||||
operasjonaliserer Layer A som en hard R7-gate, ikke bare en aktiveringsregel:
|
||||
|
||||
1. **Aktiver hooken.** Slå på `llm-security`s `post-mcp-verify` i `~/.claude/settings.json`
|
||||
(PostToolUse på `microsoft_docs_fetch`). Dette er en konfig-endring i operatørens
|
||||
`settings.json`, ikke i plugin-repoet.
|
||||
2. **Verifiser at den fyrer.** Kjør ÉN live **foreground** `microsoft_docs_fetch` og bekreft
|
||||
at `post-mcp-verify` faktisk fyrte (observer hook-output). En hook som ikke observeres fyre
|
||||
teller som IKKE aktiv.
|
||||
3. **Kompenserende skann hvis den ikke fyrer.** Fyrer den ikke (headless / GH #36071), kjøres
|
||||
den kompenserende deterministiske skannen **foreground** på den komponerte artefakten FØR
|
||||
write: Layer B (`scan-adversarial-content.mjs`) + Step 3s pre-write unicode/carrier-scan
|
||||
(`detectAdversarial` uten `path` → temp-fil, R6 Step 3). Denne er alltid-på og autoritativ
|
||||
(§5 Layer B); Layer A er kun tidlig-varsling.
|
||||
4. **Frossen judge → foreground er obligatorisk.** Det fetchede innholdet **prompt-fences IKKE**
|
||||
inn i judgen: v3.1-judgen er frosset (G1-adoptert), og å legge til fencing ville være en
|
||||
judge-bump (Non-goal §7, krever re-måling av P/R). Uten judge-side fencing er **foreground
|
||||
fetch obligatorisk** — det er det som lar Layer A-hooken (og operatørens øye) se innholdet
|
||||
før judgen. (§5s fencing-ambisjon gjelder transform-steget, ikke den frosne judgen.)
|
||||
5. **Bekreft Layer B-substratet før R7.** `llm-security`-sibling MÅ være til stede (ev. via
|
||||
`LLM_SECURITY_ROOT`) — ellers fail-closer Layer B og **BLOCKer alt** (gaten blir en vegg).
|
||||
Bekreft at `../llm-security/scanners/…` løser før første fetch.
|
||||
|
||||
**R7 starter KUN når 1–5 er grønne.**
|
||||
|
||||
**Rollback:** aktiveringen i pkt. 1 er reverserbar — fjern `post-mcp-verify`-oppføringen fra
|
||||
`~/.claude/settings.json` (og ev. `LLM_SECURITY_ROOT`) for å ta ned Layer A igjen. Layer B
|
||||
(deterministisk, in-repo) er upåvirket av denne rollback-en.
|
||||
|
||||
## 6. No local solutions — the deterministic scanner is a shared asset
|
||||
|
||||
Per house policy (*ingen lokale løsninger*, *showcase reusable patterns*): the
|
||||
|
|
@ -162,6 +193,13 @@ Layer B scanner is **not** a bespoke ms-ai-architect script. Two horizons:
|
|||
implementation and **one** lexicon dataset, so the pattern table does not drift
|
||||
into two copies. This brief is the second consumer that justifies extracting it.
|
||||
|
||||
**Coordination (2026-07-05).** The durable cross-repo mechanism is **hub-and-spoke**:
|
||||
the canonical convergence contract lives in the hub (`llm-ingestion-pipeline-security`,
|
||||
the designated shared asset); ms-ai-architect's spoke pointer is roadmap **R19**. See the
|
||||
outgoing brief delivered to the hub repo (2026-07-05) for the proposed contract, the
|
||||
confirmed language data points (ms-ai-architect Layer B = Node), and the open
|
||||
`claude-code-llm-wiki` language brick that locks the polyglot-vs-Node-primary decision.
|
||||
|
||||
## 7. Non-goals
|
||||
|
||||
- **Not** a query-time guardrail (downstream sessions can layer their own).
|
||||
|
|
|
|||
|
|
@ -10,6 +10,18 @@ Konvensjonen er **ikke lenger definert lokalt** — den bor i `catalog/docs/okf-
|
|||
2. **Ingen gjenbrukbar OKF-*ingest*-kode finnes** — `reference_agent` er BigQuery+Gemini/GCP-bundet. Adopter *prompt-mønstrene*, ikke koden. Classify/convert = bygg-selv.
|
||||
3. **Kanonisk anbefalt feltnavn er `resource`** (ikke `source`).
|
||||
|
||||
## Implementasjonskilde: delt bibliotek, ikke lokal bygging (2026-07-19, verifisert)
|
||||
Tooling-en under «Hva som må bygges» **bygges ikke lokalt her**. Den kommer fra det delte biblioteket `~/repos/llm-ingestion-okf` (offentlig Forgejo `open/`), som syv repo skal konsumere — så standard-forbedringer arver alle, i stedet for at hver plugin drifter sin egen kopi.
|
||||
|
||||
**Tre-delt eierskap (hold dem fra hverandre):**
|
||||
- **Konvensjon** → `catalog/docs/okf-second-brain/spec.md` (uendret, se over).
|
||||
- **Implementasjon** → `llm-ingestion-okf` fase 4 = `node/`-halvdelen: zero-dep ESM, importerbar + CLI-invokerbar, **vendret per plugin** (ikke npm). Python- og Node-halvdelen deler kontrakt + fixtures, aldri kode.
|
||||
- **Sikkerhet** → alltid `llm-ingestion-guard`. Fase 4 leverer kun hook-punktet ved persist (guard-as-contract); scan-/sanitize-logikk hører ALDRI hjemme her. Vår `docs/ingestion-security-brief-2026-07.md`-ambisjon om Node Layer-B-scan er guard-territorium.
|
||||
|
||||
**Vår rolle er greenfield-konsument, ikke migrering.** `llm-ingestion-okf/docs/plan/phase-4-node-half.md` koordineringspunkt 5: pluginens designet-men-ubygde behov (writer, indeksgenerator, checker, retrieval-støtte) er **akseptanse-skissen for pakkens API-flate**. Vi har null OKF-kode i dag — det er en fordel her, ikke en gjeld.
|
||||
|
||||
**Tidshorisont og gate (per 2026-07-19):** `node/` finnes ikke ennå; biblioteket er ren Python (v0.3.0, fase 1 levert). Fase 4 ← fase 3 ← fase 2, og fase 2 er **blokkert på avklaring B2** (guard-distribusjonskanal for CI — operatørbeslutning, `phase-2-doors-b-c.md:96`). Fase 4 starter dessuten med sign-off-kjede, ikke kode, der vi er punkt 5 av 5: *«okr/ms-ai-architect adopterer i egne repo (egne sesjoner, eksplisitt instruks per repo)»*. **Konsekvens: ikke start S-OKF-implementasjon før B2 er avklart og fase 4 har levert `node/`.**
|
||||
|
||||
## Scope-grense + ambisjon (ufravikelig, operatør-bekreftet 2026-06-26)
|
||||
- **OKF gjelder KUN brukerens egen kontekst/data** — hans person- og organisasjonsspesifikke «second brain» / LLM-wiki (i dag onboarding-output i `~/.claude/ms-ai-architect/org/*.md`).
|
||||
- **For dette sporet kjøres FULL Google OKF-pakke** (ikke bare lån av mønstre): frontmatter-kontrakt + `index.md`-progressiv-disclosure + retrieval + vedlikeholds-/enrichment-mekanisme, modellert på `samples/` + `toolbox/`. **Og adopsjonen holdes oppdatert etter hvert som OKF-standarden utvikler seg** (v0.1 → senere versjoner; fang `okf_version`-bump). Operatør-direktiv 2026-06-26.
|
||||
|
|
@ -26,7 +38,22 @@ Avgjørende skille er **ikke** «er det en LLM-wiki» (begge er det) — men **
|
|||
- Implikasjon: «second brain» trenger en **retrieval-skill** (list → search → read) som er aktiv uavhengig av kommandoer, + en **vedlikeholds-mekanisme** som holder wikien oppdatert etter hvert som brukeren tilfører kontekst OG etter hvert som OKF-standarden utvikler seg.
|
||||
|
||||
## OKF v0.1 — kjernekontrakt (verifisert mot spec)
|
||||
- Bundle = katalogtre av markdown-filer, ett konsept per fil. Påkrevd frontmatter-felt: `type`. Anbefalt: `title`, `description`, `resource` (kilde-URI), `tags`, `timestamp`. Konsumenter MÅ bevare ukjente felt.
|
||||
|
||||
> ⚠️ **v0.2 ER UTE (Google, 2026-07-25) — denne seksjonen beskriver v0.1-formen.**
|
||||
> Verifisert 2026-08-03 mot `okf/SPEC.md` §13.1 (ikke mot coord-melding): to felt er
|
||||
> retirert, begge med fallback, så en v0.1-formet bundle er fortsatt KONFORM:
|
||||
> - **`timestamp` er avløst av `generated.at`** — siste innholdsendring føres som
|
||||
> `generated: { by, at }`. «Consumers MAY fall back to a legacy `timestamp` when
|
||||
> `generated` is absent.»
|
||||
> - **Body-seksjonen `# Citations` er avløst av `sources`** — proveniens flyttet til
|
||||
> frontmatter. «Consumers SHOULD read `sources` and MAY still parse a legacy
|
||||
> `# Citations` body list for v0.1 documents.»
|
||||
>
|
||||
> **Vi er greenfield: de 6 filslotene finnes ikke på disk, så vi sikter v0.2 fra
|
||||
> første byte** og arver ingen migreringsbyrde. Der `timestamp` og `# Citations`
|
||||
> nevnes nedenfor, les dem som v0.1-form som IKKE skal kopieres inn i nytt arbeid.
|
||||
|
||||
- Bundle = katalogtre av markdown-filer, ett konsept per fil. Påkrevd frontmatter-felt: `type`. Anbefalt: `title`, `description`, `resource` (kilde-URI), `tags`, `timestamp` (**v0.2: `generated.at`**). Konsumenter MÅ bevare ukjente felt.
|
||||
- Reserverte filnavn: `index.md` (katalog-enumerasjon, ingen frontmatter, progressiv disclosure), `log.md` (endringslogg).
|
||||
- Kryss-lenking: bundle-relativ (`/...`) eller relativ markdown; relasjonstype utledes av prosa. Konsumenter må tolerere brutte lenker.
|
||||
- v0.1 (12. juni 2026), «starting point, not finished standard» → hold OKF-adopsjonen oppdatert ettersom standarden bumpes (`okf_version` i rot-`index.md`).
|
||||
|
|
@ -41,7 +68,7 @@ Repo: `GoogleCloudPlatform/knowledge-catalog`. Mye er GCP/Dataplex/Gemini-bundet
|
|||
- `samples/discovery/*` — Dataplex-bundet søke-agent (Google ADK + CatalogServiceClient). IKKE gjenbrukbart; kun mønster-referanse.
|
||||
|
||||
**Struktur / frontmatter-eksempler:**
|
||||
- `okf/bundles/ga4/index.md` + `okf/bundles/ga4/references/metrics/avg_pageviews.md` — ekte `index.md`-hierarki + leaf-konsept med frontmatter (`type`/`resource`/`title`/`description`/`tags`/`timestamp`) + `# Citations`-seksjon. Kopier som mal.
|
||||
- `okf/bundles/ga4/index.md` + `okf/bundles/ga4/references/metrics/avg_pageviews.md` — ekte `index.md`-hierarki + leaf-konsept med frontmatter (`type`/`resource`/`title`/`description`/`tags`/`timestamp`) + `# Citations`-seksjon. ⚠️ **IKKE kopier ordrett — dette upstream-eksempelet er v0.1-formet på nøyaktig de to aksene v0.2 retirerte** (`timestamp` → `generated.at`; `# Citations` → `sources` i frontmatter). Kopier hierarkiet og felt-disiplinen, ikke de to feltene.
|
||||
|
||||
**Produksjon / vedlikehold (ADOPTER MØNSTERET, ikke koden):**
|
||||
- `okf/src/reference_agent/prompts/web_ingestion_instruction.md` — kilde-drevet oppdatering: `list_concepts → fetch → enrich/mint/skip` med strenge frontmatter-/heading-bevaringsregler. Moden mal for «hold wikien oppdatert fra kilder».
|
||||
|
|
@ -50,7 +77,9 @@ Repo: `GoogleCloudPlatform/knowledge-catalog`. Mye er GCP/Dataplex/Gemini-bundet
|
|||
- ~~`toolbox/mdcode/...`~~ — **DØDT SPOR (spec §9.1):** `mdcode`/`kcmd` er IKKE et OKF-verktøy, men et Dataplex git-sync-verktøy med et annet frontmatter-schema (`id`/`resource.name`/`createTime`/`links`). Ikke planlegg OKF-emit/sync via det.
|
||||
|
||||
## Hva som må bygges (fremtidige faser — ikke nå)
|
||||
1. **Frontmatter-/struktur-konvensjon** for second brain: bygg mot delt spec §3 (`type` påkrevd + `index.md` per nivå + `okf_version` i rot). Anbefalte felt (spec §4): **`resource`** (kanonisk kilde-URI — IKKE `source`), `title`, `description`, `tags`, `timestamp`. Migrer dagens `org/*.md` (de mangler bare `type:` — nær OKF allerede).
|
||||
> **Leses sammen med «Implementasjonskilde» over:** punkt 1–3 beskriver *behovet*, ikke et byggeoppdrag her. Tooling-en leveres av `llm-ingestion-okf` fase 4 (writer/inbox-primitiver, `okf index`, `okf check`, retrieval-støtte). Vår jobb ved adopsjon er konfigurasjon + vendring + å prøve API-flaten mot behovene under — ikke implementasjon.
|
||||
|
||||
1. **Frontmatter-/struktur-konvensjon** for second brain: bygg mot delt spec §3 (`type` påkrevd + `index.md` per nivå + `okf_version` i rot — **sett den til `"0.2"`**). Anbefalte felt (spec §4): **`resource`** (kanonisk kilde-URI — IKKE `source`), `title`, `description`, `tags`, **`generated: { by, at }`** (v0.2; erstatter `timestamp`) og **`sources`** i frontmatter (v0.2; erstatter body-seksjonen `# Citations`). Migrer dagens `org/*.md` (de mangler bare `type:` — nær OKF allerede).
|
||||
2. **Retrieval-skill** («second-brain-search» e.l.): list → search → read over `~/.claude/ms-ai-architect/org/`, aktiv i fri chat (ikke bare kommandoer). Avgjør: ren SKILL+Grep/Glob/Read vs. dedikert MCP-server (build-both-and-measure-kandidat — se under).
|
||||
3. **Vedlikeholds-mekanisme**: hvordan wikien oppdateres når brukeren tilfører kontekst (onboarding-agent skriver OKF-konform), + en oppdaterings-rutine når OKF-standarden bumpes.
|
||||
|
||||
|
|
@ -62,6 +91,74 @@ Repo: `GoogleCloudPlatform/knowledge-catalog`. Mye er GCP/Dataplex/Gemini-bundet
|
|||
## Suksesskriterium (per operatør 2026-06-26)
|
||||
Det viktigste er **ikke teknologien**, men at second brain blir **så bra som mulig for brukeren** og at **oppdateringsmekanismene fungerer veldig bra**. Mål mot brukerverdi (henter pluginen riktig personlig/org-kontekst i chat?) + vedlikeholds-pålitelighet — ikke mot formell OKF-konformitet i seg selv.
|
||||
|
||||
## Fase-4-kartlegging: hva vi faktisk trenger fra biblioteket (2026-07-20, verifisert)
|
||||
|
||||
Svar på koordineringspunkt 5 (vi = akseptanse-skisse for Node-API-flaten). Markørlinje satt i `STATE.md`: **`planned`**.
|
||||
|
||||
**Ground truth denne kartleggingen hviler på (verifisert, ikke antatt):**
|
||||
- `~/repos/llm-ingestion-okf`: v0.3.1, `node/` finnes **ikke** (kun `src/` = Python). `docs/plan/phase-4-node-half.md:12,49-62` planlegger `okf check|index|inbox|convert` — altså writer/index/checker uten connector-krav.
|
||||
- `~/.claude/ms-ai-architect/org/` **finnes ikke på denne maskinen** — second brain er designet (6 filslots konsumert av `research-agent`, `adr-writer-agent`, `architecture-review-agent` m.fl.) men aldri materialisert. Vi er greenfield i bokstavelig forstand.
|
||||
- `scripts/kb-update/` er **ikke** en OKF-flate (se under).
|
||||
|
||||
### kb-update-pipelinen: generisk mekanikk vs. MS Learn-domenelogikk
|
||||
Deteksjonslaget er 100 % deterministisk node (null modellkall, strukturelt garantert i `run-detection.mjs:6-9`); apply-laget er LLM/MCP og alltid manuelt in-session.
|
||||
|
||||
| Generisk ingestion-mekanikk | MS Learn-domenelogikk |
|
||||
|---|---|
|
||||
| `lib/atomic-write.mjs` (tmp+rename), `lib/backup.mjs` (scoped restore + sentinel) | `lib/sitemap-stream.mjs` (hardkodet `learn.microsoft.com/_sitemaps/`), `lib/url-normalize.mjs` (locale-stripping) |
|
||||
| `lib/registry-io.mjs`, `lib/decisions-io.mjs` (manifest + ledger/idempotens) | `data/domain-taxonomy.json` (sitemap-prefikser → category → skill-routing) |
|
||||
| `lib/verified-staleness.mjs`, `lib/full-pass-worklist.mjs` (kilde-timestamp vs. verifisert-timestamp) | `lib/kb-headers.mjs` + `lib/transform.mjs` (bold-label-header-kontrakten), `lib/verify-out.mjs` (GA/preview/pris-regexer), `lib/learn-api.mjs` |
|
||||
| Invarianten «pure lib skriver aldri; kun gated caller skriver» | `transform-prompt.md` + `kb-eval/judge-claim-prompt-v3.1.md` (claim/judge-format) |
|
||||
|
||||
Den generiske kolonnen overlapper reelt med det fase 4 skal levere (atomisk write, manifest, ledger, ferskhet). **Men den flytter vi ikke** — se «Aldri hit» under.
|
||||
|
||||
### Vår adopsjonsflate = second brain, og den trenger ikke dør A
|
||||
Second brain-filene fødes av et **onboarding-intervju** (LLM → `Write`), ikke av en kildekonnektor. Det finnes ingen manifest, ingen CSV, ingen SQL, ingen URL å hente. Dør A (manifest → connector → materialisering) er derfor **irrelevant for oss** — vi trenger den *bakre halvdelen* av dør A, frikoblet fra connector-halvdelen.
|
||||
|
||||
**Konkret krav til Node-API-flaten (punkt (b) i rapporteringen):**
|
||||
1. **Writer uten connector.** `writeConcept({path, frontmatter, body})` må være kallbar direkte med LLM-forfattet innhold — ikke bak et manifest/connector-krav. Deterministisk frontmatter-emit (`type` påkrevd, `resource`/`title`/`description`/`tags`, **`generated: { by, at }` — `generated.by` er PÅKREVD når `generated` finnes — og `verified` som liste**; v0.2-formen, ikke `timestamp`), ukjente felt bevart, atomisk skriv, idempotent re-kjøring.
|
||||
2. **`okf index` som bibliotekfunksjon, ikke bare CLI.** Vi må regenerere `index.md` per nivå fra en in-process hook (onboarding-agent skriver én fil → indeks oppdateres i samme operasjon), uten å shelle ut.
|
||||
3. **`okf check` med maskinlesbar exit + funn-struktur.** Vi har allerede create-guard-mønsteret (`validate-kb-file.mjs`, exit≠0 = stopp) og vil speile det for OKF-flaten.
|
||||
4. **Guard-hook ved persist, ikke i biblioteket.** `free-context.md` er den ene slotten der brukeren limer inn vilkårlig eksternt innhold — der må `llm-ingestion-guard` kalles på kallstedet. Vi leser IKKE «security is delegated» som «trygt by default» (jf. v0.3.1-presiseringen).
|
||||
5. ~~**Fritekst-kravet (F1) gjelder oss, men via dør B — ikke dør A.**~~ **Trukket 2026-07-20 (trinn D-review).** Premisset var at dørene er monolitter; når de blir komposisjoner over offentlige primitiver, forsvinner det. Vi har ingen preferanse for hvordan F1 løses, så lenge writeren ikke låses bak connector-laget (krav 1).
|
||||
6. **Vendring, ikke npm** — zero-dep ESM som kan sjekkes inn under `scripts/`, konsistent med at pluginen distribueres offentlig og må kjøre uten install-steg.
|
||||
7. **Bundlen bærer ingen fullstendighetsegenskap** (trinn D-review 2026-07-20, **skjerpet trinn E 2026-07-20**). Onboarding er resumbar på tvers av økter — `agents/onboarding-agent.md:29-31` gjenopptar fra første fil uten `completed: true`, og sesjonshooken rapporterer delvis tilstand som gyldig (`hooks/scripts/session-start-context.mjs:160-161`). `check` må derfor skille **ufullstendig-men-gyldig** fra **ugyldig**, og `index` må være kallbar etter hver enkeltfil-skriving uten å feile på at resten mangler. Speiler vi `check` inn i create-guard-mønsteret (krav 3) og den feiler midt i et intervju, blokkerer vi vår egen onboarding.
|
||||
- **Korreksjon (trinn E):** trinn D-formuleringen sa at hooken rapporterer «Onboarding 3/6». Den rapporterer `X/5` — `session-start-context.mjs:160-161` gater på `ORG_FILES.length`, og `scripts/kb-update/lib/user-data.mjs:27-33` definerer `ORG_FILES` som fem filer. `free-context.md` er **bevisst utenfor tellingen** (`user-data.mjs:35-38`). En fullført bundle har derfor **enten 5 eller 6 filer**, og ingenting i bundlen skiller de to. Riktig krav er ikke «delvis er også gyldig», men: fullstendighet er en egenskap ved **kalleren**, ikke ved bundlen. Biblioteket må aldri utlede et forventet entry-sett fra annet enn filene som faktisk finnes.
|
||||
8. **Skriveren må avvise stempelfeltene** (nytt, trinn E 2026-07-20 — svar på bibliotekets D1). `write_concept` skal ta frontmatter verbatim **med nøyaktig ett unntak**: `generated` og `ingest_manifest` MÅ avvises i innlevert frontmatter og kun kunne emitteres via ingest-stien. Begrunnelse: de to feltene *er* §7-stempelet, og `ingest-spec.md` §3 nøkler en unwaivable slette-regel på dem. Gjøres §5s nøkkelliste til et obligatorisk prefiks for konsepter som sådan, må våre kuraterte bruker-eide onboarding-filer bære stempelet som gjør dem slettbare ved re-materialisering — vi ville måttet **forfalske stempelet for å være compliant**. Vår frontmatter (`category`/`completed`/`last_updated`, `onboarding-agent.md:54-58`, `:147-151`) har null overlapp med §5-listen, ikke engang `type`.
|
||||
|
||||
**Kollisjon med `portfolio-optimiser`s P1 — løsning foreslått (trinn E 2026-07-20):** P1 krever at `write_index(mode="regenerate")` **feiler høyt** på en uidentifisert linje; krav 7 krever at den **ikke feiler** på manglende konsepter. Predikatene kvantifiserer over disjunkte inndata (P1 leser indeksfila, krav 7 leser bundle-katalogen), og løses av pinningsregelen i krav 7-korreksjonen. Hos oss blir P1s predikat aldri sant — vi skriver aldri en indekslinje før fila finnes. **Tvinges valget, viker vi:** vårt behov kan dekkes på kallstedet (utsett indeksskriving), et stille tapt linje kan per definisjon ikke oppdages av kalleren. Vi ber i stedet om separate funn-koder + én negativ fixture per kode. Merk at P1 kun gjelder `write_index`; på `check_bundle` finnes ingen kollisjon.
|
||||
|
||||
**Index-semantikk (avklart trinn D-review):** vi legger ikke til et fjerde uforenlig krav. Ingen konsument leser en indeks — alle 12 agenter leser navngitte filer direkte via absolutt sti. Indeksen er for oss en kontrakts-/oppdagelsesforpliktelse (markøren `okf_version`), ikke en retrieval-mekanisme; bibliotekets default holder.
|
||||
|
||||
### Aldri hit (punkt (c))
|
||||
- **De 389 skill-reference-filene.** Stående operatør-direktiv (§ «Scope-grense»); native Claude Code-mekanisme dekker allerede OKFs list/search/read.
|
||||
- **Hele MS Learn KB-refresh-pipelinen.** Header-kontrakt, judge/claim-format, GA/preview-semantikk og Layer A/B-plassering er domenelogikk med egen testsuite — en delt ingestion-standard ville verken forstå eller forbedre den.
|
||||
- **All sikkerhetslogikk.** Eies av `llm-ingestion-guard` (vårt repo er `guard: active`, Layer A+B live). Ambisjonen i `docs/ingestion-security-brief-2026-07.md` om Node Layer-B-scan er guard-territorium, ikke OKF-territorium.
|
||||
|
||||
### Bundle-plassering: vi er allerede compliant (operatørbeslutning 2026-07-20)
|
||||
Beslutningen «bruker-initierte OKF-bundles bor utenfor repoet» krever **null endring** hos oss. Second brain har aldri bodd i plugin-treet: `agents/onboarding-agent.md:23` instruerer absolutt bruker-eid sti og eksplisitt «aldri plugin-roten». KB-referansene (389 filer) er plugin-eide og distribueres med pluginen — de blir.
|
||||
|
||||
**To funn meldt tilbake til biblioteket:**
|
||||
1. **Taksonomi-hull.** Beslutningstabellen sier at pluginens rolle for bruker-eide bundles er «leser via referanse». For oss er det feil — onboarding **skriver** til den bruker-eide katalogen. Klassen «bruker-eid katalog, plugin-skrevet innhold» har ingen boks.
|
||||
2. **Gate-tap (sikkerhet).** Vår Layer B commit-gate filtrerer til kun staged `skills/**/*.md` (`pre-commit-scan.mjs:11,34`). Innhold i `~/.claude/ms-ai-architect/org/` passerer aldri en commit og ligger utenfor ethvert git-tre. For bruker-eide bundles er guard-kallet ved persist derfor **det eneste laget**, ikke et ekstra — særlig for `free-context.md`. Krav 4 over er dermed bærende, ikke «nice to have».
|
||||
|
||||
### KB-korpuset er bevisst aldri-OKF (avklart 2026-07-20)
|
||||
Biblioteket leste «S-OKF åpent punkt» som at korpus-spørsmålet var ubesluttet. Det er det ikke — direktivet fra 2026-06-26 står; det åpne punktet gjelder second-brain-bygget. Presisering verdt å bære: **passformen er bedre enn tidligere formulert** (`VALID_TYPES` i `classify-ref-type.mjs:14` er allerede et lukket type-vokabular; `Source`/`Last updated` mapper til `resource`/`timestamp`). Det som blokkerer er kost/nytte — bold-label-headeren er bærende for judge-/stemplingsmaskineriet, og retrieval-gevinsten over native skills er null. Vi ville dessuten aldri brukt dør A-s `http`-connector: innholds-fetch går via den offisielle `microsoft-learn` MCP-serveren, rå HTTP kun mot sitemaps (metadata).
|
||||
|
||||
### Gate før implementasjon (uendret)
|
||||
Fase 4 ← fase 3 ← fase 2, og fase 2 er blokkert på **avklaring B2** (guard-distribusjonskanal for CI, operatørbeslutning). Ingen S-OKF-kode skrives her før `node/` er levert.
|
||||
|
||||
### Rundeutfall: konsensus ratifisert, runden avsluttet (2026-07-21)
|
||||
Kunngjøring fra koordineringsdriveren (`llm-ingestion-okf`): de fire beslutningsaksene er avgjort og medunderskrevet; runden er avsluttet og ikke lenger åpen for innspill. Beslutningsrecord: `llm-ingestion-okf/docs/beslutninger-okf-runden.local.md` §10 + `docs/plan/2026-07-21-trinn-f-konsensus-arkitektur.local.md`. **ms-ai-architect er ikke tildelt noe utførelsessteg** (konsument-stegene navngir `okr` og `portfolio-optimiser`, ikke oss); våre posisjoner konvergerte.
|
||||
|
||||
**D1 landet i vår favør.** Krav 8-innvendingen bortfaller: §5-prefikset (de sju stempelnøklene) er obligatorisk **kun på stemplede dør-A-filer, aldri på konsepter som sådan** — våre kuraterte onboarding-filer tvinges dermed aldri til å bære stempelet. Håndhevingen er flyttet til `check_bundle` (per-fil-utfall) = nøyaktig krav 7-mekanismen. Writer-avvisningen er skjerpet til en **dør C-garanti**: skriveren avviser det **komplette eierskapsstempelet** (`generated:true` **og** `ingest_manifest` sammen), ikke de enkelte navnene — round-trip for legitim dør C bevart. **Immaterielt for oss** (vår frontmatter `category`/`completed`/`last_updated` bærer ingen av feltene). Merk: krav 8-formuleringen «avvis begge navnene verbatim» er **superseded** av den ratifiserte «avvis kombinasjonen». `type` er ikke lenger et påkrevd felt pålagt oss av prefikset — det forblir vårt eget valg for våre filer (krav 1). Restrisiko (uforfalskbar mot uhell, ikke mot vilje): `ingest-spec.md:69-72`.
|
||||
|
||||
**D3** (`partial`/bevisst-terminal-skillet) tatt inn som innspill; commons avgjør endelig vokab-opptak. Vår sannsynlige sluttilstand (second brain gjennom biblioteket, 389-korpuset bevisst utenfor) er nettopp et bevisst-terminalt `partial`.
|
||||
|
||||
**D2-friksjonen ble tatt opp av runden som ÅS#5 (2026-07-21) — interim avgjort, endelig form delegert.** Ratifisert bærer for koordineringsregisteret = en **committet, ikke-gitignored fil ved siden av STATE.md** hos hvert repo (fordi STATE.md er gitignored og ikke fjern-hentbar), og tverrsnittet **genereres** fra de ni per-repo-filene og «må kunne hente oppføringen». For vårt **offentlig-speil-repo** (`open/ms-ai-architect`, distribuert til hele Norge) betyr «committet + fjern-hentbar» = **publisert offentlig** — nøyaktig den flaten vi i trinn E §4 sa vi ikke lager (markøren navngir søsken-repo + adopsjonsstatus = intern koordineringsmetadata). Vårt funn er **formelt anerkjent som en klassefeil**, ikke en enkeltsak: treffer minst `ms-ai-architect`, `llm-ingestion-okf` og `catalog`. Eierskap for endelig løsning: **commons + catalog**, avgjøres ved register-build, med tre kandidatformer — (i) generator tolererer fraværende bærer, (ii) minimal redigert offentlig bærer (status + repo-navn, uten søsken/arkitektur), eller (iii) privat sidekanal. **Interim ratifisert (og allerede vår tilstand):** hold markøren **LOCAL-ONLY** og aksepter eksklusjon fra tverrsnittet. Reåpner ikke D1–D4. **Ingen gjenstående operatørbeslutning hos oss** — vi reagerer når commons+catalog velger form; vi er uansett gated på B2 med umaterialisert second brain.
|
||||
|
||||
**Postkasse-teardown:** `~/repos/_okf-interim/` (transport) slettes. Vår svarfil `svar/ms-ai-architect.md` var transport — alt varig innhold ligger her i briefen (verifisert: krav 1–8 + P1-løsning + trinn E-korreksjoner). **Intet datatap for oss.**
|
||||
|
||||
## Referanser
|
||||
- **Delt konvensjon (KILDEN):** `catalog/docs/okf-second-brain/spec.md` (v0.1) + `log.md` (koordinering/rollout) — les FØRST.
|
||||
- OKF upstream-spec: https://github.com/GoogleCloudPlatform/knowledge-catalog/blob/main/okf/SPEC.md
|
||||
|
|
|
|||
|
|
@ -1,107 +0,0 @@
|
|||
# Plugin-roadmap: sesjonsplan R0–R18 (2026-07)
|
||||
|
||||
_Styrende sesjonsplan for ms-ai-architect fra 2026-07-02, basert på full statusanalyse (visjon/planer, ground-truth-audit m/testkjøring, Spor 1-planlesing). Erstatter IKKE eksisterende planer (korrekthetsprogrammet, Spor 1-planen, briefer) — **sekvenserer** dem og fyller hullene analysen avdekket. Operatør-detaljer som ikke hører hjemme i repoet ligger i en LOCAL-ONLY-annex (`.claude/projects/2026-07-02-helhetlig-roadmap/`)._
|
||||
|
||||
**Status 2026-07-02:** måle-/hardningsfasen av korrekthetsprogrammet er ferdig (judge v3.1 P100/R100, alle §8-gap lukket, suite 641/641, alle skills ≥90) — men mekanismen er dormant til Spor 1 kjøres. Gapet mellom visjon («reference-filene 100 % til å stole på» som prosessgaranti) og realitet er nøyaktig de ueksekverte sporene under.
|
||||
|
||||
## Modelldirektiv
|
||||
|
||||
**Alle sesjoner i denne roadmapen + alle deres subagenter kjøres på Opus 4.8 med xhigh reasoning effort** (operatørbeslutning 2026-07-02). Aldri Sonnet/Haiku. Agent-tool: `model: "opus"`.
|
||||
|
||||
## Prinsipper
|
||||
|
||||
1. **Én sesjon = én fullført enhet:** hver sesjon avsluttes med suite grønn (exit-kode), commit + push, STATE.md overskrevet med neste sesjons-ID.
|
||||
2. **Split-regel:** blir en sesjon for stor midtveis → fullfør en HEL delmengde (per skill / per batch), commit, la STATE peke på resten som egen sesjon. Aldri halvferdig tilstand.
|
||||
3. **TDD Iron Law** for all produksjonskode; **aldri auto-fiks KB** (judge/audit flagger → menneske bekrefter mot kilde → fiks).
|
||||
4. **Scope-fence:** sesjonens innhold er grensen; muligheter utover foreslås i sluttrapporten, utføres ikke.
|
||||
5. **Provenans-regel (ny):** hvert tall som skrives i STATE/docs bærer scope + målemetode i én linje (analysen 2026-07-02 brukte en hel avstemmingsrunde på å re-utlede at «45» var non-advisor-scoped).
|
||||
|
||||
## Sekvens og avhengigheter
|
||||
|
||||
```
|
||||
R0 (tillit+hygiene+klargjøring)
|
||||
└─→ R1 → R2 → R3 → R4 (Spor 1 substrat via /trekexecute; release v1.17.0 i R4)
|
||||
└─→ R5 (Spor 0-fiksing)
|
||||
└─→ R6 (judge-pass brief+plan) → R7…R10 (korpus-pass; antall fastsettes i R6)
|
||||
└─→ R11 (flag→fiks) → R12 (§7 fersk-gull — programmets sluttbevis)
|
||||
└─→ R13 → R14 (S-Cosmo, sist i programmet)
|
||||
└─→ R15 (skill-eval v2 plan) → R16+ (eksekvering)
|
||||
Fleksible (uavhengige, kan interleaves): R17 (dekningsaudit), R18 (onboarding-uplift)
|
||||
Parkert (eget operatør-go): OKF second brain · pris/unsourced-verifisering
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## R0 — Tillit, dok-hygiene og Spor 1-klargjøring (kort økt)
|
||||
|
||||
**Mål:** fjern alt som kan villede fremtidige sesjoner eller svekke tillit, og gjør Spor 1-planen eksekverbar.
|
||||
|
||||
1. **Sanering:** fjern identifiserende personkontekst fra ett tracked dokument (utpekt i LOCAL-ONLY-annexen) + søk hele tracked-treet for samme mønster; historikk-håndtering = operatørbeslutning dokumentert i annexen.
|
||||
2. **Utracket brief:** `docs/skill-validation-routine-brief-2026-07.md` — public-egnethetssjekk → commit (eller flytt til `.claude/projects/` hvis intern).
|
||||
3. **CLAUDE.md:** rett de to stale pekerne → `skills/ms-ai-advisor/references/architecture/recommended-mcp-servers.md`.
|
||||
4. **Tellinger:** fjern «398 inkl. plugin-nivå» overalt (plugin-nivå `references/` har aldri eksistert; 389 er totalen).
|
||||
5. **Programdok:** §6-statustabell à jour med §8 (Spor 2a/2b/3 lukket 2026-06-30); historisk-merknad på 84,2-tallene i intro.
|
||||
6. **Workflow-plan:** superseded-peker øverst → programdokumentet.
|
||||
7. **development.md:** fjern «For Cosmo»-instruksen — produserer aktivt ny drift mot aldri-ny-Cosmo-regelen.
|
||||
8. **STATE-korreksjoner:** registry-semantikk (2 disk-filer mangler *i* registry — ikke «stale cache»); Status-tall 31→29 (målt 2026-07-02, non-advisor); provenans-regelen innføres.
|
||||
9. **P0-beslutninger (operatør, minutter):** velsign Spor 1-planens Assumption #1 (fjern stale `**Verified:** MCP` på de 14 filene) + bekreft TOC-scope-default (alle 324 store filer).
|
||||
10. **Plan-patch:** legg manifest-re-entrans-vakt inn i Spor 1-planen (klassifiserer-CLI `--write` skal nekte/merge når manifestet er beriket med `resolvedBy`/`source`) + re-valider planen.
|
||||
|
||||
**Verifisering:** `grep -rn 'references/architecture' CLAUDE.md` viser kun skills-prefiks · `grep -rn '398' CLAUDE.md docs/` → 0 relevante treff · sanering bekreftet med søk → 0 treff i tracked filer · plan-validator strict exit 0 · suite exit 0 · commit + push, `git status --short` ren.
|
||||
|
||||
## R1–R4 — Spor 1 substrat-migrering (4 økter, strengt sekvensiell)
|
||||
|
||||
Kjøres via `/trekexecute --project .claude/projects/2026-06-30-spor1-corpus-migration` (lokalt planprosjekt, 11 steg / plan v1.7+). Roadmapen dupliserer ikke stegene — kun sesjonsgrenser og sluttkriterier:
|
||||
|
||||
- **R1 = Session 1 (Steg 1–4):** header-audit → typeklassifiserer-core (TDD) → CLI + manifest persistert (`--write`, 389 entries, union == uavhengig målt live-tall) → edge-review via subagenter, manifest finalisert. **Verifisering:** manifest med 389 entries, `reviewFlag:false` overalt; suite exit 0.
|
||||
- **R2 = Session 2 (Steg 5–6):** `insertHeaderFields` + kirurgisk `normalizeStaleVerified` (TDD; first-match/`---`-aware/pipe-safe) → authority-URL per reference-fil (deterministiske tie-breaks → subagent → defer-not-guess). **Verifisering:** suite exit 0; manifest `source`-beriket; defer-listen eksplisitt (aldri gjettet URL).
|
||||
- **R3 = Session 3 (Steg 7–8):** `migrate-corpus.mjs` unified applier (TDD: backup + atomisk skriv + advisor-fence + per-felt-idempotens + post-write-assert) → akseptansetester skrevet FØR kjøring. **Verifisering:** suite exit 0 inkl. nye akseptansetester.
|
||||
- **R4 = Session 4 (Steg 9–11):** kjør migreringen (~327 Type / ~253 Source / 324 TOC / 14 stale-verified fjernet) → re-score (CT5 `available:true`, N4 ~1) + worklist-aktivering + `docs/spor1-migration-outcome.md` (transient Spor D sub-90 merkes som policy-artefakt) → hygiene + full suite. **Deretter batch-release v1.17.0** (`release-plugin.mjs`, `check-versions.mjs` 0 ERROR).
|
||||
|
||||
**Eskaleringsregel (fra plananalysen):** per-fil post-write-assert-feil i Steg 9 → restore + stopp + rapporter; «logg som defer og fortsett» er ikke tillatt uten plan-revisjon.
|
||||
|
||||
## R5 — Spor 0: de 38+4 kjente innholds-fiksene (1 økt, split per skill ved behov)
|
||||
|
||||
Påfør `spor0-fix-manifest.json` (38 fikser / 25 filer) + G5b-fillisten (`adr-template.md`, `multi-region-azure-openai-deployment.md`, `network-resilience-patterns-ai.md`, `vector-storage-cost-optimization.md`). Mønster: subagent-fan-out henter/bekrefter kilde per fiks (v3.1-fanout-runbook som mal); hovedkontekst redigerer; tvilstilfeller → operatør. **Forutsetning:** R4 ferdig (substratet i ro). **Verifisering:** manifest 38/38 applied m/kilde-URL per fiks; suite exit 0.
|
||||
|
||||
## R6 — Korpus-judge-pass: brief + plan (1 planleggingsøkt, Voyage)
|
||||
|
||||
Selvbærende brief + adversarielt reviewet plan for v3.1-judgen over alle verifiserbare påstander i non-advisor-korpuset (~2146 fetches, ikke resume-safe → sesjons-batchet design med per-batch-manifest). Briefen skal inkludere: (a) **G2-restgapet** — generaliser stempel-guarden i `transform.mjs` FØR evt. judge-minor-bump; (b) beslutning **retire vs. harden `backfill-toc.mjs`** (unified applier har overtatt; anbefaling: retire); (c) flaggformat som mater R11 direkte; (d) **G6 ingestion-gate (§8):** design to-lags sikkerhetsgate for all henting av eksternt innhold — llm-security-pluginen aktivert + verifisert fyrende (`post-mcp-verify`) i alle fetch-økter (headless-caveat GH #36071 → foreground eller kompenserende skann) + deterministisk node-skann (unicode/decode/injection) over endrede `skills/**/*.md` før commit. **Verifisering:** brief- + plan-validator strict exit 0; antall R7–R10-økter fastsatt; G6-gate-design eksplisitt i briefen.
|
||||
|
||||
## R7–R10 — Korpus-judge-pass eksekvering (antall fra R6; estimat 4 økter à ~80 filer)
|
||||
|
||||
Per økt: én batch judged komplett → flagg akkumulert i manifest. Judge-passet endrer ALDRI KB-innhold — kun `verified`-stempling av grønne filer + flagging. **G6-gate (ufravikelig):** økten starter ALDRI uten llm-security aktiv med hooks verifisert fyrende, eller R6-definert kompenserende deterministisk skann. **Verifisering per økt:** G6-gate bekreftet ved øktstart; 0 udømte i batchen; `git status` viser kun manifest-/stempel-endringer; suite exit 0.
|
||||
|
||||
## R11 — Flag→fiks-loop (1 økt, split ved stort flaggvolum)
|
||||
|
||||
Operatør-bekreftede fikser mot kilde + `verified`-bump per fikset fil (§3c-disiplin). **Verifisering:** alle flagg lukket (fikset eller avvist m/begrunnelse); suite exit 0.
|
||||
|
||||
## R12 — §7 fersk-gull-måling: programmets sluttbevis (1 økt)
|
||||
|
||||
Trekk NYTT blindt gull-utvalg (G5-protokoll: re-adjuder mot live før fasiten stoles på), mål flagget feilrate. **<~2 % → prosessgarantien er dokumentert virksom** — visjonens suksesskriterium. Utvalget blir neste kalibreringssett; periodisk gull-re-adjudering forankres her. **Verifisering:** rapport med målt feilrate + konfidensintervall committed; ≥2 % → ny R11-runde planlegges, ikke bortforklares.
|
||||
|
||||
## R13 — S-Cosmo del 1: mekanisk korpus-rens (1 økt)
|
||||
|
||||
Scripted transform (TDD) for Cosmo-forekomster i ref-korpuset (378 ref-filer; heading-nøytralisering) + absorberte rester: redirects (1a-C) og store-fil-TOC (1b). **Verifisering:** `grep -rl "Cosmo" skills/*/references | wc -l` → 0; suite exit 0; stikkprøve-diff-review.
|
||||
|
||||
## R14 — S-Cosmo del 2: persona, advisor og legacy (1 økt)
|
||||
|
||||
Fjerning i SKILL.md-er, 23 commands, 11 docs, rot-filer + `generate-skills.md`/`transform-prompt.md`; **advisor frontmatter-backfill + advisor judge-pass** (advisor judged etter Cosmo); retir legacy bash `MODEL=sonnet`. Release-bump. **Verifisering:** `grep -rli "cosmo" --include='*.md' .` (unntak per cosmo-removal-brief) → 0; advisor stemplet; suite exit 0; release grønn.
|
||||
|
||||
## R15 → R16+ — Skill-eval v2 (1 planleggingsøkt + est. 2–3 eksekveringsøkter)
|
||||
|
||||
`/trekplan --brief docs/skill-validation-routine-brief-2026-07.md` (committet i R0). Trinn 2 (live triggering m/precision+recall, 3× repetisjon, train/held-out), Trinn 3 (uplift vs. baseline), Trinn 4 (behavioral gate + `generate-skills`-wiring). Briefens §7-beslutninger (adopt vs. reimplement, kostnadsgating) avgjøres av operatør ved planreview. Re-score ETTER Spor 1 + S-Cosmo (kanonisert rekkefølge, programdok §6). **Verifisering:** briefens 7 testbare suksesskriterier.
|
||||
|
||||
## R17 — Dekningsaudit (1 økt — fleksibel, uavhengig)
|
||||
|
||||
Svar på det aldri-målte spørsmålet: **dekker de 389 filene de riktige temaene?** Fan-out: Microsoft AI-landskapet (Foundry / Copilot / Power Platform / Agent Framework / AI Act-flater) vs. faktisk KB-dekning → dekningsrapport + prioritert gap-liste som mates inn som fremtidige KB-emner (født-verifisert generering). **Verifisering:** rapport committed med tema-matrise + gap-liste.
|
||||
|
||||
## R18 — Onboarding uplift-måling (1 økt — fleksibel, uavhengig)
|
||||
|
||||
Maskineriet er bygget (krav 1+2+3 + per-kategori-hint); målingen er aldri kjørt. Design: samme oppgavesett med/uten onboardet kontekst, mål merkbar letthet (tid/kvalitet). **Verifisering:** uplift-rapport committed; beslutning videre loggført.
|
||||
|
||||
## Parkert (krever eget operatør-go)
|
||||
|
||||
- **OKF second brain** (spec v0.1 ratifisert; bygging = separat go).
|
||||
- **Pris/unsourced-verifisering** (74 % av prispåstander ikke maskinverifiserbare; operatør-gated).
|
||||
- **Historikk-håndtering** for saneringen i R0 hvis operatør ønsker mer enn fremoverrettet fiks (egen beslutning).
|
||||
92
docs/r11-flag-format-2026-07.md
Normal file
92
docs/r11-flag-format-2026-07.md
Normal file
|
|
@ -0,0 +1,92 @@
|
|||
# R11 flag format — the judge-pass → fix-work-list contract
|
||||
|
||||
**Spec for the flag record the R7–R10 judge pass emits and R11 (the fix step)
|
||||
consumes. A flag is a judged claim the pass could NOT ground against its cited
|
||||
Microsoft Learn source. The pass never fixes — it stamps (born-verified) or
|
||||
flags; R11 does the human-confirmed fix.**
|
||||
|
||||
Status: design spec (R6 Step 5). Builds on the frozen v3.1 claim judge
|
||||
(`scripts/kb-eval/judge-claim-prompt-v3.1.md`) and the judge-pass ledger
|
||||
(`scripts/kb-eval/judge-pass-manifest.mjs`, R6 Step 4). Consumed by R11.
|
||||
|
||||
---
|
||||
|
||||
## Where a flag comes from
|
||||
|
||||
The v3.1 judge returns, per claim, a strict-JSON result (no fence):
|
||||
|
||||
```
|
||||
{"file":"<FILE>","results":[
|
||||
{"id":"<claim id>","judge_verdict":"grounded|not_grounded|source_silent",
|
||||
"rule":"<R1-R8 or empty>","evidence_url":"<url actually used>",
|
||||
"evidence_quote":"<verbatim quote or empty>","reason":"<one sentence>"}
|
||||
]}
|
||||
```
|
||||
|
||||
The pass augments each **non-grounded** result (`judge_verdict !== 'grounded'`)
|
||||
into a flag record. This is a **minimal augmentation, not a pass-through** — the
|
||||
judge output is a subset of the flag record; the pass adds four fields from the
|
||||
fan-out context and the claim manifest:
|
||||
|
||||
| Added field | Source |
|
||||
|-------------|--------|
|
||||
| `file` | the fan-out context (one subagent per file) — also echoed by the judge output's top-level `file` |
|
||||
| `line` | the claim manifest (`scripts/kb-eval/extract-judge-claims.mjs` output) |
|
||||
| `claim` | the claim manifest (the verbatim claim text that was judged) |
|
||||
| `disposition` | mapped from `judge_verdict` (see the vocabulary mapping below) |
|
||||
|
||||
## The flag record
|
||||
|
||||
A flag record is the union of the judge result and the four augmented fields:
|
||||
|
||||
```
|
||||
{
|
||||
"id": "<claim id>",
|
||||
"judge_verdict": "not_grounded" | "source_silent", // 'grounded' is never flagged
|
||||
"rule": "R1".."R8" | "",
|
||||
"evidence_url": "<url the judge actually used>",
|
||||
"evidence_quote":"<verbatim quote or empty>",
|
||||
"reason": "<one sentence: what the source said vs the claim>",
|
||||
"file": "skills/.../x.md",
|
||||
"line": <number>,
|
||||
"claim": "<verbatim claim text>",
|
||||
"disposition": "outdated" | "wrong" | "unsourced" // R11 fix disposition
|
||||
}
|
||||
```
|
||||
|
||||
In the ledger (R6 Step 4), a flagged file is a manifest record with
|
||||
`per_file_verdict: "flagged"` and one `flags[]` entry per non-grounded claim —
|
||||
the structural mirror of `spor0-fix-manifest.json`'s "fix-record with
|
||||
`applied:false`". A file is `flagged` if **any** judgeable claim is non-grounded
|
||||
(the born-verified rule is AND: `pass` requires EVERY claim grounded).
|
||||
|
||||
## The two vocabularies + the mapping
|
||||
|
||||
There are two distinct enumerations, and conflating them is the trap this spec
|
||||
exists to prevent:
|
||||
|
||||
1. **`judge_verdict`** — the judge's per-claim output (v3.1 prompt):
|
||||
- `grounded` — the source confirms the claim → never flagged.
|
||||
- `not_grounded` — the source contradicts / does not support the claim.
|
||||
- `source_silent` — the source does not address the claim at all.
|
||||
|
||||
2. **`disposition`** — the R11 fix disposition, the correctness vocabulary from
|
||||
`scripts/kb-eval/lib/base-rate.mjs` (`VERDICTS`): `correct`, `outdated`,
|
||||
`wrong`, `unsourced`. Its `ERROR_VERDICTS` subset — `outdated` and `wrong` —
|
||||
are the only ones R11 treats as **fix targets**. `unsourced` is recorded but
|
||||
is not, on its own, a content error (no MS source confirms it either way).
|
||||
|
||||
**Mapping** (`judge_verdict` → `disposition`):
|
||||
|
||||
| `judge_verdict` | `disposition` | R11 fix target? |
|
||||
|-----------------|---------------|-----------------|
|
||||
| `grounded` | `correct` | no (never flagged) |
|
||||
| `not_grounded` | `outdated` or `wrong` | **yes** — the human assigns which at R11 |
|
||||
| `source_silent` | `unsourced` | no (flagged, but not a fix target) |
|
||||
|
||||
`not_grounded → {outdated, wrong}` is deliberately a one-to-two mapping: the
|
||||
judge establishes the claim is not grounded; whether it is stale-but-once-true
|
||||
(`outdated`) or never-true (`wrong`) is a human call at R11, re-verified against
|
||||
the live source (verification duty). The pass never auto-fixes — every flag is
|
||||
human-confirmed in R11, so a judge false-positive becomes a human review, never
|
||||
silent corruption of a public file.
|
||||
2098
docs/r11-pilot-results.md
Normal file
2098
docs/r11-pilot-results.md
Normal file
File diff suppressed because it is too large
Load diff
333
docs/r11-tiered-fix-design.md
Normal file
333
docs/r11-tiered-fix-design.md
Normal file
|
|
@ -0,0 +1,333 @@
|
|||
# R11 execution design — tiered fixes over an already-evidenced flag population
|
||||
|
||||
**How the R11 fix step consumes the judge-pass flags. Extends
|
||||
`docs/r11-flag-format-2026-07.md` (the record contract) with the execution
|
||||
contract: what the fix operation is per flag, what is machine-provable, what
|
||||
requires human judgement, and what may never be automated.**
|
||||
|
||||
Status: design spec. Written 2026-08-03 against a ledger snapshot of **222
|
||||
records / 902 flags** (`scripts/kb-eval/data/judge-pass-manifest.json`). The
|
||||
population is **not yet complete** — batch R7.5 stood at 30 of 51 files — so
|
||||
every count below is a snapshot, not a final figure. The design does not depend
|
||||
on the exact counts; the classifier re-derives them at run time.
|
||||
|
||||
---
|
||||
|
||||
## 1. The measurement this design is built on
|
||||
|
||||
**All 712 `not_grounded` flags carry a non-empty verbatim `evidence_quote`.**
|
||||
712 of 712, measured. Every flag also carries the `evidence_url` the judge
|
||||
actually fetched, the rule that fired, and a one-sentence `reason` stating what
|
||||
the source said versus what the claim said.
|
||||
|
||||
The consequence is the whole point of this document: **R11 does not start from
|
||||
zero evidence.** The expensive half — locating the authoritative page, reading
|
||||
it, and extracting the passage that decides the claim — was already paid for by
|
||||
the judge pass. A fix step designed as "fetch the source, confirm, correct"
|
||||
re-does work that is already on disk.
|
||||
|
||||
What actually remains per flag is narrower:
|
||||
|
||||
1. is the quote still current (freshness, §7), and
|
||||
2. what should the sentence say instead (§3).
|
||||
|
||||
## 2. Measured shape of the flag population
|
||||
|
||||
| Measurement | Value |
|
||||
|---|---|
|
||||
| Flags total | 902 |
|
||||
| `not_grounded` (fix targets per flag-format spec) | 712 |
|
||||
| `source_silent` (recorded, not a fix target) | 190 |
|
||||
| Flags carrying a verbatim `evidence_quote` | **712 / 712** |
|
||||
| Distinct `evidence_url` across the 712 | **540** |
|
||||
| Files carrying at least one `not_grounded` flag | 199 |
|
||||
| Files with 1 flag | 45 |
|
||||
| Files with 2–6 flags | 133 (489 flags) |
|
||||
| Files with ≥7 flags | 21 (**178 flags — 25 % of the volume in 10 % of the files**) |
|
||||
|
||||
Rule distribution over the 712: R8 339 · R2 176 · (no rule) 72 · R4 41 · R3 40 ·
|
||||
R7 28 · R1 16.
|
||||
|
||||
**540 distinct source URLs for 712 flags** is the number that forecloses the
|
||||
obvious optimisation: there is no batching win hiding in shared sources. The
|
||||
largest cluster is 12 flags on one page. This is ~700 separate facts, and the
|
||||
work is irreducibly per-claim.
|
||||
|
||||
## 3. Three fix operations — the partition is by operation, not by rule
|
||||
|
||||
The tiering is defined by **what the fix does to the file**, because that is what
|
||||
determines whether a machine can prove it and whether a human must decide it.
|
||||
Rule codes are a signal, not the partition.
|
||||
|
||||
| Op | Fix operation | Provable? | Who decides |
|
||||
|---|---|---|---|
|
||||
| **O1** | **Value swap** — the claim states X, the cited source states Y; replace the token. | **Yes** (§4) | Machine proposes, human reviews a diff list |
|
||||
| **O2** | **Subtraction** — remove or generalise the specificity the source does not support. Covers: source-silent claims, absent entities (retired SKUs/models/services), and multi-part claims where the failing sub-assertion can be dropped while the grounded part survives. | Partly — the *invariant* is that no new fact is introduced | Human ratifies the policy once (§5), then reviews proposals |
|
||||
| **O3** | **Rewrite** — the corrected sentence requires a judgement about what to assert. | No | Human, per claim |
|
||||
|
||||
**Do not read the §2 rule counts as tier sizes.** They overlap: of the 202 flags
|
||||
whose claim *and* quote contain a numeric token — the naive O1 signal — **94 are
|
||||
R8 multi-part claims**, which are not clean swaps. Heuristics over this
|
||||
population give upper bounds in several directions at once, never a partition.
|
||||
|
||||
**The classifier fails closed to O3.** If it cannot prove an item is O1 or O2, the
|
||||
item is O3 and a human sees it. A misrouted O3 costs one review; a misrouted O1
|
||||
ships a wrong edit to a public file.
|
||||
|
||||
R8 deserves a specific note, because it is the largest rule class and it is *not*
|
||||
automatically the most expensive one: the judge's `reason` names **which**
|
||||
sub-assertion failed. Where the grounded part stands on its own, the fix is O2
|
||||
(drop the unsupported part), not O3 (rewrite the sentence). ~~How much of R8 falls
|
||||
that way is unknown and is a primary pilot measurement (§10).~~
|
||||
|
||||
**MEASURED 2026-08-03 — `docs/r11-pilot-results.md` §9. It falls that way for a
|
||||
minority.** Over the 46 pilot flags with the O2 shape (R8 ∧ `MULTI_PART_CLAIM`):
|
||||
17 O2 candidates, 29 O3, and **all 29 are blocked by §5 condition 3** — the
|
||||
source supplies a corrected value, so the fix is a swap or a rewrite and
|
||||
subtraction would destroy true information. Only 2 of the 46 clear both
|
||||
human-judged conditions affirmatively. The paragraph above is not wrong, but the
|
||||
case it describes is the exception in R8, not the rule.
|
||||
|
||||
## 4. The O1 invariant (what makes a value swap provable)
|
||||
|
||||
An O1 proposal is only valid if, after the edit:
|
||||
|
||||
1. the new value appears **verbatim inside the cited `evidence_quote`**, and
|
||||
2. the rest of the line is **byte-identical** to before, and
|
||||
3. exactly one line in the file changed.
|
||||
|
||||
A driver that cannot establish all three for an item **aborts before writing** and
|
||||
routes the item to O3. This is the same discipline already proven in the
|
||||
header-backfill drivers (frozen manifest, hard per-file invariant, abort before
|
||||
write, idempotent re-run, `atomicWriteSync`) — see `scripts/kb-update/backfill-*.mjs`.
|
||||
|
||||
~~This invariant is deliberately stronger than human review at scale.~~
|
||||
**FALSIFIED 2026-08-03 — see `docs/r11-pilot-results.md` §2.** Run exactly as
|
||||
written it admitted 6 swaps of which **4 were wrong edits**. Conditions 1–3
|
||||
constrain where the value *came from* and what the edit *looks like*, and nothing
|
||||
about whether the two tokens denote the same quantity. Conditions 1–3 are
|
||||
necessary; they are not sufficient.
|
||||
|
||||
### 4a. Condition 4 — context correspondence (added 2026-08-03)
|
||||
|
||||
4. the token must sit under **the same label, or the same trailing unit, on both
|
||||
sides** (`contextCorresponds()` in `scripts/kb-eval/lib/fix-op.mjs`).
|
||||
|
||||
Deliberately lexical, with no translation table beyond §4b: a swap is therefore
|
||||
provable essentially only where the context is language-neutral — a URL, a code
|
||||
sample, a parameter key.
|
||||
|
||||
**Condition 5 — the applied class is `iso_date` only** (operator decision,
|
||||
2026-08-03). Condition 4 is still not sufficient: a matching identifier *prefix*
|
||||
(`AI-`, `gpt-`, `Agent `) satisfies it while the digit is part of a **name**
|
||||
rather than a quantity, which produced `AI-900` → `AI-901`, `gpt-4o` →
|
||||
`gpt-5.1o` (twice) and a Java-agent downgrade. Hand-verification over the whole
|
||||
population: **`iso_date` 9/9 correct, `number` and `version` 0/6**. A driver may
|
||||
apply `iso_date` proposals and **must never apply `number` or `version` ones**.
|
||||
The classifier keeps reporting all admitted types — that is the measurement — and
|
||||
marks the applicable set as `o1_recommended`.
|
||||
|
||||
### 4b. The ratified status-synonym table (operator decision, 2026-08-03)
|
||||
|
||||
`STATUS_SYNONYM` — 15 pilot flags, **54 corpus-wide** — is the class where the
|
||||
corpus writes `**Preview**` / `**GA**` while the source writes *"generally
|
||||
available"*. A narrow, **closed** equivalence table is ratified:
|
||||
|
||||
| Corpus-side label | Source-side phrasing (must appear verbatim in `evidence_quote`) |
|
||||
|---|---|
|
||||
| `GA` | `generally available`, `general availability` |
|
||||
| `Preview`, `Public Preview` | `public preview`, `preview` |
|
||||
| `Private Preview` | `private preview` |
|
||||
| `Deprecated`, `Utfaset` | `deprecated`, `retired` |
|
||||
|
||||
Three constraints, because this is the **one** place where the value written into
|
||||
the file does not itself appear verbatim in the quote:
|
||||
|
||||
- The table is **closed**. Any pair not listed aborts to O3; it is never extended
|
||||
by inference at run time.
|
||||
- The file-side token must be a **complete lifecycle label** (a whole table cell
|
||||
or emphasised token), never a substring of a longer sentence.
|
||||
- The written value is the **corpus-side** equivalent with the file's own markup
|
||||
preserved (`**Preview**` → `**GA**`), never the English phrase pasted in.
|
||||
|
||||
**IMPLEMENTED 2026-08-03** in `classifyStatusSynonym()` / `fileStatusLabel()` /
|
||||
`sourceStatusRows()` (`scripts/kb-eval/lib/fix-op.mjs`), 19 tests. Two
|
||||
implementation decisions the ratified text left open, both resolved towards
|
||||
failing closed:
|
||||
|
||||
- **The status locator is LINE-scoped**, not block-scoped like the numeric one.
|
||||
Lifecycle vocabulary repeats down every column of a status table, so a block
|
||||
window is ambiguous by construction; all 15 pilot flags in this class point at
|
||||
the row that carries the claim.
|
||||
- **A quote asserting two different rows aborts** (`SOURCE_STATUS_AMBIGUOUS`), and
|
||||
a row with two corpus-side labels writes the **first** — the least specific one,
|
||||
so a source saying only "preview" can never produce "Public Preview".
|
||||
|
||||
Aborts keep the `STATUS_SYNONYM` code and name their cause in `detail.reason`, so
|
||||
the §10 abort taxonomy stays comparable across the implementation.
|
||||
|
||||
**Measured: 15 pilot / 54 corpus-wide flags → 5 and 8 proven; 5 of the 8 are
|
||||
correct** (`docs/r11-pilot-results.md` §8 + appendix B). The three defects are one
|
||||
family: §4b binds the table, the completeness of the file label and the written
|
||||
value, and **nothing about whether the source phrasing refers to the row's own
|
||||
subject** — provenance without referent, the same defect that falsified §4. The
|
||||
class is therefore **review-grade, not apply-grade**: `status` is absent from
|
||||
`o1_recommended` and no driver applies it.
|
||||
|
||||
**Open operator decision — a referent condition for §4b.** Two candidates are
|
||||
costed over the eight in `r11-pilot-results.md` §8; both kill wrong proposals and
|
||||
no correct one, and neither is implemented, because extending a table ratified as
|
||||
*closed* is an operator decision, exactly as condition 5 was in §4a.
|
||||
|
||||
## 5. The O2 policy — RATIFIED 2026-08-03, with a remainder check
|
||||
|
||||
O2 fixes by **subtraction**: the unsupported specificity is removed or
|
||||
generalised rather than replaced with a researched value.
|
||||
|
||||
**Ratified by the operator on 2026-08-03, with one condition: the remainder
|
||||
check.** Subtraction is not admitted as a blanket rule, because the pilot found
|
||||
two ways it fails (`docs/r11-pilot-results.md` §5):
|
||||
|
||||
- It can leave a **misleading remainder**. Removing `er GA (juni 2025)` from a
|
||||
claim about a tool the source calls *deprecated* leaves that tool standing in a
|
||||
list of available ones. Strictly less asserted, still misleading.
|
||||
- It can **destroy true information**. Dropping `prebuilt-check` from a model list
|
||||
removes a model that exists — its ID is `prebuilt-check.us`, so the correct fix
|
||||
is a swap.
|
||||
|
||||
**An O2 proposal is valid only if all three hold, and a human confirms them:**
|
||||
|
||||
1. the edited sentence asserts **strictly less** than before;
|
||||
2. the **remainder carries no false or misleading standing implication** — read as
|
||||
a reader would read it, not as a logician would;
|
||||
3. nothing the source **confirms** is removed. Where the source supports a
|
||||
corrected value, the fix is O1 or O3, never subtraction.
|
||||
|
||||
Conditions 2 and 3 require a human to read the remainder. O2 is therefore
|
||||
**cheaper than O3 — no fact-finding — but not mechanical**, and the §10
|
||||
throughput assumption should be re-measured against that.
|
||||
|
||||
- It requires **no new fact-finding**, which is what makes it cheap.
|
||||
- It **reduces information density**. That is the real cost, and it is an
|
||||
operator decision, not an engineering one.
|
||||
|
||||
The position this design recommends: a knowledge base that says less and says
|
||||
nothing false is worth more than one carrying stale precision. The corpus is
|
||||
publicly distributed; an incorrect specific number is a worse failure than an
|
||||
honest general statement.
|
||||
|
||||
Ratifying O2 also resolves the standing `source_silent` question as **one class
|
||||
decision** instead of 190 individual ones. ~~Until it is ratified, every O2
|
||||
candidate falls to O3.~~ Ratified — O2 is in use, subject to the remainder check
|
||||
above. **Not implemented in the classifier, and now measured to be the right
|
||||
call:** the classifier still routes every non-O1 item to O3, because O2 candidacy
|
||||
turns on the judge's prose `reason`. The prose classification was run separately
|
||||
(`docs/r11-pilot-results.md` §9) and found that condition 3 forecloses **every**
|
||||
non-candidate in the class — a mechanical O2 driver would therefore have proposed
|
||||
deletions where the source hands over a corrected value, which is precisely the
|
||||
F2 failure. O2 output is a human review list, like §4b's.
|
||||
|
||||
## 6. What stays human, permanently
|
||||
|
||||
- **Never auto-fix.** No fix reaches a file without human confirmation of the
|
||||
class (O1/O2) or the item (O3). A judge false positive must become a human
|
||||
review, never silent corruption of a public file.
|
||||
- **Subagents never write.** Proposal generation is read-only fan-out; all
|
||||
writes happen in one place (§8).
|
||||
- **O3 is not a backlog to automate later.** It is the class where the corrected
|
||||
assertion is a judgement call, and it is the reason the loop reaches ~100 % fix
|
||||
precision on top of a fallible judge.
|
||||
|
||||
## 7. Evidence freshness
|
||||
|
||||
Each `evidence_quote` is current as of the judge's fetch date, not the fix date.
|
||||
Before any file is touched, the evidence base is refreshed by **re-fetching per
|
||||
distinct URL — 540, not 712** — as read-only fan-out. A refreshed quote that no
|
||||
longer supports the flag re-routes the item (possibly closing it as no longer an
|
||||
error). This preserves the verification duty — fresh confirmation before a public
|
||||
file changes — without paying for 712 separate fetches.
|
||||
|
||||
## 8. Concurrency: one writer, many readers
|
||||
|
||||
R11 phases that are machine-bound (evidence refresh, proposal generation) may run
|
||||
across concurrent sessions. The corpus and the ledger may not.
|
||||
|
||||
**Single-writer state** — never written by more than one session:
|
||||
`scripts/kb-eval/data/judge-pass-manifest.json`, the corpus files themselves, and
|
||||
the repo's state file.
|
||||
|
||||
**Protocol:**
|
||||
|
||||
- Worker sessions write **only** disjointly-named artefacts to a local, untracked
|
||||
working directory (one file per unit of work). They do not ingest, stamp, or
|
||||
commit.
|
||||
- One integrator session ingests serially, stamps, and commits.
|
||||
- If two sessions must write corpus files concurrently, shard **by file, never by
|
||||
claim**, one git worktree per shard. Disjoint file sets merge without conflict;
|
||||
the ledger is still written only by the integrator, after merge.
|
||||
|
||||
Git worktrees share the main `.git` directory, so the Layer B `pre-commit` scan
|
||||
symlink applies inside every worktree — verified 2026-08-03 by creating a
|
||||
worktree and resolving `git rev-parse --git-common-dir` plus the hook target.
|
||||
The security gate does not weaken under sharding.
|
||||
|
||||
The reason this protocol is explicit: whole-tree backup/restore in the write
|
||||
drivers previously caused silent data loss across concurrent sessions (closed in
|
||||
the write-safety hardening pass — scoped restore + atomic writes). Sharded writes
|
||||
without a single-writer rule would reintroduce that class through a different door.
|
||||
|
||||
## 9. No full re-judge after fixing
|
||||
|
||||
A fixed file does **not** require a fresh judge pass over all of its claims.
|
||||
|
||||
- Claims already judged `grounded` keep that verdict; the fix did not touch them.
|
||||
- A fixed claim's evidence is the O1 invariant (§4) or the human confirmation
|
||||
(§5–6), recorded with the fix.
|
||||
- The programme's end-proof is a **fresh blind gold sample** measured after the
|
||||
fixes, not a re-run of the corpus pass.
|
||||
|
||||
This is stated explicitly because assuming otherwise would silently add a second
|
||||
full corpus pass to the plan.
|
||||
|
||||
## 10. Pilot and acceptance criterion
|
||||
|
||||
Before any scaling, run the classifier and the O1 driver against the **24 files
|
||||
carrying ≥7 `not_grounded` flags (202 flags)** — a quarter of the volume in a
|
||||
tenth of the files, and the densest available sample.
|
||||
|
||||
> Recount 2026-08-03 after R7.5 completed (ledger 222 → 243 records). The rule is
|
||||
> unchanged — densest ≥7 sample — only the count moved: 21 files / 178 flags was
|
||||
> measured against the 222-record ledger, before the last 21 R7.5 files were
|
||||
> ingested. Three files entered the sample (`semantic-caching-patterns.md`,
|
||||
> `small-language-models-economics.md`, `vector-storage-cost-optimization.md`).
|
||||
> The threshold counts `not_grounded` only, not `source_silent`; on all-flags it
|
||||
> would be 39 files / 352.
|
||||
|
||||
Measure and record:
|
||||
|
||||
1. the actual O1 / O2 / O3 split (this design's central unknown);
|
||||
2. how much of R8 resolves as O2 rather than O3 (§3);
|
||||
3. the O1 driver's abort rate — items it could not prove, which must land in O3;
|
||||
4. review throughput per operation class, measured rather than assumed.
|
||||
|
||||
Scale to the remaining files only on measured numbers. If the split is materially
|
||||
worse than assumed, that is known after one session rather than after ten.
|
||||
|
||||
> **RUN 2026-08-03 — results in `docs/r11-pilot-results.md`.** The split is
|
||||
> materially worse than assumed: **9 provable, correct value swaps in the whole
|
||||
> 776-flag `not_grounded` population (1.2 %)**, all of them `api-version` bumps.
|
||||
> The pilot also falsifies §4 as written — run exactly as specified it admitted 6
|
||||
> swaps of which **4 were wrong edits** (unit crossing, metric crossing, two
|
||||
> mutilated identifiers), so the invariant is *not* "stronger than human review at
|
||||
> scale". A context-correspondence condition was added; read §4 together with the
|
||||
> results doc, not on its own. §5 (O2) and the `STATUS_SYNONYM` class are the open
|
||||
> operator decisions, and they now carry the whole programme's leverage.
|
||||
|
||||
## 11. Out of scope
|
||||
|
||||
- **Rebuild instead of repair.** Regenerating flagged files from source rather
|
||||
than editing them is a settled decision: the corpus is repaired, not rebuilt.
|
||||
Reopening it is an operator call, not a design choice made here.
|
||||
- **Auto-fix in any form** (§6).
|
||||
- **Price and other unsourced claims** beyond the O2 class decision — these
|
||||
remain operator-gated as a separate matter.
|
||||
|
|
@ -104,7 +104,9 @@ Status-nøkkel: 🔴 ikke startet · 🟡 pågår · 🟢 lukket.
|
|||
| **G4** | Nedre-grense-policyen lever kun i prosa (denne dok + reconciliation-logg) — ikke kodet i judge-prompt ELLER `build-gold-set`-instruks | Re-introdusert nedre-grense-ambivalens i fremtidige gull-bygg + judge-kjøringer | Kod policyen inn i judge-prompt-v3 (G1) + build-gold-set-instruks | 🟢 **kodet 2026-06-30** (build-instruks + v3 R1); håndheving rir på G1/G2-adopsjon | Spor 1 / §7 friskt utvalg |
|
||||
| **G5** | Gull-fasiten kan aldre — ingen friskhets-/re-adjuderings-vakt på selve svarnøkkelen. v3-målingen avdekket at flere judge-«feil» trolig er *utdatert gull*, ikke judge-feil (`genaiops-llm-specific#2`: claim «1600+», live=1900 ⇒ 1,19× tett nedre grense, R1 sier korrekt `grounded`, gull sier `outdated` — gull-standarden er her for streng) | Feil adopsjonsbeslutning bygd på aldrende baseline; falsk feilrate i §7-nordstjernen | Friskhets-mikropass: re-adjuder de ~5 omstridte v3-vs-v2-claims mot live MS Learn (avgjør gull-feil vs judge-feil) + periodisk gull-re-adjudering knyttet til §7 friskt utvalg | 🟢 **lukket 2026-06-30** (G5: 2 gull-feil rettet, 3 judge-feil bekreftet, **reverserte adopsjonsbeslutningen**; **G5b: completeness-caveat lukket** — de 4 v3-FP re-sjekket, ALLE 4 stale gull, v3 → P 100 % / R 92,9 % / 0 FP — se lukke-logg) | v3.1-adopsjon (baseline må være til å stole på FØR ny prompt måles mot den) |
|
||||
|
||||
| **G6** | Ingen sikkerhetsgate på ingestion-kjeden (hentet eksternt innhold → korpus): llm-security-pluginen er **deaktivert globalt** (verifisert 2026-07-03 i `~/.claude/settings.json`), så `post-mcp-verify`-hooken (injection-skann på all tool-output, inkl. `microsoft_docs_fetch`) fyrer ikke; commit-gaten dekker kun secrets (gitleaks), ikke injeksjon/steganografi i `.md`-innhold. Tillit til MS Learn dekker faktisk korrekthet — ikke adversarielt innhold i kanalen eller i kodeeksempler/lokalisert stoff | Indirekte prompt-injeksjon/steganografi persistert i offentlig distribuert KB: references-filene blir instruksjonsnær kontekst i fremtidige agent-sesjoner, én forgiftet fil re-serveres til alle brukere (hele Norge) | **R6-briefen designer gaten, to lag** (`docs/ingestion-security-brief-2026-07.md`, committet 2026-07-04): (a) llm-security AKTIV + verifisert fyrende i enhver fetch-økt (kb-update, research, generate-skills, judge-pass); headless-caveat GH #36071 → foreground eller kompenserende skann; (b) deterministisk node-skann (unicode/decode/injection — de DELTE llm-security-detektorene importert in-process, ikke kopiert; operatør-valg 2026-07-04) over endrede `skills/**/*.md` før commit. Håndheves fra R7 og i kb-update-kadensen | 🟡 **Layer B (b) LUKKET 2026-07-04 (TDD).** Bærende gaten bygget: `scan-adversarial-content.mjs` (+ `lib/adversarial-scan.mjs` disposition-kjerne, `lib/adversarial-detect.mjs` llm-security-bro), wiret som sibling til `validate-kb-file.mjs` ved det eneste skrive-chokepunktet (kb-update §3b.d.7/§4/§5 + generate-skills per-batch/pre-commit). Provenance-tiered BLOCK/WARN; injection-flagg → samme menneske-i-loop som status-påstand. 30 tester (692/0). Premiss-korr.: research/research-agent skriver ingenting (kun Layer A); CLI-scan alene misset injection+base64 → importerer rene primitiver. **Layer A (a) gjenstår:** aktiver llm-security i `~/.claude/settings.json` + verifiser `post-mcp-verify` fyrer i én live fetch-økt (§8-verifikasjon). Aktiveringsregelen gjelder STRAKS. | Layer A: R7 (første judge-pass-fetch-økt) / enhver `/architect:kb-update`/`generate-skills`-fetch-økt |
|
||||
| **G6** | Ingen sikkerhetsgate på ingestion-kjeden (hentet eksternt innhold → korpus): llm-security-pluginen er **deaktivert globalt** (verifisert 2026-07-03 i `~/.claude/settings.json`), så `post-mcp-verify`-hooken (injection-skann på all tool-output, inkl. `microsoft_docs_fetch`) fyrer ikke; commit-gaten dekker kun secrets (gitleaks), ikke injeksjon/steganografi i `.md`-innhold. Tillit til MS Learn dekker faktisk korrekthet — ikke adversarielt innhold i kanalen eller i kodeeksempler/lokalisert stoff | Indirekte prompt-injeksjon/steganografi persistert i offentlig distribuert KB: references-filene blir instruksjonsnær kontekst i fremtidige agent-sesjoner, én forgiftet fil re-serveres til alle brukere (hele Norge) | **R6-briefen designer gaten, to lag** (`docs/ingestion-security-brief-2026-07.md`, committet 2026-07-04): (a) llm-security AKTIV + verifisert fyrende i enhver fetch-økt (kb-update, research, generate-skills, judge-pass); headless-caveat GH #36071 → foreground eller kompenserende skann; (b) deterministisk node-skann (unicode/decode/injection — de DELTE llm-security-detektorene importert in-process, ikke kopiert; operatør-valg 2026-07-04) over endrede `skills/**/*.md` før commit. Håndheves fra R7 og i kb-update-kadensen | 🟢 **Layer B (b) LUKKET 2026-07-04 (TDD).** Bærende gaten bygget: `scan-adversarial-content.mjs` (+ `lib/adversarial-scan.mjs` disposition-kjerne, `lib/adversarial-detect.mjs` llm-security-bro), wiret som sibling til `validate-kb-file.mjs` ved det eneste skrive-chokepunktet (kb-update §3b.d.7/§4/§5 + generate-skills per-batch/pre-commit). Provenance-tiered BLOCK/WARN; injection-flagg → samme menneske-i-loop som status-påstand. 30 tester (692/0). Premiss-korr.: research/research-agent skriver ingenting (kun Layer A); CLI-scan alene misset injection+base64 → importerer rene primitiver. **Baseline-adjudikering (Enhet A2) LUKKET 2026-07-18 (TDD):** korpus-baseline 55/389 flagget (83 funn) → 4 ekte defekter fikset (2 Cyrillic-homoglyph-ord, 2 filer med U+00AD) + 75 funn human-adjudikert inn i innholds-basert allowlist (`scripts/kb-update/data/layerb-allowlist.json`: class+evidence+tier+eksakt trimmet linjeinnhold per entry, med begrunnelse; flyttet linje forblir grønn, endret innhold GJENOPPSTÅR som flagg — scanneren selv usvekket, korrupt/manglende allowlist → tom = full strenghet). Korpus 389/389 OK exit 0; commit-gaten (pre-commit-scan) leser samme allowlist. **Layer A (a) LUKKET 2026-07-18 (Enhet B):** `post-mcp-verify` aktivert kirurgisk i `~/.claude/settings.json` (PostToolUse-matcher `mcp__microsoft-learn__.*`) + live-verifisert fyrende med 2 foreground `microsoft_docs_fetch` — og en **ekte hook-defekt funnet+fikset underveis** (hooken leste `tool_output`, live-protokollen sender `tool_response` → hooken skannet ingenting; fiks TDD i llm-security `44aa390` v7.8.3). Se lukke-logg. | Layer A: R7 (første judge-pass-fetch-økt) / enhver `/architect:kb-update`/`generate-skills`-fetch-økt |
|
||||
|
||||
| **G7** | **Ingen rute for korreksjoner som er RIKTIGE, men større enn O2-konvolutten.** O2 er definert som én lokator + kun-sletting. R11 §9.3/§9.4 produserte fire funn der den korrekte fiksen beviselig ligger utenfor: idx 18 (`rag-caching-optimization.md:29` — den overlevende påstanden er en hel titulert seksjon 303-318 **pluss** en `**Verified**`-rad på 510; ingen sletting begrenset til linje 29 kan reparere fila), idx 36 (`ai-threat-modeling-stride.md:38` — companion-edit på 310 kreves for at prosaen skal matche den innsnevrede severity-tabellen), idx 17 (`**Verified**`-stemplet på 258 stempler etter editen kun retnings-utsagnet), idx 33 (innholdet overlever på 357 under CAF-attribusjon, så editens gevinst er mindre enn den ser ut). Alle fire er i dag kun prosa i `r11-pilot-results.md` | **Sanne defekter som stille faller ut av programmet fordi ingen mekanisme eier dem.** O2-triagen avviser dem (utenfor konvolutt), O3 dekker dem ikke (fiksen er ikke en verdi-swap), og det menneskelige review-sporet har ingen inngangskø. Nettoeffekten er at den *vanskeligste* klassen — der fila motsier seg selv — er den eneste uten eier | **DESIGNET + BYGGET 2026-08-03 (form b — navngitt kø).** Valget ble tatt på måling, ikke på form-preferanse: av de fire subtraksjonene i `957ebef` etterlot **to** en rest (§9.6), så rester er delete-only-konvoluttens normale biprodukt, ikke et unntak. Og **to av de fem medlemmene (idx 17, 33) er erstatninger, ikke fler-lokator** — en delete-orientert O4-klasse med egen retur-kontrakt ville ikke fikset dem, altså vært feil dimensjonert mot evidensen. Køen absorberer begge klasser. Artefakter: `scripts/kb-eval/data/g7-review-queue.json` (tracked, 6 entries) + `lib/g7-queue.mjs` + `check-g7-queue.mjs` + 15 tester. **Ankere er ordrette strenger, ALDRI linjenummer** (`line` ≠ `real_line` i 9 av 17 R11-records); en åpen entry hvis anker slutter å matche gir `anchor_drift` og exit 1 — den kan ikke falle stille ut. ⚠️ **Presisering om hva som faktisk HÅNDHEVER dette:** `check-g7-queue.mjs` har ingen runner og kjører kun når noen skriver kommandoen. Den bindende gaten er **testen** — `test-g7-queue.test.mjs` siste case (`the real queue file validates against the live corpus`) kjører den ekte køen mot live korpus i hver suite-kjøring. CLI-en er for lesbar status; suiten er gaten. Ingenting i køen er maskin-anvendbart per definisjon; lukking er en menneskelig review-handling som MÅ føre `resolution`. Kobles fra ÅPEN OPERATØRBESLUTNING #2: køen står uansett hvordan den lander | 🟡 **pågår — mekanismen står, køen er ikke tømt.** 5 åpne (idx 26, 27, 33, 36, 18), 1 lukket (idx 17: dinglende ledetekst → kolon-til-punktum, operatør-ratifisert). idx 27 kom hit ved å FALLE UT av O2 på cond 3 (§9.6), idx 26 ved operatørens avvisning av delvis fiks | Før R11s menneskelige review-fase erklæres ferdig. **Mekanismen er nå lukke-vilkåret oppfylt for; det som gjenstår er innholdet i køen** |
|
||||
|
||||
**Ikke mekanisme-gap, men sporet backlog (innhold, ikke loop):** reference-`.md`-fil-fiksene fra Spor 2b (FP1 11000+/40+, FP2 «kun», FP6 Preview/Norway-East, FN2–FN6 utdaterte tall) **+ G5b** (`adr-template.md` fjern «zero permission management»; `multi-region-azure-openai-deployment.md` bytt retired `gpt-35-turbo` → gjeldende modell; `network-resilience-patterns-ai.md` «obligatorisk» → «anbefalt»; `vector-storage-cost-optimization.md` GA-dato `2024-11-01` → `2024-07-01`) er **Spor 0/1**-innholdsarbeid — pekt per-claim i `notes`, ikke gjentakelses-mekanisme. Føres i Spor 0-manifest / Spor 1-korpus-pass, ikke her.
|
||||
|
||||
|
|
@ -147,3 +149,8 @@ Status-nøkkel: 🔴 ikke startet · 🟡 pågår · 🟢 lukket.
|
|||
- **Designvalg — version-label-streng, ikke integer (`judge-v3` forkastet):** v3 er en distinkt, målt, *forkastet* versjon (R 92,9 / 3 FN) med eget navn i programmets artefakter; et `judge-v3`-stempel ville navne feil judge og kollidere. Streng `'3.1'` lar provenance navne adoptert judge eksakt. Blast-radius null: `verified_by` lagres/parses kun som `\S+`-token + presence-sjekk (`parseVerifiedByHeader`), ingen kode trekker ut integeren; parseren tar `judge-v3.1` uendret (ende-til-ende-testen bekrefter det gjennom `composeKbFile`). Default-stien interpolerer strengen direkte (utenom `Number.isInteger`-guarden), så integer-override-stien (`judgeVersion:3 → judge-v3`) består.
|
||||
- **Prompt-/command-wiring:** `transform-prompt.md` (46/101/105), `commands/kb-update.md` (130 — BÅDE Port 2 born-verified OG Port 3-kadens-inngang), `commands/generate-skills.md` (139/143/305/307) byttet `judge-claim-prompt-v2.md → -v3.1.md` + `judge-v2 → judge-v3.1`. `generate-skills.md`: kun kirurgisk judge-ref (Cosmo-heading urørt — «gjøres sist» per [[cosmo-persona-deprecated]]). Suite **641/641** (kun 2 eksisterende tester flippet, ingen lagt til).
|
||||
- **Restgap (guard-minor-nit, §8-oppfølging):** stempel-guarden (`transform.mjs:165`) honorerer *integer*-override men ikke en minor-bærende streng-override (`judgeVersion:'3.2'` faller tilbake til default pga `Number.isInteger`). Harmløst — pipelinen bruker alltid default ('3.1'); en fremtidig judge-revisjon som vil *overstyre* til en minor må generalisere guarden til en version-label-regex. Logget her, ikke lukket (utenfor G2-scope; ingen failing behov i dag).
|
||||
- **G6 🟢 LUKKET (2026-07-18) — Layer A aktivert + live-verifisert (Enhet B); begge lag i drift.** Layer B var lukket 2026-07-04 (gate) + 2026-07-18 (baseline-adjudikering, Enhet A2 `af6c31c`); dette lukker Layer A og dermed hele G6.
|
||||
- **Aktivering (kirurgisk):** PostToolUse-hook i `~/.claude/settings.json` med matcher `mcp__microsoft-learn__.*` → direkte node-kall mot `../llm-security/hooks/scripts/post-mcp-verify.mjs` (ikke global plugin-reaktivering — minst mulig flate). Backup av settings i scratchpad; rollback = fjern blokken. Edit/Write mot settings er pathguard-blokkert → endringen gjort via Bash/python3 med jq-validering (STATE-mandatert, se «Åpne spørsmål» i STATE for sanksjonert rute fremover).
|
||||
- **Live-verifikasjon avdekket ekte defekt (verifikasjonens verdi bevist):** hooken leste `tool_output` fra stdin, men live hook-protokollen sender **`tool_response`** — hooken kjørte grønt og skannet INGENTING (stille no-op). Bevist ved stdin-nøkkel-dump i live fyring. Fiks i llm-security (TDD, 3 nye tester, 73/73): `tool_response ?? tool_output`. Bevis etter fiks: volum-state-filen viser `microsoft_docs_fetch: 8252` = eksakt resp_len av testfetchen. Fiksen landet i llm-securitys `44aa390` (v7.8.3; parallell release-økt feide de stagete filene med i sin commit — multisession-race, innhold komplett).
|
||||
- **Håndhevelses-kjede nå komplett:** Layer A (post-mcp-verify på all MS Learn-fetch-output, foreground obligatorisk per mandat) + Layer B (scan-adversarial-content ved skrive-chokepunktet, 75-entry adjudikert allowlist) + commit-gate (`pre-commit-scan.mjs` git-hook-installert som `.git/hooks/pre-commit`-symlink `372a922`, e2e-testet: ren→exit 0, forgiftet staget fil→BLOCK exit 1). NB: symlinken er per-klon — fersk klon må re-installere (dokumentert i skriptets header).
|
||||
- **Kjent søsken-defekt (utenfor scope, llm-security-økt):** `post-session-guard.mjs` leser trolig også `tool_output` — samme defektklasse, ikke fikset her. Suite 942/942 exit 0.
|
||||
|
|
|
|||
|
|
@ -168,7 +168,7 @@ Fase 0 ✅ lukket; gaten sa **BYGG Fase 3 (scoped)**. Gjenstående arbeid har **
|
|||
|
||||
**S-Cosmo — Cosmo-utfasing (GODKJENT, gjøres SIST).** Fjern persona helt. Absorbér i samme pass: 1a-C (44+3 redirects), advisor-filenes frontmatter/backfill (fra S2) + TOC (fra SH). LES `docs/cosmo-removal-brief-2026-06.md`. Aldri ny Cosmo-innhold. **Gate:** 0 Cosmo-referanser igjen; advisor judged etter Cosmo.
|
||||
|
||||
**S-OKF — OKF auto-inbox-pipeline (LAV PRIO).** Skjul OKF fra bruker. `docs/okf-second-brain-brief-2026-06.md`.
|
||||
**S-OKF — OKF auto-inbox-pipeline (LAV PRIO, EKSTERNT BLOKKERT).** Skjul OKF fra bruker. `docs/okf-second-brain-brief-2026-06.md`. **Tooling bygges IKKE her** — leveres av `~/repos/llm-ingestion-okf` fase 4 (`node/`, vendres per plugin); vi er greenfield-konsument og akseptanse-skisse for API-flaten. **Gate:** avklaring B2 (guard-distribusjonskanal, operatørbeslutning) → fase 2 → 3 → 4 levert `node/` → sign-off-kjedens punkt 5. Ikke start før den kjeden er grønn. Sikkerhet eies alltid av `llm-ingestion-guard`, aldri av dette sporet.
|
||||
|
||||
### Operatør-beslutninger låst inn av denne planen
|
||||
1. **Mål-først vs bygg-begge** → ✅ LØST av gaten: bygg-begge-og-mål på volatil populasjon (S3).
|
||||
|
|
|
|||
|
|
@ -8,6 +8,12 @@ v3 er adoptert baseline, målt **P 100,0 % / R 92,9 % / 0 FP** på G5b-korrigert
|
|||
|
||||
**Forhåndsregistrert adopsjonsgate (låst FØR fan-out):** adopter v3.1 KUN hvis den **holder P = 100 % OG løfter R over 92,9 %** (mot G5b-gull). Enhver ny FP feller den → behold v3. Rapportens egen «GATE: PASS» (R≥0,70/P≥0,60) er kun gulvet, IKKE adopsjonsbaren.
|
||||
|
||||
## Forutsetning 0 (R7-gate — FØR alt annet)
|
||||
|
||||
Denne kjøringen fetcher untrusted innhold (MS Learn → judge-subagenter). **Layer A-aktiveringsprotokollen `docs/ingestion-security-brief-2026-07.md` §5b (pkt. 1–5) MÅ være grønn før første fetch** — den er en hard R7-gate, ikke bare en aktiveringsregel. Kort: aktiver `post-mcp-verify`-hooken, verifiser at den fyrer på ÉN foreground-fetch (ellers kompenserende foreground-skann), og bekreft at `llm-security`-substratet løser (ellers fail-closer Layer B og BLOCKer alt).
|
||||
|
||||
**Untrusted-data-ramme (hvorfor judgen IKKE fences):** det fetchede innholdet prompt-fences **ikke** inn i den frosne v3.1-judgen. Å legge fencing rundt `<FILE>`/`<CLAIMS>` (som ligger *inne i* den frosne malen — det finnes intet skall utenfor judgens synsfelt) ville endre teksten judgen prosesserer = en judge-bump (Non-goal §7, ville kreve re-måling av P/R). De kompenserende kontrollene er i stedet: (a) **foreground-fetch obligatorisk** (lar Layer A-hooken + operatørens øye se innholdet før judgen), og (b) **Layer B ved commit-grensen** — `scripts/kb-update/pre-commit-scan.mjs` kjører den deterministiske adversarial-skannen over stagede `skills/**/*.md` og BLOCKer commit på BLOCK/WARN, uansett modell-adferd. Judgen forblir frosset.
|
||||
|
||||
## Forutsetninger (verifiser FØRST — premiss-sjekk)
|
||||
|
||||
```bash
|
||||
|
|
|
|||
|
|
@ -20,6 +20,7 @@ import {
|
|||
FREE_CONTEXT_FILE,
|
||||
buildOrgSummary,
|
||||
} from '../../scripts/kb-update/lib/user-data.mjs';
|
||||
import { loadAiActDeadlines } from '../../scripts/kb-update/lib/ai-act-deadlines.mjs';
|
||||
|
||||
const pluginRoot = process.env.CLAUDE_PLUGIN_ROOT || join(process.cwd());
|
||||
const cwd = process.cwd();
|
||||
|
|
@ -93,20 +94,12 @@ if (shouldRunDetection(scheduleConfig, lastPollDaysAgo).run) {
|
|||
}
|
||||
}
|
||||
|
||||
// --- 3. Check EU AI Act deadlines ---
|
||||
const AI_ACT_DEADLINES = [
|
||||
// NB: Digital Omnibus (prov. enighet 2026-05-07) utsatte høyrisiko; datoer foreløpige til OJ-publisering.
|
||||
{ date: new Date('2025-02-02'), label: 'Forbudte AI-praksiser (Art. 5)' },
|
||||
{ date: new Date('2025-08-02'), label: 'GPAI-krav + governance/sanksjoner (Art. 99)' },
|
||||
{ date: new Date('2026-08-02'), label: 'Transparens Art. 50 (syntetisk innhold)' },
|
||||
{ date: new Date('2026-12-02'), label: 'Art. 50(2) merking — frist for eksisterende generative systemer (Omnibus)' },
|
||||
{ date: new Date('2027-12-02'), label: 'Annex III høyrisiko (provisorisk, Omnibus — utsatt fra 2026-08-02)' },
|
||||
{ date: new Date('2028-08-02'), label: 'Annex I høyrisiko innebygd (provisorisk, Omnibus)' },
|
||||
];
|
||||
// --- 3. Check EU AI Act deadlines (single source: scripts/kb-update/data/ai-act-deadlines.json) ---
|
||||
const aiActSource = loadAiActDeadlines();
|
||||
|
||||
let nearestDeadline = null;
|
||||
for (const dl of AI_ACT_DEADLINES) {
|
||||
const daysLeft = Math.ceil((dl.date.getTime() - now) / DAY_MS);
|
||||
for (const dl of aiActSource ? aiActSource.deadlines : []) {
|
||||
const daysLeft = Math.ceil((new Date(dl.date).getTime() - now) / DAY_MS);
|
||||
if (daysLeft > 0 && daysLeft <= 180) {
|
||||
if (!nearestDeadline || daysLeft < nearestDeadline.daysLeft) {
|
||||
nearestDeadline = { ...dl, daysLeft };
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@
|
|||
|
||||
import { readdirSync, statSync, existsSync } from 'node:fs';
|
||||
import { join } from 'node:path';
|
||||
import { loadAiActDeadlines } from '../../scripts/kb-update/lib/ai-act-deadlines.mjs';
|
||||
|
||||
const cwd = process.cwd();
|
||||
const workDir = join(cwd, '.work');
|
||||
|
|
@ -60,12 +61,18 @@ const suggestions = [
|
|||
'/architect:summary — lag beslutningsnotat',
|
||||
];
|
||||
|
||||
// Add AI Act suggestion if deadline is within 180 days
|
||||
// Add AI Act suggestion if the nearest deadline (from the shared source) is within 180 days
|
||||
const DAY_MS = 24 * 60 * 60 * 1000;
|
||||
const gpaiDeadline = new Date('2026-08-02');
|
||||
const daysToGpai = Math.ceil((gpaiDeadline.getTime() - now) / DAY_MS);
|
||||
if (daysToGpai > 0 && daysToGpai <= 180) {
|
||||
suggestions.push(`/architect:classify — EU AI Act-klassifisering (${daysToGpai}d til GPAI-frist)`);
|
||||
const aiActSource = loadAiActDeadlines();
|
||||
let nearestDeadline = null;
|
||||
for (const dl of aiActSource ? aiActSource.deadlines : []) {
|
||||
const daysLeft = Math.ceil((new Date(dl.date).getTime() - now) / DAY_MS);
|
||||
if (daysLeft > 0 && daysLeft <= 180 && (!nearestDeadline || daysLeft < nearestDeadline.daysLeft)) {
|
||||
nearestDeadline = { ...dl, daysLeft };
|
||||
}
|
||||
}
|
||||
if (nearestDeadline) {
|
||||
suggestions.push(`/architect:classify — EU AI Act-klassifisering (${nearestDeadline.daysLeft}d til ${nearestDeadline.label})`);
|
||||
}
|
||||
|
||||
const sessionList = recentSessions.join(', ');
|
||||
|
|
|
|||
|
|
@ -246,11 +246,11 @@
|
|||
"reports": {
|
||||
"classify": {
|
||||
"input": {},
|
||||
"raw_markdown": "# EU AI Act — Klassifisering: Acme Kunde-chatbot\n\nSystem: Acme Kunde-chatbot (Acme Kommune)\nBeskrivelse: AI-system som identifiserer objekter som krever oppfølging via sensordata + objektregister\n\n## Risikonivå\n\nRisk-level: høy\n\n## Rolle\n\nRolle: Provider og Deployer (utvikler internt + drifter selv)\n\n## Begrunnelse\n\nReasoning: Systemet brukes av offentlig myndighet for håndheving av lov, og påvirker individers rettigheter direkte gjennom automatisert beslutningsstøtte for håndtering. Dette plasserer systemet under Annex III, punkt 6 (rettshåndhevelse) og krever full høyrisiko-compliance per Art. 6(2).\n\n## Forpliktelser\n\n- Risk management system per Art. 9\n- Data governance og -kvalitet per Art. 10\n- Teknisk dokumentasjon per Art. 11\n- Logging og sporbarhet per Art. 12\n- Transparens overfor deployer per Art. 13\n- Menneskelig oversikt per Art. 14\n- Robusthet, sikkerhet og nøyaktighet per Art. 15\n- FRIA (Fundamental Rights Impact Assessment) per Art. 27 — obligatorisk for offentlig sektor\n- Registrering i EU-database per Art. 49\n- Conformity assessment per Art. 43\n\n## Frist\n\nFull compliance innen 2027-08-02 (Annex III høyrisiko full compliance).\n"
|
||||
"raw_markdown": "# EU AI Act — Klassifisering: Acme Kunde-chatbot\n\nSystem: Acme Kunde-chatbot (Acme Kommune)\nBeskrivelse: AI-system som identifiserer objekter som krever oppfølging via sensordata + objektregister\n\n## Risikonivå\n\nRisk-level: høy\n\n## Rolle\n\nRolle: Provider og Deployer (utvikler internt + drifter selv)\n\n## Begrunnelse\n\nReasoning: Systemet brukes av offentlig myndighet for håndheving av lov, og påvirker individers rettigheter direkte gjennom automatisert beslutningsstøtte for håndtering. Dette plasserer systemet under Annex III, punkt 6 (rettshåndhevelse) og krever full høyrisiko-compliance per Art. 6(2).\n\n## Forpliktelser\n\n- Risk management system per Art. 9\n- Data governance og -kvalitet per Art. 10\n- Teknisk dokumentasjon per Art. 11\n- Logging og sporbarhet per Art. 12\n- Transparens overfor deployer per Art. 13\n- Menneskelig oversikt per Art. 14\n- Robusthet, sikkerhet og nøyaktighet per Art. 15\n- FRIA (Fundamental Rights Impact Assessment) per Art. 27 — obligatorisk for offentlig sektor\n- Registrering i EU-database per Art. 49\n- Conformity assessment per Art. 43\n\n## Frist\n\nFull compliance innen 2027-12-02 (Annex III høyrisiko full compliance, utsatt fra 2026-08-02).\n"
|
||||
},
|
||||
"requirements": {
|
||||
"input": {},
|
||||
"raw_markdown": "# EU AI Act — Krav for høyrisiko provider+deployer\n\nSystem: Acme Kunde-chatbot (Acme Kommune)\nKlassifisering: høy risiko, rolle Provider+Deployer\n\n## Krav\n\n| Krav | Status | Kilde |\n|------|--------|-------|\n| Risk Management System etablert og dokumentert | partial | Art. 9 |\n| Treningsdata-governance med kvalitetssjekker | met | Art. 10 |\n| Teknisk dokumentasjon (Annex IV) komplett | partial | Art. 11 |\n| Automatisk logging av hendelser implementert | met | Art. 12 |\n| Transparens-instruksjoner for deployer skrevet | missing | Art. 13 |\n| Human-in-the-loop på alle sanksjonsavgjørelser | met | Art. 14 |\n| Nøyaktighetsmål med stratifisert testing | partial | Art. 15 |\n| Cybersikkerhetstiltak verifisert (NSM Grunnprinsipper) | met | Art. 15 |\n| FRIA gjennomført før idriftsettelse | missing | Art. 27 |\n| Registrering i EU-database planlagt | missing | Art. 49 |\n| Conformity assessment per Annex VI gjennomført | missing | Art. 43 |\n| CE-merking utført før markedsføring | missing | Art. 48 |\n| Post-market monitoring system etablert | partial | Art. 72 |\n| Avviksrapportering til myndigheter rutinert | partial | Art. 73 |\n\n## Sammendrag\n\n- 4 krav er møtt (met)\n- 4 krav er delvis møtt (partial)\n- 6 krav mangler implementering (missing)\n\nPrioritering: FRIA og transparens-instruksjoner må adresseres før idriftsettelse 2027-08-02.\n"
|
||||
"raw_markdown": "# EU AI Act — Krav for høyrisiko provider+deployer\n\nSystem: Acme Kunde-chatbot (Acme Kommune)\nKlassifisering: høy risiko, rolle Provider+Deployer\n\n## Krav\n\n| Krav | Status | Kilde |\n|------|--------|-------|\n| Risk Management System etablert og dokumentert | partial | Art. 9 |\n| Treningsdata-governance med kvalitetssjekker | met | Art. 10 |\n| Teknisk dokumentasjon (Annex IV) komplett | partial | Art. 11 |\n| Automatisk logging av hendelser implementert | met | Art. 12 |\n| Transparens-instruksjoner for deployer skrevet | missing | Art. 13 |\n| Human-in-the-loop på alle sanksjonsavgjørelser | met | Art. 14 |\n| Nøyaktighetsmål med stratifisert testing | partial | Art. 15 |\n| Cybersikkerhetstiltak verifisert (NSM Grunnprinsipper) | met | Art. 15 |\n| FRIA gjennomført før idriftsettelse | missing | Art. 27 |\n| Registrering i EU-database planlagt | missing | Art. 49 |\n| Conformity assessment per Annex VI gjennomført | missing | Art. 43 |\n| CE-merking utført før markedsføring | missing | Art. 48 |\n| Post-market monitoring system etablert | partial | Art. 72 |\n| Avviksrapportering til myndigheter rutinert | partial | Art. 73 |\n\n## Sammendrag\n\n- 4 krav er møtt (met)\n- 4 krav er delvis møtt (partial)\n- 6 krav mangler implementering (missing)\n\nPrioritering: FRIA og transparens-instruksjoner må adresseres før idriftsettelse 2027-12-02.\n"
|
||||
},
|
||||
"transparency": {
|
||||
"input": {},
|
||||
|
|
@ -262,7 +262,7 @@
|
|||
},
|
||||
"conformity": {
|
||||
"input": {},
|
||||
"raw_markdown": "# Samsvarsvurdering (Art. 43) — Acme Kunde-chatbot\n\nSystem: Acme Kunde-chatbot (Acme Kommune)\nVurderingsprosedyre: Annex VI (intern kontroll)\n\n## Sjekkliste\n\n| Krav | Status | Bevis |\n|------|--------|-------|\n| Risk Management System dokumentert | bestått | RMS-rapport v2.1 (2026-04-15) |\n| Treningsdata-governance med kvalitetskriterier | bestått | Data-governance handbook §4.2 |\n| Teknisk dokumentasjon Annex IV komplett | betinget | Mangler ytelsesmål per stratum |\n| Logging av hendelser implementert | bestått | OpenTelemetry-spans i Azure Monitor |\n| Transparens-instruksjoner skrevet | avvist | Skal leveres innen 2026-09-01 |\n| Menneskelig oversikt på saksbehandler | bestått | Workflow-design godkjent av juridisk |\n| Nøyaktighetsmål dokumentert | betinget | 96.3% overall, men ikke per objekt-ID-region |\n| Robusthet under adversarielle forhold | betinget | Test-suite mangler skitne plater og natt-scenarier |\n| Cybersikkerhetstiltak per Art. 15 | bestått | NSM Grunnprinsipper-vurdering bestått |\n| Conformity assessment underskrevet | avvist | Avhengig av FRIA-resultat |\n| EU declaration of conformity utstedt | avvist | Avhenger av Art. 47 |\n| CE-merking påført | avvist | Markedsplassering ikke aktuell (intern bruk) — vurder om Art. 48 gjelder |\n\n## Frister\n\n| Dato | Milepæl | Status |\n|------|---------|--------|\n| 2026-08-02 | GPAI-krav + Annex III høyrisiko | upcoming |\n| 2026-09-01 | Transparens-instruksjoner ferdigstilt | upcoming |\n| 2027-02-01 | FRIA og DPIA-revisjon | upcoming |\n| 2027-08-02 | Full Annex III høyrisiko-compliance | upcoming |\n\n## Konklusjon\n\n5 av 12 krav er fullt møtt; 4 er delvis møtt; 3 mangler implementering. Critical path: transparens-instruksjoner (Art. 13) blokkerer conformity declaration.\n"
|
||||
"raw_markdown": "# Samsvarsvurdering (Art. 43) — Acme Kunde-chatbot\n\nSystem: Acme Kunde-chatbot (Acme Kommune)\nVurderingsprosedyre: Annex VI (intern kontroll)\n\n## Sjekkliste\n\n| Krav | Status | Bevis |\n|------|--------|-------|\n| Risk Management System dokumentert | bestått | RMS-rapport v2.1 (2026-04-15) |\n| Treningsdata-governance med kvalitetskriterier | bestått | Data-governance handbook §4.2 |\n| Teknisk dokumentasjon Annex IV komplett | betinget | Mangler ytelsesmål per stratum |\n| Logging av hendelser implementert | bestått | OpenTelemetry-spans i Azure Monitor |\n| Transparens-instruksjoner skrevet | avvist | Skal leveres innen 2026-09-01 |\n| Menneskelig oversikt på saksbehandler | bestått | Workflow-design godkjent av juridisk |\n| Nøyaktighetsmål dokumentert | betinget | 96.3% overall, men ikke per objekt-ID-region |\n| Robusthet under adversarielle forhold | betinget | Test-suite mangler skitne plater og natt-scenarier |\n| Cybersikkerhetstiltak per Art. 15 | bestått | NSM Grunnprinsipper-vurdering bestått |\n| Conformity assessment underskrevet | avvist | Avhengig av FRIA-resultat |\n| EU declaration of conformity utstedt | avvist | Avhenger av Art. 47 |\n| CE-merking påført | avvist | Markedsplassering ikke aktuell (intern bruk) — vurder om Art. 48 gjelder |\n\n## Frister\n\n| Dato | Milepæl | Status |\n|------|---------|--------|\n| 2026-08-02 | Transparens (Art. 50) | upcoming |\n| 2026-09-01 | Transparens-instruksjoner ferdigstilt | upcoming |\n| 2027-02-01 | FRIA og DPIA-revisjon | upcoming |\n| 2027-12-02 | Full Annex III høyrisiko-compliance (utsatt fra 2026-08-02) | upcoming |\n\n## Konklusjon\n\n5 av 12 krav er fullt møtt; 4 er delvis møtt; 3 mangler implementering. Critical path: transparens-instruksjoner (Art. 13) blokkerer conformity declaration.\n"
|
||||
},
|
||||
"dpia": {
|
||||
"input": {},
|
||||
|
|
@ -278,7 +278,7 @@
|
|||
},
|
||||
"review": {
|
||||
"input": {},
|
||||
"raw_markdown": "# Arkitekturgjennomgang — Acme Kunde-chatbot\n\nSystem: Acme Kunde-chatbot (Acme Kommune)\nVurderingsdato: 2026-04-30\nReviewers: AI-arkitekt, sikkerhetsarkitekt, Datatilsynet\n\n## Funn\n\n| ID | Severity | Status | Lokasjon | Anbefaling |\n|----|----------|--------|----------|------------|\n| F-01 | critical | remove | Authentication layer | Tilgang til AI-forklaringer mangler attribute-based access control — alle saksbehandler ser alle saker. Implementer ABAC basert på sak-tildeling. |\n| F-02 | high | review | Data pipeline | Treningsdata oppdateres månedlig, men ingen formell drift-deteksjon. Etabler statistisk drift-monitoring i Azure Monitor. |\n| F-03 | high | review | Model serving | Modellen serves fra en enkelt regional endpoint uten failover. Replikér til en sekundær region for RTO < 1t. |\n| F-04 | high | review | Logging | Audit-logg lagres 30 dager — under arkivlovens krav for sak-relevant info. Endre retensjon til 7 år for sak-knyttede oppslag. |\n| F-05 | medium | keep | Cost management | Ingen budsjettalarmer på Azure AI Services — prediction-kostnaden kan øke med 4× ved belastnings-topper uten varsel. |\n| F-06 | medium | review | Compliance | FRIA-rapport ikke vedlikeholdt etter modell-endring 2026-03-12. Re-evaluering trengs. |\n| F-07 | medium | keep | UX | saksbehandler-grensesnitt viser ikke konfidensgrad tydelig nok — risiko for over-trust på AI-output. |\n| F-08 | low | suppressed | Documentation | README mangler oppdatert arkitekturdiagram (siste fra 2025-11). |\n| F-09 | low | suppressed | Testing | Manglende E2E-test for utenlandske objekt-ID. |\n\n## Sammendrag\n\nCritical (1): ABAC mangler — må fikses før idriftsettelse.\nHigh (3): Drift-deteksjon, failover, logg-retensjon — må fikses innen 6 mnd.\nMedium (3): Budsjett, FRIA-revisjon, UX-konfidens — bør fikses innen 12 mnd.\nLow (2): Dokumentasjon, testing — opportunity-quality.\n\n## Anbefaling\n\nIdriftsettelse anbefales IKKE før F-01 er løst. F-02 til F-04 må adresseres innen 2026-09-01 for å holde 2027-08-02-fristen.\n"
|
||||
"raw_markdown": "# Arkitekturgjennomgang — Acme Kunde-chatbot\n\nSystem: Acme Kunde-chatbot (Acme Kommune)\nVurderingsdato: 2026-04-30\nReviewers: AI-arkitekt, sikkerhetsarkitekt, Datatilsynet\n\n## Funn\n\n| ID | Severity | Status | Lokasjon | Anbefaling |\n|----|----------|--------|----------|------------|\n| F-01 | critical | remove | Authentication layer | Tilgang til AI-forklaringer mangler attribute-based access control — alle saksbehandler ser alle saker. Implementer ABAC basert på sak-tildeling. |\n| F-02 | high | review | Data pipeline | Treningsdata oppdateres månedlig, men ingen formell drift-deteksjon. Etabler statistisk drift-monitoring i Azure Monitor. |\n| F-03 | high | review | Model serving | Modellen serves fra en enkelt regional endpoint uten failover. Replikér til en sekundær region for RTO < 1t. |\n| F-04 | high | review | Logging | Audit-logg lagres 30 dager — under arkivlovens krav for sak-relevant info. Endre retensjon til 7 år for sak-knyttede oppslag. |\n| F-05 | medium | keep | Cost management | Ingen budsjettalarmer på Azure AI Services — prediction-kostnaden kan øke med 4× ved belastnings-topper uten varsel. |\n| F-06 | medium | review | Compliance | FRIA-rapport ikke vedlikeholdt etter modell-endring 2026-03-12. Re-evaluering trengs. |\n| F-07 | medium | keep | UX | saksbehandler-grensesnitt viser ikke konfidensgrad tydelig nok — risiko for over-trust på AI-output. |\n| F-08 | low | suppressed | Documentation | README mangler oppdatert arkitekturdiagram (siste fra 2025-11). |\n| F-09 | low | suppressed | Testing | Manglende E2E-test for utenlandske objekt-ID. |\n\n## Sammendrag\n\nCritical (1): ABAC mangler — må fikses før idriftsettelse.\nHigh (3): Drift-deteksjon, failover, logg-retensjon — må fikses innen 6 mnd.\nMedium (3): Budsjett, FRIA-revisjon, UX-konfidens — bør fikses innen 12 mnd.\nLow (2): Dokumentasjon, testing — opportunity-quality.\n\n## Anbefaling\n\nIdriftsettelse anbefales IKKE før F-01 er løst. F-02 til F-04 må adresseres innen 2026-09-01 for å holde 2027-12-02-fristen.\n"
|
||||
},
|
||||
"cost": {
|
||||
"input": {},
|
||||
|
|
@ -294,11 +294,11 @@
|
|||
},
|
||||
"adr": {
|
||||
"input": {},
|
||||
"raw_markdown": "# ADR-001 — Velg Azure AI Foundry som primær AI-plattform for Acme Kunde-chatbot\n\nStatus: accepted\nDate: 2026-04-30\nDeciders: AI-arkitekt, sikkerhetsarkitekt, seksjonsleder\nConsulted: Datatilsynet, juridisk rådgiver, Drift\nInformed: prosjekteierskap, AI-teamet\n\n## Context and Problem Statement\n\nAcme Kommune skal modernisere Acme Kunde-chatbot fra on-prem OCR-løsning til skybasert AI-plattform. Plattformen må støtte custom modell-trening, audit-logging på inferens-nivå, real-time inferens (<100ms P95), og full compliance med EU AI Act + GDPR + sikkerhetsloven.\n\n## Decision Drivers\n\n- Compliance med EU AI Act høyrisiko-krav (Art. 9-15)\n- Norsk dataresidens-krav\n- Customer-managed keys og Private Endpoints\n- Custom modell-trening kapabilitet\n- Total cost of ownership over 3 år\n- Driftbarhet for AI-teamet\n\n## Considered Options\n\n1. **Azure AI Foundry** — Enterprise AI-plattform med full compliance-pakke\n2. **Azure ML + AKS** — Mer kontroll, men høyere driftskost\n3. **AWS SageMaker** — Konkurransedyktig, men mangler norske compliance-sertifiseringer\n4. **On-prem GPU-cluster** — Maks kontroll, men krever betydelig CapEx og driftskompetanse\n\n## Decision Outcome\n\nChosen option: **Azure AI Foundry**, fordi det balanserer compliance, driftbarhet, og fleksibilitet best for vår bemanning og tidsramme.\n\n### Consequences\n\n- Good: full compliance-pakke for leverandøren, raskere time-to-prod, integrert med eksisterende Entra ID\n- Good: customer-managed keys og Customer Lockbox tilgjengelig\n- Bad: lock-in til Azure, men mitigert via standardiserte modell-formater (ONNX) og data-portabilitet\n- Bad: høyere månedlig kostnad enn ren Azure ML — kompenseres ved redusert egen-drift\n\n## Validation\n\nBeslutning evalueres etter 12 måneder mot KPI-er:\n- Saksbehandlingstid (mål: -40%)\n- Modell-nøyaktighet (mål: ≥96% F1)\n- Total cost (mål: ≤ NOK 1.7M/år)\n- Compliance-status (mål: 100% av krav dekket innen 2027-08-02)\n\n## More Information\n\n- Compare-rapport: see `compare-foundry-vs-aml.md`\n- Cost-analyse: see `cost-tco-3year.md`\n- Security-vurdering: see `security-foundry-baseline.md`\n"
|
||||
"raw_markdown": "# ADR-001 — Velg Azure AI Foundry som primær AI-plattform for Acme Kunde-chatbot\n\nStatus: accepted\nDate: 2026-04-30\nDeciders: AI-arkitekt, sikkerhetsarkitekt, seksjonsleder\nConsulted: Datatilsynet, juridisk rådgiver, Drift\nInformed: prosjekteierskap, AI-teamet\n\n## Context and Problem Statement\n\nAcme Kommune skal modernisere Acme Kunde-chatbot fra on-prem OCR-løsning til skybasert AI-plattform. Plattformen må støtte custom modell-trening, audit-logging på inferens-nivå, real-time inferens (<100ms P95), og full compliance med EU AI Act + GDPR + sikkerhetsloven.\n\n## Decision Drivers\n\n- Compliance med EU AI Act høyrisiko-krav (Art. 9-15)\n- Norsk dataresidens-krav\n- Customer-managed keys og Private Endpoints\n- Custom modell-trening kapabilitet\n- Total cost of ownership over 3 år\n- Driftbarhet for AI-teamet\n\n## Considered Options\n\n1. **Azure AI Foundry** — Enterprise AI-plattform med full compliance-pakke\n2. **Azure ML + AKS** — Mer kontroll, men høyere driftskost\n3. **AWS SageMaker** — Konkurransedyktig, men mangler norske compliance-sertifiseringer\n4. **On-prem GPU-cluster** — Maks kontroll, men krever betydelig CapEx og driftskompetanse\n\n## Decision Outcome\n\nChosen option: **Azure AI Foundry**, fordi det balanserer compliance, driftbarhet, og fleksibilitet best for vår bemanning og tidsramme.\n\n### Consequences\n\n- Good: full compliance-pakke for leverandøren, raskere time-to-prod, integrert med eksisterende Entra ID\n- Good: customer-managed keys og Customer Lockbox tilgjengelig\n- Bad: lock-in til Azure, men mitigert via standardiserte modell-formater (ONNX) og data-portabilitet\n- Bad: høyere månedlig kostnad enn ren Azure ML — kompenseres ved redusert egen-drift\n\n## Validation\n\nBeslutning evalueres etter 12 måneder mot KPI-er:\n- Saksbehandlingstid (mål: -40%)\n- Modell-nøyaktighet (mål: ≥96% F1)\n- Total cost (mål: ≤ NOK 1.7M/år)\n- Compliance-status (mål: 100% av krav dekket innen 2027-12-02)\n\n## More Information\n\n- Compare-rapport: see `compare-foundry-vs-aml.md`\n- Cost-analyse: see `cost-tco-3year.md`\n- Security-vurdering: see `security-foundry-baseline.md`\n"
|
||||
},
|
||||
"summary": {
|
||||
"input": {},
|
||||
"raw_markdown": "# Beslutningsnotat — Acme Kunde-chatbot\n\nSystem: Acme Kunde-chatbot (Acme Kommune)\nDato: 2026-04-30\nTil: Direktør for Digital og IT\nFra: AI-teamet\n\n## Verdict\n\nVerdict: warning\nSub: Pilot anbefalt med betingelser\n\n## Rationale\n\nArkitekturen er teknisk solid og økonomisk forsvarlig (P50 NOK 1.7M/år), men compliance-arbeidet ligger 6 måneder bak ideell tidslinje. Pilot kan starte etter at FRIA og transparens-instruksjoner er ferdigstilt; full produksjonssetting krever lukking av alle critical funn fra arkitekturgjennomgang.\n\n## Key Metrics\n\n| Metric | Verdi | Mål |\n|--------|-------|-----|\n| Compliance-dekning | 33% (4/12 fullt møtt) | 100% innen 2027-08-02 |\n| Sikkerhetsscore | 22/30 (73%) | ≥27/30 (90%) |\n| TCO 3 år | NOK 6.7M | ≤ NOK 7M |\n| Saksbehandlingstid (pilot) | -32% (estimert) | -40% |\n| ROS-restrisiko | medium | low-medium |\n\n## Next Steps\n\n- Lukk F-01 (ABAC) innen 2026-06-15\n- Gjennomfør FRIA innen 2026-07-15 (Art. 27-frist)\n- Produksjonsdokumentere transparens-instruksjoner innen 2026-09-01\n- Pilot 3 regioner (Oslo, Bergen, Trondheim) Q4 2026\n- Full utrulling Q2 2027\n\n## Restrisiko\n\nEtter foreslåtte tiltak: medium. Hovedeksponering: bias mot utenlandske objekt-ID krever løpende monitoring.\n\n## Anbefaling\n\nGodkjenn pilot-fase med tydelig stage-gate til full produksjonssetting. Avstem med Datatilsynet før fase 4.\n"
|
||||
"raw_markdown": "# Beslutningsnotat — Acme Kunde-chatbot\n\nSystem: Acme Kunde-chatbot (Acme Kommune)\nDato: 2026-04-30\nTil: Direktør for Digital og IT\nFra: AI-teamet\n\n## Verdict\n\nVerdict: warning\nSub: Pilot anbefalt med betingelser\n\n## Rationale\n\nArkitekturen er teknisk solid og økonomisk forsvarlig (P50 NOK 1.7M/år), men compliance-arbeidet ligger 6 måneder bak ideell tidslinje. Pilot kan starte etter at FRIA og transparens-instruksjoner er ferdigstilt; full produksjonssetting krever lukking av alle critical funn fra arkitekturgjennomgang.\n\n## Key Metrics\n\n| Metric | Verdi | Mål |\n|--------|-------|-----|\n| Compliance-dekning | 33% (4/12 fullt møtt) | 100% innen 2027-12-02 |\n| Sikkerhetsscore | 22/30 (73%) | ≥27/30 (90%) |\n| TCO 3 år | NOK 6.7M | ≤ NOK 7M |\n| Saksbehandlingstid (pilot) | -32% (estimert) | -40% |\n| ROS-restrisiko | medium | low-medium |\n\n## Next Steps\n\n- Lukk F-01 (ABAC) innen 2026-06-15\n- Gjennomfør FRIA innen 2026-07-15 (Art. 27-frist)\n- Produksjonsdokumentere transparens-instruksjoner innen 2026-09-01\n- Pilot 3 regioner (Oslo, Bergen, Trondheim) Q4 2026\n- Full utrulling Q2 2027\n\n## Restrisiko\n\nEtter foreslåtte tiltak: medium. Hovedeksponering: bias mot utenlandske objekt-ID krever løpende monitoring.\n\n## Anbefaling\n\nGodkjenn pilot-fase med tydelig stage-gate til full produksjonssetting. Avstem med Datatilsynet før fase 4.\n"
|
||||
},
|
||||
"poc": {
|
||||
"input": {},
|
||||
|
|
|
|||
149
scripts/kb-eval/apply-o2-ratified.mjs
Normal file
149
scripts/kb-eval/apply-o2-ratified.mjs
Normal file
|
|
@ -0,0 +1,149 @@
|
|||
#!/usr/bin/env node
|
||||
// apply-o2-ratified.mjs — R11 §9.4. Applies the O2 subtractions the operator
|
||||
// ratified 2026-08-03, and only those.
|
||||
//
|
||||
// Ratified: idx 17, 19, 33 as attested; idx 14 as the REDUCED subtraction
|
||||
// (`/ SharePoint` only). Deliberately NOT ratified and therefore absent from
|
||||
// RATIFIED: idx 26 (renumbering artifact unresolved), idx 27 (cond 3 still
|
||||
// `human_must_confirm`), idx 36 (needs the out-of-envelope companion edit at
|
||||
// line 310, now owned by gap G7), idx 18 (no reduction exists — also G7).
|
||||
//
|
||||
// Every string comes from the tracked evidence in data/r11-o2-returns/, never
|
||||
// from transcription. The one amendment (idx 14) is expressed as a derivation
|
||||
// over the attested verbatim and asserts its own effect, so a drifted record
|
||||
// aborts rather than silently writing something else.
|
||||
//
|
||||
// Anchoring is on `file_text_verbatim`, NEVER on a line number: `line` differs
|
||||
// from `real_line` in 9 of 17 records, and idx 17 shifts idx 19's lines in the
|
||||
// file they share. The verbatim must occur EXACTLY once or the run aborts.
|
||||
//
|
||||
// Recovery contract: writes are crash-safe (atomicWriteSync tmp+rename — a reader
|
||||
// sees the old file or the new one, never a partial). An interrupted run is
|
||||
// recovered by re-running: an already-applied edit no longer finds its verbatim,
|
||||
// which aborts the run, writing nothing, rather than corrupting the file.
|
||||
//
|
||||
// Usage: node scripts/kb-eval/apply-o2-ratified.mjs [--dry]
|
||||
import { readFileSync, readdirSync, realpathSync } from 'node:fs';
|
||||
import { join, dirname } from 'node:path';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
|
||||
import { isDeletionOnly, novelWordForms } from './lib/o2-return-check.mjs';
|
||||
import { atomicWriteSync } from '../kb-update/lib/atomic-write.mjs';
|
||||
|
||||
const __dirname = dirname(fileURLToPath(import.meta.url));
|
||||
const PLUGIN_ROOT = join(__dirname, '..', '..');
|
||||
const RETURNS = join(PLUGIN_ROOT, 'scripts/kb-eval/data/r11-o2-returns');
|
||||
|
||||
/**
|
||||
* The ratified reduction for idx 14: delete only the `/ SharePoint` alternative.
|
||||
* `Automatically` MUST survive — line 566 of the same file restates automaticity
|
||||
* in Norwegian, so deleting it would leave a remainder the file contradicts.
|
||||
* @param {string} verbatim
|
||||
* @returns {string}
|
||||
*/
|
||||
export function reduceSharePointOnly(verbatim) {
|
||||
const out = verbatim.replace('Dataverse / SharePoint', 'Dataverse');
|
||||
if (out === verbatim) throw new Error('idx 14 reduction is a no-op — record drifted');
|
||||
if (!out.includes('Automatically add')) throw new Error('idx 14: `Automatically` must survive');
|
||||
return out;
|
||||
}
|
||||
|
||||
// Frozen manifest — the operator's ratification, 2026-08-03. `amend: null` means
|
||||
// apply the attested `proposed_remainder` byte-for-byte.
|
||||
export const RATIFIED = [
|
||||
{ idx: 17, amend: null },
|
||||
{ idx: 19, amend: null },
|
||||
{ idx: 33, amend: null },
|
||||
{ idx: 14, amend: reduceSharePointOnly },
|
||||
];
|
||||
|
||||
/**
|
||||
* The remainder actually written for a record: attested, or the ratified amendment.
|
||||
* @param {object} row
|
||||
* @param {{amend: ((v: string) => string) | null}} entry
|
||||
* @returns {string}
|
||||
*/
|
||||
export function amendedRemainder(row, entry) {
|
||||
return entry.amend ? entry.amend(row.file_text_verbatim) : row.proposed_remainder;
|
||||
}
|
||||
|
||||
/**
|
||||
* Replace the anchored block with its remainder. Pure — no I/O. Throws on any
|
||||
* condition that would make the write unsafe rather than writing something else.
|
||||
* @param {string} content
|
||||
* @param {string} verbatim
|
||||
* @param {string} remainder
|
||||
* @returns {string}
|
||||
*/
|
||||
export function applyEdit(content, verbatim, remainder) {
|
||||
const occurrences = content.split(verbatim).length - 1;
|
||||
if (occurrences !== 1) {
|
||||
throw new Error(`ABORT — anchor occurs ${occurrences} times, expected exactly 1`);
|
||||
}
|
||||
if (!isDeletionOnly(verbatim, remainder)) {
|
||||
throw new Error('ABORT — remainder is not deletion-only w.r.t. the anchor');
|
||||
}
|
||||
const novel = novelWordForms(verbatim, remainder);
|
||||
if (novel.length) {
|
||||
throw new Error(`ABORT — remainder introduces novel word form(s): ${novel.join(', ')}`);
|
||||
}
|
||||
return content.replace(verbatim, remainder);
|
||||
}
|
||||
|
||||
function loadRows() {
|
||||
return readdirSync(RETURNS).filter((f) => f.endsWith('.json')).sort()
|
||||
.flatMap((f) => JSON.parse(readFileSync(join(RETURNS, f), 'utf8')));
|
||||
}
|
||||
|
||||
export function run({ dry = false } = {}) {
|
||||
const byIdx = new Map(loadRows().map((r) => [r.idx, r]));
|
||||
|
||||
// Group by file so two edits sharing a file compose in memory and write once.
|
||||
const perFile = new Map();
|
||||
for (const entry of RATIFIED) {
|
||||
const row = byIdx.get(entry.idx);
|
||||
if (!row) throw new Error(`ABORT — no return record for idx ${entry.idx}`);
|
||||
if (row.verdict !== 'O2_CANDIDATE') {
|
||||
throw new Error(`ABORT — idx ${entry.idx} is ${row.verdict}, not an O2 candidate`);
|
||||
}
|
||||
if (!perFile.has(row.file)) perFile.set(row.file, []);
|
||||
perFile.get(row.file).push({ entry, row });
|
||||
}
|
||||
|
||||
const planned = [];
|
||||
for (const [rel, edits] of perFile) {
|
||||
const abs = join(PLUGIN_ROOT, rel);
|
||||
const before = readFileSync(abs, 'utf8');
|
||||
let out = before;
|
||||
for (const { entry, row } of edits) {
|
||||
out = applyEdit(out, row.file_text_verbatim, amendedRemainder(row, entry));
|
||||
}
|
||||
// Post-condition: every anchor is gone, and the file actually changed.
|
||||
for (const { row } of edits) {
|
||||
if (out.includes(row.file_text_verbatim)) {
|
||||
throw new Error(`ABORT — idx ${row.idx} anchor still present after edit`);
|
||||
}
|
||||
}
|
||||
if (out === before) throw new Error(`ABORT — ${rel} unchanged`);
|
||||
planned.push({ rel, abs, out, idxs: edits.map((e) => e.entry.idx) });
|
||||
}
|
||||
|
||||
console.log(`Ratified edits: ${RATIFIED.length} across ${planned.length} files`);
|
||||
for (const p of planned) console.log(` ~ ${p.rel} (idx ${p.idxs.join(', ')})`);
|
||||
if (dry) {
|
||||
console.log('\n(dry run — no writes)');
|
||||
return planned;
|
||||
}
|
||||
for (const p of planned) atomicWriteSync(p.abs, p.out);
|
||||
console.log(`\nWrote ${planned.length} files.`);
|
||||
return planned;
|
||||
}
|
||||
|
||||
const isMain = (() => {
|
||||
try {
|
||||
return realpathSync(process.argv[1]) === realpathSync(fileURLToPath(import.meta.url));
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
})();
|
||||
if (isMain) run({ dry: process.argv.includes('--dry') });
|
||||
|
|
@ -10,14 +10,15 @@
|
|||
// apples-to-apples and a fresh session can resume with one command, no improvising.
|
||||
//
|
||||
// Pure string assembly — no LLM, no network, no math. Reads:
|
||||
// data/judge-bakeoff-claims.json (blind manifest from extract-judge-claims.mjs)
|
||||
// <--claims> (default data/judge-bakeoff-claims.json — the bake-off blind manifest;
|
||||
// R7–R10 corpus batches pass their per-batch extracted claims manifest here)
|
||||
// <--prompt> (judge-claim-prompt-vN.md, with <FILE>/<CLAIMS>)
|
||||
// Writes (with --write):
|
||||
// data/<--out> (array of {file, claim_count, prompt})
|
||||
//
|
||||
// Usage:
|
||||
// node scripts/kb-eval/build-judge-payloads.mjs --prompt judge-claim-prompt-v3.1.md \
|
||||
// --out judge-bakeoff-payloads-v3.1.json [--write]
|
||||
// [--claims <path>] --out judge-bakeoff-payloads-v3.1.json [--write]
|
||||
// (default: print per-file claim counts + a sanity sample; --write persists)
|
||||
|
||||
import fs from 'node:fs';
|
||||
|
|
@ -48,7 +49,15 @@ if (!template.includes('<FILE>') || !template.includes('<CLAIMS>')) {
|
|||
process.exit(2);
|
||||
}
|
||||
|
||||
const manifest = JSON.parse(fs.readFileSync(path.join(DATA, 'judge-bakeoff-claims.json'), 'utf8'));
|
||||
const claimsFlag = flag('--claims');
|
||||
const claimsPath = claimsFlag
|
||||
? (path.isAbsolute(claimsFlag) ? claimsFlag : path.resolve(process.cwd(), claimsFlag))
|
||||
: path.join(DATA, 'judge-bakeoff-claims.json');
|
||||
if (!fs.existsSync(claimsPath)) {
|
||||
console.error(`error: claims manifest not found: ${claimsPath}`);
|
||||
process.exit(2);
|
||||
}
|
||||
const manifest = JSON.parse(fs.readFileSync(claimsPath, 'utf8'));
|
||||
const claims = manifest.claims || [];
|
||||
|
||||
// Group by file, preserving manifest order (deterministic).
|
||||
|
|
|
|||
64
scripts/kb-eval/check-g7-queue.mjs
Normal file
64
scripts/kb-eval/check-g7-queue.mjs
Normal file
|
|
@ -0,0 +1,64 @@
|
|||
#!/usr/bin/env node
|
||||
/**
|
||||
* Check the G7 review queue against the live corpus.
|
||||
*
|
||||
* Exit 0 = every open entry still anchors to real text; exit 1 = drift or a
|
||||
* schema fault. Drift is a finding, never a silent pass: an entry that stops
|
||||
* matching is exactly the case G7 exists to prevent — a real defect leaving the
|
||||
* programme unnoticed because someone edited around it.
|
||||
*
|
||||
* Read-only. Nothing in the queue is machine-appliable; resolution is a human
|
||||
* review act (form b, ratified 2026-08-03).
|
||||
*
|
||||
* node scripts/kb-eval/check-g7-queue.mjs [--json]
|
||||
*/
|
||||
import { readFileSync } from 'node:fs';
|
||||
import { validateQueue } from './lib/g7-queue.mjs';
|
||||
|
||||
const QUEUE = 'scripts/kb-eval/data/g7-review-queue.json';
|
||||
|
||||
const asJson = process.argv.includes('--json');
|
||||
|
||||
let queue;
|
||||
try {
|
||||
queue = JSON.parse(readFileSync(QUEUE, 'utf8'));
|
||||
} catch (err) {
|
||||
console.error(`cannot read ${QUEUE}: ${err.message}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const entries = queue.entries ?? [];
|
||||
const { ok, findings } = validateQueue(entries, (p) => readFileSync(p, 'utf8'));
|
||||
|
||||
const open = entries.filter((e) => e.status === 'open');
|
||||
const resolved = entries.filter((e) => e.status === 'resolved');
|
||||
|
||||
if (asJson) {
|
||||
console.log(JSON.stringify({ ok, open: open.length, resolved: resolved.length, findings }, null, 2));
|
||||
process.exit(ok ? 0 : 1);
|
||||
}
|
||||
|
||||
console.log(`G7 review queue — ${open.length} open, ${resolved.length} resolved\n`);
|
||||
|
||||
const byClass = (cls) => open.filter((e) => e.class === cls);
|
||||
for (const cls of ['multi-locator', 'replacement']) {
|
||||
const rows = byClass(cls);
|
||||
if (rows.length === 0) continue;
|
||||
console.log(` ${cls} (${rows.length}):`);
|
||||
for (const e of rows) {
|
||||
console.log(` ${e.id.padEnd(8)} ${e.file.split('/').pop()}`);
|
||||
}
|
||||
console.log();
|
||||
}
|
||||
|
||||
if (findings.length > 0) {
|
||||
console.log(`FINDINGS (${findings.length}):`);
|
||||
for (const f of findings) {
|
||||
console.log(` [${f.kind}] ${f.id ?? ''} ${f.message}`);
|
||||
}
|
||||
console.log('\nA drifted anchor means the file changed under a queued defect.');
|
||||
console.log('Re-derive the anchor from the live file — do not delete the entry.');
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
console.log('All open entries still anchor to live corpus text. exit 0');
|
||||
74
scripts/kb-eval/check-o2-returns.mjs
Normal file
74
scripts/kb-eval/check-o2-returns.mjs
Normal file
|
|
@ -0,0 +1,74 @@
|
|||
#!/usr/bin/env node
|
||||
// check-o2-returns.mjs — R11 §10 measurement #2: verify and tally the O2/O3
|
||||
// classification returns. READ-ONLY; writes nothing.
|
||||
//
|
||||
// node scripts/kb-eval/check-o2-returns.mjs
|
||||
//
|
||||
// Runs the V1/V2/V2b/V3 checks (scripts/kb-eval/lib/o2-return-check.mjs) over
|
||||
// scripts/kb-eval/data/r11-o2-returns/*.json and prints the measurement: the
|
||||
// O2/O3 split, the machine-clean candidate count, and — the actionable part —
|
||||
// WHICH of §5's three conditions forecloses each O3. The top-level split alone
|
||||
// says nothing; the blocking condition is where the decision lives.
|
||||
|
||||
import { readFileSync, readdirSync } from 'node:fs';
|
||||
import { join, dirname, resolve } from 'node:path';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
|
||||
import { checkRow } from './lib/o2-return-check.mjs';
|
||||
|
||||
const REPO = resolve(dirname(fileURLToPath(import.meta.url)), '../..');
|
||||
const RETURNS = join(REPO, 'scripts/kb-eval/data/r11-o2-returns');
|
||||
|
||||
const cache = new Map();
|
||||
const readRepoFile = (rel) => {
|
||||
if (!cache.has(rel)) cache.set(rel, readFileSync(join(REPO, rel), 'utf8'));
|
||||
return cache.get(rel);
|
||||
};
|
||||
|
||||
const files = readdirSync(RETURNS).filter((f) => f.endsWith('.json')).sort();
|
||||
const rows = files.flatMap((f) =>
|
||||
JSON.parse(readFileSync(join(RETURNS, f), 'utf8')).map((r) => ({ ...r, _batch: f })));
|
||||
|
||||
const findings = rows.flatMap((r) =>
|
||||
checkRow(r, readRepoFile).map((f) => ({ ...f, idx: r.idx, file: r.file, line: r.line, batch: r._batch })));
|
||||
|
||||
const flaggedIdx = new Set(findings.map((f) => f.idx));
|
||||
const o2 = rows.filter((r) => r.verdict === 'O2_CANDIDATE');
|
||||
const o3 = rows.filter((r) => r.verdict === 'O3');
|
||||
const clean = o2.filter((r) => !flaggedIdx.has(r.idx));
|
||||
|
||||
// Which condition forecloses O2. "human_must_confirm" counts as NOT held: the
|
||||
// point of the tri-state is that an unresolved condition is not a satisfied one.
|
||||
const held = (v) => v === true || v === 'yes';
|
||||
const blockTally = {};
|
||||
for (const r of o3) {
|
||||
const failed = [
|
||||
!held(r.cond1_strictly_less?.holds) && 'cond1',
|
||||
!held(r.cond2_remainder_not_misleading?.holds) && 'cond2',
|
||||
!held(r.cond3_nothing_confirmed_removed?.holds) && 'cond3',
|
||||
].filter(Boolean);
|
||||
const key = failed.length ? failed.join('+') : 'none-stated';
|
||||
blockTally[key] = (blockTally[key] || 0) + 1;
|
||||
}
|
||||
const cond3Blocked = o3.filter((r) => !held(r.cond3_nothing_confirmed_removed?.holds)).length;
|
||||
|
||||
const tally = (xs, key) => xs.reduce((a, x) => ({ ...a, [x[key]]: (a[x[key]] || 0) + 1 }), {});
|
||||
const pct = (n) => `${((n / rows.length) * 100).toFixed(1)} %`;
|
||||
|
||||
console.log(`R11 §10 #2 — R8 multi-part claims (pilot): ${rows.length} items from ${files.length} batches`);
|
||||
console.log(` O2_CANDIDATE ${o2.length} (${pct(o2.length)}) · O3 ${o3.length} (${pct(o3.length)})`);
|
||||
console.log(` locator_failed: ${rows.filter((r) => r.locator_failed).length}`);
|
||||
console.log(` machine-clean O2 candidates: ${clean.length}/${o2.length}`);
|
||||
console.log(` O2 confidence: ${JSON.stringify(tally(o2, 'confidence'))}`);
|
||||
console.log(`\nO3 blocking conditions: ${JSON.stringify(blockTally)}`);
|
||||
console.log(`O3 where condition 3 fails (source supplies a corrected value → swap/rewrite): ${cond3Blocked}/${o3.length}`);
|
||||
|
||||
console.log(`\nmachine findings: ${findings.length}`);
|
||||
for (const f of findings) console.log(` [${f.check}] idx=${f.idx} ${f.file}:${f.line} — ${f.detail}`);
|
||||
|
||||
console.log('\nO2 candidates (review-grade — conditions 2 and 3 are human-confirmed):');
|
||||
for (const r of o2) {
|
||||
console.log(` ${r.idx}. ${r.file}:${r.real_line ?? r.line}${flaggedIdx.has(r.idx) ? ' [MACHINE-FLAGGED]' : ''}`);
|
||||
}
|
||||
|
||||
process.exitCode = 0;
|
||||
205
scripts/kb-eval/classify-fix-ops.mjs
Normal file
205
scripts/kb-eval/classify-fix-ops.mjs
Normal file
|
|
@ -0,0 +1,205 @@
|
|||
#!/usr/bin/env node
|
||||
// classify-fix-ops.mjs — R11 pilot runner (docs/r11-tiered-fix-design.md §10).
|
||||
//
|
||||
// Runs the fix-operation classifier over the pilot population: the files
|
||||
// carrying >= 7 `not_grounded` flags, the densest available sample. Produces the
|
||||
// four §10 measurements — the O1/O3 split, the R8 breakdown, the typed abort
|
||||
// distribution, and the per-flag record needed to re-analyse without re-running.
|
||||
//
|
||||
// READ-ONLY over the corpus and the ledger. It never edits a KB file and never
|
||||
// touches judge-pass-manifest.json — §8's single-writer state is untouched. The
|
||||
// only write is its own report, and only with --write.
|
||||
//
|
||||
// Usage: node scripts/kb-eval/classify-fix-ops.mjs [--write] [--threshold N] [--examples N]
|
||||
|
||||
import fs from 'node:fs';
|
||||
import path from 'node:path';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
|
||||
import { ABORT_CODES, classifyFlag } from './lib/fix-op.mjs';
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
const REPO = path.resolve(__dirname, '..', '..');
|
||||
const DATA = path.join(__dirname, 'data');
|
||||
|
||||
const argv = process.argv.slice(2);
|
||||
const flagArg = (name, fallback) => {
|
||||
const i = argv.indexOf(name);
|
||||
return i === -1 ? fallback : Number(argv[i + 1]);
|
||||
};
|
||||
const THRESHOLD = flagArg('--threshold', 7);
|
||||
const EXAMPLES = flagArg('--examples', 3);
|
||||
|
||||
const ledger = JSON.parse(fs.readFileSync(path.join(DATA, 'judge-pass-manifest.json'), 'utf8'));
|
||||
|
||||
const ng = (rec) => (rec.flags || []).filter((f) => f.judge_verdict === 'not_grounded');
|
||||
const population = ledger.files.filter((rec) => ng(rec).length >= THRESHOLD);
|
||||
|
||||
// Two passes over the same population. The canonical one applies the context
|
||||
// condition; the §4-only pass exists purely to MEASURE what that condition
|
||||
// removes — it is never a source of proposals, because four of the six swaps it
|
||||
// admits on this population are wrong edits (see lib/fix-op.mjs).
|
||||
const items = [];
|
||||
const s4Only = [];
|
||||
for (const rec of population) {
|
||||
const text = fs.readFileSync(path.join(REPO, rec.file), 'utf8');
|
||||
for (const flag of ng(rec)) {
|
||||
const verdict = classifyFlag(flag, text);
|
||||
s4Only.push(classifyFlag(flag, text, { contextCheck: false }));
|
||||
items.push({
|
||||
id: flag.id,
|
||||
file: flag.file,
|
||||
line: flag.line,
|
||||
rule: flag.rule || '(none)',
|
||||
claim: flag.claim,
|
||||
evidence_url: flag.evidence_url,
|
||||
evidence_quote: flag.evidence_quote,
|
||||
reason: flag.reason,
|
||||
op: verdict.op,
|
||||
code: verdict.code,
|
||||
detail: verdict.detail,
|
||||
proposal: verdict.proposal,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// ------------------------------------------------------------------ measurements
|
||||
|
||||
const tally = (rows, key) =>
|
||||
rows.reduce((acc, r) => {
|
||||
const k = typeof key === 'function' ? key(r) : r[key];
|
||||
acc[k] = (acc[k] || 0) + 1;
|
||||
return acc;
|
||||
}, {});
|
||||
|
||||
const o1 = items.filter((i) => i.op === 'O1');
|
||||
const o3 = items.filter((i) => i.op === 'O3');
|
||||
const byCode = tally(o3, 'code');
|
||||
const byRule = tally(items, 'rule');
|
||||
const r8 = items.filter((i) => i.rule === 'R8');
|
||||
|
||||
// LOCATOR_MISS / LOCATOR_AMBIGUOUS are a FIXABLE engineering gap (the locator did
|
||||
// not find the value the claim asserts). Every other abort is intrinsic to the
|
||||
// flag: no swappable value, no same-type replacement in the cited quote, or a
|
||||
// claim that is not a value swap at all. The distinction is what tells the
|
||||
// operator whether more engineering would move the O1 number.
|
||||
const LOCATOR_CODES = new Set([ABORT_CODES.LOCATOR_MISS, ABORT_CODES.LOCATOR_AMBIGUOUS]);
|
||||
const locatorAborts = o3.filter((i) => LOCATOR_CODES.has(i.code)).length;
|
||||
|
||||
// §4-as-written is a NUMERIC-path measurement: it is the baseline the context
|
||||
// condition (§4a) was added against. §4b status swaps do not run through the
|
||||
// context condition at all, so counting them here would silently inflate the
|
||||
// baseline and break the comparison with the pilot's hand-verified 6.
|
||||
const isStatus = (v) => v.op === 'O1' && v.proposal.type === 'status';
|
||||
const s4O1 = s4Only.filter((v) => v.op === 'O1' && !isStatus(v)).length;
|
||||
const o1Numeric = o1.filter((i) => i.proposal.type !== 'status').length;
|
||||
|
||||
// §4b (ratified 2026-08-03): the STATUS_SYNONYM class, split into what the closed
|
||||
// synonym table proves and why the remainder still aborts. The abort REASON
|
||||
// sub-distribution is the actionable part — NO_COMPLETE_FILE_LABEL is a corpus
|
||||
// shape, SOURCE_STATUS_AMBIGUOUS is a quote shape, FILE_ALREADY_MATCHES means the
|
||||
// flag was never a status mismatch in the first place.
|
||||
const statusProposals = items.filter((i) => i.op === 'O1' && i.proposal.type === 'status');
|
||||
const statusAborts = o3.filter((i) => i.code === ABORT_CODES.STATUS_SYNONYM);
|
||||
|
||||
// O1 precision is NOT uniform across token types, and this split is the pilot's
|
||||
// operational conclusion. Hand-verified over the whole not_grounded population:
|
||||
// every iso_date swap is an `api-version=` bump in a URL or code sample and all
|
||||
// were correct; the number/version swaps mutilated identifiers instead
|
||||
// ("AI-900" -> "AI-901", "gpt-4o" -> "gpt-5.1o" twice, a Java agent DOWNgrade),
|
||||
// because a matching identifier prefix ("AI-", "gpt-") satisfies the context
|
||||
// condition while the digit is part of a name rather than a quantity.
|
||||
const O1_HAND_VERIFIED_TYPES = new Set(['iso_date']);
|
||||
const byType = tally(o1, (i) => i.proposal.type);
|
||||
const recommended = o1.filter((i) => O1_HAND_VERIFIED_TYPES.has(i.proposal.type));
|
||||
const pct = (n) => `${((n / items.length) * 100).toFixed(1)} %`;
|
||||
|
||||
const report = {
|
||||
_meta: {
|
||||
purpose:
|
||||
'R11 pilot measurement (§10): fix-operation classification over the densest not_grounded sample. Read-only — no KB file and no ledger record was written.',
|
||||
contract: 'docs/r11-tiered-fix-design.md §3/§4/§10',
|
||||
classifier: 'scripts/kb-eval/lib/fix-op.mjs (the O1 driver with writes disabled)',
|
||||
ledger: 'scripts/kb-eval/data/judge-pass-manifest.json',
|
||||
ledger_records: ledger.files.length,
|
||||
threshold: `not_grounded >= ${THRESHOLD} (source_silent excluded, per §10)`,
|
||||
generated_from: 'ledger snapshot at run time — counts are re-derived, never read from a plan',
|
||||
disclaimer_two_202s:
|
||||
"This population is 202 flags. §3's '202 flags whose claim and quote contain a numeric token' is a DIFFERENT 202, measured over the full 712-flag population. Do not conflate them.",
|
||||
},
|
||||
population: { files: population.length, flags: items.length },
|
||||
s4_as_written: {
|
||||
O1: s4O1,
|
||||
note:
|
||||
'What §4 exactly as written would admit on the NUMERIC path (§4b status swaps excluded — they never run through the context condition). NOT a source of proposals: on the >=7 pilot all 6 were hand-verified and 4 were wrong edits (unit crossing, metric crossing, two mutilated identifiers) — measured precision 2/6. Runs at other thresholds carry no hand-verification.',
|
||||
},
|
||||
status_synonym: {
|
||||
contract: '§4b — the closed synonym table, ratified 2026-08-03',
|
||||
class_total: statusProposals.length + statusAborts.length,
|
||||
proven: statusProposals.length,
|
||||
aborts: tally(statusAborts, (i) => (i.detail && i.detail.reason) || '(unspecified)'),
|
||||
hand_verified:
|
||||
THRESHOLD === 7
|
||||
? 'All 5 hand-judged 2026-08-03 (docs/r11-pilot-results.md appendix B). Four carry the source phrasing on the row\'s OWN subject and are correct. One (security-copilot-integration.md:94) harvests a "(Preview)" marker that belongs to a DIFFERENT agent in an enumerated quote — the same provenance-without-referent defect that falsified §4. Its outcome is plausibly right; its proof is not.'
|
||||
: 'hand-verification was done on the >=7 pilot only',
|
||||
applicability:
|
||||
'REVIEW-GRADE, NOT APPLY-GRADE. status is deliberately absent from o1_recommended: §4b binds the table, the completeness of the file label and the written value, and nothing about whether the source phrasing refers to the row\'s subject. A referent condition is an open operator decision.',
|
||||
},
|
||||
o1_by_type: byType,
|
||||
o1_recommended: {
|
||||
count: recommended.length,
|
||||
types: [...O1_HAND_VERIFIED_TYPES],
|
||||
note:
|
||||
'The only O1 class that survived hand-verification: iso_date, which in this corpus is always an api-version bump inside a URL or code sample. number/version proposals are NOT safe to apply — they mutilate product, model and certification identifiers.',
|
||||
},
|
||||
split: { O1: o1.length, O2: 0, O3: o3.length, O2_note: 'O2 requires operator ratification (§5); until then every non-O1 item is O3 by design.' },
|
||||
abort_codes: byCode,
|
||||
locator_aborts: { count: locatorAborts, note: 'fixable engineering gap — every other abort is intrinsic to the flag' },
|
||||
by_rule: byRule,
|
||||
r8: { total: r8.length, O1: r8.filter((i) => i.op === 'O1').length, codes: tally(r8.filter((i) => i.op === 'O3'), 'code') },
|
||||
items,
|
||||
};
|
||||
|
||||
// ---------------------------------------------------------------------- output
|
||||
|
||||
console.log(`R11 pilot — ${population.length} files / ${items.length} not_grounded flags (threshold >= ${THRESHOLD})`);
|
||||
console.log(`ledger: ${ledger.files.length} records\n`);
|
||||
console.log(`O1 (provable value swap): ${o1.length} (${pct(o1.length)})`);
|
||||
console.log(`O3 (human): ${o3.length} (${pct(o3.length)})`);
|
||||
console.log(`O2: 0 (unratified — §5)`);
|
||||
const handNote =
|
||||
THRESHOLD === 7
|
||||
? ' — all 6 hand-verified: 4 are wrong edits (unit crossing, metric crossing, two mutilated identifiers)'
|
||||
: ' (hand-verification was done on the >=7 pilot only)';
|
||||
console.log(`\n§4 as written would admit ${s4O1} on the numeric path${handNote}. Context condition removes ${s4O1 - o1Numeric}.`);
|
||||
console.log(
|
||||
`§4b status class: ${report.status_synonym.class_total} flags -> ${report.status_synonym.proven} proven, ` +
|
||||
`${JSON.stringify(report.status_synonym.aborts)} — REVIEW-grade, not applied by any driver.\n`,
|
||||
);
|
||||
console.log('abort codes:');
|
||||
for (const [code, n] of Object.entries(byCode).sort((a, b) => b[1] - a[1])) {
|
||||
console.log(` ${code.padEnd(20)} ${String(n).padStart(4)} ${pct(n)}`);
|
||||
}
|
||||
console.log(`\nO1 by token type: ${JSON.stringify(byType)}`);
|
||||
console.log(`O1 hand-verified-safe class (iso_date / api-version): ${recommended.length} — the rest mutilate identifiers, do NOT apply`);
|
||||
console.log(`\nlocator aborts (fixable): ${locatorAborts} intrinsic aborts: ${o3.length - locatorAborts}`);
|
||||
console.log(`\nrule distribution: ${JSON.stringify(byRule)}`);
|
||||
console.log(`R8: ${r8.length} flags — O1 ${report.r8.O1}, aborts ${JSON.stringify(report.r8.codes)}`);
|
||||
|
||||
if (EXAMPLES > 0 && o1.length > 0) {
|
||||
console.log(`\n--- ${Math.min(EXAMPLES, o1.length)} proven O1 proposals ---`);
|
||||
for (const i of o1.slice(0, EXAMPLES)) {
|
||||
console.log(`\n${i.file}:${i.proposal.line} [${i.rule}] ${i.token || i.proposal.token} -> ${i.proposal.replacement}`);
|
||||
console.log(` - ${i.proposal.before}`);
|
||||
console.log(` + ${i.proposal.after}`);
|
||||
console.log(` quote: ${i.proposal.evidence_quote.slice(0, 160)}`);
|
||||
}
|
||||
}
|
||||
|
||||
if (argv.includes('--write')) {
|
||||
const out = path.join(DATA, 'r11-pilot-classification.json');
|
||||
fs.writeFileSync(out, JSON.stringify(report, null, 2) + '\n');
|
||||
console.log(`\nwrote ${out}`);
|
||||
} else {
|
||||
console.log('\n(dry run — pass --write to persist r11-pilot-classification.json)');
|
||||
}
|
||||
1053
scripts/kb-eval/data/g7-review-queue.json
Normal file
1053
scripts/kb-eval/data/g7-review-queue.json
Normal file
File diff suppressed because one or more lines are too long
14479
scripts/kb-eval/data/judge-pass-manifest.json
Normal file
14479
scripts/kb-eval/data/judge-pass-manifest.json
Normal file
File diff suppressed because it is too large
Load diff
575
scripts/kb-eval/data/r11-footer-class-2026-08-11.json
Normal file
575
scripts/kb-eval/data/r11-footer-class-2026-08-11.json
Normal file
|
|
@ -0,0 +1,575 @@
|
|||
{
|
||||
"_meta": {
|
||||
"measured": "2026-08-11",
|
||||
"population": "skills/**/references/**/*.md (389 files)",
|
||||
"class": "a stated count of MCP tool calls made while researching THIS document",
|
||||
"buckets": {
|
||||
"konsistent": "total stated + breakdown stated + they agree under the only reasonable reading",
|
||||
"inkonsistent": "total stated + breakdown stated + they disagree under every reasonable reading",
|
||||
"tvetydig": "total stated + breakdown stated, verdict flips between two defensible readings",
|
||||
"ikke_sjekkbar": "no internal cross-check exists (bare total, breakdown without total, or a component with no number)"
|
||||
},
|
||||
"flags": "orthogonal to bucket: provenance_mix, label_value_mismatch, prose_repeat, direction_stated_gt_enumerated"
|
||||
},
|
||||
"members": [
|
||||
{
|
||||
"f": "skills/ms-ai-advisor/references/copilot-extensibility/adaptive-cards-copilot-responses.md",
|
||||
"l": 517,
|
||||
"a": "**MCP calls:** 3 docs_search, 2 docs_fetch, 1 code_sample_search",
|
||||
"b": "ikke_sjekkbar",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-advisor/references/copilot-extensibility/copilot-api-rate-limiting-resilience.md",
|
||||
"l": 496,
|
||||
"a": "**MCP Calls:** 6 (3 searches, 2 fetches, 1 code sample search)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-advisor/references/copilot-extensibility/declarative-agents-grounding-strategies.md",
|
||||
"l": 461,
|
||||
"a": "**MCP-kall:** 7 (3 search, 3 fetch, 1 code sample search)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-advisor/references/copilot-extensibility/enterprise-governance-copilot-deployment.md",
|
||||
"l": 920,
|
||||
"a": "- 3 microsoft_docs_search calls",
|
||||
"b": "ikke_sjekkbar",
|
||||
"flags": [],
|
||||
"form": "block",
|
||||
"span": [
|
||||
919,
|
||||
922
|
||||
]
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-advisor/references/copilot-extensibility/microsoft-graph-api-copilot-integration.md",
|
||||
"l": 544,
|
||||
"a": "**MCP calls:** 7 (3x docs_search, 2x docs_fetch, 1x code_sample_search, 1x ToolSearch)",
|
||||
"b": "tvetydig",
|
||||
"flags": [
|
||||
"provenance_mix"
|
||||
],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-advisor/references/copilot-extensibility/sharepoint-copilot-agents.md",
|
||||
"l": 356,
|
||||
"a": "**MCP-calls:** 5 (3 search + 2 fetch).",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-advisor/references/copilot-extensibility/teams-copilot-message-extensions.md",
|
||||
"l": 469,
|
||||
"a": "**MCP-kall utført:** 6 (3 search, 2 fetch, 1 code sample search)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-advisor/references/prompt-engineering/chain-of-thought-prompting.md",
|
||||
"l": 500,
|
||||
"a": "**Totalt:** 4 MCP-kall, 3 unike Microsoft Learn-kilder.",
|
||||
"b": "inkonsistent",
|
||||
"flags": [
|
||||
"direction_stated_gt_enumerated"
|
||||
],
|
||||
"form": "block",
|
||||
"span": [
|
||||
494,
|
||||
500
|
||||
],
|
||||
"stated": 4,
|
||||
"enumerated": 3,
|
||||
"delta": 1
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-advisor/references/prompt-engineering/domain-specific-prompt-optimization.md",
|
||||
"l": 589,
|
||||
"a": "- **MCP-søk** — 3 søk mot microsoft-learn (2026-02-04)",
|
||||
"b": "ikke_sjekkbar",
|
||||
"flags": [],
|
||||
"form": "block",
|
||||
"span": [
|
||||
587,
|
||||
591
|
||||
]
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-advisor/references/prompt-engineering/multi-turn-conversation-management.md",
|
||||
"l": 683,
|
||||
"a": "**MCP calls:** 5 (search + fetch)",
|
||||
"b": "ikke_sjekkbar",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-advisor/references/prompt-engineering/prompt-testing-and-evaluation.md",
|
||||
"l": 1080,
|
||||
"a": "**MCP Calls:** 3 (microsoft_docs_search × 2, microsoft_docs_fetch × 2, microsoft_code_sample_search × 1)",
|
||||
"b": "inkonsistent",
|
||||
"flags": [],
|
||||
"form": "line",
|
||||
"stated": 3,
|
||||
"enumerated": 5,
|
||||
"delta": -2
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/agent-orchestration/agent-evaluation-testing-frameworks.md",
|
||||
"l": 563,
|
||||
"a": "- **MCP calls:** 3 (microsoft_docs_search) + 2 (microsoft_docs_fetch) + 1 (microsoft_code_sample_search) = 6",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/agent-orchestration/agent-memory-and-context-management.md",
|
||||
"l": 535,
|
||||
"a": "**MCP calls**: 6 (3x microsoft_docs_search, 2x microsoft_docs_fetch, 1x microsoft_code_sample_search)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/agent-orchestration/foundry-workflows-visual-orchestration.md",
|
||||
"l": 651,
|
||||
"a": "**MCP calls**: 4 (2x docs_search, 2x docs_fetch)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/agent-orchestration/multi-agent-orchestration-patterns.md",
|
||||
"l": 720,
|
||||
"a": "**Total MCP calls:** 6 (3 microsoft_docs_search + 2 microsoft_docs_fetch + 1 microsoft_code_sample_search)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/azure-ai-services/ai-services-api-best-practices.md",
|
||||
"l": 762,
|
||||
"a": "**MCP call summary:** 7 microsoft_docs_search + 4 microsoft_docs_fetch + 1 microsoft_code_sample_search = 12 total MCP calls",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/azure-ai-services/ai-services-cost-optimization.md",
|
||||
"l": 396,
|
||||
"a": "**Total MCP calls:** 6",
|
||||
"b": "ikke_sjekkbar",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/azure-ai-services/ai-services-governance-compliance.md",
|
||||
"l": 753,
|
||||
"a": "**Total antall MCP-kall:** 8 (4 docs_search + 4 docs_fetch)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/azure-ai-services/ai-services-networking-security.md",
|
||||
"l": 620,
|
||||
"a": "**MCP calls:** 7 (microsoft_docs_search, microsoft_docs_fetch, microsoft_code_sample_search)",
|
||||
"b": "ikke_sjekkbar",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/azure-ai-services/azure-ai-vision-image-analysis.md",
|
||||
"l": 392,
|
||||
"a": "**MCP-kall totalt:** 4 (3 docs_search + 1 code_sample_search)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/azure-ai-services/document-intelligence-prebuilt-models.md",
|
||||
"l": 565,
|
||||
"a": "**Totalt MCP-kall:** 5 (3× search, 2× fetch, 1× code samples)",
|
||||
"b": "inkonsistent",
|
||||
"flags": [],
|
||||
"form": "line",
|
||||
"stated": 5,
|
||||
"enumerated": 6,
|
||||
"delta": -1
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/azure-ai-services/speech-services-text-to-speech.md",
|
||||
"l": 538,
|
||||
"a": "**Totalt antall MCP-kall:** 7 (4 × docs_search, 3 × docs_fetch, 1 × code_sample_search)",
|
||||
"b": "inkonsistent",
|
||||
"flags": [],
|
||||
"form": "line",
|
||||
"stated": 7,
|
||||
"enumerated": 8,
|
||||
"delta": -1
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/azure-ai-services/translator-document-translation.md",
|
||||
"l": 402,
|
||||
"a": "**Total MCP calls:** 4 (docs_search) + 3 (docs_fetch) = **7**",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/mlops-genaiops/data-drift-monitoring-detection.md",
|
||||
"l": 378,
|
||||
"a": "**MCP Calls:** 5 (3 × microsoft_docs_search, 1 × microsoft_docs_fetch, 1 × microsoft_code_sample_search)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/mlops-genaiops/genaiops-llm-specific-practices.md",
|
||||
"l": 383,
|
||||
"a": "**Totalt:** 18 kilder, 8 MCP-kall.",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/mlops-genaiops/inferencing-optimization-caching.md",
|
||||
"l": 1005,
|
||||
"a": "**Total MCP-kall:** 7 (docs search) + 3 (docs fetch) + 2 (code samples) = **12**",
|
||||
"b": "konsistent",
|
||||
"flags": [
|
||||
"prose_repeat"
|
||||
],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/mlops-genaiops/infrastructure-as-code-mlops.md",
|
||||
"l": 933,
|
||||
"a": "- **microsoft_docs_search calls:** 4",
|
||||
"b": "ikke_sjekkbar",
|
||||
"flags": [],
|
||||
"form": "block",
|
||||
"span": [
|
||||
932,
|
||||
935
|
||||
]
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/mlops-genaiops/mlops-security-access-control.md",
|
||||
"l": 744,
|
||||
"a": "**MCP Calls:** 8 (microsoft-learn docs search + fetch, code samples)",
|
||||
"b": "ikke_sjekkbar",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/mlops-genaiops/mlops-teams-collaboration-tools.md",
|
||||
"l": 738,
|
||||
"a": "**Totalt antall MCP-kall:** 6 (3x search, 2x fetch, 1x code samples)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/mlops-genaiops/model-deployment-strategies-azure.md",
|
||||
"l": 1067,
|
||||
"a": "**MCP-kall utført:** 8 (microsoft_docs_search × 5, microsoft_docs_fetch × 2, microsoft_code_sample_search × 1)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/mlops-genaiops/model-versioning-registry-management.md",
|
||||
"l": 580,
|
||||
"a": "- **Total searches:** 3 (Azure ML registry, AI Foundry, MLOps lifecycle)",
|
||||
"b": "ikke_sjekkbar",
|
||||
"flags": [],
|
||||
"form": "block",
|
||||
"span": [
|
||||
579,
|
||||
582
|
||||
]
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/mlops-genaiops/responsible-ai-mlops-integration.md",
|
||||
"l": 733,
|
||||
"a": "**MCP-calls brukt:** 6 (microsoft_docs_search x 3, microsoft_docs_fetch x 2, microsoft_code_sample_search x 1)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/rag-architecture/rag-cost-optimization.md",
|
||||
"l": 558,
|
||||
"a": "**MCP calls:** 3 (search) + 2 (fetch) = 5 total",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-engineering/references/rag-architecture/rag-document-preprocessing.md",
|
||||
"l": 791,
|
||||
"a": "**Totalt antall MCP-kilder:** 3 docs_search calls + 2 docs_fetch calls = **5 MCP-kall**",
|
||||
"b": "konsistent",
|
||||
"flags": [
|
||||
"label_value_mismatch"
|
||||
],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-governance/references/monitoring-observability/real-time-streaming-monitoring.md",
|
||||
"l": 560,
|
||||
"a": "**MCP calls:** 6 (3 × search, 2 × fetch, 1 × code search)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-governance/references/monitoring-observability/response-quality-metrics-rag.md",
|
||||
"l": 637,
|
||||
"a": "**MCP research calls:** 3 (microsoft_docs_search × 3, microsoft_docs_fetch × 2, microsoft_code_sample_search × 1)",
|
||||
"b": "inkonsistent",
|
||||
"flags": [],
|
||||
"form": "line",
|
||||
"stated": 3,
|
||||
"enumerated": 6,
|
||||
"delta": -3
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-governance/references/responsible-ai/ai-act-annex-iii-checklist.md",
|
||||
"l": 523,
|
||||
"a": "- `microsoft_docs_search`: 2 queries (EU AI Act compliance, Purview AI governance)",
|
||||
"b": "ikke_sjekkbar",
|
||||
"flags": [
|
||||
"provenance_mix"
|
||||
],
|
||||
"form": "block",
|
||||
"span": [
|
||||
521,
|
||||
525
|
||||
]
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-governance/references/responsible-ai/ai-act-compliance-guide.md",
|
||||
"l": 733,
|
||||
"a": "- `microsoft_docs_search`: 3 queries (EU AI Act compliance, governance, risk classification)",
|
||||
"b": "ikke_sjekkbar",
|
||||
"flags": [
|
||||
"provenance_mix"
|
||||
],
|
||||
"form": "block",
|
||||
"span": [
|
||||
732,
|
||||
735
|
||||
]
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-governance/references/responsible-ai/ai-impact-assessment-framework.md",
|
||||
"l": 653,
|
||||
"a": "**Antall dokumenter søkt:** 4 (search queries) + 2 (deep fetch)",
|
||||
"b": "ikke_sjekkbar",
|
||||
"flags": [
|
||||
"label_value_mismatch"
|
||||
],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-governance/references/responsible-ai/ai-risk-taxonomy-classification.md",
|
||||
"l": 460,
|
||||
"a": "- **microsoft_docs_search:** 3 calls (AI risk classification, AI Act levels, Azure framework)",
|
||||
"b": "ikke_sjekkbar",
|
||||
"flags": [],
|
||||
"form": "block",
|
||||
"span": [
|
||||
458,
|
||||
463
|
||||
]
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-governance/references/responsible-ai/algorithmic-accountability-auditability.md",
|
||||
"l": 562,
|
||||
"a": "### MCP Calls: 6",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "block",
|
||||
"span": [
|
||||
562,
|
||||
565
|
||||
]
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-governance/references/responsible-ai/continuous-improvement-feedback-loops.md",
|
||||
"l": 599,
|
||||
"a": "**Total MCP calls:** 6 (3 searches + 2 fetches + 1 code sample search)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-governance/references/responsible-ai/human-in-the-loop-oversight.md",
|
||||
"l": 832,
|
||||
"a": "**MCP Calls:** 6 (3 searches + 2 fetches + 1 code sample search)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-governance/references/responsible-ai/model-monitoring-drift-detection.md",
|
||||
"l": 774,
|
||||
"a": "### Total MCP Calls: 4",
|
||||
"b": "inkonsistent",
|
||||
"flags": [],
|
||||
"form": "block",
|
||||
"span": [
|
||||
774,
|
||||
777
|
||||
],
|
||||
"stated": 4,
|
||||
"enumerated": 6,
|
||||
"delta": -2
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-governance/references/responsible-ai/responsible-ai-framework-overview.md",
|
||||
"l": 374,
|
||||
"a": "**MCP-søk utført:** 3 søk (microsoft-learn)",
|
||||
"b": "ikke_sjekkbar",
|
||||
"flags": [],
|
||||
"form": "block",
|
||||
"span": [
|
||||
374,
|
||||
375
|
||||
]
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-governance/references/responsible-ai/responsible-ai-policy-development.md",
|
||||
"l": 561,
|
||||
"a": "**MCP Calls:** 4 (microsoft_docs_search x3, microsoft_docs_fetch x2)",
|
||||
"b": "inkonsistent",
|
||||
"flags": [],
|
||||
"form": "line",
|
||||
"stated": 4,
|
||||
"enumerated": 5,
|
||||
"delta": -1
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-security/references/ai-security-engineering/ai-security-scoring-framework.md",
|
||||
"l": 514,
|
||||
"a": "**MCP calls:** 5 (3 søk + 2 fetch)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-security/references/ai-security-engineering/content-safety-filter-calibration.md",
|
||||
"l": 536,
|
||||
"a": "**MCP-kall:** 6 (3x microsoft_docs_search, 2x microsoft_docs_fetch, 1x microsoft_code_sample_search)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-security/references/ai-security-engineering/norwegian-content-safety.md",
|
||||
"l": 537,
|
||||
"a": "**MCP-kall:** 6 (microsoft_docs_search x6)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-security/references/ai-security-engineering/output-validation-grounding-verification.md",
|
||||
"l": 694,
|
||||
"a": "**MCP-kall utført:** 4 (2x docs_search, 1x code_sample_search, 2x docs_fetch)",
|
||||
"b": "inkonsistent",
|
||||
"flags": [],
|
||||
"form": "line",
|
||||
"stated": 4,
|
||||
"enumerated": 5,
|
||||
"delta": -1
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-security/references/ai-security-engineering/prompt-injection-defense-patterns.md",
|
||||
"l": 477,
|
||||
"a": "- 3 MCP microsoft-learn docs_search calls",
|
||||
"b": "ikke_sjekkbar",
|
||||
"flags": [],
|
||||
"form": "block",
|
||||
"span": [
|
||||
476,
|
||||
478
|
||||
]
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-security/references/cost-optimization/azure-ai-foundry-cost-governance.md",
|
||||
"l": 903,
|
||||
"a": "**Total MCP Calls:** 4 (3x microsoft_docs_search, 1x microsoft_docs_fetch, 1x microsoft_code_sample_search)",
|
||||
"b": "inkonsistent",
|
||||
"flags": [],
|
||||
"form": "line",
|
||||
"stated": 4,
|
||||
"enumerated": 5,
|
||||
"delta": -1
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-security/references/cost-optimization/budget-forecasting-ai-projects.md",
|
||||
"l": 529,
|
||||
"a": "**Total MCP calls:** 3 (docs_search) + 2 (docs_fetch) + 1 (code_sample_search) = 6",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-security/references/cost-optimization/inference-endpoint-cost-optimization.md",
|
||||
"l": 616,
|
||||
"a": "**Totalt MCP-kall:** 3 (microsoft_docs_search) + 2 (microsoft_docs_fetch) + 1 (microsoft_code_sample_search) = 6",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-security/references/cost-optimization/model-selection-price-performance.md",
|
||||
"l": 570,
|
||||
"a": "**MCP-kall brukt:** 6 (4x docs_search, 2x docs_fetch)",
|
||||
"b": "konsistent",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-security/references/cost-optimization/reserved-capacity-planning.md",
|
||||
"l": 557,
|
||||
"a": "**MCP Calls:** 3",
|
||||
"b": "ikke_sjekkbar",
|
||||
"flags": [],
|
||||
"form": "line"
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-security/references/cost-optimization/small-language-models-economics.md",
|
||||
"l": 644,
|
||||
"a": "**Total MCP-kall:** 4 (3x search, 2x fetch, 1x code samples)",
|
||||
"b": "inkonsistent",
|
||||
"flags": [],
|
||||
"form": "line",
|
||||
"stated": 4,
|
||||
"enumerated": 6,
|
||||
"delta": -2
|
||||
},
|
||||
{
|
||||
"f": "skills/ms-ai-security/references/cost-optimization/token-counting-optimization.md",
|
||||
"l": 611,
|
||||
"a": "**MCP Calls:** 4 (microsoft_docs_search × 3, microsoft_docs_fetch × 2, microsoft_code_sample_search × 1)",
|
||||
"b": "inkonsistent",
|
||||
"flags": [],
|
||||
"form": "line",
|
||||
"stated": 4,
|
||||
"enumerated": 6,
|
||||
"delta": -2
|
||||
}
|
||||
]
|
||||
}
|
||||
104
scripts/kb-eval/data/r11-o2-returns/batch-01.json
Normal file
104
scripts/kb-eval/data/r11-o2-returns/batch-01.json
Normal file
|
|
@ -0,0 +1,104 @@
|
|||
[
|
||||
{
|
||||
"idx": 1,
|
||||
"id": "ms-ai-engineering/agent-orchestration/foundry-agent-service-ga.md#13",
|
||||
"file": "skills/ms-ai-engineering/references/agent-orchestration/foundry-agent-service-ga.md",
|
||||
"line": 187,
|
||||
"real_line": 198,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| Verktøy | Type | Formål | Tilgjengelighet |\n|---------|------|--------|-----------------|\n| **Code Interpreter** | Action | Kjøre Python-kode i sandkasse, generere filer og visualiseringer | GA |\n| **File Search** | Knowledge | RAG over opplastede filer via Azure AI Search | GA (ikke tilgjengelig i Italy North, Brazil South) |\n| **Grounding with Bing Search** | Knowledge | Webgrunnlag via Bing | GA |\n| **Bing Custom Search** | Knowledge | Webgrunnlag begrenset til definerte domener | GA |\n| **SharePoint** | Knowledge | Tilgang til interne dokumenter via SharePoint | Preview |\n| **Azure Functions** | Action | Kalle serverless-funksjoner (synkron via MCP eller asynkron via Queue) | GA |\n| **Azure Logic Apps** | Action/Trigger | Over 1400 forhåndsbygde koblinger, event-trigget invokasjon | GA |\n| **OpenAPI tool** | Action | Kalle HTTP-endepunkter beskrevet med OpenAPI 3.0-spec | GA |\n| **MCP tool** | Action/Knowledge | Koble til MCP-servere (remote) | GA (juni 2025) |\n| **Deep Research tool** | Knowledge | Flerstegs research via o3-deep-research + Bing | GA (juni 2025) |\n| **Fabric Data Agent** | Knowledge | Chat med strukturert data i Microsoft Fabric | GA |\n| **Morningstar tool** | Knowledge | Finansdata fra Morningstar | GA |",
|
||||
"failing_part": "The table presents Deep Research (line 198, stated as 'GA (juni 2025)') and Morningstar (line 200, stated as 'GA') as current built-in tools; the source's migration table shows Deep Research as classic-only Public Preview with no equivalent in new Foundry, and Morningstar is said to be absent from the current catalog.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting the Deep Research and Morningstar rows leaves a table that asserts a strict subset of the original twelve tool claims."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "The judge's reason is that the table as a whole describes the superseded classic tool set, so the surviving ten rows would still stand under the heading 'Innebygde verktøy' as the current new-Foundry catalog with their unqualified GA labels — the same standing-implication failure the contract names for this exact file."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence_quote 'Deep Research | Yes (Public Preview) | No (Recommendation: Deep Research model with Web Search tool)' confirms the tool does exist in classic and names its new-Foundry replacement, so the source supports a corrected value rather than deletion; and the payload carries no evidence at all about Morningstar, so its removal cannot be justified without new fact-finding."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Conditions 2 and 3 both fail: the source confirms Deep Research's existence and its replacement (a value/qualifier fix, not subtraction), Morningstar's absence is unevidenced in the payload, and the classic-vs-new framing of the whole table cannot be fixed by deleting rows.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 2,
|
||||
"id": "ms-ai-engineering/agent-orchestration/foundry-agent-service-ga.md#22",
|
||||
"file": "skills/ms-ai-engineering/references/agent-orchestration/foundry-agent-service-ga.md",
|
||||
"line": 340,
|
||||
"real_line": 352,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "Foundry Agent Service er tilgjengelig i følgende Azure-regioner (per februar 2026):\n\n| Region | Status |\n|--------|--------|\n| **Norway East** | **Tilgjengelig** |\n| Sweden Central | Tilgjengelig |\n| West Europe | Tilgjengelig |\n| Germany West Central | Tilgjengelig |\n| France Central | Tilgjengelig |\n| Switzerland North | Tilgjengelig |\n| UK South | Tilgjengelig |\n| East US / East US 2 | Tilgjengelig |\n| ... (19 regioner totalt) | Se docs for full liste |",
|
||||
"failing_part": "The total region count '19 regioner totalt' on line 352; the eight individually named regions are all confirmed available.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Replacing '... (19 regioner totalt)' with a countless '... (flere regioner)' would drop the numeric assertion while keeping every surviving region claim intact."},
|
||||
"cond2_remainder_not_misleading": {"holds": "yes", "evidence": "The eight listed regions are confirmed present in the Agents column, and the retained 'Se docs for full liste' pointer prevents a reader from taking the shown rows as the complete set."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The judge states the source's Agents column is Yes for 30 regions, so the source supports a corrected value (30) for exactly the failing part — which the contract routes to O1 value swap, never subtraction."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Condition 3 fails: the source supports a corrected total (30), making this a value swap (O1) rather than a subtraction; deciding whether to swap the count, re-date the 'per februar 2026' qualifier, or generalise is a decision about what to assert.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 3,
|
||||
"id": "ms-ai-engineering/azure-ai-services/document-intelligence-prebuilt-models.md#10",
|
||||
"file": "skills/ms-ai-engineering/references/azure-ai-services/document-intelligence-prebuilt-models.md",
|
||||
"line": 391,
|
||||
"real_line": 391,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| Tier | Pris per side (USD) | Inkludert |\n|------|---------------------|-----------|\n| **Free (F0)** | $0 | 500 sider/måned, 2 sider per dokument, 20 calls/min |\n| **Standard (S0)** | $1.50 per 1000 sider (prebuilt models) | 2,000 sider per dokument, 15 TPS |",
|
||||
"failing_part": "The F0 rate limit '20 calls/min'; F0's 2 pages/document, S0's 2,000 pages and S0's 15 TPS are all confirmed, and '500 sider/måned' is neither confirmed nor contradicted by the fetched source.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting ', 20 calls/min' from the F0 row would leave the remaining page-allowance assertions untouched and assert strictly less."},
|
||||
"cond2_remainder_not_misleading": {"holds": "human_must_confirm", "evidence": "In a two-row tier table where the S0 row still states '15 TPS', an F0 row with no throughput figure can read as 'F0 has no throughput cap' — the opposite of the source's 1 transaction/second — and the unverified '500 sider/måned' would remain standing."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence_quote 'Analyze transactions Per Second limit | 1 | 15 (default value)' confirms the correct F0 value (1 TPS), so the source supports a corrected value for the failing part and the contract routes it to O1, not subtraction."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Condition 3 fails (source confirms F0 = 1 analyze transaction/second, a swap target), and condition 2 is at best unresolved because dropping the F0 throughput next to S0's stated 15 TPS implies an absent limit.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 4,
|
||||
"id": "ms-ai-engineering/azure-ai-services/document-intelligence-prebuilt-models.md#4",
|
||||
"file": "skills/ms-ai-engineering/references/azure-ai-services/document-intelligence-prebuilt-models.md",
|
||||
"line": 40,
|
||||
"real_line": 46,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| **Check** | `prebuilt-check` | Sjekkbehandling | Check number, amount, payee, date |",
|
||||
"failing_part": "The model ID string `prebuilt-check`; the source's model ID is `prebuilt-check.us`. The other six IDs in the table are confirmed.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting the Check row (or the ID cell) from the Financial Services table would assert strictly less than the current seven-model list."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "A Financial Services table without a check model implies to a reader that Document Intelligence has no prebuilt bank-check model, which the source contradicts."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence_quote 'prebuilt-check.us | ✓ | ✓' confirms the model exists under a corrected ID, and this is the exact failure case the ratified contract names for subtraction."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Conditions 2 and 3 fail: the source confirms a corrected copy-into-code SKU string (`prebuilt-check.us`), so the fix is a value swap (O1), never subtraction.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 5,
|
||||
"id": "ms-ai-engineering/azure-ai-services/document-intelligence-prebuilt-models.md#5",
|
||||
"file": "skills/ms-ai-engineering/references/azure-ai-services/document-intelligence-prebuilt-models.md",
|
||||
"line": 52,
|
||||
"real_line": 56,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| **Marriage Certificate** | `prebuilt-marriageCertificate` | Vigselattester |",
|
||||
"failing_part": "The model ID string `prebuilt-marriageCertificate`; the source's model ID is `prebuilt-marriageCertificate.us`. The ID and tax model IDs in the same table are confirmed.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting the Marriage Certificate row from the Identity & Tax table would assert strictly less than the current eight-model list."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "Removing the row leaves the Identity & Tax table implying no prebuilt marriage-certificate model exists, which the source contradicts."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence_quote 'prebuilt-marriageCertificate.us | ✓ | ✓' confirms the model under a corrected ID, so a supported value exists and subtraction would destroy true information."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Conditions 2 and 3 fail: the source supports a corrected SKU string (`prebuilt-marriageCertificate.us`), making this an O1 value swap.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 6,
|
||||
"id": "ms-ai-engineering/azure-ai-services/document-intelligence-prebuilt-models.md#6",
|
||||
"file": "skills/ms-ai-engineering/references/azure-ai-services/document-intelligence-prebuilt-models.md",
|
||||
"line": 65,
|
||||
"real_line": 71,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| **Disclosure** | `prebuilt-mortgage.us.disclosure` | Endelige lånevilkår |",
|
||||
"failing_part": "The model ID string `prebuilt-mortgage.us.disclosure`; the source's model ID is `prebuilt-mortgage.us.closingDisclosure`. The four other mortgage IDs are confirmed.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting the Disclosure row from the US Mortgage table would assert strictly less than the current five-model list."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "Removing the row leaves the US Mortgage table implying no closing-disclosure model exists, which the source contradicts."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence_quote '| Closing Disclosure | Extract closing, transaction costs, and loan details. | prebuilt-mortgage.us.closingDisclosure |' confirms the model under a corrected ID, so the supported fix is a swap, not deletion."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Conditions 2 and 3 fail: the source supports a corrected SKU string (`prebuilt-mortgage.us.closingDisclosure`), making this an O1 value swap.",
|
||||
"confidence": "high"
|
||||
}
|
||||
]
|
||||
104
scripts/kb-eval/data/r11-o2-returns/batch-02.json
Normal file
104
scripts/kb-eval/data/r11-o2-returns/batch-02.json
Normal file
|
|
@ -0,0 +1,104 @@
|
|||
[
|
||||
{
|
||||
"idx": 7,
|
||||
"id": "ms-ai-engineering/azure-ai-services/document-intelligence-prebuilt-models.md#7",
|
||||
"file": "skills/ms-ai-engineering/references/azure-ai-services/document-intelligence-prebuilt-models.md",
|
||||
"line": 77,
|
||||
"real_line": 79,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "### Grunnleggende modeller\n\n| Modell | Model ID | Formål |\n|--------|----------|--------|\n| **Read** | `prebuilt-read` | OCR: tekst, linjer, ord, språkdeteksjon |\n| **Layout** | `prebuilt-layout` | Struktur: tabeller, selection marks, seksjoner, key-value pairs (valgfritt) |\n| **General Document** | `prebuilt-document` | Key-value pairs, tabeller, selection marks fra generiske dokumenter |",
|
||||
"failing_part": "The third table row presenting `prebuilt-document` (General Document) as a current basic model — the source states the general document model is no longer supported and its capabilities live in the layout model.",
|
||||
"proposed_remainder": "### Grunnleggende modeller\n\n| Modell | Model ID | Formål |\n|--------|----------|--------|\n| **Read** | `prebuilt-read` | OCR: tekst, linjer, ord, språkdeteksjon |\n| **Layout** | `prebuilt-layout` | Struktur: tabeller, selection marks, seksjoner, key-value pairs (valgfritt) |",
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "One whole table row is deleted and nothing is added, so the table asserts the existence of two basic models instead of three."},
|
||||
"cond2_remainder_not_misleading": {"holds": "yes", "evidence": "The remaining two rows (`prebuilt-read`, `prebuilt-layout`) are both judged grounded and the deleted row is the only mention of `prebuilt-document` in the file (grep: line 79 only), so no dangling reference or implied-availability trap is left behind."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "human_must_confirm", "evidence": "The source confirms no replacement value for this cell — it says the model is no longer supported and its capabilities are in the layout model — so subtraction destroys no confirmed current fact, but a human may prefer a rewrite that explicitly records the deprecation and the redirect to layout."},
|
||||
"verdict": "O2_CANDIDATE",
|
||||
"o3_reason": null,
|
||||
"confidence": "medium"
|
||||
},
|
||||
{
|
||||
"idx": 8,
|
||||
"id": "ms-ai-engineering/mlops-genaiops/data-drift-monitoring-detection.md#14",
|
||||
"file": "skills/ms-ai-engineering/references/mlops-genaiops/data-drift-monitoring-detection.md",
|
||||
"line": 191,
|
||||
"real_line": 191,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "**Azure Machine Learning Workspace** (Verified)\nData drift monitoring krever:\n- Azure ML workspace (v2 API)\n- Compute resources (serverless Spark eller managed compute cluster)\n- Datastore for production inference data (Azure Blob Storage eller ADLS Gen2)\n- Optional: Application Insights for custom metrics logging",
|
||||
"failing_part": "Two sub-assertions: (a) `eller managed compute cluster` as an alternative compute option — the how-to page and the monitor schema require a Spark pool; and (b) the `Optional: Application Insights for custom metrics logging` prerequisite line, which is not among the v2 model-monitoring prerequisites.",
|
||||
"proposed_remainder": "**Azure Machine Learning Workspace** (Verified)\nData drift monitoring krever:\n- Azure ML workspace (v2 API)\n- Compute resources (serverless Spark)\n- Datastore for production inference data (Azure Blob Storage eller ADLS Gen2)",
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "One alternative is struck from a disjunction and one whole bullet is deleted, with no word added, so the prerequisite list asserts a strict subset of what it asserted before."},
|
||||
"cond2_remainder_not_misleading": {"holds": "yes", "evidence": "The remainder states serverless Spark as the compute requirement, which is exactly what the evidence quote supports (\"Schedule model monitoring jobs to run on serverless Spark compute pools\"), and dropping the Application Insights bullet leaves no false implication because Application Insights is still covered on its own terms at lines 202-203."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "yes", "evidence": "The judge states the source does not support `managed compute cluster` as an option and does not list Application Insights in the prerequisites, so neither deletion removes anything the source confirms."},
|
||||
"verdict": "O2_CANDIDATE",
|
||||
"o3_reason": null,
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 9,
|
||||
"id": "ms-ai-engineering/mlops-genaiops/data-drift-monitoring-detection.md#18",
|
||||
"file": "skills/ms-ai-engineering/references/mlops-genaiops/data-drift-monitoring-detection.md",
|
||||
"line": 218,
|
||||
"real_line": 218,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "**Microsoft Foundry (tidligere Azure AI Studio)** (Baseline + Verified)\nFor generative AI workloads: Microsoft Foundry har egen monitoring med observability features og generation quality metrics (groundedness, relevance, fluency). Støtter også drift detection for grounding data i RAG scenarios.",
|
||||
"failing_part": "The second sentence, `Støtter også drift detection for grounding data i RAG scenarios.` — the canonical observability page lists only Evaluation, Monitoring and Tracing, and covers `system drift` via scheduled evaluation on test datasets, not drift detection over RAG grounding data.",
|
||||
"proposed_remainder": "**Microsoft Foundry (tidligere Azure AI Studio)** (Baseline + Verified)\nFor generative AI workloads: Microsoft Foundry har egen monitoring med observability features og generation quality metrics (groundedness, relevance, fluency).",
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "A complete sentence is deleted and the surviving sentence is untouched, so the paragraph asserts strictly less."},
|
||||
"cond2_remainder_not_misleading": {"holds": "yes", "evidence": "The judge states the first sentence holds on its own (Foundry has its own gen-AI monitoring with groundedness, relevance and fluency), and it makes no claim about drift that the deletion would leave half-standing."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "human_must_confirm", "evidence": "The deleted sentence is not confirmed by the source, so nothing confirmed is lost; however the source does confirm an adjacent true fact (scheduled evaluation detects system drift) that a human may prefer to assert instead, which would make the fix a rewrite rather than a subtraction."},
|
||||
"verdict": "O2_CANDIDATE",
|
||||
"o3_reason": null,
|
||||
"confidence": "medium"
|
||||
},
|
||||
{
|
||||
"idx": 10,
|
||||
"id": "ms-ai-engineering/mlops-genaiops/data-drift-monitoring-detection.md#21",
|
||||
"file": "skills/ms-ai-engineering/references/mlops-genaiops/data-drift-monitoring-detection.md",
|
||||
"line": 395,
|
||||
"real_line": 396,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "**Setup options**:\n- **Out-of-box**: Automatically configured for Azure ML online endpoints (no configuration required)\n- **Advanced**: Custom monitoring for models deployed outside Azure ML (batch endpoints, external)\n- **Azure Event Grid integration**: Route monitoring alerts for automated response",
|
||||
"failing_part": "The `Advanced` bullet's category mapping: the source defines advanced setup as more signals, training/validation data as the reference dataset and top-N features, while models deployed outside Azure ML and batch endpoints belong to a separate setup path (\"Set up model monitoring for production data\").",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting the `Advanced` bullet outright would assert strictly less, so condition 1 is not what blocks this item."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "The heading `Setup options` frames the list as the option space, so a remainder of only `Out-of-box` plus an Event Grid bullet would imply that out-of-box is the only real setup path, which is exactly the standing-implication failure the contract warns about; and trimming only the parenthetical `(batch endpoints, external)` would leave the false `Advanced = models deployed outside Azure ML` mapping intact."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The source confirms both that an advanced setup exists (with a different meaning) and that models outside Azure ML / batch endpoints can be monitored, so removing the bullet destroys confirmed information rather than merely dropping unsupported specificity."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Conditions 2 and 3 both fail: fixing this requires deciding what `Advanced` denotes and re-splitting the option space (advanced signal configuration vs. the separate production-data setup), which is a rewrite, not a subtraction.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 11,
|
||||
"id": "ms-ai-engineering/mlops-genaiops/data-drift-monitoring-detection.md#22",
|
||||
"file": "skills/ms-ai-engineering/references/mlops-genaiops/data-drift-monitoring-detection.md",
|
||||
"line": 400,
|
||||
"real_line": 400,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "**Statistical methods used**:\n- Jensen-Shannon divergence for categorical features\n- Wasserstein distance (Earth Mover's Distance) for numerical features\n- Population Stability Index (PSI) for feature stability",
|
||||
"failing_part": "Both the metric names and the feature-type mapping: the allowed metric is `jensen_shannon_distance` (Jensen-Shannon Distance, valid for numerical and categorical features, not categorical alone) and `normalized_wasserstein_distance` (Normalized Wasserstein Distance), and the per-feature-type mapping the claim constructs does not exist in the reference.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Stripping the `for ... features` qualifiers would assert strictly less, so condition 1 alone would not block the item."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "A remainder of bare names would still read `Jensen-Shannon divergence` and `Wasserstein distance (Earth Mover's Distance)`, which are the wrong metric names, so the misleading part survives the subtraction."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence quote confirms the corrected values verbatim (`jensen_shannon_distance`, `normalized_wasserstein_distance`, `population_stability_index`), and the contract says that where the source supports a corrected value the fix is a swap, never subtraction."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Condition 3 fails outright — the source supplies corrected metric names, making this a value-swap/rewrite; condition 2 also fails because subtraction leaves the wrong names standing.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 12,
|
||||
"id": "ms-ai-engineering/mlops-genaiops/feedback-loops-continuous-improvement.md#8",
|
||||
"file": "skills/ms-ai-engineering/references/mlops-genaiops/feedback-loops-continuous-improvement.md",
|
||||
"line": 443,
|
||||
"real_line": 443,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| Komponent | Azure-tjeneste | Formål |\n|-----------|----------------|--------|\n| **Data collection** | Inference tables (managed endpoints) | Capture production inputs/outputs |\n| **Monitoring** | Model Monitor (Azure ML) | Data drift, prediction drift, performance |\n| **Alerting** | Azure Monitor Alerts | Email/webhook ved threshold breach |\n| **Retraining** | Azure ML Pipelines | Triggered retraining workflow |\n| **A/B testing** | Staging endpoints | Champion vs challenger validation |\n| **Deployment** | Managed Online Endpoints | Blue-green deployment |",
|
||||
"failing_part": "The `Azure-tjeneste` cell of the `Data collection` row: `Inference tables (managed endpoints)` — Azure ML has no inference tables (a Databricks concept); collection is done by the Azure Machine Learning Data collector, which logs to Azure Blob Storage.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": false, "evidence": "There is no subtraction that repairs this: the cell must name a service, so emptying it breaks the table and deleting the row removes a component the source confirms exists."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "Deleting the row would leave an Azure ML feedback-loop table with monitoring, alerting and retraining but no data-collection step, implying production data reaches Model Monitor by itself; the same wrong term also appears in the architecture diagram at line 296, which the subtraction would not touch."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence quote confirms the corrected value directly (\"Azure Machine Learning Data collector provides real-time logging of input and output data from models that are deployed to managed online endpoints ... stores the logged inference data in Azure blob storage\"), so the supported fix is a swap, not subtraction."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Condition 3 (and 1) fails — the source names the correct service, so this is a value swap (Inference tables -> Data collector / Azure Blob Storage), and the wrong term recurs at line 296.",
|
||||
"confidence": "high"
|
||||
}
|
||||
]
|
||||
104
scripts/kb-eval/data/r11-o2-returns/batch-03.json
Normal file
104
scripts/kb-eval/data/r11-o2-returns/batch-03.json
Normal file
|
|
@ -0,0 +1,104 @@
|
|||
[
|
||||
{
|
||||
"idx": 13,
|
||||
"id": "ms-ai-engineering/mlops-genaiops/feedback-loops-continuous-improvement.md#9",
|
||||
"file": "skills/ms-ai-engineering/references/mlops-genaiops/feedback-loops-continuous-improvement.md",
|
||||
"line": 471,
|
||||
"real_line": 473,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "### Microsoft Foundry (GenAI)\n\n**Feedback loop-komponenter:**\n\n| Komponent | Azure-tjeneste | Formål |\n|-----------|----------------|--------|\n| **Production tracing** | MLflow Tracing (Databricks) | Span-level telemetry |\n| **User feedback** | Review App | Thumbs up/down, textual feedback |\n| **LLM judges** | Agent Evaluation | Automated quality scoring |\n| **Monitoring dashboard** | Microsoft Foundry Observability | Quality trends, latency, errors |\n| **Eval datasets** | MLflow Datasets (Unity Catalog) | Versioned test sets |\n| **Red teaming** | AI Red Teaming Agent | Adversarial testing for safety |",
|
||||
"failing_part": "Three of the six rows name Databricks-MLflow components (Review App, Agent Evaluation, MLflow Datasets in Unity Catalog) as the Azure services of a table headed '### Microsoft Foundry (GenAI)'; the row 'MLflow Tracing (Databricks)' is of the same family though the judge does not name it.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting the three misattributed rows would leave a table asserting three service mappings instead of six, which is strictly less."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "The surviving table would still head a section titled 'Microsoft Foundry (GenAI)' while carrying the row 'Production tracing | MLflow Tracing (Databricks)', so a Databricks component keeps standing as a Foundry component — the exact misleading-remainder pattern condition 2 forbids."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence quote confirms Foundry itself provides Evaluation, Monitoring and Tracing as core capabilities, so deleting the 'LLM judges' and 'Production tracing' rows destroys true information whose correct fix is to name Foundry's own evaluation/tracing services (a swap), not subtraction."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Conditions 2 and 3 both fail: the remainder still attributes a Databricks component to Foundry, and the source confirms Foundry has evaluation/tracing capabilities, so correcting the table means deciding which Foundry services to name.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 14,
|
||||
"id": "ms-ai-engineering/mlops-genaiops/feedback-loops-continuous-improvement.md#11",
|
||||
"file": "skills/ms-ai-engineering/references/mlops-genaiops/feedback-loops-continuous-improvement.md",
|
||||
"line": 552,
|
||||
"real_line": 555,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| Komponent | Power Platform-tjeneste | Formål |\n|-----------|-------------------------|--------|\n| **Automated feedback collection** | Power Automate | Route low-confidence predictions til human review |\n| **Storage** | Dataverse / SharePoint | Lagre feedback data |\n| **Model improvement** | AI Builder Feedback Loop | Automatically add reviewed samples to training set |\n| **Retraining** | AI Builder | Manual/scheduled retraining |",
|
||||
"failing_part": "Two sub-assertions: 'SharePoint' as a feedback storage service, and the word 'Automatically' in 'Automatically add reviewed samples to training set' (the source requires the documents to be selected, tagged and the model retrained).",
|
||||
"proposed_remainder": "| Komponent | Power Platform-tjeneste | Formål |\n|-----------|-------------------------|--------|\n| **Automated feedback collection** | Power Automate | Route low-confidence predictions til human review |\n| **Storage** | Dataverse | Lagre feedback data |\n| **Model improvement** | AI Builder Feedback Loop | Add reviewed samples to training set |\n| **Retraining** | AI Builder | Manual/scheduled retraining |",
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Dropping the '/ SharePoint' alternative and the adverb 'Automatically' removes two assertions and adds none, so the edited rows assert strictly less."},
|
||||
"cond2_remainder_not_misleading": {"holds": "human_must_confirm", "evidence": "The edited rows themselves read correctly against the source (Dataverse storage; the feedback loop is the mechanism for adding reviewed samples before retraining), but line 566 of the same file still states 'Reviewed documents automatisk tilgjengelige i \"Feedback loop\" data source når modellen retraines', so the automaticity claim keeps standing a few lines below unless it receives the same subtraction."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "yes", "evidence": "The source names only the Dataverse 'AI Builder Feedback Loop' table and describes a select-tag-retrain flow, so neither 'SharePoint' nor the automaticity is confirmed, while the confirmed parts (Dataverse, adding reviewed samples, manual/scheduled retraining) are all retained."},
|
||||
"verdict": "O2_CANDIDATE",
|
||||
"o3_reason": null,
|
||||
"confidence": "medium"
|
||||
},
|
||||
{
|
||||
"idx": 15,
|
||||
"id": "ms-ai-engineering/mlops-genaiops/model-evaluation-frameworks.md#3",
|
||||
"file": "skills/ms-ai-engineering/references/mlops-genaiops/model-evaluation-frameworks.md",
|
||||
"line": 39,
|
||||
"real_line": 39,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| **Risk & Safety** | Self-harm, Hateful content, Violence, Sexual content, Protected material, Indirect attack | Nei | Nei (Foundry-hosted GPT-4) | Content moderation og sikkerhetsvurdering |",
|
||||
"failing_part": "The parenthetical 'Foundry-hosted GPT-4' in the 'Krever judge model?' column — the source says these evaluators run against Microsoft's hosted safety models and contrasts them explicitly with GPT-based LLM-as-judge evaluators.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting the parenthetical would leave a bare 'Nei' in the judge-model column, asserting strictly less than before."},
|
||||
"cond2_remainder_not_misleading": {"holds": "yes", "evidence": "A bare 'Nei' is exactly what the source supports, since risk & safety evaluators need no user-supplied judge model."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The source supports a corrected value for the very slot being emptied — Microsoft's hosted safety models — so per the contract the fix is a value swap ('Foundry-hosted GPT-4' to 'Microsoft-hostede sikkerhetsmodeller'), not subtraction."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Condition 3: the source states the replacement fact (hosted safety models), which makes this an O1-style swap rather than a subtraction.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 16,
|
||||
"id": "ms-ai-engineering/rag-architecture/rag-caching-optimization.md#7",
|
||||
"file": "skills/ms-ai-engineering/references/rag-architecture/rag-caching-optimization.md",
|
||||
"line": 186,
|
||||
"real_line": 188,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "**Tiers:**\n- **Premium tier** — 99.9% SLA, up to 120GB per shard\n- **Enterprise tier** — 99.99% SLA, active-active geo-replication, Flash storage support\n- **Enterprise Flash tier** — Up to 13TB cache size, 20% RAM + 80% NVMe Flash",
|
||||
"failing_part": "'Up to 13TB cache size' for the Enterprise Flash tier — the source states 300 GB – 4.5 TB.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting the capacity clause would leave the Enterprise Flash bullet asserting only the 20% RAM / 80% NVMe split, which is strictly less."},
|
||||
"cond2_remainder_not_misleading": {"holds": "yes", "evidence": "A tier bullet describing only the RAM/Flash composition carries no false standing implication, since the two sibling bullets keep their own accurate SLA and capacity statements."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence quote gives the documented Enterprise Flash range (300 GB – 4.5 TB), so the source supports a corrected value and the contract routes this to a swap rather than subtraction."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Condition 3: an explicit corrected capacity range exists in the cited source, making this an O1-style value swap.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 17,
|
||||
"id": "ms-ai-engineering/rag-architecture/rag-caching-optimization.md#12",
|
||||
"file": "skills/ms-ai-engineering/references/rag-architecture/rag-caching-optimization.md",
|
||||
"line": 253,
|
||||
"real_line": 254,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "**Score Threshold Tuning** (APIM `score-threshold` er en DISTANSE: lavere = strengere, krever høyere semantisk likhet):\n- 0.1-0.2 → Strict matching, lavere hit rate, høy relevance\n- 0.3-0.5 → Balanced, medium hit rate, god relevance\n- 0.6-0.8 → Liberal matching, høyere hit rate, noe lavere relevance",
|
||||
"failing_part": "The three-band rubric (0.1-0.2 strict / 0.3-0.5 balanced / 0.6-0.8 liberal) — undocumented, and the two upper bands contradict the source's warning that a threshold above 0.2 may lead to cache mismatch.",
|
||||
"proposed_remainder": "**Score Threshold Tuning** (APIM `score-threshold` er en DISTANSE: lavere = strengere, krever høyere semantisk likhet):",
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "The three bullet lines are deleted and the surviving heading line is kept byte-for-byte, so the passage asserts only the direction of the threshold and nothing about band values."},
|
||||
"cond2_remainder_not_misleading": {"holds": "yes", "evidence": "The retained direction statement is exactly what the source grounds ('lower values require higher semantic similarity for a match'), and the doc's own operative recommendations elsewhere (line 164 'Start med 0.15' and the policy sample at line 239 using score-threshold=\"0.15\") stay below the source's 0.2 mismatch warning; the only cosmetic residue is the now-dangling colon on the heading line, which a human may drop."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "yes", "evidence": "The source documents only a recommended starting point (0.05) and a mismatch warning above 0.2 — it confirms none of the three bands, so no confirmed information is lost, and it offers no replacement rubric that a swap could install."},
|
||||
"verdict": "O2_CANDIDATE",
|
||||
"o3_reason": null,
|
||||
"confidence": "medium"
|
||||
},
|
||||
{
|
||||
"idx": 18,
|
||||
"id": "ms-ai-engineering/rag-architecture/rag-caching-optimization.md#2",
|
||||
"file": "skills/ms-ai-engineering/references/rag-architecture/rag-caching-optimization.md",
|
||||
"line": 29,
|
||||
"real_line": 29,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "Microsoft-stakken tilbyr flere tjenester optimalisert for AI-workloads: Azure Cache for Redis (traditional og semantic caching), Azure Cosmos DB (semantic cache med vektorsøk), Azure AI Search (built-in caching av search results), og Azure API Management (semantic caching for LLM APIs). Valget av løsning avhenger av cache-type, scale-requirements, og compliance-krav.",
|
||||
"failing_part": "The list item 'Azure AI Search (built-in caching av search results)' — the source states each query operates on the current index view with no caching or snapshot of results.",
|
||||
"proposed_remainder": "Microsoft-stakken tilbyr flere tjenester optimalisert for AI-workloads: Azure Cache for Redis (traditional og semantic caching), Azure Cosmos DB (semantic cache med vektorsøk), og Azure API Management (semantic caching for LLM APIs). Valget av løsning avhenger av cache-type, scale-requirements, og compliance-krav.",
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "One of four enumerated services is dropped and the remaining sentence is otherwise untouched, so it asserts strictly less."},
|
||||
"cond2_remainder_not_misleading": {"holds": "human_must_confirm", "evidence": "The edited sentence itself is clean — the three surviving services are all genuine caching services — but the same file still carries the section '### Azure AI Search - Built-in Caching' (lines 303-318, 'Azure AI Search cacher automatisk content etter første query') and the verification row 'Azure AI Search caching | **Verified**' at line 510, so removing only the intro mention leaves the contradicted claim standing further down."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "yes", "evidence": "The cited source denies result caching outright rather than supplying a corrected value, and AI Search's only documented cache (the preview enrichment cache for skillset output in Azure Storage) is not query-result caching, so no confirmed fact is lost by removing the item from a list of RAG response caches."},
|
||||
"verdict": "O2_CANDIDATE",
|
||||
"o3_reason": null,
|
||||
"confidence": "medium"
|
||||
}
|
||||
]
|
||||
104
scripts/kb-eval/data/r11-o2-returns/batch-04.json
Normal file
104
scripts/kb-eval/data/r11-o2-returns/batch-04.json
Normal file
|
|
@ -0,0 +1,104 @@
|
|||
[
|
||||
{
|
||||
"idx": 19,
|
||||
"id": "ms-ai-engineering/rag-architecture/rag-caching-optimization.md#14",
|
||||
"file": "skills/ms-ai-engineering/references/rag-architecture/rag-caching-optimization.md",
|
||||
"line": 296,
|
||||
"real_line": 297,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "**Fordeler:**\n- Globally distributed, multi-region writes\n- Automatic indexing av vectors\n- 99.999% SLA med multi-region setup\n- Built-in TTL support",
|
||||
"failing_part": "The bullet \"Automatic indexing av vectors\" — the judge states vector indexes must be declared explicitly in the indexing policy (only at container creation) and the vector path is placed in excludedPaths, i.e. vectors are deliberately kept out of automatic indexing.",
|
||||
"proposed_remainder": "**Fordeler:**\n- Globally distributed, multi-region writes\n- 99.999% SLA med multi-region setup\n- Built-in TTL support",
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "One of four bullets is deleted and the remaining three are untouched, so the passage asserts a strict subset of what it asserted before."},
|
||||
"cond2_remainder_not_misleading": {"holds": "human_must_confirm", "evidence": "The three surviving bullets (multi-region writes, 99.999% SLA, TTL) are each grounded per the judge's reason and none of them implies anything about vector indexing, but the preceding code sample (lines 275-292) issues VectorDistance queries without showing index creation, so a human should confirm that silence about the required explicit vector index is acceptable rather than an implicit \"no setup needed\"."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "yes", "evidence": "The evidence_quote contradicts rather than corrects the deleted bullet — there is no confirmed \"advantage\" value to swap in, since the source's fact (explicit vector index in the indexing policy, vector path in excludedPaths) is the negation of the deleted assertion, not a corrected form of it."},
|
||||
"verdict": "O2_CANDIDATE",
|
||||
"o3_reason": null,
|
||||
"confidence": "medium"
|
||||
},
|
||||
{
|
||||
"idx": 20,
|
||||
"id": "ms-ai-engineering/rag-architecture/rag-caching-optimization.md#19",
|
||||
"file": "skills/ms-ai-engineering/references/rag-architecture/rag-caching-optimization.md",
|
||||
"line": 373,
|
||||
"real_line": 377,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| Tier | Size | Kapasitet | Månedskostnad (NOK) | Best For |\n|------|------|-----------|---------------------|----------|\n| Basic C0 | 250 MB | N/A (no SLA) | ~400 | Dev/Test |\n| Standard C1 | 1 GB | 2 replicas, 99.9% SLA | ~1,200 | Small production |\n| Premium P1 | 6 GB | Clustering, geo-replication | ~7,000 | Enterprise |\n| Enterprise E10 | 12 GB | Active-active, 99.99% SLA | ~25,000 | Mission-critical |\n| Enterprise Flash F300 | 345 GB | 20% RAM + 80% Flash | ~60,000 | Large-scale AI |",
|
||||
"failing_part": "The Size cell of the Enterprise Flash F300 row: \"345 GB\" (source table gives F300 = 384 GB); secondarily the Kapasitet cell of the Standard C1 row: \"2 replicas\" (Learn describes Standard as two VMs in a replicated configuration, i.e. primary + one replica).",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Emptying the Size cell of the F300 row would technically assert less, but that is not the operative test here because condition 3 already forecloses subtraction."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "A pricing table whose Size column is populated for every tier except Enterprise Flash reads as an omission the reader must fill in, and the ~60,000 NOK cost line would stand with no capacity to justify it."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence_quote positively confirms the corrected value (\"| F300 | 384 GB |\"), so the supported fix is a value swap 345 GB -> 384 GB (and correspondingly 2 replicas -> 1 replica for Standard), never deletion."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Condition 3 fails: the source confirms the corrected figure (384 GB), which makes this an O1 value swap — subtraction would destroy true information. A second independent failing sub-assertion (Standard C1 \"2 replicas\") is likewise source-corrected, not source-negated.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 21,
|
||||
"id": "ms-ai-governance/responsible-ai/content-safety-implementation.md#1",
|
||||
"file": "skills/ms-ai-governance/references/responsible-ai/content-safety-implementation.md",
|
||||
"line": 37,
|
||||
"real_line": 42,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| **Protected Material (Code)** | Oppdager kopiert kode fra public repos | LLM-generert kode | Match med source citation URL | GA |",
|
||||
"failing_part": "The Status cell \"GA\" on the Protected Material (Code) row — What's new and the current quickstart both title the feature \"(preview)\", and the Aug 2024 GA covered only Prompt Shields and Protected Material for text.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Blanking the Status cell would assert strictly less than asserting \"GA\", so condition 1 alone is not what disqualifies this item."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "This is exactly the contract's documented failure case: a row with an empty Status cell sitting in a table where every other row is labelled GA or Preview leaves Protected Material (Code) standing among generally available features, which is the misleading standing implication the judge flagged."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence_quote \"Protected material detection for code (preview)\" confirms the corrected status outright, so the supported fix is the swap GA -> Preview."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Conditions 2 and 3 both fail: the source confirms the replacement status (Preview), making this an O1 swap, and a blanked Status cell would still read as \"available\" beside the GA rows.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 22,
|
||||
"id": "ms-ai-governance/responsible-ai/content-safety-implementation.md#2",
|
||||
"file": "skills/ms-ai-governance/references/responsible-ai/content-safety-implementation.md",
|
||||
"line": 37,
|
||||
"real_line": 40,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| **Groundedness Detection** | Verifiserer at LLM-svar er grunnlagt i kildemateriale | Query + grounding sources (maks 55K tegn) | Grounded/ungrounded score | Preview |",
|
||||
"failing_part": "The Input cell's scoping of the limit: \"Query + grounding sources (maks 55K tegn)\" — the live overview caps grounding sources at 55,000 characters and text/query separately at 7,500 characters, so the 55K ceiling is misattributed to the combined input.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Dropping the parenthetical to leave \"Query + grounding sources\" would assert strictly less, since the quantitative ceiling would simply be gone."},
|
||||
"cond2_remainder_not_misleading": {"holds": "human_must_confirm", "evidence": "A blank limit in a column where every neighbouring row states a hard limit (10K tegn, 4MB, min 110 tegn) invites the reader to assume none applies, though it asserts nothing false on its own."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence_quote confirms \"Maximum length for grounding sources: 55,000 characters (per API call)\" — the 55,000 figure is true information about this feature, and the source additionally supplies the missing 7,500-character query cap, so the supported fix restates the two scoped limits rather than deleting the number."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Condition 3 fails in the contract's documented pattern (the prebuilt-check case): the number is real but mis-scoped, and the source hands over both corrected values (55,000 for sources, 7,500 for text/query), so correcting requires asserting the split, not subtracting.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 23,
|
||||
"id": "ms-ai-governance/responsible-ai/stakeholder-communication-ai-decisions.md#6",
|
||||
"file": "skills/ms-ai-governance/references/responsible-ai/stakeholder-communication-ai-decisions.md",
|
||||
"line": 374,
|
||||
"real_line": 379,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| **Causal Inference** | \"What if\" analysis for counterfactuals | Business: Inform strategy. End users: \"What can I change to get different outcome?\" |",
|
||||
"failing_part": "The Funksjon gloss on the Causal Inference row: \"'What if' analysis for counterfactuals\" — counterfactual what-if is a separate dashboard component (DiCE) while causal inference (EconML) concerns causal treatment effects; the row merges the two and the table omits counterfactual what-if as a tool in its own right.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": false, "evidence": "The Stakeholder-verdi cell of the same row carries the counterfactual framing independently (\"What can I change to get different outcome?\"), so deleting only the Funksjon gloss leaves the same conflation asserted, and deleting the whole row would remove Causal Inference, a component the source confirms exists."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "Any subtraction leaves a Responsible AI tool table that still presents six tools with counterfactual what-if absent, preserving the very omission the judge named."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence_quote confirms both components and what each does (causal inference = causal treatment effects from historical data; counterfactual what-if = what to change for a different outcome), so the source supports a corrected gloss plus an added row, not a deletion."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "All three conditions fail: fixing this requires deciding what to assert — a corrected causal-inference gloss and a separate counterfactual what-if entry — which is a rewrite, not a subtraction.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 24,
|
||||
"id": "ms-ai-governance/responsible-ai/stakeholder-communication-ai-decisions.md#9",
|
||||
"file": "skills/ms-ai-governance/references/responsible-ai/stakeholder-communication-ai-decisions.md",
|
||||
"line": 437,
|
||||
"real_line": 437,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "**Stakeholder communication features**:\n\n1. **Agent observability**: Alle agenter har unik identitet (owner, version, lifecycle status)\n - **Verdi**: Governance team kan tracke hvem som er ansvarlig for hvilke agenter\n\n2. **Centralized logging**: Key events logges til Azure Log Analytics\n - **Verdi**: Audit trail for compliance\n\n3. **Cost tracking**: Token consumption og compute usage per agent\n - **Verdi**: CFO/Finance kan allokere kostnader til avdelinger\n\n4. **User disclosure**: Agents identifiserer seg som AI (ikke menneske)\n - **Verdi**: Etisk transparency overfor sluttbrukere",
|
||||
"failing_part": "Items 1-3 under the \"### Copilot Studio\" heading: the attribution of unique agent identity to Copilot Studio (the source attributes it to Microsoft Entra Agent ID, and the inventory/registry to Agent 365), the identity field list (source tracks ownership, purpose, platform, access scope — not owner, version, lifecycle status), and the telemetry/cost mechanisms (Copilot Studio's own facilities are Application Insights telemetry and Copilot Credits analytics, not an Azure Log Analytics key-event log with token/compute per agent).",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting items 1-3 and leaving only item 4 (user disclosure) would assert strictly less, but the remaining conditions foreclose that route."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "The heading \"### Copilot Studio\" plus \"**Stakeholder communication features**\" would survive, and the governance workflow immediately below still instructs \"Assign agent identity (owner, cost center, compliance tags)\" at line 451, so the mis-attributed identity capability would keep standing in the section even after the bullets go."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence_quote confirms that agent identity, lifecycle controls and an organizational registry genuinely exist (via Entra Agent ID and Agent 365), and the judge names Copilot Studio's real equivalents (Application Insights, Copilot Credits), so the supported fix is re-attribution and substitution, not deletion."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Conditions 2 and 3 fail: correcting requires deciding what to assert (re-attributing identity/registry to Entra Agent ID and Agent 365, and replacing Log Analytics/token-compute with Application Insights and Copilot Credits), and the surrounding heading and workflow keep the wrong attribution alive after any subtraction.",
|
||||
"confidence": "high"
|
||||
}
|
||||
]
|
||||
104
scripts/kb-eval/data/r11-o2-returns/batch-05.json
Normal file
104
scripts/kb-eval/data/r11-o2-returns/batch-05.json
Normal file
|
|
@ -0,0 +1,104 @@
|
|||
[
|
||||
{
|
||||
"idx": 25,
|
||||
"id": "ms-ai-governance/responsible-ai/stakeholder-communication-ai-decisions.md#2",
|
||||
"file": "skills/ms-ai-governance/references/responsible-ai/stakeholder-communication-ai-decisions.md",
|
||||
"line": 91,
|
||||
"real_line": 94,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| Nivå | Målgruppe | Eksempel | Microsoft-verktøy |\n|------|-----------|----------|-------------------|\n| **Global explanations** | Business ledere, produkteiere | \"Hvilke faktorer påvirker lånegodkjenning generelt?\" | Azure ML Interpretability component |\n| **Local explanations** | Sluttbrukere, saksbehandlere | \"Hvorfor ble *min* lånesøknad avslått?\" | Counterfactual What-If |\n| **Cohort explanations** | Compliance, fairness officers | \"Påvirker modellen lavlønnede søkere ulikt?\" | Responsible AI Dashboard |",
|
||||
"failing_part": "The tool assignment in the 'Microsoft-verktøy' column: local explanations are attributed to 'Counterfactual What-If', whereas the source attributes both local and cohort explanations to the interpretability component itself (counterfactual what-if is a separate DiCE-based component for feature perturbations).",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting the 'Microsoft-verktøy' column (or emptying the offending cell) is pure character deletion and would leave the row asserting only the level/audience/example mapping."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "Emptying only the offending cell leaves a blank tool cell in a tool column, which stands as the false implication that no Microsoft tool produces local explanations, while deleting the whole column also erases the two mappings the evidence supports."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence_quote confirms that the interpretability views (global, local, and cohort) belong to the same dashboard/interpretability component, so the source supports a corrected value for the local row rather than its removal."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Condition 3 fails: the source supports a corrected tool value for the local-explanations row (the interpretability component / Responsible AI dashboard), which makes this a value swap or rewrite, never a subtraction; condition 2 also fails for the narrow variant.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 26,
|
||||
"id": "ms-ai-governance/responsible-ai/transparency-documentation-standards.md#4",
|
||||
"file": "skills/ms-ai-governance/references/responsible-ai/transparency-documentation-standards.md",
|
||||
"line": 114,
|
||||
"real_line": 117,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "**Komponenter i Scorecard:**\n\n1. **Model overview**: Architecture, training data, intended use\n2. **Fairness assessment**: Performance disparities across sensitive groups (gender, ethnicity, age)\n3. **Model interpretability**: Feature importance (global/local explanations)\n4. **Error analysis**: Error rates per cohort, confusion matrices\n5. **Counterfactual analysis**: What-if scenarios (e.g., \"loan approved if income +10k\")\n6. **Causal inference**: Causal vs correlational relationships i features\n7. **Data quality**: Dataset statistics, missing values, outlier analysis",
|
||||
"failing_part": "Items 4 (Error analysis) and 5 (Counterfactual analysis) are listed as Responsible AI Scorecard components, but the canonical scorecard segment enumeration does not contain them (they are dashboard components, not scorecard segments).",
|
||||
"proposed_remainder": "**Komponenter i Scorecard:**\n\n1. **Model overview**: Architecture, training data, intended use\n2. **Fairness assessment**: Performance disparities across sensitive groups (gender, ethnicity, age)\n3. **Model interpretability**: Feature importance (global/local explanations)\n6. **Causal inference**: Causal vs correlational relationships i features\n7. **Data quality**: Dataset statistics, missing values, outlier analysis",
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Two list items are deleted whole and nothing else changes, so the passage asserts five scorecard components instead of seven."},
|
||||
"cond2_remainder_not_misleading": {"holds": "human_must_confirm", "evidence": "The five surviving items all map onto canonical segments named in the judge's reason (model overview, fairness insights, top important factors, causal insights, data analysis) so no false implication stands, but the raw numbering becomes 1,2,3,6,7 (renumbering would exceed delete-only) and the human must accept that artifact."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "yes", "evidence": "The judge states plainly that neither Error analysis nor Counterfactual analysis is a scorecard component, and the evidence_quote confirms nothing about them, so no source-confirmed information is destroyed and neither is a mangled form of a canonical segment name that a swap could repair."},
|
||||
"verdict": "O2_CANDIDATE",
|
||||
"o3_reason": null,
|
||||
"confidence": "medium"
|
||||
},
|
||||
{
|
||||
"idx": 27,
|
||||
"id": "ms-ai-governance/responsible-ai/transparency-documentation-standards.md#12",
|
||||
"file": "skills/ms-ai-governance/references/responsible-ai/transparency-documentation-standards.md",
|
||||
"line": 426,
|
||||
"real_line": 426,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "**Built-in disclosures:**\n\n| Component | Disclosure |\n|-----------|------------|\n| **Chat interface** | \"Powered by AI\" badge i chat window |\n| **Generative answers** | Attribution links til source documents |\n| **Plugin actions** | Confirmation prompts før sensitive actions (send email, delete file) |\n| **Data usage** | Privacy statement link i bot settings |",
|
||||
"failing_part": "The 'Chat interface' row (a \"Powered by AI\" badge in the chat window) and the 'Plugin actions' row (confirmation prompts before sensitive actions) are asserted as built-in Copilot Studio disclosures but are absent from the canonical enumeration of built-in safety components.",
|
||||
"proposed_remainder": "**Built-in disclosures:**\n\n| Component | Disclosure |\n|-----------|------------|\n| **Generative answers** | Attribution links til source documents |\n| **Data usage** | Privacy statement link i bot settings |",
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Two full table rows are deleted and the header, separator and remaining rows are untouched, so the table asserts two built-in disclosures instead of four."},
|
||||
"cond2_remainder_not_misleading": {"holds": "yes", "evidence": "The judge confirms both surviving rows (citations for generative answers and the privacy statement link), and a shorter list of built-in disclosures carries no standing implication that the removed features are unavailable or deprecated."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "human_must_confirm", "evidence": "The evidence_quote documents only human-oversight guidance ('review AI-generated outputs and automated actions before applying them'), which is adjacent to but does not confirm a confirmation-prompt feature, so the human should confirm that dropping the 'Plugin actions' row destroys nothing the source actually establishes."},
|
||||
"verdict": "O2_CANDIDATE",
|
||||
"o3_reason": null,
|
||||
"confidence": "medium"
|
||||
},
|
||||
{
|
||||
"idx": 28,
|
||||
"id": "ms-ai-governance/responsible-ai/transparency-documentation-standards.md#3",
|
||||
"file": "skills/ms-ai-governance/references/responsible-ai/transparency-documentation-standards.md",
|
||||
"line": 82,
|
||||
"real_line": 83,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "**Microsoft implementasjon:**\n- Microsoft Foundry: Model catalog med built-in model cards for pretrained models\n- Hugging Face integration: Model cards synces automatisk\n- Custom models: Template for å generere egne model cards",
|
||||
"failing_part": "The second and third bullets — that Hugging Face model cards are synchronised automatically, and that a template exists for generating model cards for custom models — are not covered by the canonical model card enumeration.",
|
||||
"proposed_remainder": "**Microsoft implementasjon:**\n- Microsoft Foundry: Model catalog med built-in model cards for pretrained models",
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Two whole bullets are deleted with no other change, leaving only the model-catalog assertion the judge says holds."},
|
||||
"cond2_remainder_not_misleading": {"holds": "yes", "evidence": "The single surviving bullet is exactly the part the judge confirms, and a one-line 'Microsoft implementasjon' list states less without implying anything false about Hugging Face or custom models."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "human_must_confirm", "evidence": "The evidence_quote only covers what a model card contains, so it confirms neither removed bullet, but deleting the whole Hugging Face bullet also removes the bare notion of a Hugging Face integration, and the human must confirm that the integration itself is not separately source-confirmed (which would make it a narrower edit)."},
|
||||
"verdict": "O2_CANDIDATE",
|
||||
"o3_reason": null,
|
||||
"confidence": "medium"
|
||||
},
|
||||
{
|
||||
"idx": 29,
|
||||
"id": "ms-ai-infrastructure/bcdr/monitoring-alerting-failover-detection.md#5",
|
||||
"file": "skills/ms-ai-infrastructure/references/bcdr/monitoring-alerting-failover-detection.md",
|
||||
"line": 180,
|
||||
"real_line": 185,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "AzureDiagnostics\n| where ResourceProvider == \"MICROSOFT.COGNITIVESERVICES\"\n| where Category == \"RequestResponse\"\n| where TimeGenerated > ago(1h)\n| extend\n deploymentName = tostring(properties_s.modelDeploymentName),\n latencyMs = duration_s * 1000,\n statusCode = resultCode_d\n| summarize\n P50 = percentile(latencyMs, 50),\n P95 = percentile(latencyMs, 95),\n P99 = percentile(latencyMs, 99),\n SuccessRate = round(countif(statusCode < 400) * 100.0 / count(), 2),\n TotalRequests = count()\n by bin(TimeGenerated, 5m), deploymentName",
|
||||
"failing_part": "The column names duration_s and resultCode_d in the extend clause: the documented AzureDiagnostics columns for Azure OpenAI are DurationMs and ResultSignature.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": false, "evidence": "Removing the two extend assignments would also force deletion of the summarize lines that consume latencyMs and statusCode, gutting the query rather than weakening one assertion, and the surviving KQL would no longer be a runnable statement without rewording."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "A code block presented as a working KQL query that references undefined or removed identifiers stands as a broken artifact readers would copy and run, which is a worse standing implication than the original."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence_quote explicitly projects DurationMs and ResultSignature, so the source supplies the corrected values and the contract routes this to O1/O3, never subtraction."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Condition 3 fails outright (the source supports corrected column names DurationMs and ResultSignature, making this a value swap), and condition 1 fails because no delete-only edit leaves a coherent query.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 30,
|
||||
"id": "ms-ai-security/ai-security-engineering/ai-incident-response-procedures.md#6",
|
||||
"file": "skills/ms-ai-security/references/ai-security-engineering/ai-incident-response-procedures.md",
|
||||
"line": 103,
|
||||
"real_line": 103,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "workflow Isolate-CompromisedVM {\n param([string]$VMResourceId, [string]$IncidentId)\n\n $nsg = Get-AzNetworkSecurityGroup -ResourceId $VMResourceId\n Add-AzNetworkSecurityRuleConfig -NetworkSecurityGroup $nsg `\n -Name \"Block-All-Incident-$IncidentId\" `\n -Priority 100 -Access Deny -Protocol * -Direction Inbound `\n -SourceAddressPrefix * -DestinationAddressPrefix *\n Set-AzNetworkSecurityGroup -NetworkSecurityGroup $nsg",
|
||||
"failing_part": "The -ResourceId parameter on Get-AzNetworkSecurityGroup, which exists in no parameter set of that cmdlet (only -Name, -ResourceGroupName, -ExpandResource, -DefaultProfile).",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting the trailing ' -ResourceId $VMResourceId' is pure character deletion and leaves the line asserting only that Get-AzNetworkSecurityGroup is called."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "A bare Get-AzNetworkSecurityGroup returns every NSG in the subscription, so the runbook would read as isolating the named VM while actually piping a collection into a deny-all rule addition, and the now-unused $VMResourceId parameter leaves a visibly broken script."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence_quote gives the real parameter set including -Name and -ResourceGroupName, so the source supports a corrected targeting mechanism and the contract routes this to O1/O3 rather than subtraction."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Conditions 2 and 3 both fail: the delete-only remainder is an actively dangerous script that targets all NSGs, and the source supports a corrected parameterisation, so the fix requires deciding what to assert.",
|
||||
"confidence": "high"
|
||||
}
|
||||
]
|
||||
104
scripts/kb-eval/data/r11-o2-returns/batch-06.json
Normal file
104
scripts/kb-eval/data/r11-o2-returns/batch-06.json
Normal file
|
|
@ -0,0 +1,104 @@
|
|||
[
|
||||
{
|
||||
"idx": 31,
|
||||
"id": "ms-ai-security/ai-security-engineering/ai-incident-response-procedures.md#8",
|
||||
"file": "skills/ms-ai-security/references/ai-security-engineering/ai-incident-response-procedures.md",
|
||||
"line": 132,
|
||||
"real_line": 139,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "{\n \"storageAccount\": \"forensicstorage\",\n \"immutabilityPolicy\": {\n \"immutabilityPeriodSinceCreationInDays\": 2190,\n \"allowProtectedAppendWrites\": false,\n \"state\": \"Locked\"\n },\n \"legalHold\": {\n \"tags\": [\"incident-2026-02-001\", \"model-theft-investigation\"],\n \"enabled\": true\n }\n}",
|
||||
"failing_part": "The legalHold object is given a field named \"enabled\"; the Storage API's LegalHold model exposes tags and hasLegalHold, so the literal field name \"enabled\" does not exist.",
|
||||
"proposed_remainder": "{\n \"storageAccount\": \"forensicstorage\",\n \"immutabilityPolicy\": {\n \"immutabilityPeriodSinceCreationInDays\": 2190,\n \"allowProtectedAppendWrites\": false,\n \"state\": \"Locked\"\n },\n \"legalHold\": {\n \"tags\": [\"incident-2026-02-001\", \"model-theft-investigation\"]\n }\n}",
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Only characters are deleted (the trailing comma on the tags line plus the whole \"enabled\": true line), so the block asserts the same immutabilityPolicy fields but no longer asserts any boolean field on legalHold."},
|
||||
"cond2_remainder_not_misleading": {"holds": "yes", "evidence": "A legalHold object carrying only tags is exactly what the evidence describes as the operative state, since the quote says hasLegalHold is set to true by SRP whenever at least one tag exists, so nothing in the remainder implies the hold is inactive."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "human_must_confirm", "evidence": "The source confirms a neighbouring field name (hasLegalHold) which could argue for a swap rather than a deletion, but the quote also states hasLegalHold is set by SRP rather than by the caller, so writing it into a desired-state config payload would assert something the API does not accept — a human should confirm that deletion, not swap, is the right call here."},
|
||||
"verdict": "O2_CANDIDATE",
|
||||
"o3_reason": null,
|
||||
"confidence": "medium"
|
||||
},
|
||||
{
|
||||
"idx": 32,
|
||||
"id": "ms-ai-security/ai-security-engineering/ai-incident-response-procedures.md#17",
|
||||
"file": "skills/ms-ai-security/references/ai-security-engineering/ai-incident-response-procedures.md",
|
||||
"line": 474,
|
||||
"real_line": 474,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| **Defender XDR** | M365 E5 Security or E5 | Includes Defender for Endpoint, Identity, M365 |",
|
||||
"failing_part": "Two parts: the Required License cell asserting that M365 E5 Security or E5 is what Defender XDR requires, and the product name \"M365\" in the component list (the real component is Microsoft Defender for Office 365).",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": false, "evidence": "The failing text sits in a mandatory table cell under the column header Required License, so deleting it leaves an empty cell that a reader parses as a positive claim (no licence needed / unknown), not as a narrower claim."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "Even the survivable deletion of \", M365\" from the third cell would leave the licence cell standing with the contradicted E5-only requirement, which the evidence directly refutes by listing M365 E3 with the Defender Suite add-on and M365 E3 with EMS E5 as qualifying licences."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence confirms a corrected licence set and the judge names the correct product name (Microsoft Defender for Office 365), so both failing parts have supported replacement values and must be swapped, not subtracted."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Conditions 1 and 3 both fail: a mandatory table cell cannot be emptied without asserting something new, and the source supplies corrected values for both the licence list and the product name.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 33,
|
||||
"id": "ms-ai-security/ai-security-engineering/ai-threat-modeling-stride.md#11",
|
||||
"file": "skills/ms-ai-security/references/ai-security-engineering/ai-threat-modeling-stride.md",
|
||||
"line": 211,
|
||||
"real_line": 211,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "**Capabilities:** *(Verified MCP 2026-04)*\n- Automated detection of AI workloads across Azure subscriptions (via Azure Resource Graph)\n- AI security posture management: automate detection and remediation of generative AI risks\n- Security recommendations for AI models, data stores, network isolation\n- Integration with Purview for data classification, DLP og Insider Risk Management for prompt-based data exfiltration",
|
||||
"failing_part": "Two parts attributed to Defender for Cloud AI Security Posture Management that the AISPM page does not support: the discovery mechanism \"(via Azure Resource Graph)\" and the entire Purview-integration bullet.",
|
||||
"proposed_remainder": "**Capabilities:** *(Verified MCP 2026-04)*\n- Automated detection of AI workloads across Azure subscriptions\n- AI security posture management: automate detection and remediation of generative AI risks\n- Security recommendations for AI models, data stores, network isolation",
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "The parenthetical and the fourth bullet are removed by deleting characters only, leaving the surviving bullets byte-identical, so the section attributes strictly fewer capabilities to AISPM."},
|
||||
"cond2_remainder_not_misleading": {"holds": "yes", "evidence": "The surviving first bullet matches the evidence quote, which states that Defender for Cloud automatically and continuously discovers deployed AI workloads, and the remainder makes no claim at all about how discovery is implemented or about Purview."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "human_must_confirm", "evidence": "Azure Resource Graph and Purview are real tools the CAF page describes as separate from Defender for Cloud, so deleting the bullet drops content that is true of Purview itself even though it is false of AISPM — a human should confirm that relocating rather than deleting it is not required."},
|
||||
"verdict": "O2_CANDIDATE",
|
||||
"o3_reason": null,
|
||||
"confidence": "medium"
|
||||
},
|
||||
{
|
||||
"idx": 34,
|
||||
"id": "ms-ai-security/ai-security-engineering/ai-threat-modeling-stride.md#14",
|
||||
"file": "skills/ms-ai-security/references/ai-security-engineering/ai-threat-modeling-stride.md",
|
||||
"line": 281,
|
||||
"real_line": 281,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| **Microsoft Defender for Cloud (AI)** | ~$15/server/month (standard tier) | AI workload discovery, security posture management, threat detection |",
|
||||
"failing_part": "The License/Cost cell: both the plan name \"standard tier\" and the per-server unit price, since the AI capabilities come from the Defender CSPM and Defender for AI Services plans and are billed per resource and per scanned tokens respectively.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": false, "evidence": "The whole cell is the failing assertion, and blanking a cell under the column header License/Cost reads to a table user as a positive statement about price rather than as silence."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "A row that names an AI capability set in a cost table but shows no licence or cost implies the capability is free or licence-free, which the evidence contradicts by naming the Defender CSPM plan as the securing plan."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The judge's reason cites supported replacement values (Defender CSPM plus Defender for AI Services, resource-based and token-based billing capped at 75 billion tokens scanned), so the correct fix supplies a value rather than removing one."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "All three conditions fail; the cell cannot be emptied without asserting something new, and the source supports corrected plan names and billing units.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 35,
|
||||
"id": "ms-ai-security/ai-security-engineering/ai-threat-modeling-stride.md#2",
|
||||
"file": "skills/ms-ai-security/references/ai-security-engineering/ai-threat-modeling-stride.md",
|
||||
"line": 37,
|
||||
"real_line": 37,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| **Spoofing** | Neural Net Reprogramming, Malicious ML Providers | Important-Critical | Strong API authentication, access control, client-server mutual auth |",
|
||||
"failing_part": "The placement of Malicious ML Providers under Spoofing, and the Important-Critical severity band applied to it, since the source treats it as information disclosure with severity Important if data is PII and Moderate otherwise.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting \", Malicious ML Providers\" is a pure character deletion and removes one of the two threats the Spoofing row asserts."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "After the deletion the Important-Critical band stands alone against Neural Net Reprogramming, whose severity and Spoofing placement the evidence never establishes and whose treatment in the source as \"an abuse scenario\" is what the judge calls into question, so the subtraction leaves an unsupported severity assertion looking newly precise."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The source confirms Malicious ML Providers is a real threat and supplies its correct home (information disclosure) and its correct severity, and the file already has an Information Disclosure row at line 40 to receive it, so the supported fix is a relocation with a corrected severity, not a deletion."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Conditions 2 and 3 fail: the surviving severity band becomes an unsupported claim about the one remaining threat, and the source confirms both the threat and its corrected categorisation, which makes this a move/rewrite rather than a subtraction.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 36,
|
||||
"id": "ms-ai-security/ai-security-engineering/ai-threat-modeling-stride.md#3",
|
||||
"file": "skills/ms-ai-security/references/ai-security-engineering/ai-threat-modeling-stride.md",
|
||||
"line": 38,
|
||||
"real_line": 38,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| **Tampering** | Data Poisoning (targeted/indiscriminate), Backdoored Models | Critical | Training data validation, anomaly detection, RONI defense, bagging |",
|
||||
"failing_part": "The \"/indiscriminate\" qualifier, which extends the Tampering placement and the Critical severity to indiscriminate data poisoning; the source gives that variant severity Important and the traditional parallel authenticated denial of service.",
|
||||
"proposed_remainder": "| **Tampering** | Data Poisoning (targeted), Backdoored Models | Critical | Training data validation, anomaly detection, RONI defense, bagging |",
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting the eleven characters \"/indiscriminate\" narrows the row from both poisoning variants to the targeted variant only, with every other character untouched."},
|
||||
"cond2_remainder_not_misleading": {"holds": "yes", "evidence": "The judge states explicitly that targeted poisoning and Backdoored Models are in fact Critical, so the surviving row is fully supported and it makes no claim whatsoever about the indiscriminate variant."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "human_must_confirm", "evidence": "The source does confirm indiscriminate data poisoning exists with severity Important, but the row's single shared Severity cell already reads Critical for the two remaining threats, so the confirmed value cannot be swapped in place and retaining the variant would require adding a new row — a human should confirm that dropping the coverage is acceptable rather than mandating that rewrite."},
|
||||
"verdict": "O2_CANDIDATE",
|
||||
"o3_reason": null,
|
||||
"confidence": "medium"
|
||||
}
|
||||
]
|
||||
104
scripts/kb-eval/data/r11-o2-returns/batch-07.json
Normal file
104
scripts/kb-eval/data/r11-o2-returns/batch-07.json
Normal file
|
|
@ -0,0 +1,104 @@
|
|||
[
|
||||
{
|
||||
"idx": 37,
|
||||
"id": "ms-ai-security/ai-security-engineering/ai-threat-modeling-stride.md#5",
|
||||
"file": "skills/ms-ai-security/references/ai-security-engineering/ai-threat-modeling-stride.md",
|
||||
"line": 40,
|
||||
"real_line": 40,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| **Information Disclosure** | Model Inversion, Membership Inference, Model Stealing | Important-Critical | Rate limiting, access control, output obfuscation, differential privacy |",
|
||||
"failing_part": "The severity band Important-Critical asserted for all three threats, plus the placement of Membership Inference under Information Disclosure (source files it as a Data Privacy issue with no security severity, and Model Stealing is Important only in security-sensitive models, Moderate otherwise).",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting 'Membership Inference, ' from the threat cell would leave the row asserting a strict subset of the original threat list."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "Even after removing Membership Inference the remainder still asserts the severity band Important-Critical for Model Stealing, which the judge documents as Moderate outside security-sensitive models, so the surviving cell keeps an unsupported severity floor."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "human_must_confirm", "evidence": "The evidence_quote confirms Membership Inference is a real, source-documented AI threat (filed under Data Privacy), so deleting the term drops information the source does carry, merely under a different heading."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Condition 2 fails: no deletion-only edit repairs the severity cell, and correcting 'Important-Critical' requires deciding what severity to assert for the remaining threats — a rewrite, not a subtraction.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 38,
|
||||
"id": "ms-ai-security/ai-security-engineering/data-leakage-prevention-ai.md#18",
|
||||
"file": "skills/ms-ai-security/references/ai-security-engineering/data-leakage-prevention-ai.md",
|
||||
"line": 396,
|
||||
"real_line": 396,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "**Policy templates:**\n- \"DSPM for AI - Detect risky AI usage\"\n- \"DSPM for AI - Unethical behavior in AI apps\"\n- \"DSPM for AI - Protect sensitive data from Copilot processing\"",
|
||||
"failing_part": "The listing of 'DSPM for AI - Unethical behavior in AI apps' and 'DSPM for AI - Protect sensitive data from Copilot processing' as Insider Risk Management policy templates; the source assigns the former to Communication Compliance and the judge assigns the latter to DLP.",
|
||||
"proposed_remainder": "**Policy templates:**\n- \"DSPM for AI - Detect risky AI usage\"",
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting the two mis-assigned bullet lines leaves a strict subset of the original list under the same heading, asserting one template instead of three."},
|
||||
"cond2_remainder_not_misleading": {"holds": "yes", "evidence": "The surviving bullet is the one template the judge confirms is genuinely an Insider Risk Management policy, and the heading '**Policy templates:**' under section 4.3 does not claim exhaustiveness, so nothing false is left standing."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "human_must_confirm", "evidence": "The source confirms 'DSPM for AI - Unethical behavior in AI apps' exists as a Communication Compliance template, so an operator may prefer relocating both bullets to correctly-labelled sections rather than deleting them — though nothing the source confirms *about Insider Risk Management* is lost."},
|
||||
"verdict": "O2_CANDIDATE",
|
||||
"o3_reason": null,
|
||||
"confidence": "medium"
|
||||
},
|
||||
{
|
||||
"idx": 39,
|
||||
"id": "ms-ai-security/ai-security-engineering/data-leakage-prevention-ai.md#25",
|
||||
"file": "skills/ms-ai-security/references/ai-security-engineering/data-leakage-prevention-ai.md",
|
||||
"line": 606,
|
||||
"real_line": 607,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "**Viktige cmdlets:**\n- `New-DlpCompliancePolicy`: Create DLP policy\n- `New-DlpComplianceRule`: Add rule til policy\n- `Get-DlpCompliancePolicy`: List policies\n- `Set-DlpPolicy`: Update existing policy\n- `Get-Label`: List sensitivity labels med GUIDs",
|
||||
"failing_part": "The bullet '`Set-DlpPolicy`: Update existing policy' — the cmdlet is retired from the cloud-based service and functional only in on-premises Exchange.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting the `Set-DlpPolicy` bullet would leave four cmdlets, a strict subset of the original five."},
|
||||
"cond2_remainder_not_misleading": {"holds": "human_must_confirm", "evidence": "A four-cmdlet list with New- and Get- verbs but no update verb could leave a reader to infer no supported update cmdlet exists, in a section explicitly titled 'Viktige cmdlets'."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence_quote states outright 'Use the Set-DlpCompliancePolicy and Set-DlpComplianceRule cmdlets instead', so the source supports a corrected value and the contract forbids fixing this by subtraction."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Condition 3 fails — this is the documented prebuilt-check pattern: the source names the replacement cmdlet, so the correct fix is an O1 value swap (`Set-DlpPolicy` -> `Set-DlpCompliancePolicy`), never deletion.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 40,
|
||||
"id": "ms-ai-security/ai-security-engineering/supply-chain-security-ai-models.md#5",
|
||||
"file": "skills/ms-ai-security/references/ai-security-engineering/supply-chain-security-ai-models.md",
|
||||
"line": 134,
|
||||
"real_line": 133,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "Dependency scanning genererer alerts for:\n- **Direct vulnerabilities**: Pakker i `requirements.txt`\n- **Transitive vulnerabilities**: Pakker som direkte dependencies bruker\n- **CVE severity mapping**: Critical (CVSS ≥9.0), High (7.0-9.0), Medium (4.0-7.0), Low (1.0-4.0)",
|
||||
"failing_part": "The third bullet '**CVE severity mapping**' presented as a category of alert that dependency scanning generates; severity is a property of an alert, not an alert category.",
|
||||
"proposed_remainder": "Dependency scanning genererer alerts for:\n- **Direct vulnerabilities**: Pakker i `requirements.txt`\n- **Transitive vulnerabilities**: Pakker som direkte dependencies bruker",
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting the third bullet reduces the enumeration from three alert categories to two without touching any other assertion."},
|
||||
"cond2_remainder_not_misleading": {"holds": "yes", "evidence": "The two surviving bullets map exactly onto the evidence_quote's 'any open-source component, direct or transitive, found to be vulnerable', so the remainder states precisely what the source states."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "human_must_confirm", "evidence": "The evidence_quote says nothing about CVSS bands, so no source-confirmed fact is lost, but an operator may prefer an O3 rewrite that keeps the severity bands re-framed as an alert property rather than deleting them."},
|
||||
"verdict": "O2_CANDIDATE",
|
||||
"o3_reason": null,
|
||||
"confidence": "medium"
|
||||
},
|
||||
{
|
||||
"idx": 41,
|
||||
"id": "ms-ai-security/ai-security-engineering/supply-chain-security-ai-models.md#8",
|
||||
"file": "skills/ms-ai-security/references/ai-security-engineering/supply-chain-security-ai-models.md",
|
||||
"line": 157,
|
||||
"real_line": 156,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "Defender for Containers:\n- Genererer vulnerability assessments automatisk når image pushes til Azure Container Registry\n- Blokkerer deployment av images med critical vulnerabilities (konfigurerbart via Azure Policy)\n- Integrerer med Azure Monitor for alerting",
|
||||
"failing_part": "The parenthetical '(konfigurerbart via Azure Policy)' — blocking is configured through Defender for Containers' gated-deployment security rules in Defender for Cloud, and the relevant Azure Policy definition offers only AuditIfNotExists and Disabled.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting ' (konfigurerbart via Azure Policy)' removes the mechanism attribution while leaving the blocking assertion untouched, so strictly less is asserted."},
|
||||
"cond2_remainder_not_misleading": {"holds": "human_must_confirm", "evidence": "The bare remainder 'Blokkerer deployment av images med critical vulnerabilities' reads as out-of-the-box behaviour, whereas the evidence_quote describes an admission-controller capability that must be configured with Deny rules and applies at Kubernetes admission, not at ACR push."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence_quote supplies the corrected mechanism — gated deployment via an admission controller with Deny rules — so the source supports a replacement value and the contract routes this to O1/O3 rather than subtraction."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Condition 3 fails (and condition 2 is doubtful): the source names the correct configuration mechanism, so the parenthetical should be swapped, not deleted.",
|
||||
"confidence": "medium"
|
||||
},
|
||||
{
|
||||
"idx": 42,
|
||||
"id": "ms-ai-security/ai-security-engineering/supply-chain-security-ai-models.md#9",
|
||||
"file": "skills/ms-ai-security/references/ai-security-engineering/supply-chain-security-ai-models.md",
|
||||
"line": 202,
|
||||
"real_line": 200,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "Microsoft tilbyr verifiserte modeller via:\n\n- **Azure Machine Learning Model Catalog**: Curated models med security attestation\n- **HuggingFace Registry i Azure**: Integrert med Azure ML, med provenance tracking",
|
||||
"failing_part": "The second bullet presenting the HuggingFace Registry as a Microsoft channel for verified models with provenance tracking; the source calls it a community registry Microsoft support doesn't cover, with weights not hosted on Azure.",
|
||||
"proposed_remainder": "Microsoft tilbyr verifiserte modeller via:\n\n- **Azure Machine Learning Model Catalog**: Curated models med security attestation",
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting the HuggingFace bullet leaves one channel where two were asserted, with no other text altered."},
|
||||
"cond2_remainder_not_misleading": {"holds": "yes", "evidence": "The surviving bullet is the channel the judge leaves unchallenged, and the heading 'Microsoft tilbyr verifiserte modeller via:' remains true of the Model Catalog alone — the disputed entry is removed rather than left standing in a list of vetted options."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "human_must_confirm", "evidence": "The source does confirm HuggingFace models can be deployed from Azure ML, so the 'Integrert med Azure ML' fragment is true; an operator may prefer an O3 rewrite that keeps the entry with a community-registry/no-Microsoft-support caveat instead of deleting it."},
|
||||
"verdict": "O2_CANDIDATE",
|
||||
"o3_reason": null,
|
||||
"confidence": "medium"
|
||||
}
|
||||
]
|
||||
70
scripts/kb-eval/data/r11-o2-returns/batch-08.json
Normal file
70
scripts/kb-eval/data/r11-o2-returns/batch-08.json
Normal file
|
|
@ -0,0 +1,70 @@
|
|||
[
|
||||
{
|
||||
"idx": 43,
|
||||
"id": "ms-ai-security/cost-optimization/gpt5-gpt41-pricing-models.md#9",
|
||||
"file": "skills/ms-ai-security/references/cost-optimization/gpt5-gpt41-pricing-models.md",
|
||||
"line": 203,
|
||||
"real_line": 202,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| Modell | Takst-nivå | Copilot Credits | Power Platform Credits |\n|--------|-----------|----------------|----------------------|\n| `gpt-4.1-mini` | **Basic** | Laveste forbruk | Laveste forbruk |\n| `gpt-4.1` | **Standard** | Moderat forbruk | Moderat forbruk |\n| `gpt-5-chat` (preview) | **Standard** | Moderat forbruk | Moderat forbruk |\n| `gpt-5-reasoning` (preview) | **Premium** | Høyeste forbruk | Høyeste forbruk |\n| `o3` | **Premium** | Høyeste forbruk | Høyeste forbruk |\n| `Claude Sonnet 4.5` (experimental) | **Standard** | Moderat forbruk | Moderat forbruk |\n| `Claude Opus 4.5` (experimental) | **Premium** | Høyeste forbruk | Høyeste forbruk |",
|
||||
"failing_part": "The `Claude Sonnet 4.5` / `Claude Opus 4.5` rows (source lists 4.6 versions) and the `o3` row (absent from the source rate table).",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting the o3 row and the two Claude 4.5 rows would leave a table asserting rate tiers only for the four gpt-* models, which is strictly fewer assertions."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "The surrounding prose is not deletable in the same stroke — line 214 still names o3, line 216 still names 'Claude Opus 4.5' and line 217 still states 'Claude Sonnet 4.5 og Opus 4.5 er nå tilgjengelig i Copilot Studio', so a table without those rows reads as a contradiction of its own section."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence_quote positively confirms 'Claude Sonnet 4.6 | Standard rate', so the source supports a corrected value (4.5 -> 4.6) and deleting the row destroys true information — by contract that is O1/O3, never subtraction."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Conditions 2 and 3 both fail: the source confirms corrected Claude version/tier values (value swap territory), and any table-only subtraction leaves the section's prose asserting the deleted models.",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 44,
|
||||
"id": "ms-ai-security/cost-optimization/gpt5-gpt41-pricing-models.md#16",
|
||||
"file": "skills/ms-ai-security/references/cost-optimization/gpt5-gpt41-pricing-models.md",
|
||||
"line": 468,
|
||||
"real_line": 468,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| Modell | Tilgjengelighet | Registrering |\n|--------|----------------|-------------|\n| `gpt-5` | GA (begrenset) | Krever godkjenning (aka.ms/oai/gpt5access) |\n| `gpt-5-mini` | GA | Ikke nødvendig |\n| `gpt-5-nano` | GA | Ikke nødvendig |\n| `gpt-5-chat` | Preview (2 versjoner) | Ikke nødvendig |\n| `gpt-5-codex` | GA (begrenset) | Krever godkjenning |\n| `gpt-5-pro` | GA (begrenset) | Kun MCA-E/Default-abonnementer |",
|
||||
"failing_part": "The '(begrenset)' availability qualifier plus the Registrering cells 'Krever godkjenning (aka.ms/oai/gpt5access)', 'Krever godkjenning' and 'Kun MCA-E/Default-abonnementer' for gpt-5, gpt-5-codex and gpt-5-pro.",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting ' (begrenset)' and the three Registrering cell contents leaves the table asserting only that the models are GA, which is strictly less than before."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "Three empty Registrering cells sitting beside rows that explicitly say 'Ikke nødvendig' reads as 'registration status unknown/omitted' rather than 'no registration required', which is precisely the fact the source establishes."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "'Access is no longer restricted for this model.' is an affirmative source statement supporting the corrected values 'GA' and 'Ikke nødvendig', so the correct fix is a value swap, not a deletion."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Condition 3 fails (source supports a corrected value for both columns, making this O1-shaped) and condition 2 fails (blank cells in a column whose other rows are filled in are misleading).",
|
||||
"confidence": "high"
|
||||
},
|
||||
{
|
||||
"idx": 45,
|
||||
"id": "ms-ai-security/cost-optimization/semantic-caching-patterns.md#14",
|
||||
"file": "skills/ms-ai-security/references/cost-optimization/semantic-caching-patterns.md",
|
||||
"line": 436,
|
||||
"real_line": 436,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "| **Data residency** | Bruk Norway East/West for Redis og OpenAI for å sikre data forblir i Norge/EU. |",
|
||||
"failing_part": "The '/West' half of the region pair, i.e. the standing implication that Azure OpenAI can be deployed in Norway West.",
|
||||
"proposed_remainder": "| **Data residency** | Bruk Norway East for Redis og OpenAI for å sikre data forblir i Norge/EU. |",
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting the five characters '/West' narrows a two-region recommendation to a one-region recommendation and asserts nothing new."},
|
||||
"cond2_remainder_not_misleading": {"holds": "yes", "evidence": "'Bruk Norway East for Redis og OpenAI' is true for both products (Azure Managed Redis is available in Norway East and the source's regional table has a norwayeast column), and it is consistent with line 451 of the same file, 'Schrems II: Azure OpenAI i EU-region (Norway East)'."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "human_must_confirm", "evidence": "The deletion incidentally drops the true fact that Redis is also available in Norway West, but the line is a single bundled recommendation for both products — keeping '/West' is impossible without re-asserting the false OpenAI half, and no researched value is needed to obtain the remainder."},
|
||||
"verdict": "O2_CANDIDATE",
|
||||
"o3_reason": null,
|
||||
"confidence": "medium"
|
||||
},
|
||||
{
|
||||
"idx": 46,
|
||||
"id": "ms-ai-security/cost-optimization/vector-storage-cost-optimization.md#11",
|
||||
"file": "skills/ms-ai-security/references/cost-optimization/vector-storage-cost-optimization.md",
|
||||
"line": 73,
|
||||
"real_line": 73,
|
||||
"locator_failed": false,
|
||||
"file_text_verbatim": "Azure AI Search lagrer vektorer i to kopier:\n1. **Index copy** (i minne, brukes til query execution)\n2. **Stored copy** (på disk, brukes til retrieval i query response)\n\nVed å sette `stored: false` kan man spare opptil 50 % disklagring, men man mister muligheten til å returnere vektorer i query-responser. Dette er akseptabelt i de fleste RAG-scenarier der kun tekst/metadata returneres.",
|
||||
"failing_part": "The count 'to kopier' and the implied exhaustiveness of the two-item enumeration — the source lists a third stored instance (original full-precision vectors retained for rescoring).",
|
||||
"proposed_remainder": null,
|
||||
"cond1_strictly_less": {"holds": true, "evidence": "Deleting 'to ' would drop the numeric assertion, so strictly less would be asserted — but no deletion-only edit yields grammatical Norwegian ('lagrer vektorer i kopier:')."},
|
||||
"cond2_remainder_not_misleading": {"holds": "no", "evidence": "The numbered list 1./2. survives any deletion of the count and still reads as a complete inventory, and the dependent sentence at line 77 ('spare opptil 50 % disklagring') derives its arithmetic from exactly two copies."},
|
||||
"cond3_nothing_confirmed_removed": {"holds": "no", "evidence": "The evidence_quote enumerates three storage instances, so the source supports a corrected count and a third list item — supplying them is a rewrite, not a subtraction."},
|
||||
"verdict": "O3",
|
||||
"o3_reason": "Conditions 2 and 3 fail: no character-only deletion produces a grammatical, non-exhaustive remainder, the downstream 50 % cost claim depends on the wrong count, and the source confirms the corrected three-copy value.",
|
||||
"confidence": "high"
|
||||
}
|
||||
]
|
||||
3009
scripts/kb-eval/data/r11-pilot-classification.json
Normal file
3009
scripts/kb-eval/data/r11-pilot-classification.json
Normal file
File diff suppressed because it is too large
Load diff
227
scripts/kb-eval/data/r71-payloads.json
Normal file
227
scripts/kb-eval/data/r71-payloads.json
Normal file
File diff suppressed because one or more lines are too long
247
scripts/kb-eval/data/r72-payloads.json
Normal file
247
scripts/kb-eval/data/r72-payloads.json
Normal file
File diff suppressed because one or more lines are too long
247
scripts/kb-eval/data/r73-payloads.json
Normal file
247
scripts/kb-eval/data/r73-payloads.json
Normal file
File diff suppressed because one or more lines are too long
247
scripts/kb-eval/data/r74-payloads.json
Normal file
247
scripts/kb-eval/data/r74-payloads.json
Normal file
File diff suppressed because one or more lines are too long
257
scripts/kb-eval/data/r75-payloads.json
Normal file
257
scripts/kb-eval/data/r75-payloads.json
Normal file
File diff suppressed because one or more lines are too long
|
|
@ -7,7 +7,8 @@
|
|||
"note": "This is a SAMPLE of corpus errors (255 volatile claims). Spor 1 surfaces the full set corpus-wide. [2026-06-29] Spor 0 executed: 37 fixes applied (each value live-verified per source before edit; recurrences of each wrong fact fixed file-wide), 1 rejected (#8 model-router-GA: file already correct). Adjacent (non-38) findings captured for Spor 1.",
|
||||
"applied": 37,
|
||||
"rejected": 1,
|
||||
"applied_date": "2026-06-29"
|
||||
"applied_date": "2026-06-29",
|
||||
"g5b_note": "[2026-07-04] R5 G5b-restanse: 4 innholds-fikser påført (adr zero-permission-overclaim, multi-region retired gpt-35-turbo→gpt-4o-mini, network obligatorisk→anbefalt, vector-storage GA 2024-11-01→2024-07-01). Hver live-verifisert per kilde (Opus xhigh verify-agent) før edit. De 38 kjerne-fiksene var allerede applied 2026-06-29. Se g5b_fixes[]."
|
||||
},
|
||||
"fixes": [
|
||||
{
|
||||
|
|
@ -619,5 +620,67 @@
|
|||
"verified": "2026-06-29",
|
||||
"verified_by": "per-source verify-agent (Opus xhigh, live microsoft_docs_fetch) + main-context apply"
|
||||
}
|
||||
],
|
||||
"g5b_fixes": [
|
||||
{
|
||||
"id": "ms-ai-advisor/architecture/adr-template.md#g5b",
|
||||
"file": "skills/ms-ai-advisor/references/architecture/adr-template.md",
|
||||
"skill": "ms-ai-advisor",
|
||||
"claim_type": "capability-overclaim",
|
||||
"verdict": "confirm-fix",
|
||||
"wrong_assertion": "Zero permission management - SharePoint permissions respekteres automatisk",
|
||||
"correction": "Overclaim: Copilot/Graph håndhever eksisterende tilganger (sant), men oversharing må fortsatt styres aktivt (Restricted SharePoint Search, SharePoint Advanced Management, sensitivitetsmerker, Purview). \"Zero permission management\" fjernet.",
|
||||
"source": "https://learn.microsoft.com/microsoft-365/copilot/security-microsoft-365-copilot",
|
||||
"reverify_required": true,
|
||||
"fixed": true,
|
||||
"applied": true,
|
||||
"verified": "2026-07-04",
|
||||
"verified_by": "per-source verify-agent (Opus xhigh, live MS Learn) + main-context apply"
|
||||
},
|
||||
{
|
||||
"id": "ms-ai-infrastructure/bcdr/multi-region-azure-openai-deployment.md#g5b",
|
||||
"file": "skills/ms-ai-infrastructure/references/bcdr/multi-region-azure-openai-deployment.md",
|
||||
"skill": "ms-ai-infrastructure",
|
||||
"claim_type": "retired-model",
|
||||
"verdict": "confirm-fix",
|
||||
"wrong_assertion": "gpt-35-turbo listet som tilgjengelig modell (regionstabeller + kvoteeksempler, 5 forekomster)",
|
||||
"correction": "Alle gpt-35-turbo-versjoner retired (0301/0613 2025-02-13, 16k 2025-04-30, 1106/0125 forbi \"no earlier than 2025-09-01\"-vindu) → byttet til gpt-4o-mini (bekreftet Standard-tilgjengelig i Sweden Central / West Europe / UK South).",
|
||||
"source": "https://learn.microsoft.com/azure/ai-foundry/openai/concepts/model-retirements",
|
||||
"reverify_required": true,
|
||||
"fixed": true,
|
||||
"applied": true,
|
||||
"verified": "2026-07-04",
|
||||
"verified_by": "per-source verify-agent (Opus xhigh, live MS Learn) + main-context apply"
|
||||
},
|
||||
{
|
||||
"id": "ms-ai-infrastructure/bcdr/network-resilience-patterns-ai.md#g5b",
|
||||
"file": "skills/ms-ai-infrastructure/references/bcdr/network-resilience-patterns-ai.md",
|
||||
"skill": "ms-ai-infrastructure",
|
||||
"claim_type": "overstatement",
|
||||
"verdict": "confirm-fix",
|
||||
"wrong_assertion": "Circuit Breaker + Retry med exponential backoff er OBLIGATORISK for alle Azure AI API-kall — dette er ikke valgfritt.",
|
||||
"correction": "Overstatement: MS rammer retry/circuit-breaker som anbefalt beste praksis (Well-Architected reliability patterns), ikke krav. Endret til \"sterkt anbefalt ... ikke et absolutt krav\".",
|
||||
"source": "https://learn.microsoft.com/azure/well-architected/reliability/design-patterns",
|
||||
"reverify_required": true,
|
||||
"fixed": true,
|
||||
"applied": true,
|
||||
"verified": "2026-07-04",
|
||||
"verified_by": "per-source verify-agent (Opus xhigh, live MS Learn) + main-context apply"
|
||||
},
|
||||
{
|
||||
"id": "ms-ai-security/cost-optimization/vector-storage-cost-optimization.md#g5b",
|
||||
"file": "skills/ms-ai-security/references/cost-optimization/vector-storage-cost-optimization.md",
|
||||
"skill": "ms-ai-security",
|
||||
"claim_type": "ga-date",
|
||||
"verdict": "confirm-fix",
|
||||
"wrong_assertion": "Vector quantization (GA siden 2024-11-01)",
|
||||
"correction": "GA-dato feil: scalar + binary quantization ble GA i 2024-07-01 stable Search-API. 2024-11-01 var en preview (rescoring/syntaks). Rettet til 2024-07-01.",
|
||||
"source": "https://learn.microsoft.com/azure/search/search-api-migration",
|
||||
"reverify_required": true,
|
||||
"fixed": true,
|
||||
"applied": true,
|
||||
"verified": "2026-07-04",
|
||||
"verified_by": "per-source verify-agent (Opus xhigh, live MS Learn) + main-context apply"
|
||||
}
|
||||
]
|
||||
}
|
||||
|
|
|
|||
201
scripts/kb-eval/judge-pass-manifest.mjs
Executable file
201
scripts/kb-eval/judge-pass-manifest.mjs
Executable file
|
|
@ -0,0 +1,201 @@
|
|||
#!/usr/bin/env node
|
||||
// judge-pass-manifest.mjs — R7–R10 full-pass judge LEDGER (Spor 3 pre-pass infra, R6 Step 4).
|
||||
//
|
||||
// The durable per-file record of the corpus judge pass: one entry per never-verified reference
|
||||
// file, carrying its per-file verdict (born-verified `pass` → surgical stamp, or `flagged` →
|
||||
// R11 fix work-list), the judge provenance, and any not-grounded claim flags. Mirrors
|
||||
// scripts/kb-eval/data/spor0-fix-manifest.json in shape ({_meta, <records[]>}) and discipline
|
||||
// (git-tracked under data/ so drift is diff-reviewable; hand/subagent-maintained).
|
||||
//
|
||||
// Durability + resume are the whole point (brief NFR): R7 calls appendJudgedFile after EVERY
|
||||
// judged file and writes atomically, so a crash on file 200/243 loses at most ONE file; a
|
||||
// resume calls pendingFiles(manifest, worklist) and re-processes only the remainder. The
|
||||
// aggregate-in-memory-write-once alternative loses a whole økt on a crash — rejected.
|
||||
//
|
||||
// Pure core: validateManifest / mergeBatch / appendJudgedFile / pendingFiles are disk-free and
|
||||
// deterministic (path-sorted output). IO-shell: --json / --write [--merge] [--out <path>].
|
||||
//
|
||||
// Exit codes: 0 ok · 2 usage error · 3 re-entrance guard (judged ledger, no --merge)
|
||||
|
||||
import { readFileSync, existsSync, realpathSync } from 'node:fs';
|
||||
import { join, dirname } from 'node:path';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
import { atomicWriteJson } from '../kb-update/lib/atomic-write.mjs';
|
||||
|
||||
const __dirname = dirname(fileURLToPath(import.meta.url));
|
||||
|
||||
// Pinned: the single git-tracked ledger downstream R7–R10 batches append to (verified NOT
|
||||
// under .gitignore, so the judged ledger is review-visible — brief preference).
|
||||
export const DEFAULT_MANIFEST_PATH = join(__dirname, 'data', 'judge-pass-manifest.json');
|
||||
|
||||
// A file's per-pass verdict: `pass` = every judgeable claim grounded (AND) → born-verified
|
||||
// stamp; `flagged` = ≥1 not-grounded claim OR unstampable → R11 work-list. Nothing else.
|
||||
export const VALID_VERDICTS = ['pass', 'flagged'];
|
||||
|
||||
const KNOWN_FLAGS = new Set(['--json', '--write', '--merge', '--out']);
|
||||
|
||||
const byFilePath = (a, b) => (a.file < b.file ? -1 : a.file > b.file ? 1 : 0);
|
||||
|
||||
/** A fresh, empty ledger scaffold. Deterministic (no live timestamp) so --write is stable. */
|
||||
export function scaffoldManifest() {
|
||||
return {
|
||||
_meta: {
|
||||
purpose:
|
||||
'R7–R10 full-pass judge ledger: one record per never-verified reference file — ' +
|
||||
'born-verified pass (surgical stamp) or flagged (R11 work-list). Durable per file.',
|
||||
derived_from: 'scripts/kb-update/data/full-pass-worklist.json (243 due, all never-verified)',
|
||||
cadence: 'R7–R10 (5 økter × ~49, 8–10 samtidige)',
|
||||
batch: null,
|
||||
count: 0,
|
||||
generated: null,
|
||||
},
|
||||
files: [],
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate the ledger shape. Pure. Returns {valid, errors[]}. Enforces the born-verified
|
||||
* contract at rest: a `pass` record MUST carry verified + verified_by; a `flagged` record MUST
|
||||
* carry ≥1 flag. Duplicate file records are an error (one record per file).
|
||||
*/
|
||||
export function validateManifest(manifest) {
|
||||
const errors = [];
|
||||
if (!manifest || typeof manifest !== 'object' || Array.isArray(manifest)) {
|
||||
return { valid: false, errors: ['manifest must be an object'] };
|
||||
}
|
||||
if (!manifest._meta || typeof manifest._meta !== 'object') errors.push('_meta must be an object');
|
||||
if (!Array.isArray(manifest.files)) {
|
||||
errors.push('files must be an array');
|
||||
return { valid: false, errors };
|
||||
}
|
||||
const seen = new Set();
|
||||
for (const rec of manifest.files) {
|
||||
if (!rec || typeof rec !== 'object') { errors.push('file record must be an object'); continue; }
|
||||
if (typeof rec.file !== 'string' || rec.file.trim() === '') { errors.push("record missing 'file'"); continue; }
|
||||
if (seen.has(rec.file)) errors.push(`duplicate file record: ${rec.file}`);
|
||||
seen.add(rec.file);
|
||||
if (!VALID_VERDICTS.includes(rec.per_file_verdict)) {
|
||||
errors.push(`${rec.file}: per_file_verdict must be one of ${VALID_VERDICTS.join('|')}`);
|
||||
}
|
||||
if (!Array.isArray(rec.flags)) errors.push(`${rec.file}: flags must be an array`);
|
||||
if (rec.per_file_verdict === 'pass' && (!rec.verified || !rec.verified_by)) {
|
||||
errors.push(`${rec.file}: a pass record must carry verified + verified_by`);
|
||||
}
|
||||
if (rec.per_file_verdict === 'flagged' && !(Array.isArray(rec.flags) && rec.flags.length > 0)) {
|
||||
errors.push(`${rec.file}: a flagged record must carry ≥1 flag`);
|
||||
}
|
||||
}
|
||||
return { valid: errors.length === 0, errors };
|
||||
}
|
||||
|
||||
/**
|
||||
* Add or update ONE file record, returning a NEW manifest (pure). The durability primitive:
|
||||
* R7 calls this after each judged file, then writes atomically → max one file lost on crash.
|
||||
* An existing record for the same `file` is replaced (idempotent re-judge). Files stay
|
||||
* path-sorted and _meta.count is refreshed, so the on-disk ledger is deterministic + diff-clean.
|
||||
*/
|
||||
export function appendJudgedFile(manifest, record) {
|
||||
if (!record || typeof record.file !== 'string' || record.file.trim() === '') {
|
||||
throw new Error('appendJudgedFile: record.file is required');
|
||||
}
|
||||
const base = manifest && typeof manifest === 'object' && !Array.isArray(manifest) ? manifest : scaffoldManifest();
|
||||
const kept = (Array.isArray(base.files) ? base.files : []).filter((r) => r.file !== record.file);
|
||||
const files = [...kept, record].sort(byFilePath);
|
||||
return { ...base, _meta: { ...(base._meta ?? {}), count: files.length }, files };
|
||||
}
|
||||
|
||||
/**
|
||||
* Merge a batch of file records into an existing ledger, preserving already-judged records
|
||||
* verbatim (a judged entry is authoritative — a re-run never silently re-judges it). Records
|
||||
* for not-yet-judged files are added/updated. Pure; output path-sorted, _meta.count refreshed.
|
||||
* @param {object} existing — the current ledger
|
||||
* @param {Array|object} incoming — a batch (array of records, or a manifest with files[])
|
||||
*/
|
||||
export function mergeBatch(existing, incoming) {
|
||||
const existingFiles = Array.isArray(existing?.files) ? existing.files : [];
|
||||
const judged = new Set(
|
||||
existingFiles.filter((r) => VALID_VERDICTS.includes(r?.per_file_verdict)).map((r) => r.file),
|
||||
);
|
||||
const byFile = new Map(existingFiles.map((r) => [r.file, r]));
|
||||
const incomingFiles = Array.isArray(incoming) ? incoming : Array.isArray(incoming?.files) ? incoming.files : [];
|
||||
for (const rec of incomingFiles) {
|
||||
if (!rec || typeof rec.file !== 'string') continue;
|
||||
if (judged.has(rec.file)) continue; // judged wins — preserve verbatim
|
||||
byFile.set(rec.file, rec);
|
||||
}
|
||||
const files = [...byFile.values()].sort(byFilePath);
|
||||
return { ...(existing ?? {}), _meta: { ...(existing?._meta ?? {}), count: files.length }, files };
|
||||
}
|
||||
|
||||
/**
|
||||
* Resume primitive: given the full worklist of file paths, return those NOT yet judged (no
|
||||
* ledger record with a per_file_verdict set). R7 resume processes only these — a re-run after
|
||||
* a crash skips every already-judged file. Pure; preserves worklist order.
|
||||
*/
|
||||
export function pendingFiles(manifest, worklistPaths) {
|
||||
const judged = new Set(
|
||||
(manifest?.files ?? [])
|
||||
.filter((r) => r && VALID_VERDICTS.includes(r.per_file_verdict))
|
||||
.map((r) => r.file),
|
||||
);
|
||||
return (worklistPaths ?? []).filter((p) => !judged.has(p));
|
||||
}
|
||||
|
||||
function hasJudged(manifest) {
|
||||
return Array.isArray(manifest?.files) && manifest.files.some((r) => VALID_VERDICTS.includes(r?.per_file_verdict));
|
||||
}
|
||||
|
||||
function usageExit(msg, code) {
|
||||
process.stderr.write(`${msg}\nusage: judge-pass-manifest.mjs [--json] [--write] [--merge] [--out <path>]\n`);
|
||||
process.exit(code);
|
||||
}
|
||||
|
||||
function main(argv) {
|
||||
const args = argv.slice(2);
|
||||
let outPath = DEFAULT_MANIFEST_PATH;
|
||||
for (let i = 0; i < args.length; i++) {
|
||||
if (args[i] === '--out') {
|
||||
if (!args[i + 1]) usageExit('--out requires a path', 2);
|
||||
outPath = args[++i];
|
||||
continue;
|
||||
}
|
||||
if (!KNOWN_FLAGS.has(args[i])) usageExit(`unknown flag: ${args[i]}`, 2);
|
||||
}
|
||||
const json = args.includes('--json');
|
||||
const write = args.includes('--write');
|
||||
const merge = args.includes('--merge');
|
||||
|
||||
const existing = existsSync(outPath) ? JSON.parse(readFileSync(outPath, 'utf8')) : null;
|
||||
if (existing) {
|
||||
const check = validateManifest(existing);
|
||||
if (!check.valid) usageExit(`invalid manifest at ${outPath}:\n ${check.errors.join('\n ')}`, 2);
|
||||
}
|
||||
let manifest = existing ?? scaffoldManifest();
|
||||
|
||||
if (write) {
|
||||
if (existing && hasJudged(existing) && !merge) {
|
||||
process.stderr.write(
|
||||
`refusing --write: ledger at ${outPath} already carries judged record(s); a fresh write ` +
|
||||
'would clobber the judged pass. Re-run with --merge to preserve it.\n',
|
||||
);
|
||||
process.exit(3);
|
||||
}
|
||||
manifest = mergeBatch(manifest, []); // normalize: path-sort + refresh count, judged preserved
|
||||
atomicWriteJson(outPath, manifest);
|
||||
}
|
||||
|
||||
if (json) {
|
||||
process.stdout.write(JSON.stringify(manifest, null, 2) + '\n');
|
||||
} else {
|
||||
console.log(`judge-pass-manifest: ${manifest.files.length} record(s)` + (write ? ` → ${outPath}` : ''));
|
||||
}
|
||||
}
|
||||
|
||||
const isMain = (() => {
|
||||
try {
|
||||
return realpathSync(process.argv[1]) === realpathSync(fileURLToPath(import.meta.url));
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
})();
|
||||
if (isMain) main(process.argv);
|
||||
481
scripts/kb-eval/lib/fix-op.mjs
Normal file
481
scripts/kb-eval/lib/fix-op.mjs
Normal file
|
|
@ -0,0 +1,481 @@
|
|||
// fix-op.mjs — R11 fix-operation classifier over judge-pass flags.
|
||||
//
|
||||
// Contract: docs/r11-tiered-fix-design.md §3 (the O1/O2/O3 partition is by
|
||||
// OPERATION, not by rule code) and §4 (the O1 invariant).
|
||||
//
|
||||
// This module IS the O1 driver with writes disabled. It attempts the value swap
|
||||
// and checks §4's three conditions; an item it cannot prove is O3 with a typed
|
||||
// abort code. That is deliberate: a proxy heuristic would have measured
|
||||
// something other than the mechanism that will later touch a public corpus.
|
||||
//
|
||||
// Two properties the callers depend on:
|
||||
// - PURE. No fs, no network, no mutation of the input flag. The caller reads
|
||||
// the file and passes its text.
|
||||
// - FAILS CLOSED. Every path returns O1-with-proof or O3-with-a-known-code.
|
||||
// A misrouted O3 costs one human review; a misrouted O1 ships a wrong edit
|
||||
// to a publicly distributed file.
|
||||
//
|
||||
// What this module deliberately does NOT do: decide O2. Subtraction candidacy
|
||||
// turns on which sub-assertion the judge's prose `reason` names as failing, and
|
||||
// no regex reads prose. O2 requires operator ratification (§5) before it exists
|
||||
// as a class at all; until then every non-O1 item is O3 by design.
|
||||
|
||||
/**
|
||||
* Abort codes. The taxonomy is part of the contract, not diagnostics: the pilot's
|
||||
* measurement #3 (§10) is the DISTRIBUTION of these, because "abort rate 85 %"
|
||||
* is not actionable while "60 % LOCATOR_MISS" is an engineering gap and "60 %
|
||||
* NOT_VERBATIM" is intrinsic to the corpus.
|
||||
*/
|
||||
export const ABORT_CODES = {
|
||||
MULTI_PART_CLAIM: 'MULTI_PART_CLAIM', // enumeration / several assertions in one claim (§3, the R8 class)
|
||||
NO_VALUE_TOKEN: 'NO_VALUE_TOKEN', // nothing swappable — the claim asserts prose
|
||||
STATUS_SYNONYM: 'STATUS_SYNONYM', // GA/Preview class: file vocabulary != source vocabulary (operator question)
|
||||
MULTI_VALUE_TOKEN: 'MULTI_VALUE_TOKEN', // several distinct values — which one is wrong is a judgement
|
||||
LOCATOR_MISS: 'LOCATOR_MISS', // value not found in the block the flag points at
|
||||
LOCATOR_AMBIGUOUS: 'LOCATOR_AMBIGUOUS', // value occurs more than once in that block
|
||||
NOT_VERBATIM: 'NOT_VERBATIM', // no same-type replacement occurs verbatim in evidence_quote (§4.1)
|
||||
MULTI_REPLACEMENT: 'MULTI_REPLACEMENT', // quote offers several candidate values
|
||||
CONTEXT_MISMATCH: 'CONTEXT_MISMATCH', // §4 held but the tokens do not denote the same quantity (see below)
|
||||
INVARIANT_FAIL: 'INVARIANT_FAIL', // swap constructed but §4 did not hold — must never happen silently
|
||||
};
|
||||
|
||||
/** Verdict code for a proven swap. Kept out of ABORT_CODES so `op === 'O1' <=> code === 'PROVEN'`. */
|
||||
export const PROVEN = 'PROVEN';
|
||||
|
||||
// Value types, most specific first. Matching is non-overlapping and priority
|
||||
// ordered, so `2.3.0` is one version rather than two numbers, and `20 %` is a
|
||||
// percent rather than the number 20. Types never cross in a swap: a percent may
|
||||
// only be replaced by a percent.
|
||||
const TOKEN_PATTERNS = [
|
||||
['iso_date', /\d{4}-\d{2}-\d{2}/g],
|
||||
['percent', /\d+(?:[.,]\d+)?\s?%/g],
|
||||
['version', /v?\d+\.\d+\.\d+/g],
|
||||
['number', /\d+(?:[.,]\d+)?/g],
|
||||
];
|
||||
|
||||
// Lifecycle vocabulary. Present in a claim without any numeric token, this is the
|
||||
// GA/Preview class: the corpus writes `**Preview**` / `**GA**` while the cited
|
||||
// source writes "generally available". A swap satisfies §4 literally while
|
||||
// pasting English prose into a Norwegian table, which is why the class was put to
|
||||
// the operator as a design question — answered by the §4b table below.
|
||||
const STATUS_RE =
|
||||
/\b(?:GA|generally available|allment tilgjengelig|public preview|private preview|preview|deprecated|utfaset|retired|avviklet)\b/i;
|
||||
|
||||
// ------------------------------------------------------- §4b synonym table
|
||||
//
|
||||
// RATIFIED 2026-08-03 (docs/r11-tiered-fix-design.md §4b). This is the one place
|
||||
// where the value written into the file does NOT appear verbatim in the quote —
|
||||
// §4 condition 1 is structurally unsatisfiable for it, because the corpus writes
|
||||
// a label and the source writes a phrase. The table is what the operator ratified
|
||||
// in its place, and it is CLOSED: a pair not listed here aborts, and nothing
|
||||
// extends it at run time.
|
||||
//
|
||||
// `corpus[0]` is the canonical value written back. The order is the table's own:
|
||||
// the row's least specific label wins, so a source that says only "preview" can
|
||||
// never produce the more specific "Public Preview" — that would assert something
|
||||
// the source does not.
|
||||
export const STATUS_TABLE = [
|
||||
{ row: 'GA', corpus: ['GA'], source: ['generally available', 'general availability'] },
|
||||
{ row: 'PREVIEW', corpus: ['Preview', 'Public Preview'], source: ['public preview', 'preview'] },
|
||||
{ row: 'PRIVATE_PREVIEW', corpus: ['Private Preview'], source: ['private preview'] },
|
||||
{ row: 'DEPRECATED', corpus: ['Deprecated', 'Utfaset'], source: ['deprecated', 'retired'] },
|
||||
];
|
||||
|
||||
// Longest first, so "private preview" is consumed as its own row before the
|
||||
// "preview" row can claim the tail of it.
|
||||
const SOURCE_PHRASES = STATUS_TABLE.flatMap((r) => r.source.map((p) => ({ row: r.row, phrase: p }))).sort(
|
||||
(a, b) => b.phrase.length - a.phrase.length,
|
||||
);
|
||||
|
||||
const CORPUS_LABELS = new Map(
|
||||
STATUS_TABLE.flatMap((r) => r.corpus.map((label) => [label.toLowerCase(), r])),
|
||||
);
|
||||
|
||||
// A hyphen counts as a word character HERE, unlike WORD elsewhere in this module:
|
||||
// `2025-11-15-preview` is an api-version identifier, not a statement that the
|
||||
// feature is in preview. Treating `-` as a boundary would harvest lifecycle rows
|
||||
// out of URLs and code samples.
|
||||
const isPhraseChar = (ch) => ch !== undefined && /[A-Za-z0-9-]/.test(ch);
|
||||
|
||||
/**
|
||||
* Which lifecycle rows does the cited quote assert, per the closed §4b table?
|
||||
* Matching is case-insensitive, non-overlapping, longest phrase first.
|
||||
* @returns {string[]} distinct row keys, in table order
|
||||
*/
|
||||
export function sourceStatusRows(quote) {
|
||||
const text = (quote || '').toLowerCase();
|
||||
const taken = [];
|
||||
const rows = new Set();
|
||||
for (const { row, phrase } of SOURCE_PHRASES) {
|
||||
let from = 0;
|
||||
for (;;) {
|
||||
const at = text.indexOf(phrase, from);
|
||||
if (at === -1) break;
|
||||
const end = at + phrase.length;
|
||||
from = end;
|
||||
if (taken.some(([s, e]) => at < e && end > s)) continue;
|
||||
if (isPhraseChar(text[at - 1]) || isPhraseChar(text[end])) continue;
|
||||
taken.push([at, end]);
|
||||
rows.add(row);
|
||||
}
|
||||
}
|
||||
return STATUS_TABLE.map((r) => r.row).filter((r) => rows.has(r));
|
||||
}
|
||||
|
||||
// Markup wrappers the corpus actually uses around a lifecycle label. The label is
|
||||
// replaced INSIDE the wrapper, which is how §4b constraint 3 ("the file's own
|
||||
// markup preserved") is satisfied without any markup handling at write time.
|
||||
const WRAPPERS = ['**', '__', '*', '_', '`'];
|
||||
|
||||
/**
|
||||
* Is this segment exactly a lifecycle label, once whitespace and one or more
|
||||
* markup wrappers are peeled off? Returns the label's span in the ORIGINAL line.
|
||||
*/
|
||||
function labelInSegment(segment, segStart) {
|
||||
let text = segment;
|
||||
let off = 0;
|
||||
const trim = () => {
|
||||
const lead = text.length - text.trimStart().length;
|
||||
off += lead;
|
||||
text = text.trim();
|
||||
};
|
||||
trim();
|
||||
for (;;) {
|
||||
const w = WRAPPERS.find((x) => text.length > 2 * x.length && text.startsWith(x) && text.endsWith(x));
|
||||
if (!w) break;
|
||||
off += w.length;
|
||||
text = text.slice(w.length, text.length - w.length);
|
||||
trim();
|
||||
}
|
||||
const rec = CORPUS_LABELS.get(text.toLowerCase());
|
||||
return rec ? { row: rec.row, label: text, index: segStart + off, length: text.length } : null;
|
||||
}
|
||||
|
||||
// An emphasised run anywhere on the line. `|` is excluded from the inner text so a
|
||||
// run can never span two table cells.
|
||||
const EMPHASIS_RE = /(\*\*|__|\*|_|`)([^*_`|]+?)\1/g;
|
||||
|
||||
/**
|
||||
* The single complete lifecycle label on `line`, or null when there is none or
|
||||
* more than one (§4b constraint 2: a whole table cell or an emphasised token,
|
||||
* never a substring of a longer sentence).
|
||||
*
|
||||
* Deliberately LINE-scoped rather than block-scoped, unlike the numeric locator.
|
||||
* Lifecycle vocabulary repeats down every column of a status table, so a block
|
||||
* window is ambiguous by construction — and all 15 pilot flags in this class
|
||||
* point at the row that carries the claim, not at the table header.
|
||||
*
|
||||
* @returns {{row: string, label: string, index: number, length: number}|null}
|
||||
*/
|
||||
export function fileStatusLabel(line) {
|
||||
const text = line || '';
|
||||
const hits = [];
|
||||
if (text.includes('|')) {
|
||||
let at = 0;
|
||||
for (const cell of text.split('|')) {
|
||||
const hit = labelInSegment(cell, at);
|
||||
if (hit) hits.push(hit);
|
||||
at += cell.length + 1;
|
||||
}
|
||||
}
|
||||
EMPHASIS_RE.lastIndex = 0;
|
||||
let m;
|
||||
while ((m = EMPHASIS_RE.exec(text)) !== null) {
|
||||
const hit = labelInSegment(m[0], m.index);
|
||||
if (hit && !hits.some((h) => h.index === hit.index)) hits.push(hit);
|
||||
}
|
||||
return hits.length === 1 ? hits[0] : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* §4b: classify a status-vocabulary flag against the ratified table.
|
||||
*
|
||||
* Every abort keeps the STATUS_SYNONYM code and names its cause in `detail.reason`
|
||||
* — the top-level abort taxonomy is the pilot's measurement #3 and stays
|
||||
* comparable across the implementation, with the reasons reported as a
|
||||
* sub-distribution.
|
||||
*/
|
||||
function classifyStatusSynonym(flag, lines) {
|
||||
const detail = (reason, extra = {}) => abort(ABORT_CODES.STATUS_SYNONYM, { detail: { reason, ...extra } });
|
||||
|
||||
if (flag.line < 1 || flag.line > lines.length) return detail('LINE_OUT_OF_RANGE', { line: flag.line });
|
||||
|
||||
const rows = sourceStatusRows(flag.evidence_quote || '');
|
||||
if (rows.length === 0) return detail('NO_SOURCE_STATUS');
|
||||
if (rows.length > 1) return detail('SOURCE_STATUS_AMBIGUOUS', { rows });
|
||||
|
||||
const before = lines[flag.line - 1];
|
||||
const hit = fileStatusLabel(before);
|
||||
if (!hit) return detail('NO_COMPLETE_FILE_LABEL');
|
||||
if (hit.row === rows[0]) return detail('FILE_ALREADY_MATCHES', { row: hit.row });
|
||||
|
||||
const target = STATUS_TABLE.find((r) => r.row === rows[0]);
|
||||
const replacement = target.corpus[0];
|
||||
const after = before.slice(0, hit.index) + replacement + before.slice(hit.index + hit.length);
|
||||
|
||||
// §4 conditions 2 and 3 still bind. Condition 1 is replaced by the table: what
|
||||
// must appear verbatim in the quote is the SOURCE phrasing, not the written value.
|
||||
const rest =
|
||||
before.slice(0, hit.index) === after.slice(0, hit.index) &&
|
||||
before.slice(hit.index + hit.length) === after.slice(hit.index + replacement.length);
|
||||
const rebuilt = lines.slice();
|
||||
rebuilt[flag.line - 1] = after;
|
||||
const changedLines = rebuilt.reduce((n, l, i) => n + (l === lines[i] ? 0 : 1), 0);
|
||||
if (!rest || changedLines !== 1) {
|
||||
return abort(ABORT_CODES.INVARIANT_FAIL, { detail: { restIdentical: rest, changedLines } });
|
||||
}
|
||||
|
||||
return {
|
||||
op: 'O1',
|
||||
code: PROVEN,
|
||||
proposal: {
|
||||
file: flag.file,
|
||||
line: flag.line,
|
||||
token: hit.label,
|
||||
replacement,
|
||||
type: 'status',
|
||||
status_row_from: hit.row,
|
||||
status_row_to: rows[0],
|
||||
before,
|
||||
after,
|
||||
evidence_url: flag.evidence_url,
|
||||
evidence_quote: flag.evidence_quote,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract swappable value tokens, non-overlapping and priority ordered.
|
||||
* Status words are NOT value tokens — see STATUS_RE.
|
||||
* @returns {Array<{type: string, value: string, index: number}>} in order of appearance
|
||||
*/
|
||||
export function extractValueTokens(text) {
|
||||
if (!text) return [];
|
||||
const taken = []; // [start, end) ranges already consumed by a higher-priority type
|
||||
const out = [];
|
||||
for (const [type, re] of TOKEN_PATTERNS) {
|
||||
re.lastIndex = 0;
|
||||
let m;
|
||||
while ((m = re.exec(text)) !== null) {
|
||||
const start = m.index;
|
||||
const end = start + m[0].length;
|
||||
if (taken.some(([s, e]) => start < e && end > s)) continue;
|
||||
taken.push([start, end]);
|
||||
out.push({ type, value: m[0], index: start });
|
||||
}
|
||||
}
|
||||
return out.sort((a, b) => a.index - b.index);
|
||||
}
|
||||
|
||||
/** True if the text carries lifecycle-status vocabulary. */
|
||||
export function hasStatusWord(text) {
|
||||
return STATUS_RE.test(text || '');
|
||||
}
|
||||
|
||||
/**
|
||||
* The contiguous non-blank block containing `line` (1-indexed).
|
||||
*
|
||||
* This is the search window, and it is structural rather than a magic ±N: claims
|
||||
* are LLM-extracted restatements whose `line` often points at a table HEADER
|
||||
* while the asserted value sits in a row below. A block is exactly that table,
|
||||
* list, or paragraph. A blank line degenerates to itself.
|
||||
*/
|
||||
export function blockWindow(lines, line) {
|
||||
if (line < 1 || line > lines.length) return { start: line, end: line };
|
||||
if (lines[line - 1].trim() === '') return { start: line, end: line };
|
||||
let start = line;
|
||||
let end = line;
|
||||
while (start > 1 && lines[start - 2].trim() !== '') start -= 1;
|
||||
while (end < lines.length && lines[end].trim() !== '') end += 1;
|
||||
return { start, end };
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------- context condition
|
||||
//
|
||||
// MEASURED, NOT ASSUMED: §4 alone admits wrong edits. On the pilot it proved six
|
||||
// swaps of which four were false — "30-dagers" -> "24" from a quote saying 24
|
||||
// HOURS (unit crossing), an indexing rate replaced by a query throttle (metric
|
||||
// crossing), and two identifiers mutilated by digits harvested out of "E7" and a
|
||||
// table cell ("Microsoft Agent 365" -> "Agent 7", "text-embedding-ada-002" ->
|
||||
// "ada-2"). §4 constrains where the new value CAME FROM and what the edit LOOKS
|
||||
// LIKE; it constrains nothing about whether the two tokens denote the same
|
||||
// quantity.
|
||||
//
|
||||
// The condition below adds that, and it is deliberately lexical rather than
|
||||
// semantic: the token must sit under the same label, or the same trailing unit,
|
||||
// on both sides. No translation table — "dokumenter" is not taught to equal
|
||||
// "documents", because a synonym/translation table introduces a new fact source
|
||||
// and is an operator decision (§5-class), not an engineering one. The consequence
|
||||
// is measured and reported: a swap is provable essentially only where the context
|
||||
// is language-neutral (a URL, a code sample, a parameter key).
|
||||
|
||||
const WORD = /[A-Za-z0-9_.\-æøåÆØÅ]/;
|
||||
|
||||
/** Normalise a context run for comparison: lowercase, punctuation stripped. */
|
||||
const normContext = (s) => s.toLowerCase().replace(/[^a-z0-9æøå]/g, '');
|
||||
|
||||
/** The word run immediately left of [index], skipping any separator run first. */
|
||||
function leftContext(text, index) {
|
||||
let i = index - 1;
|
||||
// A separator run may be skipped; a word character adjacent to the token may
|
||||
// NOT be — that adjacency is what makes "7" part of the identifier "E7".
|
||||
if (i >= 0 && !WORD.test(text[i])) {
|
||||
while (i >= 0 && !WORD.test(text[i])) i -= 1;
|
||||
}
|
||||
let end = i + 1;
|
||||
while (i >= 0 && WORD.test(text[i])) i -= 1;
|
||||
return normContext(text.slice(i + 1, end));
|
||||
}
|
||||
|
||||
/** The word run immediately right of [index], skipping any separator run first. */
|
||||
function rightContext(text, index) {
|
||||
let i = index;
|
||||
if (i < text.length && !WORD.test(text[i])) {
|
||||
while (i < text.length && !WORD.test(text[i])) i += 1;
|
||||
}
|
||||
const start = i;
|
||||
while (i < text.length && WORD.test(text[i])) i += 1;
|
||||
return normContext(text.slice(start, i));
|
||||
}
|
||||
|
||||
/**
|
||||
* Do the two occurrences sit in corresponding context? True when a non-empty
|
||||
* label matches on the left, or a non-empty unit matches on the right.
|
||||
*/
|
||||
export function contextCorresponds(fileLine, fileIndex, fileLen, quote, quoteIndex, quoteLen) {
|
||||
const lf = leftContext(fileLine, fileIndex);
|
||||
const lq = leftContext(quote, quoteIndex);
|
||||
if (lf && lf === lq) return true;
|
||||
const rf = rightContext(fileLine, fileIndex + fileLen);
|
||||
const rq = rightContext(quote, quoteIndex + quoteLen);
|
||||
return Boolean(rf) && rf === rq;
|
||||
}
|
||||
|
||||
/** Distinct by type+value, preserving order. */
|
||||
function distinct(tokens) {
|
||||
const seen = new Set();
|
||||
return tokens.filter((t) => {
|
||||
const k = `${t.type} | ||||