fix(ms-ai-architect): registry-herding Fase 1a — +5 sitemap-prefiks + skjemaløs URL-ekstraksjon

Update-mekanismen lot siterte URL-er stå som not_in_sitemap fordi deres docset
aldri ble pollet, og skjemaløse citater (tabellceller: bare learn.microsoft.com/...
uten https://) ble aldri ekstrahert.

- Taxonomy: +5 docsets (graph, ai-builder, power-apps, power-automate,
  microsoftsearch) — hver ett child-sitemap, verifisert live mot indeksen.
  Bevist: 21/24 not_in_sitemap-URL-er i disse docsetene blir nå tracked.
- url-normalize: extractUrls fanger skjemaløse citater-med-sti (krever /path
  etter domenet → avviser bare-domene-prosa og JSON-eksempler); normalizeUrl
  kanonikaliserer scheme til https. Bevist: +19 nye URL-er ekstraherbare.
- Bugfix: backtick fra inline-kode-citat (`learn.microsoft.com/...`) lekket inn
  i URL-en — ekskludert i regex + trailing-strip.

TDD: tests/kb-update/test-url-normalize.test.mjs (ny, 14) + test-taxonomy
prefiks-count 18→23 + Fase 1a-assertion. Full suite 338/338, 0 regresjon.
Registry-refresh (build-registry --merge + poll) bevisst utsatt — unngår
552KB re-order-churn; effekten er empirisk verifisert read-only.
This commit is contained in:
Kjell Tore Guttormsen 2026-06-26 10:15:46 +02:00
commit e74646d395
4 changed files with 152 additions and 7 deletions

View file

@ -24,9 +24,9 @@ import {
const tax = loadTaxonomy();
// --- (a) sitemap prefixes — poll/discover parity ---
test('sitemap_prefixes — 18 entries with poll/discover parity', () => {
test('sitemap_prefixes — 23 entries with poll/discover parity', () => {
const prefixes = getSitemapPrefixes(tax);
assert.equal(prefixes.length, 18);
assert.equal(prefixes.length, 23);
// poll superset includes the 6 that discover previously lacked
for (const p of ['microsoftteams_en-us_', 'sharepoint_en-us_', 'microsoft-365_en-us_',
'training_en-us_', 'cloud-computing_en-us_', 'privacy_en-us_']) {
@ -37,6 +37,17 @@ test('sitemap_prefixes — 18 entries with poll/discover parity', () => {
assert.ok(!prefixes.includes('dotnet_en-us_'));
});
// Fase 1a registry-herding: 5 stack-relevant docsets whose cited URLs were
// stranded as not_in_sitemap because their child sitemaps were never polled.
// Each is a single child sitemap (verified live against the sitemap index).
test('sitemap_prefixes — Fase 1a additions (graph / power / ai-builder / search)', () => {
const prefixes = getSitemapPrefixes(tax);
for (const p of ['graph_en-us_', 'ai-builder_en-us_', 'power-apps_en-us_',
'power-automate_en-us_', 'microsoftsearch_en-us_']) {
assert.ok(prefixes.includes(p), `missing Fase 1a prefix ${p}`);
}
});
// --- (c) category → skill — PHYSICAL DISK is canonical ---
test('category_skill — disk-truth canonical (the 4 formerly stale entries)', () => {
// category-skill-map.json had these as ms-ai-engineering; disk says otherwise.

View file

@ -0,0 +1,115 @@
// tests/kb-update/test-url-normalize.test.mjs
// Unit tests for scripts/kb-update/lib/url-normalize.mjs.
// Pins the canonical-URL contract that build-registry ↔ poll-sitemaps matching
// depends on, AND the Fase 1a registry-herding extension: schema-less citations
// (learn.microsoft.com/... without the https:// prefix) must be extracted and
// canonicalised to https://, while bare-domain prose mentions (no path) and
// JSON example strings must NOT be captured.
import { test } from 'node:test';
import assert from 'node:assert/strict';
import { normalizeUrl, extractUrls } from '../../scripts/kb-update/lib/url-normalize.mjs';
// --- normalizeUrl: existing contract (regression guard) ---
test('normalizeUrl — strips locale, fragment, query, trailing slash; lowercases', () => {
assert.equal(
normalizeUrl('https://learn.microsoft.com/en-us/azure/AI-Foundry/Overview/?view=x#sec'),
'https://learn.microsoft.com/azure/ai-foundry/overview'
);
});
test('normalizeUrl — strips trailing markdown punctuation', () => {
assert.equal(
normalizeUrl('https://learn.microsoft.com/azure/foo).'),
'https://learn.microsoft.com/azure/foo'
);
});
test('normalizeUrl — non-learn URL returns null', () => {
assert.equal(normalizeUrl('https://example.com/azure/foo'), null);
assert.equal(normalizeUrl(''), null);
assert.equal(normalizeUrl(null), null);
});
// --- normalizeUrl: Fase 1a — schema-less canonicalisation ---
test('normalizeUrl — schema-less citation gets https:// prepended', () => {
assert.equal(
normalizeUrl('learn.microsoft.com/azure/api-management/genai-gateway-capabilities'),
'https://learn.microsoft.com/azure/api-management/genai-gateway-capabilities'
);
});
test('normalizeUrl — schema-less with en-us locale is also canonicalised', () => {
assert.equal(
normalizeUrl('learn.microsoft.com/en-us/compliance/assurance/assurance-artificial-intelligence'),
'https://learn.microsoft.com/compliance/assurance/assurance-artificial-intelligence'
);
});
test('normalizeUrl — idempotent on schema-less input', () => {
const once = normalizeUrl('learn.microsoft.com/azure/foo');
assert.equal(normalizeUrl(once), once);
});
// --- extractUrls: existing contract (regression guard) ---
test('extractUrls — extracts https markdown-link URL', () => {
assert.deepEqual(
extractUrls('See [docs](https://learn.microsoft.com/azure/foo) here.'),
['https://learn.microsoft.com/azure/foo']
);
});
test('extractUrls — dedups same URL across forms', () => {
const out = extractUrls(
'https://learn.microsoft.com/azure/foo and learn.microsoft.com/azure/foo'
);
assert.deepEqual(out, ['https://learn.microsoft.com/azure/foo']);
});
// --- extractUrls: Fase 1a — schema-less table-cell citations (the 2 named files) ---
test('extractUrls — captures schema-less table-cell citation', () => {
const row = '| **GenAI Gateway** | desc | learn.microsoft.com/azure/api-management/genai-gateway-capabilities |';
assert.deepEqual(
extractUrls(row),
['https://learn.microsoft.com/azure/api-management/genai-gateway-capabilities']
);
});
test('extractUrls — captures multiple schema-less rows', () => {
const text = [
'| A | learn.microsoft.com/en-us/compliance/assurance/assurance-artificial-intelligence |',
'| B | learn.microsoft.com/en-us/azure/cloud-adoption-framework/scenarios/ai/govern |',
].join('\n');
assert.deepEqual(extractUrls(text), [
'https://learn.microsoft.com/compliance/assurance/assurance-artificial-intelligence',
'https://learn.microsoft.com/azure/cloud-adoption-framework/scenarios/ai/govern',
]);
});
// --- extractUrls: must NOT capture bare-domain prose / JSON examples ---
test('extractUrls — ignores bare-domain prose mention (no path)', () => {
assert.deepEqual(
extractUrls('All info fra offisiell dokumentasjon (learn.microsoft.com, blogs.microsoft.com).'),
[]
);
});
test('extractUrls — ignores JSON example string with no path', () => {
assert.deepEqual(extractUrls('{"url": "learn.microsoft.com"}'), []);
});
test('extractUrls — strips wrapping backticks from inline-code citation', () => {
// `learn.microsoft.com/agent-framework/` inside prose backticks must not leak
// the backtick into the URL (would never match a sitemap entry).
assert.deepEqual(
extractUrls('verifisert mot `learn.microsoft.com/agent-framework/`).'),
['https://learn.microsoft.com/agent-framework']
);
});
test('normalizeUrl — strips trailing backtick', () => {
assert.equal(
normalizeUrl('https://learn.microsoft.com/azure/foo`'),
'https://learn.microsoft.com/azure/foo'
);
});