linkedin-studio/scripts/analytics/tests/csv-parser.test.ts
Kjell Tore Guttormsen 63506f7d5c feat(linkedin-studio): N16 — out-of-network-andel + patterns-oppdatering + boundary-map [skip-docs]
Reach-splitten (in/out-of-network) er native i LinkedIns post-analytics siden juni
2026, men vises som PROSENT og finnes ikke i CSV-eksporten. Planen antok to
manuelle antall; verifiseringen viste prosent, så modellen er ett felt —
outOfNetworkPct — og in-network er komplementet.

- parseOptionalPercent: egen parser, ikke parseOptionalCount. Komma er desimal
  (36,5 -> 36.5, aldri 365), og verdi >100 avvises: i én kolonne kan ikke et
  absolutt antall skilles fra en andel, så svaret er unknown, ikke en gjetning.
  Blank/ikke-numerisk/negativ -> unknown; ekte 0 beholdes.
- Ett lagret halvpart, kryssjekket: In-network godtas og lagres som komplement;
  et transkribert par som ikke summerer til ~100 (±1 avrunding) forkastes som
  unknown i stedet for å bli halvveis trodd.
- weightedOutOfNetworkPct: impressions-vektet roll-up (avgOutOfNetworkPct, uke +
  måned). Flatt snitt lar en 50-visnings-post slå en på 10 000; poster uten
  avlesning ekskluderes, og null vekt gir undefined — aldri 0, aldri NaN.
- Reach inngår ALDRI i engagementRate (distribusjon != engasjement). Rapporten
  leser den som akvisisjon (ut) vs resonans (inn), og sier «ikke ført for denne
  perioden» framfor å estimere. En reach-innsikt går inn i N15s do-next-kanal.
- Step 7c (A2-F11): rapporten tilbyr diff mot brukerens engagement-patterns.md
  med eksplisitt go — aldri stille skriving, aldri inn i den shippede malen.
- Boundary-map (E#9): dwell eksplisitt umålbar, saves partner-gated, reach
  native men CSV-eksport uverifisert.
- Reach-frie importer er byte-identiske med før, på skjerm og på disk.

TDD: rødt bevist først (10 feilende), analytics 119 -> 144 tester, tsc ren.
test-runner 232 -> 247 (Section 16w, gulv 213 -> 228). Alle suiter grønne.
CHANGELOG: N15-oppføringen manglet og er backfilt sammen med N16.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01QxvWAjte7vPcF79QeSRvRJ
2026-07-25 15:56:04 +02:00

379 lines
14 KiB
TypeScript

import { describe, it } from "node:test";
import assert from "node:assert/strict";
import { parseLinkedInCSV } from "../src/parsers/csv-parser.js";
import { join, dirname } from "node:path";
import { fileURLToPath } from "node:url";
const __dirname = dirname(fileURLToPath(import.meta.url));
const fixturesDir = join(__dirname, "fixtures");
describe("CSV Parser", () => {
it("should parse standard CSV export", () => {
const filePath = join(fixturesDir, "sample-export.csv");
const batch = parseLinkedInCSV(filePath, "sample-export.csv");
assert.equal(batch.postCount, 8, "Should have 8 posts");
assert.equal(batch.posts.length, 8, "Posts array should have 8 items");
assert.equal(batch.exportFilename, "sample-export.csv");
assert.ok(batch.batchId, "Should have a batchId");
assert.ok(batch.importedAt, "Should have importedAt timestamp");
// Check first post
const firstPost = batch.posts[0];
assert.ok(firstPost.id, "Post should have an ID");
assert.ok(
firstPost.title.includes("uncomfortable truth"),
"Title should match"
);
assert.equal(firstPost.publishedDate, "2026-01-28");
assert.equal(firstPost.metrics.impressions, 4523);
assert.equal(firstPost.metrics.reactions, 87);
assert.equal(firstPost.metrics.comments, 23);
assert.equal(firstPost.metrics.shares, 12);
assert.equal(firstPost.metrics.clicks, 156);
assert.ok(firstPost.metrics.engagementRate > 0, "Should have engagement rate");
});
it("should handle European format", () => {
const filePath = join(fixturesDir, "european-export.csv");
const batch = parseLinkedInCSV(filePath, "european-export.csv");
assert.equal(batch.postCount, 2, "Should have 2 posts");
// Check that European number format is parsed correctly
const firstPost = batch.posts[0];
assert.equal(firstPost.metrics.impressions, 4523, "Should parse 4.523 as 4523");
assert.equal(firstPost.publishedDate, "2026-01-28", "Should normalize date from DD.MM.YYYY");
const secondPost = batch.posts[1];
assert.equal(secondPost.metrics.impressions, 2891, "Should parse 2.891 as 2891");
assert.equal(secondPost.publishedDate, "2026-01-26", "Should normalize date from DD.MM.YYYY");
});
it("should handle empty CSV", () => {
const filePath = join(fixturesDir, "empty-export.csv");
const batch = parseLinkedInCSV(filePath, "empty-export.csv");
assert.equal(batch.postCount, 0, "Should have 0 posts");
assert.equal(batch.posts.length, 0, "Posts array should be empty");
assert.equal(batch.dateRange.from, "", "Date range from should be empty");
assert.equal(batch.dateRange.to, "", "Date range to should be empty");
});
it("should handle BOM", () => {
const filePath = join(fixturesDir, "bom-export.csv");
const batch = parseLinkedInCSV(filePath, "bom-export.csv");
assert.equal(batch.postCount, 8, "Should parse BOM file correctly");
assert.ok(
batch.posts[0].title.includes("uncomfortable truth"),
"Should parse first post correctly despite BOM"
);
});
it("should calculate engagement rate", () => {
const filePath = join(fixturesDir, "sample-export.csv");
const batch = parseLinkedInCSV(filePath, "sample-export.csv");
const firstPost = batch.posts[0];
// (87+23+12+156)/4523 * 100 = 6.14...
const expectedRate = ((87 + 23 + 12 + 156) / 4523) * 100;
assert.ok(
Math.abs(firstPost.metrics.engagementRate - expectedRate) < 0.01,
`Engagement rate should be ~${expectedRate}, got ${firstPost.metrics.engagementRate}`
);
});
it("should generate deterministic post IDs", () => {
const filePath = join(fixturesDir, "sample-export.csv");
const batch1 = parseLinkedInCSV(filePath, "sample-export.csv");
const batch2 = parseLinkedInCSV(filePath, "sample-export.csv");
// Same post should have same ID
assert.equal(
batch1.posts[0].id,
batch2.posts[0].id,
"Same post should generate same ID"
);
// Different posts should have different IDs
assert.notEqual(
batch1.posts[0].id,
batch1.posts[1].id,
"Different posts should have different IDs"
);
});
it("should normalize dates to YYYY-MM-DD", () => {
const filePath = join(fixturesDir, "sample-export.csv");
const batch = parseLinkedInCSV(filePath, "sample-export.csv");
// All dates should be in YYYY-MM-DD format
batch.posts.forEach((post) => {
assert.match(
post.publishedDate,
/^\d{4}-\d{2}-\d{2}$/,
`Date ${post.publishedDate} should be in YYYY-MM-DD format`
);
});
// Check date range
assert.equal(batch.dateRange.from, "2026-01-13", "Date range from should be earliest date");
assert.equal(batch.dateRange.to, "2026-01-28", "Date range to should be latest date");
});
});
describe("Saves (manual-entry, optional)", () => {
it("should parse a Saves column when the user augments the CSV with it", () => {
const filePath = join(fixturesDir, "saves-export.csv");
const batch = parseLinkedInCSV(filePath, "saves-export.csv");
assert.equal(batch.postCount, 2, "Should have 2 posts");
// Row 1 carries a saves count read from native LinkedIn analytics.
assert.equal(batch.posts[0].metrics.saves, 42, "Should parse the Saves cell value");
});
it("should leave saves undefined when the Saves cell is blank (unknown != zero)", () => {
const filePath = join(fixturesDir, "saves-export.csv");
const batch = parseLinkedInCSV(filePath, "saves-export.csv");
// Row 2's Saves cell is empty — saves is unknown, NOT zero.
assert.equal(
batch.posts[1].metrics.saves,
undefined,
"Blank Saves cell must stay undefined, never coerced to 0"
);
});
it("should leave saves undefined for a standard export with no Saves column (backward-compat)", () => {
const filePath = join(fixturesDir, "sample-export.csv");
const batch = parseLinkedInCSV(filePath, "sample-export.csv");
for (const post of batch.posts) {
assert.equal(
post.metrics.saves,
undefined,
"Existing CSV exports without a Saves column must round-trip unchanged"
);
}
});
it("should NOT fold saves into engagementRate (kept comparable to historical data)", () => {
const filePath = join(fixturesDir, "saves-export.csv");
const batch = parseLinkedInCSV(filePath, "saves-export.csv");
// Row 1: (100+30+15+200)/5000 * 100 = 6.9 — saves (42) must NOT be in the numerator.
const expectedRate = ((100 + 30 + 15 + 200) / 5000) * 100;
assert.ok(
Math.abs(batch.posts[0].metrics.engagementRate - expectedRate) < 0.01,
`engagementRate should exclude saves (~${expectedRate}), got ${batch.posts[0].metrics.engagementRate}`
);
});
it("should treat an explicit '0' Saves cell as a genuine zero (not undefined)", () => {
const filePath = join(fixturesDir, "saves-edge-export.csv");
const batch = parseLinkedInCSV(filePath, "saves-edge-export.csv");
// A literal 0 in the Saves column is a real reading — zero saves, not unknown.
assert.equal(batch.posts[0].metrics.saves, 0, "Explicit '0' must stay 0, not collapse to undefined");
});
it("should leave saves undefined for a non-numeric Saves cell (unknown, never coerced to 0)", () => {
const filePath = join(fixturesDir, "saves-edge-export.csv");
const batch = parseLinkedInCSV(filePath, "saves-edge-export.csv");
// "n/a" is not a count — saves stays unknown, NOT silently flattened to 0.
assert.equal(
batch.posts[1].metrics.saves,
undefined,
"Non-numeric Saves cell must stay undefined — never coerced to 0"
);
});
});
/**
* Out-of-network reach (manual-entry, optional) — N16.
*
* LinkedIn surfaces the in-network/out-of-network split natively in post
* analytics (Discovery section, under Impressions; progressive global rollout
* from June 2026) as a PERCENTAGE split — not as two absolute counts, and not
* in the CSV export. So the ingest is a percent cell the operator transcribes,
* and the stored field is a single share: `outOfNetworkPct`. In-network is its
* complement by definition, so storing both halves would only invite a
* self-contradicting record.
*/
describe("Out-of-network reach (manual-entry, optional)", () => {
it("should parse an Out-of-network percent cell written with a % suffix", () => {
const filePath = join(fixturesDir, "reach-export.csv");
const batch = parseLinkedInCSV(filePath, "reach-export.csv");
assert.equal(batch.postCount, 4, "Should have 4 posts");
assert.equal(
batch.posts[0].metrics.outOfNetworkPct,
37,
"'37%' must parse to the number 37"
);
});
it("should leave outOfNetworkPct undefined when the cell is blank (unknown != zero)", () => {
const filePath = join(fixturesDir, "reach-export.csv");
const batch = parseLinkedInCSV(filePath, "reach-export.csv");
assert.equal(
batch.posts[1].metrics.outOfNetworkPct,
undefined,
"Blank Out-of-network cell must stay undefined, never coerced to 0"
);
});
it("should treat an explicit '0' as a genuine zero share (nothing left the network)", () => {
const filePath = join(fixturesDir, "reach-export.csv");
const batch = parseLinkedInCSV(filePath, "reach-export.csv");
assert.equal(
batch.posts[2].metrics.outOfNetworkPct,
0,
"Explicit '0' is a real reading — must stay 0, not collapse to undefined"
);
});
it("should read a European decimal comma as a decimal, not a thousands separator", () => {
const filePath = join(fixturesDir, "reach-export.csv");
const batch = parseLinkedInCSV(filePath, "reach-export.csv");
// A share never carries a thousands separator, so "36,5" is 36.5 percent.
// parseOptionalCount's US-thousands rule would read this as 365 — wrong here.
assert.equal(
batch.posts[3].metrics.outOfNetworkPct,
36.5,
"'36,5' must parse to 36.5 percent, never 365"
);
});
it("should leave outOfNetworkPct undefined for a standard export with no reach column (backward-compat)", () => {
const filePath = join(fixturesDir, "sample-export.csv");
const batch = parseLinkedInCSV(filePath, "sample-export.csv");
for (const post of batch.posts) {
assert.equal(
post.metrics.outOfNetworkPct,
undefined,
"Existing CSV exports without a reach column must round-trip unchanged"
);
}
});
it("should leave outOfNetworkPct undefined for a non-numeric cell (unknown, never 0)", () => {
const filePath = join(fixturesDir, "reach-edge-export.csv");
const batch = parseLinkedInCSV(filePath, "reach-edge-export.csv");
assert.equal(
batch.posts[0].metrics.outOfNetworkPct,
undefined,
"Non-numeric reach cell must stay undefined — never coerced to 0"
);
});
it("should refuse a value above 100 — a count and a share are undecidable in one column", () => {
const filePath = join(fixturesDir, "reach-edge-export.csv");
const batch = parseLinkedInCSV(filePath, "reach-edge-export.csv");
// "1234" in a share column is almost certainly an absolute impression count.
// We cannot tell which, so the honest answer is unknown — never a guess.
assert.equal(
batch.posts[1].metrics.outOfNetworkPct,
undefined,
"A share above 100 must stay undefined, never stored as-is"
);
});
it("should leave outOfNetworkPct undefined for a negative cell", () => {
const filePath = join(fixturesDir, "reach-edge-export.csv");
const batch = parseLinkedInCSV(filePath, "reach-edge-export.csv");
assert.equal(
batch.posts[2].metrics.outOfNetworkPct,
undefined,
"A negative share is not a real reading — must stay undefined"
);
});
it("should accept exactly 100 as a real reading (the boundary is inclusive)", () => {
const filePath = join(fixturesDir, "reach-edge-export.csv");
const batch = parseLinkedInCSV(filePath, "reach-edge-export.csv");
assert.equal(
batch.posts[3].metrics.outOfNetworkPct,
100,
"100 percent out-of-network is possible and must be kept"
);
});
it("should derive outOfNetworkPct from an In-network column as its complement", () => {
const filePath = join(fixturesDir, "reach-in-network-export.csv");
const batch = parseLinkedInCSV(filePath, "reach-in-network-export.csv");
// The operator transcribed the other half of the same split.
assert.equal(
batch.posts[0].metrics.outOfNetworkPct,
37,
"'In-network 63%' must store out-of-network 37"
);
assert.equal(
batch.posts[1].metrics.outOfNetworkPct,
undefined,
"A blank In-network cell leaves neither half known"
);
});
it("should keep the out-of-network half when both columns agree", () => {
const filePath = join(fixturesDir, "reach-both-export.csv");
const batch = parseLinkedInCSV(filePath, "reach-both-export.csv");
assert.equal(
batch.posts[0].metrics.outOfNetworkPct,
37,
"63 + 37 = 100 is consistent — keep the out-of-network reading"
);
});
it("should refuse a contradictory split rather than pick a half", () => {
const filePath = join(fixturesDir, "reach-both-export.csv");
const batch = parseLinkedInCSV(filePath, "reach-both-export.csv");
// 63 + 20 = 83. One of the two cells is a misreading and we cannot tell
// which, so the record stays unknown instead of silently trusting one.
assert.equal(
batch.posts[1].metrics.outOfNetworkPct,
undefined,
"A split that does not sum to ~100 must stay undefined"
);
});
it("should tolerate one point of rounding slack between the two halves", () => {
const filePath = join(fixturesDir, "reach-both-export.csv");
const batch = parseLinkedInCSV(filePath, "reach-both-export.csv");
// 62 + 37 = 99: the UI rounds each half independently, so a one-point gap
// is rounding, not a misreading.
assert.equal(
batch.posts[2].metrics.outOfNetworkPct,
37,
"A 99 or 101 sum is rounding slack — keep the out-of-network reading"
);
});
it("should NOT fold out-of-network reach into engagementRate", () => {
const filePath = join(fixturesDir, "reach-export.csv");
const batch = parseLinkedInCSV(filePath, "reach-export.csv");
// Row 1: (100+30+15+200)/5000 * 100 = 6.9. Reach is a distribution signal,
// not engagement — it must not touch the rate.
const expectedRate = ((100 + 30 + 15 + 200) / 5000) * 100;
assert.ok(
Math.abs(batch.posts[0].metrics.engagementRate - expectedRate) < 0.01,
`engagementRate should exclude reach (~${expectedRate}), got ${batch.posts[0].metrics.engagementRate}`
);
});
});