feat(inbox): .docx extraction — hidden runs, comments, core metadata (stage 2d)
The docx payload hides where a human reviewing the file in Word does not look. The extractor surfaces all three regions into the concept text so the stage-2 scan catches the injection: - hidden/vanish runs (w:vanish) — still runs, so paragraph.text includes them; - review comments (doc.comments[].text); - core metadata properties (subject/keywords/comments/title/category/author). Detach-proof: the same visible body WITHOUT the hidden run ADMITs, so it is the extractor surfacing the hidden region that caught it, not merely 'a docx'. python-docx is imported lazily (dev/showcase-scoped, not a core dep). Tests 300 -> 305.
This commit is contained in:
parent
02d59efeb2
commit
26a231d6c4
2 changed files with 88 additions and 0 deletions
|
|
@ -124,6 +124,32 @@ def _ingest_csv(rel_name, text, bundle, provenance, rejected):
|
|||
_materialize_text(rel_name, text, "csv", bundle, provenance)
|
||||
|
||||
|
||||
def _extract_docx_text(fs_path) -> str:
|
||||
"""Extract text from a ``.docx``, including the regions a human reviewing the
|
||||
document in Word does not see: hidden/vanish runs (still runs, so ``paragraph
|
||||
.text`` includes them), core metadata properties, and review comments.
|
||||
|
||||
``python-docx`` is imported lazily — it is a dev/showcase-scoped parser, not a
|
||||
core dependency; a consumer would guard the import behind their own extra.
|
||||
"""
|
||||
from docx import Document
|
||||
|
||||
doc = Document(str(fs_path))
|
||||
parts = [para.text for para in doc.paragraphs if para.text]
|
||||
|
||||
cp = doc.core_properties
|
||||
for attr in ("title", "subject", "keywords", "comments", "category", "author"):
|
||||
value = getattr(cp, attr, None)
|
||||
if value:
|
||||
parts.append(str(value))
|
||||
|
||||
for comment in doc.comments:
|
||||
if comment.text:
|
||||
parts.append(comment.text)
|
||||
|
||||
return "\n".join(parts)
|
||||
|
||||
|
||||
def _ingest_regular_file(fs_path, rel_name, bundle, provenance, rejected, *, strict):
|
||||
"""Dispatch one on-disk file by suffix. ``strict`` raises on an unsupported
|
||||
suffix (a top-level drop); a folder walk passes ``strict=False`` to skip it."""
|
||||
|
|
@ -131,6 +157,8 @@ def _ingest_regular_file(fs_path, rel_name, bundle, provenance, rejected, *, str
|
|||
if suffix == ".csv":
|
||||
text = Path(fs_path).read_text(encoding="utf-8", errors="replace")
|
||||
_ingest_csv(rel_name, text, bundle, provenance, rejected)
|
||||
elif suffix == ".docx":
|
||||
_materialize_text(rel_name, _extract_docx_text(fs_path), "docx", bundle, provenance)
|
||||
elif suffix in _TEXT_SUFFIXES:
|
||||
text = Path(fs_path).read_text(encoding="utf-8", errors="replace")
|
||||
_materialize_text(rel_name, text, suffix.lstrip("."), bundle, provenance)
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue