From 8faccd61ee99a1850fdfdacbe227e9911a9eab0c Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Mon, 17 Aug 2026 18:28:31 +0000 Subject: [PATCH 1/4] test(email): add RED Slice 3 media admission fixtures Add four synthetic multipart/related .eml fixtures and failing acceptance tests for CID resolution, unresolved CID fail-closed, 1x1 tracking-pixel exclusion, and repeated base64 hash provenance. Co-authored-by: Seongho Bae --- .../cid_related_document_image.eml | 20 ++++ .../repeated_identical_base64.eml | 27 +++++ .../tracking_pixel_1x1.eml | 21 ++++ .../email_media_admission/unresolved_cid.eml | 13 +++ backend/tests/test_email_media_admission.py | 99 +++++++++++++++++++ 5 files changed, 180 insertions(+) create mode 100644 backend/tests/fixtures/email_media_admission/cid_related_document_image.eml create mode 100644 backend/tests/fixtures/email_media_admission/repeated_identical_base64.eml create mode 100644 backend/tests/fixtures/email_media_admission/tracking_pixel_1x1.eml create mode 100644 backend/tests/fixtures/email_media_admission/unresolved_cid.eml create mode 100644 backend/tests/test_email_media_admission.py diff --git a/backend/tests/fixtures/email_media_admission/cid_related_document_image.eml b/backend/tests/fixtures/email_media_admission/cid_related_document_image.eml new file mode 100644 index 000000000..0cbef177f --- /dev/null +++ b/backend/tests/fixtures/email_media_admission/cid_related_document_image.eml @@ -0,0 +1,20 @@ +MIME-Version: 1.0 +From: sender@naruon.test +To: member@naruon.test +Subject: Slice 3 admission fixture +Date: Mon, 17 Aug 2026 12:00:00 +0000 +Content-Type: multipart/related; boundary="rel-bound"; type="text/html" + +--rel-bound +Content-Type: text/html; charset="utf-8" +Content-Transfer-Encoding: 7bit + +

Quarterly chart

chart +--rel-bound +Content-Type: image/png +Content-ID: +Content-Disposition: inline +Content-Transfer-Encoding: base64 + +iVBORw0KGgoAAAANSUhEUgAAAEAAAAAwCAIAAAAuKetIAAAAQ0lEQVR42u3PQQkAAAgEsItiNKMZ1Qp+hcEKLNXzWgQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQErhZ0pIDE5ejydQAAAABJRU5ErkJggg== +--rel-bound-- diff --git a/backend/tests/fixtures/email_media_admission/repeated_identical_base64.eml b/backend/tests/fixtures/email_media_admission/repeated_identical_base64.eml new file mode 100644 index 000000000..de2479101 --- /dev/null +++ b/backend/tests/fixtures/email_media_admission/repeated_identical_base64.eml @@ -0,0 +1,27 @@ +MIME-Version: 1.0 +From: sender@naruon.test +To: member@naruon.test +Subject: Slice 3 admission fixture +Date: Mon, 17 Aug 2026 12:00:00 +0000 +Content-Type: multipart/related; boundary="rel-bound"; type="text/html" + +--rel-bound +Content-Type: text/html; charset="utf-8" +Content-Transfer-Encoding: 7bit + +

Two identical inline scans

+--rel-bound +Content-Type: image/png +Content-ID: +Content-Disposition: inline +Content-Transfer-Encoding: base64 + +iVBORw0KGgoAAAANSUhEUgAAACAAAAAYCAIAAAAUMWhjAAAAJElEQVR42mM4EaBBU8QwasGoBaMWjFowasGoBaMWjFowNCwAAOj5wC7XMt9ZAAAAAElFTkSuQmCC +--rel-bound +Content-Type: image/png +Content-ID: +Content-Disposition: inline +Content-Transfer-Encoding: base64 + +iVBORw0KGgoAAAANSUhEUgAAACAAAAAYCAIAAAAUMWhjAAAAJElEQVR42mM4EaBBU8QwasGoBaMWjFowasGoBaMWjFowNCwAAOj5wC7XMt9ZAAAAAElFTkSuQmCC +--rel-bound-- diff --git a/backend/tests/fixtures/email_media_admission/tracking_pixel_1x1.eml b/backend/tests/fixtures/email_media_admission/tracking_pixel_1x1.eml new file mode 100644 index 000000000..7ed7029c8 --- /dev/null +++ b/backend/tests/fixtures/email_media_admission/tracking_pixel_1x1.eml @@ -0,0 +1,21 @@ +MIME-Version: 1.0 +From: sender@naruon.test +To: member@naruon.test +Subject: Slice 3 admission fixture +Date: Mon, 17 Aug 2026 12:00:00 +0000 +Content-Type: multipart/related; boundary="rel-bound"; type="text/html" + +--rel-bound +Content-Type: text/html; charset="utf-8" +Content-Transfer-Encoding: 7bit + +

Newsletter

+--rel-bound +Content-Type: image/gif +Content-ID: +Content-Location: https://click.list-manage.com/track/open.php?u=fixture +Content-Disposition: inline +Content-Transfer-Encoding: base64 + +R0lGODlhAQABAIAAAAAAAP///yH5BAEAAAAALAAAAAABAAEAAAICRAEAOw== +--rel-bound-- diff --git a/backend/tests/fixtures/email_media_admission/unresolved_cid.eml b/backend/tests/fixtures/email_media_admission/unresolved_cid.eml new file mode 100644 index 000000000..43d081f66 --- /dev/null +++ b/backend/tests/fixtures/email_media_admission/unresolved_cid.eml @@ -0,0 +1,13 @@ +MIME-Version: 1.0 +From: sender@naruon.test +To: member@naruon.test +Subject: Slice 3 admission fixture +Date: Mon, 17 Aug 2026 12:00:00 +0000 +Content-Type: multipart/related; boundary="rel-bound"; type="text/html" + +--rel-bound +Content-Type: text/html; charset="utf-8" +Content-Transfer-Encoding: 7bit + +missing +--rel-bound-- diff --git a/backend/tests/test_email_media_admission.py b/backend/tests/test_email_media_admission.py new file mode 100644 index 000000000..e7569d47b --- /dev/null +++ b/backend/tests/test_email_media_admission.py @@ -0,0 +1,99 @@ +"""Acceptance tests for #1350 Slice 3 email inline-media admission.""" + +from __future__ import annotations + +from pathlib import Path + +from services.email_media_admission import admit_email_inline_media + +FIXTURE_DIRECTORY = Path(__file__).parent / "fixtures" / "email_media_admission" + + +def _fixture_bytes(file_name: str) -> bytes: + """Read one synthetic RFC 5322 fixture as raw message bytes.""" + return (FIXTURE_DIRECTORY / file_name).read_bytes() + + +def test_cid_related_document_image_resolves_with_hash_provenance() -> None: + """A cid: reference inside multipart/related admits the local PNG as document evidence.""" + result = admit_email_inline_media(_fixture_bytes("cid_related_document_image.eml")) + + assert result.remote_fetch_policy == "disabled" + assert len(result.cid_references) == 1 + cid_reference = result.cid_references[0] + assert cid_reference.error_code is None + assert cid_reference.content_id == "chart@naruon.test" + assert cid_reference.media_classification == "document_image" + assert cid_reference.evidence_boundary == "known" + assert cid_reference.content_sha256 is not None + assert len(cid_reference.content_sha256) == 64 + + assert len(result.inline_images) == 1 + admitted_image = result.inline_images[0] + assert admitted_image.source_part_index >= 0 + assert admitted_image.content_id == "chart@naruon.test" + assert admitted_image.media_classification == "document_image" + assert admitted_image.evidence_boundary == "known" + assert admitted_image.error_code is None + assert admitted_image.content_sha256 == cid_reference.content_sha256 + assert admitted_image.source_part_index == cid_reference.source_part_index + assert admitted_image.pixel_width == 64 + assert admitted_image.pixel_height == 48 + + +def test_unresolved_cid_fails_closed_and_is_not_document_evidence() -> None: + """A missing cid: target fails closed with a stable error_code and is not admitted.""" + result = admit_email_inline_media(_fixture_bytes("unresolved_cid.eml")) + + assert result.remote_fetch_policy == "disabled" + assert len(result.cid_references) == 1 + cid_reference = result.cid_references[0] + assert cid_reference.content_id == "missing@naruon.test" + assert cid_reference.error_code == "unresolved_cid_reference" + assert cid_reference.media_classification is None + assert cid_reference.content_sha256 is None + assert cid_reference.source_part_index is None + assert result.inline_images == () + + +def test_tracking_pixel_is_not_document_evidence() -> None: + """A 1x1 GIF with a tracker Content-Location is classified, not sent as a document.""" + result = admit_email_inline_media(_fixture_bytes("tracking_pixel_1x1.eml")) + + assert result.remote_fetch_policy == "disabled" + assert len(result.inline_images) == 1 + tracking_pixel = result.inline_images[0] + assert tracking_pixel.media_classification == "tracking_pixel" + assert tracking_pixel.evidence_boundary == "known" + assert tracking_pixel.pixel_width == 1 + assert tracking_pixel.pixel_height == 1 + assert tracking_pixel.declared_content_type == "image/gif" + assert tracking_pixel.content_location is not None + assert "list-manage.com" in tracking_pixel.content_location + assert tracking_pixel.error_code is None + assert all( + image.media_classification != "document_image" + for image in result.inline_images + ) + + assert len(result.cid_references) == 1 + cid_reference = result.cid_references[0] + assert cid_reference.media_classification == "tracking_pixel" + assert cid_reference.error_code is None + assert cid_reference.content_sha256 == tracking_pixel.content_sha256 + + +def test_repeated_identical_base64_parts_share_the_same_hash() -> None: + """Identical decoded source bytes keep one SHA-256 across distinct part positions.""" + result = admit_email_inline_media(_fixture_bytes("repeated_identical_base64.eml")) + + assert len(result.inline_images) == 2 + first_image, second_image = result.inline_images + assert first_image.content_sha256 == second_image.content_sha256 + assert first_image.source_part_index != second_image.source_part_index + assert {first_image.content_id, second_image.content_id} == { + "scan-one@naruon.test", + "scan-two@naruon.test", + } + assert first_image.media_classification == "document_image" + assert second_image.media_classification == "document_image" From a6847b22e009b0457e28b17bc6d06d8455a006cc Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Mon, 17 Aug 2026 18:31:35 +0000 Subject: [PATCH 2/4] feat(email): admit and classify local inline media Resolve cid: against same-message multipart/related parts, fail closed on unresolved CID, and classify tracking_pixel, unsupported_media, and document_image from local header and dimension evidence only. No remote fetch, OCR, or VLM. Co-authored-by: Seongho Bae --- backend/services/email_media_admission.py | 515 ++++++++++++++++++ .../test_email_media_admission_boundaries.py | 356 ++++++++++++ 2 files changed, 871 insertions(+) create mode 100644 backend/services/email_media_admission.py create mode 100644 backend/tests/test_email_media_admission_boundaries.py diff --git a/backend/services/email_media_admission.py b/backend/services/email_media_admission.py new file mode 100644 index 000000000..2e0a95c8d --- /dev/null +++ b/backend/services/email_media_admission.py @@ -0,0 +1,515 @@ +"""Admit and classify local email inline images without remote fetch or models. + +This module is the #1350 Slice 3 admission contract. It resolves ``cid:`` +references against the same message's ``multipart/related`` parts, classifies +admitted images into a closed set, and returns hash plus part-index +provenance. It does not call OCR, a VLM, NewsDOM, or any LLM, does not +mutate provider mail, and does not fetch ``http`` or ``https`` image URLs. +""" + +from __future__ import annotations + +import hashlib +import re +import urllib.parse +from dataclasses import dataclass +from email import policy +from email.message import Message +from email.parser import BytesParser + +DOCUMENT_IMAGE_CLASSIFICATION = "document_image" +TRACKING_PIXEL_CLASSIFICATION = "tracking_pixel" +UNSUPPORTED_MEDIA_CLASSIFICATION = "unsupported_media" + +KNOWN_EVIDENCE_BOUNDARY = "known" +UNKNOWN_EVIDENCE_BOUNDARY = "unknown" + +UNRESOLVED_CID_ERROR_CODE = "unresolved_cid_reference" +REMOTE_FETCH_POLICY = "disabled" + +TRACKING_PIXEL_MAX_EDGE = 1 +TINY_TRACKER_GIF_MAX_BYTES = 48 + +SUPPORTED_IMAGE_TYPES = frozenset( + {"image/png", "image/jpeg", "image/gif", "image/webp"} +) +IMAGE_TYPE_ALIASES = {"image/jpg": "image/jpeg"} +TRACKER_CONTENT_TYPES = frozenset({"image/gif"}) +TRACKER_HOST_SUFFIXES = ( + "list-manage.com", + "doubleclick.net", + "google-analytics.com", + "googletagmanager.com", + "adsrvr.org", +) +TRACKER_PATH_MARKERS = ( + "/track/open", + "/pixel.gif", + "/open.gif", + "/open.php", +) + +_CONTROL_CHARACTER_RE = re.compile(r"[\x00-\x1f\x7f]") +_IMG_SRC_RE = re.compile( + r"""]*\bsrc\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+))""", + flags=re.IGNORECASE | re.DOTALL, +) + + +@dataclass(frozen=True) +class InlineImageAdmission: + """Provenance and closed-set classification for one local MIME image part.""" + + source_part_index: int + content_id: str | None + content_sha256: str + media_classification: str + evidence_boundary: str + error_code: str | None + declared_content_type: str + content_location: str | None + pixel_width: int | None + pixel_height: int | None + + +@dataclass(frozen=True) +class CidReferenceAdmission: + """Outcome of one ``cid:`` HTML reference against related MIME parts.""" + + raw_reference: str + content_id: str | None + source_part_index: int | None + content_sha256: str | None + media_classification: str | None + error_code: str | None + evidence_boundary: str | None + + +@dataclass(frozen=True) +class EmailMediaAdmissionResult: + """Admission outcomes for one raw RFC 5322 message.""" + + inline_images: tuple[InlineImageAdmission, ...] + cid_references: tuple[CidReferenceAdmission, ...] + remote_fetch_policy: str = REMOTE_FETCH_POLICY + + +@dataclass(frozen=True) +class _RelatedImagePart: + """Internal related-scope image part used only during CID matching.""" + + related_scope: str | None + content_id: str | None + admission: InlineImageAdmission + + +def admit_email_inline_media(raw_message: bytes) -> EmailMediaAdmissionResult: + """Admit local inline images and resolve same-message ``cid:`` references. + + Remote ``http(s)`` image references are ignored for admission and are never + downloaded. Unresolved ``cid:`` references fail closed with + ``unresolved_cid_reference`` and are not treated as document evidence. + + Args: + raw_message: Complete RFC 5322 message bytes, including MIME headers. + + Returns: + Classified local image parts plus every HTML ``cid:`` outcome. + + Raises: + TypeError: If ``raw_message`` is not ``bytes``. + """ + if not isinstance(raw_message, bytes): + raise TypeError("raw_message must be bytes") + + message = BytesParser(policy=policy.default).parsebytes(raw_message) + images: list[_RelatedImagePart] = [] + html_parts: list[tuple[str | None, str]] = [] + part_index_holder = [0] + _collect_message_parts( + message, + path="0", + related_scope=None, + part_index_holder=part_index_holder, + images=images, + html_parts=html_parts, + ) + + cid_references: list[CidReferenceAdmission] = [] + for related_scope, html_source in html_parts: + for raw_reference in _html_image_references(html_source): + if _is_remote_reference(raw_reference): + continue + if not raw_reference.casefold().startswith("cid:"): + continue + cid_references.append( + _resolve_cid_reference( + raw_reference=raw_reference, + related_scope=related_scope, + images=images, + ) + ) + + return EmailMediaAdmissionResult( + inline_images=tuple(item.admission for item in images), + cid_references=tuple(cid_references), + ) + + +def _collect_message_parts( + part: Message, + *, + path: str, + related_scope: str | None, + part_index_holder: list[int], + images: list[_RelatedImagePart], + html_parts: list[tuple[str | None, str]], +) -> None: + """Walk one MIME subtree and collect HTML plus local image parts.""" + current_scope = ( + path if part.get_content_type() == "multipart/related" else related_scope + ) + children = _message_children(part) + if children: + for child_index, child in enumerate(children): + _collect_message_parts( + child, + path=f"{path}.{child_index}", + related_scope=current_scope, + part_index_holder=part_index_holder, + images=images, + html_parts=html_parts, + ) + return + + source_part_index = part_index_holder[0] + part_index_holder[0] += 1 + declared_content_type = _normalize_content_type(part.get_content_type()) + if declared_content_type == "text/html": + html_parts.append((current_scope, _decode_text_part(part))) + return + if part.get_content_maintype().casefold() != "image": + return + + raw_payload = part.get_payload(decode=True) + payload_bytes = raw_payload if isinstance(raw_payload, bytes) else b"" + images.append( + _build_related_image_part( + source_part_index=source_part_index, + related_scope=current_scope, + content_id=_normalize_content_id(part.get("Content-ID")), + content_location=_header_text(part.get("Content-Location")), + declared_content_type=declared_content_type, + payload_bytes=payload_bytes, + ) + ) + + +def _build_related_image_part( + *, + source_part_index: int, + related_scope: str | None, + content_id: str | None, + content_location: str | None, + declared_content_type: str, + payload_bytes: bytes, +) -> _RelatedImagePart: + """Classify one decoded image part and retain its related-scope identity.""" + classification, evidence_boundary, pixel_width, pixel_height = _classify_image( + declared_content_type=declared_content_type, + payload_bytes=payload_bytes, + content_location=content_location, + ) + admission = InlineImageAdmission( + source_part_index=source_part_index, + content_id=content_id, + content_sha256=hashlib.sha256(payload_bytes).hexdigest(), + media_classification=classification, + evidence_boundary=evidence_boundary, + error_code=None, + declared_content_type=declared_content_type, + content_location=content_location, + pixel_width=pixel_width, + pixel_height=pixel_height, + ) + return _RelatedImagePart( + related_scope=related_scope, + content_id=content_id, + admission=admission, + ) + + +def _classify_image( + *, + declared_content_type: str, + payload_bytes: bytes, + content_location: str | None, +) -> tuple[str, str, int | None, int | None]: + """Return classification, evidence boundary, and header-derived dimensions.""" + inferred_content_type = _infer_image_content_type(payload_bytes) + if ( + declared_content_type not in SUPPORTED_IMAGE_TYPES + or inferred_content_type is None + or inferred_content_type != declared_content_type + ): + return ( + UNSUPPORTED_MEDIA_CLASSIFICATION, + KNOWN_EVIDENCE_BOUNDARY, + None, + None, + ) + + dimensions = _pixel_dimensions_from_header(declared_content_type, payload_bytes) + pixel_width, pixel_height = dimensions if dimensions is not None else (None, None) + if _is_tracking_pixel( + pixel_width=pixel_width, + pixel_height=pixel_height, + declared_content_type=declared_content_type, + content_location=content_location, + payload_bytes=payload_bytes, + ): + return ( + TRACKING_PIXEL_CLASSIFICATION, + KNOWN_EVIDENCE_BOUNDARY, + pixel_width, + pixel_height, + ) + if pixel_width is None or pixel_height is None: + return ( + DOCUMENT_IMAGE_CLASSIFICATION, + UNKNOWN_EVIDENCE_BOUNDARY, + pixel_width, + pixel_height, + ) + return ( + DOCUMENT_IMAGE_CLASSIFICATION, + KNOWN_EVIDENCE_BOUNDARY, + pixel_width, + pixel_height, + ) + + +def _is_tracking_pixel( + *, + pixel_width: int | None, + pixel_height: int | None, + declared_content_type: str, + content_location: str | None, + payload_bytes: bytes, +) -> bool: + """Decide tracking-pixel admission from local evidence only. + + Evidence is header-derived pixel size, an already-present Content-Location + tracker pattern, or a typical tracker content-type with a tiny GIF payload. + Remote pixels are never downloaded. + """ + if ( + pixel_width is not None + and pixel_height is not None + and pixel_width <= TRACKING_PIXEL_MAX_EDGE + and pixel_height <= TRACKING_PIXEL_MAX_EDGE + ): + return True + if _content_location_matches_tracker(content_location): + return True + return ( + declared_content_type in TRACKER_CONTENT_TYPES + and (pixel_width is None or pixel_height is None) + and 0 < len(payload_bytes) <= TINY_TRACKER_GIF_MAX_BYTES + ) + + +def _content_location_matches_tracker(content_location: str | None) -> bool: + """Match known tracker hosts or paths on an already-present header value.""" + if content_location is None: + return False + try: + parsed_url = urllib.parse.urlsplit(content_location) + except ValueError: + return False + host_name = (parsed_url.hostname or "").casefold() + path_value = (parsed_url.path or "").casefold() + if any( + host_name == suffix or host_name.endswith(f".{suffix}") + for suffix in TRACKER_HOST_SUFFIXES + ): + return True + return any(marker in path_value for marker in TRACKER_PATH_MARKERS) + + +def _resolve_cid_reference( + *, + raw_reference: str, + related_scope: str | None, + images: list[_RelatedImagePart], +) -> CidReferenceAdmission: + """Bind one ``cid:`` URL to a unique related-scope Content-ID, or fail closed.""" + content_id = _normalize_cid_url(raw_reference) + if content_id is None or related_scope is None: + return _unresolved_cid_reference(raw_reference, content_id) + + candidates = [ + image + for image in images + if image.related_scope == related_scope and image.content_id == content_id + ] + if len(candidates) != 1: + return _unresolved_cid_reference(raw_reference, content_id) + + admission = candidates[0].admission + return CidReferenceAdmission( + raw_reference=raw_reference, + content_id=content_id, + source_part_index=admission.source_part_index, + content_sha256=admission.content_sha256, + media_classification=admission.media_classification, + error_code=None, + evidence_boundary=admission.evidence_boundary, + ) + + +def _unresolved_cid_reference( + raw_reference: str, content_id: str | None +) -> CidReferenceAdmission: + """Return the stable fail-closed outcome for a CID that cannot be bound.""" + return CidReferenceAdmission( + raw_reference=raw_reference, + content_id=content_id, + source_part_index=None, + content_sha256=None, + media_classification=None, + error_code=UNRESOLVED_CID_ERROR_CODE, + evidence_boundary=KNOWN_EVIDENCE_BOUNDARY, + ) + + +def _html_image_references(html_source: str) -> list[str]: + """Extract raw ``img`` ``src`` values without fetching or rewriting them.""" + references: list[str] = [] + for match in _IMG_SRC_RE.finditer(html_source): + raw_value = next( + group for group in match.groups() if group is not None + ) + references.append(raw_value.strip()) + return references + + +def _message_children(part: Message) -> list[Message]: + """Return multipart children, ignoring non-message payload entries.""" + payload = part.get_payload() + if not part.is_multipart() or not isinstance(payload, list): + return [] + return [child for child in payload if isinstance(child, Message)] + + +def _decode_text_part(part: Message) -> str: + """Decode a text MIME part, replacing undecodable bytes.""" + payload = part.get_payload(decode=True) + if isinstance(payload, bytes): + charset = part.get_content_charset() or "utf-8" + try: + return payload.decode(charset, errors="replace") + except LookupError: + return payload.decode("utf-8", errors="replace") + content = part.get_payload() + return content if isinstance(content, str) else "" + + +def _normalize_cid_url(reference: str) -> str | None: + """Convert a ``cid:`` URL into a Content-ID token per RFC 2392.""" + if not reference.casefold().startswith("cid:"): + return None + encoded_value = reference[4:] + try: + decoded_value = urllib.parse.unquote(encoded_value, errors="strict") + except UnicodeDecodeError: + return None + if ( + not decoded_value + or _CONTROL_CHARACTER_RE.search(decoded_value) + or any(character.isspace() for character in decoded_value) + ): + return None + return decoded_value + + +def _normalize_content_id(value: object) -> str | None: + """Normalize a Content-ID header by stripping angle brackets.""" + if value is None: + return None + normalized = str(value).strip() + if normalized.startswith("<") and normalized.endswith(">"): + normalized = normalized[1:-1] + if ( + not normalized + or _CONTROL_CHARACTER_RE.search(normalized) + or any(character.isspace() for character in normalized) + ): + return None + return normalized + + +def _normalize_content_type(value: str) -> str: + """Return a lowercase media type without parameters.""" + normalized = (value or "").split(";", 1)[0].strip().casefold() + return IMAGE_TYPE_ALIASES.get(normalized, normalized) + + +def _header_text(value: object) -> str | None: + """Return a stripped header string, or ``None`` when absent or blank.""" + if value is None: + return None + text = str(value).strip() + return text or None + + +def _infer_image_content_type(payload: bytes) -> str | None: + """Infer a supported image type from a deterministic file signature.""" + if payload.startswith(b"\x89PNG\r\n\x1a\n"): + return "image/png" + if payload.startswith(b"\xff\xd8\xff"): + return "image/jpeg" + if payload.startswith((b"GIF87a", b"GIF89a")): + return "image/gif" + if len(payload) >= 12 and payload.startswith(b"RIFF") and payload[8:12] == b"WEBP": + return "image/webp" + return None + + +def _pixel_dimensions_from_header( + content_type: str, payload: bytes +) -> tuple[int, int] | None: + """Read PNG IHDR or GIF logical-screen size without decoding pixels. + + This helper exists only for the Slice 3 tracking-pixel heuristic. It does + not implement or replace the #1376 ``EmailMediaArtifact`` pixel-dimension + contract, which is not present on protected ``develop``. + """ + if ( + content_type == "image/png" + and len(payload) >= 24 + and payload.startswith(b"\x89PNG\r\n\x1a\n") + ): + return ( + int.from_bytes(payload[16:20], "big"), + int.from_bytes(payload[20:24], "big"), + ) + if ( + content_type == "image/gif" + and len(payload) >= 10 + and payload.startswith((b"GIF87a", b"GIF89a")) + ): + return ( + int.from_bytes(payload[6:8], "little"), + int.from_bytes(payload[8:10], "little"), + ) + return None + + +def _is_remote_reference(reference: str) -> bool: + """Return True for ``http`` or ``https`` URLs that must not be fetched.""" + try: + scheme = urllib.parse.urlsplit(reference).scheme.casefold() + except ValueError: + return False + return scheme in {"http", "https"} diff --git a/backend/tests/test_email_media_admission_boundaries.py b/backend/tests/test_email_media_admission_boundaries.py new file mode 100644 index 000000000..7919a1642 --- /dev/null +++ b/backend/tests/test_email_media_admission_boundaries.py @@ -0,0 +1,356 @@ +"""Boundary and helper coverage for Slice 3 email media admission.""" + +from __future__ import annotations + +import base64 +from email.message import EmailMessage +from pathlib import Path + +import pytest + +from services import email_media_admission as admission + + +def _png_header(width: int, height: int) -> bytes: + """Return a signature-bearing PNG whose IHDR carries the given size.""" + return ( + b"\x89PNG\r\n\x1a\n" + + b"\x00\x00\x00\rIHDR" + + width.to_bytes(4, "big") + + height.to_bytes(4, "big") + + b"payload" + ) + + +def _gif_header(width: int, height: int) -> bytes: + """Return a GIF89a logical-screen header with the given size.""" + return ( + b"GIF89a" + + width.to_bytes(2, "little") + + height.to_bytes(2, "little") + + b"payload" + ) + + +def _related_message( + html_body: str, + image_parts: list[tuple[str, str, bytes, str | None]], +) -> bytes: + """Build a multipart/related message for helper-level admission cases.""" + raw = ( + "MIME-Version: 1.0\r\n" + 'Content-Type: multipart/related; boundary="rel"; type="text/html"\r\n' + "\r\n" + "--rel\r\n" + "Content-Type: text/html; charset=utf-8\r\n" + "\r\n" + f"{html_body}\r\n" + ).encode("utf-8") + for content_id, content_type, payload, content_location in image_parts: + headers = ( + f"--rel\r\nContent-Type: {content_type}\r\n" + f"Content-ID: <{content_id}>\r\n" + ) + if content_location is not None: + headers += f"Content-Location: {content_location}\r\n" + raw += ( + headers.encode("ascii") + + b"Content-Transfer-Encoding: base64\r\n\r\n" + + base64.b64encode(payload) + + b"\r\n" + ) + return raw + b"--rel--\r\n" + + +def _html_only_message(html_body: str) -> bytes: + """Return a single-part HTML message with no related image parts.""" + return ( + "MIME-Version: 1.0\r\n" + "Content-Type: text/html; charset=utf-8\r\n" + "\r\n" + f"{html_body}" + ).encode("utf-8") + + +def test_raw_message_must_be_bytes() -> None: + """Reject non-bytes input before any MIME walk.""" + with pytest.raises(TypeError, match="raw_message must be bytes"): + admission.admit_email_inline_media("not-bytes") # type: ignore[arg-type] + + +def test_unsupported_and_mismatched_images_are_not_document_evidence() -> None: + """SVG and signature-mismatched parts stay in the unsupported closed set.""" + result = admission.admit_email_inline_media( + _related_message( + "

images

", + [ + ("vector@naruon.test", "image/svg+xml", b"", None), + ("mismatch@naruon.test", "image/png", _gif_header(8, 8), None), + ], + ) + ) + + assert {image.media_classification for image in result.inline_images} == { + admission.UNSUPPORTED_MEDIA_CLASSIFICATION + } + assert all( + image.evidence_boundary == admission.KNOWN_EVIDENCE_BOUNDARY + for image in result.inline_images + ) + assert all(image.pixel_width is None for image in result.inline_images) + + +def test_relative_html_image_is_ignored_for_admission() -> None: + """Non-cid local file references are not treated as document evidence.""" + result = admission.admit_email_inline_media( + _html_only_message('') + ) + assert result.inline_images == () + assert result.cid_references == () + + +def test_remote_html_images_are_not_fetched_or_admitted() -> None: + """HTTP(S) img src values stay outside admission and never become documents.""" + result = admission.admit_email_inline_media( + _html_only_message( + '' + '' + ) + ) + + assert result.remote_fetch_policy == admission.REMOTE_FETCH_POLICY + assert result.inline_images == () + assert result.cid_references == () + + +def test_cid_outside_related_and_invalid_cid_fail_closed() -> None: + """CID without a related scope, or a malformed cid: URL, is unresolved.""" + outside = admission.admit_email_inline_media( + _html_only_message('') + ) + assert outside.cid_references[0].error_code == ( + admission.UNRESOLVED_CID_ERROR_CODE + ) + assert outside.cid_references[0].media_classification is None + + invalid = admission.admit_email_inline_media( + _html_only_message('' '') + ) + assert {item.error_code for item in invalid.cid_references} == { + admission.UNRESOLVED_CID_ERROR_CODE + } + + +def test_ambiguous_content_id_fails_closed() -> None: + """Duplicate Content-ID values in one related scope are not document evidence.""" + result = admission.admit_email_inline_media( + _related_message( + '', + [ + ("dup@naruon.test", "image/png", _png_header(16, 16), None), + ("dup@naruon.test", "image/gif", _gif_header(16, 16), None), + ], + ) + ) + cid_reference = result.cid_references[0] + assert cid_reference.error_code == admission.UNRESOLVED_CID_ERROR_CODE + assert cid_reference.content_sha256 is None + + +def test_percent_encoded_cid_and_unquoted_src_resolve() -> None: + """RFC 2392 percent-decoding and unquoted src values still bind uniquely.""" + result = admission.admit_email_inline_media( + _related_message( + "", + [("chart@naruon.test", "image/png", _png_header(20, 10), None)], + ) + ) + assert result.cid_references[0].error_code is None + assert result.cid_references[0].content_id == "chart@naruon.test" + assert result.inline_images[0].media_classification == ( + admission.DOCUMENT_IMAGE_CLASSIFICATION + ) + + +def test_jpeg_without_header_parser_is_document_image_with_unknown_boundary() -> None: + """Supported JPEG bytes without a local size parser stay unknown, not tracking.""" + jpeg_payload = b"\xff\xd8\xff" + b"jpeg-data" + result = admission.admit_email_inline_media( + _related_message( + "

scan

", + [("scan@naruon.test", "image/jpg", jpeg_payload, None)], + ) + ) + image = result.inline_images[0] + assert image.declared_content_type == "image/jpeg" + assert image.media_classification == admission.DOCUMENT_IMAGE_CLASSIFICATION + assert image.evidence_boundary == admission.UNKNOWN_EVIDENCE_BOUNDARY + assert image.pixel_width is None + assert image.pixel_height is None + + +def test_tracker_content_location_classifies_without_downloading() -> None: + """A larger GIF with a tracker Content-Location is still a tracking pixel.""" + result = admission.admit_email_inline_media( + _related_message( + "

ad

", + [ + ( + "ad@naruon.test", + "image/gif", + _gif_header(32, 32), + "https://click.list-manage.com/track/open.php?u=x", + ) + ], + ) + ) + image = result.inline_images[0] + assert image.media_classification == admission.TRACKING_PIXEL_CLASSIFICATION + assert image.evidence_boundary == admission.KNOWN_EVIDENCE_BOUNDARY + assert image.pixel_width == 32 + + +def test_tiny_gif_without_dimensions_uses_tracker_content_type() -> None: + """A typical tracker GIF with no parseable size is not document evidence.""" + tiny_gif = b"GIF89a" + b"x" + result = admission.admit_email_inline_media( + _related_message( + "

beacon

", + [("beacon@naruon.test", "image/gif", tiny_gif, None)], + ) + ) + image = result.inline_images[0] + assert image.media_classification == admission.TRACKING_PIXEL_CLASSIFICATION + assert image.pixel_width is None + + +def test_tracker_path_marker_and_host_suffix_helpers() -> None: + """Already-present Content-Location values match host suffixes or path markers.""" + assert admission._content_location_matches_tracker(None) is False + assert ( + admission._content_location_matches_tracker( + "https://pixel.doubleclick.net/open.gif" + ) + is True + ) + assert ( + admission._content_location_matches_tracker( + "https://cdn.example.test/pixel.gif" + ) + is True + ) + assert ( + admission._content_location_matches_tracker("https://cdn.example.test/logo.png") + is False + ) + assert admission._content_location_matches_tracker("http://[broken") is False + assert admission._content_location_matches_tracker("/open.php") is True + assert admission._content_location_matches_tracker("https://list-manage.com/x") is True + + +def test_remote_reference_helper_and_invalid_url() -> None: + """Only http(s) schemes are remote; broken brackets stay non-remote.""" + assert admission._is_remote_reference("https://example.test/a.png") is True + assert admission._is_remote_reference("cid:x@naruon.test") is False + assert admission._is_remote_reference("http://[broken") is False + + +def test_content_helpers_cover_empty_and_unknown_inputs() -> None: + """Normalizers reject empty, control, and whitespace identities.""" + assert admission._normalize_content_id(None) is None + assert admission._normalize_content_id(" ") == "x@naruon.test" + assert admission._normalize_content_id("") is None + assert admission._normalize_content_id("") is None + assert admission._normalize_content_id("") is None + assert admission._normalize_content_type("") == "" + assert admission._normalize_content_type("IMAGE/JPG; name=x") == "image/jpeg" + assert admission._header_text(None) is None + assert admission._header_text(" ") is None + assert admission._header_text(" https://x.test ") == "https://x.test" + assert admission._infer_image_content_type(b"not-image") is None + assert admission._infer_image_content_type( + b"RIFF\x08\x00\x00\x00WEBPdata" + ) == "image/webp" + assert admission._infer_image_content_type(b"GIF87a....") == "image/gif" + assert admission._pixel_dimensions_from_header( + "image/gif", b"GIF87a" + (1).to_bytes(2, "little") + (1).to_bytes(2, "little") + ) == (1, 1) + assert admission._pixel_dimensions_from_header("image/jpeg", b"jpeg") is None + assert admission._pixel_dimensions_from_header("image/png", b"short") is None + assert admission._pixel_dimensions_from_header("image/gif", b"GIF") is None + assert admission._normalize_cid_url("https://example.test") is None + assert admission._normalize_cid_url("cid:%FF") is None + assert admission._normalize_cid_url("cid:has space@naruon.test") is None + + +def test_decode_text_part_and_message_children_fallbacks() -> None: + """Text decode and multipart walking stay deterministic on odd payloads.""" + unknown_charset = EmailMessage() + unknown_charset.set_content("한글", charset="utf-8") + unknown_charset.set_param("charset", "not-a-real-charset") + assert "한글" in admission._decode_text_part(unknown_charset) + + empty_part = EmailMessage() + empty_part.set_payload(None) + assert admission._decode_text_part(empty_part) == "" + + string_part = EmailMessage() + string_part.set_payload("direct") + assert admission._decode_text_part(string_part) == "direct" + + mixed = EmailMessage() + mixed.set_content("plain") + assert admission._message_children(mixed) == [] + + class _NonMessageMultipart: + """Duck-typed multipart whose payload list is not MIME children.""" + + def is_multipart(self) -> bool: + return True + + def get_payload(self) -> list[str]: + return ["not-a-message"] + + assert admission._message_children(_NonMessageMultipart()) == [] # type: ignore[arg-type] + + +def test_non_image_parts_and_non_img_tags_are_ignored() -> None: + """PDF parts and non-img tags do not create inline image admissions.""" + raw = ( + b"MIME-Version: 1.0\r\nContent-Type: multipart/mixed; boundary=m\r\n\r\n" + b"--m\r\nContent-Type: application/pdf\r\nContent-ID: \r\n\r\n" + b"%PDF-1.7\r\n--m\r\nContent-Type: text/html; charset=utf-8\r\n\r\n" + b'xnone\r\n--m--\r\n' + ) + result = admission.admit_email_inline_media(raw) + assert result.inline_images == () + assert result.cid_references == () + + +def test_quoted_img_src_and_webp_document_image() -> None: + """Single-quoted src and WebP signatures admit as document images.""" + webp_payload = b"RIFF\x08\x00\x00\x00WEBPdata" + result = admission.admit_email_inline_media( + _related_message( + "", + [("shot@naruon.test", "image/webp", webp_payload, None)], + ) + ) + assert result.cid_references[0].error_code is None + assert result.inline_images[0].media_classification == ( + admission.DOCUMENT_IMAGE_CLASSIFICATION + ) + assert result.inline_images[0].evidence_boundary == ( + admission.UNKNOWN_EVIDENCE_BOUNDARY + ) + + +def test_admission_module_has_no_network_client() -> None: + """The admission module must not grow a fetch path in this slice.""" + source = admission.__file__ + assert source is not None + module_text = Path(source).read_text(encoding="utf-8") + assert "httpx" not in module_text + assert "urllib.request" not in module_text + assert "urlopen" not in module_text + assert "socket" not in module_text From f3f7d09a2e4e191230c96a3da2305c794a08ae76 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Mon, 17 Aug 2026 18:31:35 +0000 Subject: [PATCH 3/4] docs: record Slice 3 media admission doctoring Add APA 7 RFC and email-tracking citations, state the no-fetch admission boundary, and record the anti-pattern so 1x1 beacons are not treated as document evidence. Co-authored-by: Seongho Bae --- AGENTS.md | 6 ++ docs/architecture/image-content-detection.md | 8 ++ .../doctoring/email-inline-media-admission.md | 89 +++++++++++++++++++ 3 files changed, 103 insertions(+) create mode 100644 docs/doctoring/email-inline-media-admission.md diff --git a/AGENTS.md b/AGENTS.md index 9104dd1f4..eb4540f64 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -348,6 +348,12 @@ in this repo. embedding generation is unavailable. Tests must cover the local `embeddinggemma` path so Data workspace imports do not silently bypass the selected embedding model. +- Email inline-media admission must resolve `cid:` only against the same + message's `multipart/related` parts. Unresolved CID and remote `http(s)` + image references fail closed and must not be sent to OCR/VLM or treated as + document evidence. Classify 1×1 / tracker-header pixels as `tracking_pixel` + from local header and dimension evidence only; do not fetch remote pixels + and do not invent a new egress policy in this slice. - Home/Today dashboard reply-wait surfaces must read signed `/api/emails/pending-replies` data instead of inferring pending replies from generic inbox fixtures or static copy. Tests and E2E mocks must verify the diff --git a/docs/architecture/image-content-detection.md b/docs/architecture/image-content-detection.md index 16b1af641..fee3d5346 100644 --- a/docs/architecture/image-content-detection.md +++ b/docs/architecture/image-content-detection.md @@ -2,6 +2,14 @@ This document outlines the architecture and flow for processing and detecting image contents within the Naruon workspace. +**Admission gate (current Slice 3 contract):** before any OCR, VLM, or +NewsDOM step, `services.email_media_admission` classifies local MIME images +and resolves `cid:` only against the same message's `multipart/related` +parts. Tracking pixels and unresolved CID references are not document +evidence. Remote `http(s)` images remain no-fetch unless a separately +authorized egress policy already exists; this slice does not add one. See +`docs/doctoring/email-inline-media-admission.md`. + ## Flow Diagram ```mermaid diff --git a/docs/doctoring/email-inline-media-admission.md b/docs/doctoring/email-inline-media-admission.md new file mode 100644 index 000000000..32f684a83 --- /dev/null +++ b/docs/doctoring/email-inline-media-admission.md @@ -0,0 +1,89 @@ +# Email inline-media admission evidence + +## Scope and shipped-state boundary + +This note traces the deterministic admission classifier in +`backend/services/email_media_admission.py`. It is **#1350 Slice 3 admission +only** and is **not protected-`develop` shipped truth until the pull request +merges**. The slice runs before OCR, vision-language models, NewsDOM, or any +LLM. It does not mutate provider mail, does not blanket-mask business PII, and +does not add a remote-image egress path. + +Predecessor evidence is **N/A**. Open PR #1376 already covers +`EmailMediaArtifact` pixel-dimension extraction on a separate branch; that +contract is **not** on protected `develop` and is not copied or rewritten here. +This module reads PNG IHDR and GIF logical-screen sizes only as a local +tracking-pixel heuristic. + +## Buyer-visible contract + +Naruon must not send a 1×1 beacon to a model or treat it as document evidence. +Admission therefore: + +1. Resolves `cid:` URLs against Content-ID values inside the same message's + `multipart/related` entity (RFC 2392; RFC 2046). A missing, malformed, or + ambiguous target fails closed with `unresolved_cid_reference`. +2. Classifies admitted local images into the closed set `tracking_pixel`, + `unsupported_media`, and `document_image`. Screenshots, charts, and scans + share `document_image` in this slice. +3. Labels tracking pixels from local evidence only: tiny header-derived + dimensions (1×1), typical tracker content-types on tiny GIF payloads, and + known tracker hosts or paths on an already-present `Content-Location` + header. Remote pixels are never downloaded. +4. Returns provenance: source part index, Content-ID, SHA-256 of the exact + decoded source bytes, classification, and `known` / `unknown` evidence + boundary. Repeated identical base64 parts share one hash and keep distinct + part indexes. + +`http` and `https` image references remain recorded as no-fetch +(`remote_fetch_policy=disabled`). They are not admitted as document images. + +## Standards-to-code trace + +| Requirement | Primary basis | Naruon behavior | +| --- | --- | --- | +| Preserve MIME entity structure and media types | RFC 2045; RFC 2046 | Walk the parsed MIME tree, retain leaf part indexes, and classify from declared type plus file signature. | +| Treat related body parts as one aggregate | RFC 2046 multipart composite media types | Bind `cid:` only inside the nearest `multipart/related` scope. | +| Convert `cid:` URLs to Content-ID tokens | RFC 2392 | Percent-decode the URL form, reject control or whitespace corruption, and match stripped Content-ID headers. Duplicate matches fail closed. | +| Do not create external side effects while admitting evidence | Product/security policy; Englehardt et al. (2018) | No HTTP client. Tracker URL patterns are applied only to headers already present on the local part. | + +## Academic summary (PDF not attached) + +Englehardt, Han, and Narayanan (2018) measured commercial mailing-list mail +and showed that viewing a message commonly loads third-party embedded pixels +that leak recipient identity. About 30% of their corpus leaked the recipient +address to one or more third parties on view. The paper is published under +Creative Commons Attribution-NonCommercial-NoDerivs 3.0, so this repository +cites and summarizes it rather than redistributing the PDF. + +Naruon uses that finding as the buyer reason to keep 1×1 and tracker-header +images out of OCR/VLM admission. The paper's proposed client-side stripping of +remote tracking tags is **not** implemented here; this slice only refuses to +treat those local beacons as document evidence and refuses to fetch remote +ones. + +## Verification + +Focused product tests cover four realistic `.eml` fixtures: a resolving CID +chart, an unresolved CID, a 1×1 GIF with a Mailchimp-style Content-Location, +and two identical base64 PNG parts that share one SHA-256. Boundary tests +cover unsupported media, remote no-fetch, ambiguous Content-ID, and helper +fail-closed paths. Owned module statement and branch coverage is 100% on the +focused harness. + +## References + +Englehardt, S., Han, J., & Narayanan, A. (2018). I never signed up for this! +Privacy implications of email tracking. *Proceedings on Privacy Enhancing +Technologies, 2018*(1), 109–126. https://doi.org/10.1515/popets-2018-0006 + +Freed, N., & Borenstein, N. (1996a). *Multipurpose Internet Mail Extensions +(MIME) Part One: Format of Internet Message Bodies* (RFC 2045). RFC Editor. +https://doi.org/10.17487/RFC2045 + +Freed, N., & Borenstein, N. (1996b). *Multipurpose Internet Mail Extensions +(MIME) Part Two: Media Types* (RFC 2046). RFC Editor. +https://doi.org/10.17487/RFC2046 + +Levinson, E. (1998). *Content-ID and Message-ID Uniform Resource Locators* +(RFC 2392). RFC Editor. https://doi.org/10.17487/RFC2392 From 5a80583bcabc22609e8677864ae86f867d85fd45 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 20 Aug 2026 19:12:20 +0900 Subject: [PATCH 4/4] test(email-media): assert tracker host exactly --- backend/tests/test_email_media_admission.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/backend/tests/test_email_media_admission.py b/backend/tests/test_email_media_admission.py index e7569d47b..77e280141 100644 --- a/backend/tests/test_email_media_admission.py +++ b/backend/tests/test_email_media_admission.py @@ -3,6 +3,7 @@ from __future__ import annotations from pathlib import Path +from urllib.parse import urlsplit from services.email_media_admission import admit_email_inline_media @@ -69,7 +70,7 @@ def test_tracking_pixel_is_not_document_evidence() -> None: assert tracking_pixel.pixel_height == 1 assert tracking_pixel.declared_content_type == "image/gif" assert tracking_pixel.content_location is not None - assert "list-manage.com" in tracking_pixel.content_location + assert urlsplit(tracking_pixel.content_location).hostname == "click.list-manage.com" assert tracking_pixel.error_code is None assert all( image.media_classification != "document_image"