diff --git a/CHANGELOG.md b/CHANGELOG.md index 20b7289aa..5b964b72e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,7 @@ All notable changes to TEPP are documented here. The format follows Keep a Chang ### Added +- `evidence_core` embedded-image units: `data:image/;base64,...` URIs keep their original source spans and media types, and cannot be used as lexical inference text. - `persistence_postgres` entity/project target SQL now rejects empty, oversized, or hostile type/status labels before insert; interpolated codes are restricted to lowercase ASCII `snake_case` characters so membership foreign keys remain referentially safe (ADR 0003 / ADR 0013). - `persistence_postgres` live SQLx transport retains one pool-backed PostgreSQL connection per session so tenant binding and the following statement share a session, and closes the connection and owned runtime safely from another Tokio runtime (ADR 0013). - The authored-line coverage gate now filters LLVM-only literal and expression continuation records while retaining branch coverage for their executable decisions; Rust function signatures and structural branch lines are no longer counted as uncovered statements. diff --git a/crates/evidence_core/src/error.rs b/crates/evidence_core/src/error.rs index b9701c7ea..27750c033 100644 --- a/crates/evidence_core/src/error.rs +++ b/crates/evidence_core/src/error.rs @@ -44,6 +44,10 @@ pub enum EvidenceError { InvalidLayoutBounds, /// Layout coordinates exceeded the enclosing page. LayoutOutOfBounds, + /// A base64 image data URI was treated as lexical inference text. + EmbeddedImageIsNotLexicalText, + /// An embedded image data URI declared an implausible image media type. + ImplausibleImageMediaType, } impl fmt::Display for EvidenceError { @@ -70,6 +74,8 @@ impl fmt::Display for EvidenceError { Self::InvalidPageGeometry => "page geometry must be finite and positive", Self::InvalidLayoutBounds => "layout bounds must be finite, nonnegative, and nonempty", Self::LayoutOutOfBounds => "layout bounds exceed the page geometry", + Self::EmbeddedImageIsNotLexicalText => "embedded image is not lexical text", + Self::ImplausibleImageMediaType => "embedded image media type is implausible", }; formatter.write_str(message) } diff --git a/crates/evidence_core/src/image_unit.rs b/crates/evidence_core/src/image_unit.rs new file mode 100644 index 000000000..84953336c --- /dev/null +++ b/crates/evidence_core/src/image_unit.rs @@ -0,0 +1,254 @@ +//! Embedded `data:image` units that keep their original source location. + +use crate::{DocumentRecord, EvidenceError, SourceSpan}; + +const DATA_IMAGE_PREFIX: &str = "data:image/"; +const BASE64_MARK: &str = ";base64,"; + +/// Image media types accepted as plausible by [`embedded_image_units`]. +/// +/// The set is deliberately conservative and tracks widely registered or +/// de facto standard image subtypes; anything else fails closed instead of +/// yielding a bogus embedded-image unit. +const PLAUSIBLE_IMAGE_MEDIA_TYPES: [&str; 14] = [ + "image/apng", + "image/avif", + "image/bmp", + "image/gif", + "image/heic", + "image/heif", + "image/jpeg", + "image/jpg", + "image/png", + "image/svg+xml", + "image/tiff", + "image/vnd.microsoft.icon", + "image/webp", + "image/x-icon", +]; + +/// One embedded image located in a document body. +#[derive(Clone, Copy, Debug, PartialEq)] +pub struct EmbeddedImageUnit<'document> { + span: SourceSpan, + media_type: &'document str, +} + +impl<'document> EmbeddedImageUnit<'document> { + /// Exact source span of the data URI, including the `data:image/` prefix. + #[must_use] + pub const fn span(self) -> SourceSpan { + self.span + } + + /// Declared image media type (`image/png`, `image/jpeg`, …). + #[must_use] + pub const fn media_type(self) -> &'document str { + self.media_type + } +} + +/// Locate `data:image/;base64,...` units and retain their original spans. +/// +/// Only plausible image media types are accepted: a candidate URI whose +/// declared media type is not in [`PLAUSIBLE_IMAGE_MEDIA_TYPES`] fails the +/// whole parse so malformed bodies cannot produce bogus units. +/// +/// # Errors +/// +/// Returns [`EvidenceError::EmptySourceSpan`] when the document contains no +/// well-formed embedded image URI, and +/// [`EvidenceError::ImplausibleImageMediaType`] when a candidate URI +/// declares an implausible image media type. +pub fn embedded_image_units( + document: &DocumentRecord, +) -> Result>, EvidenceError> { + let text = document.text(); + let mut units = Vec::new(); + let mut search_from = 0usize; + while let Some(relative) = text[search_from..].find(DATA_IMAGE_PREFIX) { + let start = search_from + relative; + let after_prefix = start + DATA_IMAGE_PREFIX.len(); + let Some(mark_rel) = text[after_prefix..].find(BASE64_MARK) else { + search_from = after_prefix; + continue; + }; + let media_end = after_prefix + mark_rel; + let payload_start = media_end + BASE64_MARK.len(); + let payload_end = payload_start + + text[payload_start..] + .find(|ch: char| !is_base64_payload_char(ch)) + .unwrap_or(text.len() - payload_start); + if payload_end == payload_start { + search_from = payload_start; + continue; + } + let media_type = &text[start + "data:".len()..media_end]; + if media_type.contains(DATA_IMAGE_PREFIX) { + search_from = after_prefix; + continue; + } + let base_media_type = base_media_type(media_type); + if !is_plausible_image_media_type(base_media_type) { + return Err(EvidenceError::ImplausibleImageMediaType); + } + let scalar_start = text[..start].chars().count(); + let scalar_end = scalar_start + text[start..payload_end].chars().count(); + let span = SourceSpan::new(document, start, payload_end, scalar_start, scalar_end, None)?; + units.push(EmbeddedImageUnit { + span, + media_type: base_media_type, + }); + search_from = payload_end; + } + if units.is_empty() { + return Err(EvidenceError::EmptySourceSpan); + } + Ok(units) +} + +/// Refuse using a document body that still contains an embedded image as +/// lexical inference text. +/// +/// # Errors +/// +/// Returns [`EvidenceError::InvalidWirePayload`] for empty input and +/// [`EvidenceError::EmbeddedImageIsNotLexicalText`] when a `data:image` +/// base64 URI is present. +pub fn refuse_base64_image_as_lexical_text(text: &str) -> Result<(), EvidenceError> { + if text.is_empty() { + return Err(EvidenceError::InvalidWirePayload); + } + if contains_base64_image_data_uri(text) { + return Err(EvidenceError::EmbeddedImageIsNotLexicalText); + } + Ok(()) +} + +fn contains_base64_image_data_uri(text: &str) -> bool { + let mut search_from = 0usize; + while let Some(relative) = text[search_from..].find(DATA_IMAGE_PREFIX) { + let start = search_from + relative; + let after_prefix = start + DATA_IMAGE_PREFIX.len(); + let Some(mark_rel) = text[after_prefix..].find(BASE64_MARK) else { + search_from = after_prefix; + continue; + }; + let media_end = after_prefix + mark_rel; + let media_type = &text[start + "data:".len()..media_end]; + if is_image_media_type_token(media_type) { + return true; + } + search_from = after_prefix; + } + false +} + +fn is_image_media_type_token(media_type: &str) -> bool { + let Some(subtype) = base_media_type(media_type).strip_prefix("image/") else { + return false; + }; + !subtype.is_empty() + && subtype + .bytes() + .all(|byte| byte.is_ascii_alphanumeric() || matches!(byte, b'.' | b'+' | b'-')) +} + +fn base_media_type(media_type: &str) -> &str { + media_type.split(';').next().unwrap_or("") +} + +fn is_base64_payload_char(ch: char) -> bool { + ch.is_ascii_alphanumeric() || matches!(ch, '+' | '/' | '=') +} + +/// Report whether a declared media type is a plausible image media type. +fn is_plausible_image_media_type(media_type: &str) -> bool { + PLAUSIBLE_IMAGE_MEDIA_TYPES.contains(&media_type) +} + +#[cfg(test)] +mod tests { + use super::{embedded_image_units, refuse_base64_image_as_lexical_text}; + use crate::{DocumentRecord, EvidenceError, SourceArtifact}; + + #[test] + fn jpeg_uri_and_incomplete_prefix_are_classified() { + let text = "x data:image/jpeg;base64,/9j/4AA= y data:image/gif y"; + let artifact = SourceArtifact::from_bytes(text.as_bytes()).expect("artifact"); + let document = DocumentRecord::from_text(artifact.id(), text).expect("document"); + let units = embedded_image_units(&document).expect("jpeg"); + assert_eq!(units.len(), 1); + assert_eq!(units[0].media_type(), "image/jpeg"); + refuse_base64_image_as_lexical_text("plain note").expect("plain"); + refuse_base64_image_as_lexical_text("data:image/png").expect("incomplete image"); + assert_eq!( + refuse_base64_image_as_lexical_text("data:image/png;base64,AAAA"), + Err(EvidenceError::EmbeddedImageIsNotLexicalText) + ); + + let empty_text = "data:image/png;base64, following text"; + let empty_artifact = SourceArtifact::from_bytes(empty_text.as_bytes()).expect("artifact"); + let empty_document = + DocumentRecord::from_text(empty_artifact.id(), empty_text).expect("document"); + assert_eq!( + embedded_image_units(&empty_document), + Err(EvidenceError::EmptySourceSpan) + ); + } + + #[test] + fn implausible_media_types_fail_closed() { + let text = "data:image/not-a-type;base64,AAAA"; + let artifact = SourceArtifact::from_bytes(text.as_bytes()).expect("artifact"); + let document = DocumentRecord::from_text(artifact.id(), text).expect("document"); + assert_eq!( + embedded_image_units(&document), + Err(EvidenceError::ImplausibleImageMediaType) + ); + assert_eq!( + refuse_base64_image_as_lexical_text(text), + Err(EvidenceError::EmbeddedImageIsNotLexicalText) + ); + } + + #[test] + fn common_raster_media_types_are_accepted() { + let text = "a data:image/png;base64,AAAA b data:image/jpeg;base64,BBBB \ + c data:image/webp;base64,CCCC d data:image/gif;base64,DDDD e"; + let artifact = SourceArtifact::from_bytes(text.as_bytes()).expect("artifact"); + let document = DocumentRecord::from_text(artifact.id(), text).expect("document"); + let units = embedded_image_units(&document).expect("units"); + let media_types: Vec<&str> = units.iter().map(|unit| unit.media_type()).collect(); + assert_eq!( + media_types, + vec!["image/png", "image/jpeg", "image/webp", "image/gif"] + ); + } + + #[test] + fn empty_payload_is_not_an_image_unit() { + let text = "data:image/png;base64,"; + let artifact = SourceArtifact::from_bytes(text.as_bytes()).expect("artifact"); + let document = DocumentRecord::from_text(artifact.id(), text).expect("document"); + assert_eq!( + embedded_image_units(&document), + Err(EvidenceError::EmptySourceSpan) + ); + } + + #[test] + fn malformed_image_prefix_does_not_swallow_later_valid_image() { + let text = "data:image/gif then data:image/png;base64,AAAA"; + let artifact = SourceArtifact::from_bytes(text.as_bytes()).expect("artifact"); + let document = DocumentRecord::from_text(artifact.id(), text).expect("document"); + + let units = embedded_image_units(&document).expect("png"); + assert_eq!(units.len(), 1); + assert_eq!(units[0].media_type(), "image/png"); + assert_eq!( + units[0].span().byte_start(), + text.find("data:image/png").expect("png start") + ); + } +} diff --git a/crates/evidence_core/src/lib.rs b/crates/evidence_core/src/lib.rs index 0d28ab0d8..35fdcf866 100644 --- a/crates/evidence_core/src/lib.rs +++ b/crates/evidence_core/src/lib.rs @@ -7,13 +7,15 @@ //! records, source spans whose byte, Unicode-scalar, page, and layout //! coordinates are validated before entering later temporal or psychometric //! layers, and strict versioned JSON wire contracts that reconstruct records -//! only through the same domain validation boundary. +//! only through the same domain validation boundary. Embedded `data:image` +//! units keep their original offsets and are not lexical inference text. mod artifact; mod digest; mod document; mod error; mod identifier; +mod image_unit; mod span; mod wire; @@ -27,6 +29,12 @@ pub use document::DocumentRecord; pub use error::EvidenceError; /// A validated RFC 9562 `UUIDv7` evidence identifier. pub use identifier::EvidenceId; +/// One embedded image located in a document body. +pub use image_unit::EmbeddedImageUnit; +/// Locate `data:image` base64 units with exact source spans. +pub use image_unit::embedded_image_units; +/// Refuse treating an embedded image URI as lexical inference text. +pub use image_unit::refuse_base64_image_as_lexical_text; /// A validated page-relative location for source evidence. pub use span::PageLocation; /// An exact byte, Unicode-scalar, and optional page/layout span. diff --git a/crates/evidence_core/tests/embedded_image_contract.rs b/crates/evidence_core/tests/embedded_image_contract.rs new file mode 100644 index 000000000..1f56baf6f --- /dev/null +++ b/crates/evidence_core/tests/embedded_image_contract.rs @@ -0,0 +1,92 @@ +//! Embedded base64 images keep their original location and are not lexical text. + +use evidence_core::{ + DocumentRecord, EvidenceError, SourceArtifact, embedded_image_units, + refuse_base64_image_as_lexical_text, +}; + +#[test] +fn data_uri_recovers_exact_span_and_media_type() { + let uri = "data:image/png;base64,iVBORw0KGgo="; + let text = format!("Before the figure.\n\n{uri}\n\nAfter the figure. data:image/gif y"); + let artifact = SourceArtifact::from_bytes(text.as_bytes()).expect("artifact"); + let document = DocumentRecord::from_text(artifact.id(), &text).expect("document"); + + let units = embedded_image_units(&document).expect("units"); + assert_eq!(units.len(), 1); + assert_eq!(units[0].media_type(), "image/png"); + assert_eq!( + &document.text()[units[0].span().byte_start()..units[0].span().byte_end()], + uri + ); + assert_eq!( + refuse_base64_image_as_lexical_text(document.text()), + Err(EvidenceError::EmbeddedImageIsNotLexicalText) + ); + refuse_base64_image_as_lexical_text("data:image/png").expect("incomplete image"); + refuse_base64_image_as_lexical_text("Before the figure.").expect("plain text"); + refuse_base64_image_as_lexical_text("data:image/gif y").expect("incomplete image marker"); + refuse_base64_image_as_lexical_text("문서: data:image/png 형식;base64, 설명") + .expect("ordinary prose"); + assert_eq!( + refuse_base64_image_as_lexical_text("data:image/png;version=1;base64,AAAA"), + Err(EvidenceError::EmbeddedImageIsNotLexicalText) + ); +} + +#[test] +fn implausible_media_types_fail_closed_and_common_types_are_accepted() { + let malformed = "data:image/not-a-type;base64,AAAA"; + let artifact = SourceArtifact::from_bytes(malformed.as_bytes()).expect("artifact"); + let document = DocumentRecord::from_text(artifact.id(), malformed).expect("document"); + assert_eq!( + embedded_image_units(&document), + Err(EvidenceError::ImplausibleImageMediaType) + ); + + let text = "a data:image/png;base64,AAAA b data:image/jpeg;base64,BBBB \ + c data:image/webp;base64,CCCC d data:image/gif;base64,DDDD e"; + let artifact = SourceArtifact::from_bytes(text.as_bytes()).expect("artifact"); + let document = DocumentRecord::from_text(artifact.id(), text).expect("document"); + let units = embedded_image_units(&document).expect("units"); + let media_types: Vec<&str> = units.iter().map(|unit| unit.media_type()).collect(); + assert_eq!( + media_types, + vec!["image/png", "image/jpeg", "image/webp", "image/gif"] + ); + + let parameterized = "data:image/png;charset=x;base64,AAAA"; + let artifact = SourceArtifact::from_bytes(parameterized.as_bytes()).expect("artifact"); + let document = DocumentRecord::from_text(artifact.id(), parameterized).expect("document"); + let units = embedded_image_units(&document).expect("parameterized image"); + assert_eq!(units.len(), 1); + assert_eq!(units[0].media_type(), "image/png"); + assert_eq!( + &document.text()[units[0].span().byte_start()..units[0].span().byte_end()], + parameterized + ); +} + +#[test] +fn documents_without_images_and_empty_payloads_fail_closed() { + let text = "No figures in this note."; + let artifact = SourceArtifact::from_bytes(text.as_bytes()).expect("artifact"); + let document = DocumentRecord::from_text(artifact.id(), text).expect("document"); + assert_eq!( + embedded_image_units(&document), + Err(EvidenceError::EmptySourceSpan) + ); + + assert_eq!( + refuse_base64_image_as_lexical_text(""), + Err(EvidenceError::InvalidWirePayload) + ); + let empty_payload = "data:image/png;base64,"; + let empty_artifact = SourceArtifact::from_bytes(empty_payload.as_bytes()).expect("artifact"); + let empty_document = + DocumentRecord::from_text(empty_artifact.id(), empty_payload).expect("document"); + assert_eq!( + embedded_image_units(&empty_document), + Err(EvidenceError::EmptySourceSpan) + ); +} diff --git a/crates/evidence_core/tests/records_and_spans_contract.rs b/crates/evidence_core/tests/records_and_spans_contract.rs index 810729e03..ceddc79db 100644 --- a/crates/evidence_core/tests/records_and_spans_contract.rs +++ b/crates/evidence_core/tests/records_and_spans_contract.rs @@ -329,6 +329,14 @@ fn every_record_validation_error_has_a_stable_message() { EvidenceError::LayoutOutOfBounds, "layout bounds exceed the page geometry", ), + ( + EvidenceError::EmbeddedImageIsNotLexicalText, + "embedded image is not lexical text", + ), + ( + EvidenceError::ImplausibleImageMediaType, + "embedded image media type is implausible", + ), ]; for (error, expected) in cases { diff --git a/docs/research/embedded-image-units.md b/docs/research/embedded-image-units.md new file mode 100644 index 000000000..82c9b358a --- /dev/null +++ b/docs/research/embedded-image-units.md @@ -0,0 +1,30 @@ +# Embedded image source units + +## Scope + +This note doctors the `evidence_core` contract for `data:image/;base64,...` payloads that appear in document bodies: + +1. each well-formed data URI becomes an `EmbeddedImageUnit` with an exact source span; +2. the declared media type is retained; +3. the original image location is preserved so later object/OCR search can attach to that span; +4. the base64 payload is not lexical inference text. + +No OCR/object model is executed here. No database migration is allocated. + +## Authoritative sources + +Masinter, L. (1998). *The "data" URL scheme* (RFC 2397). RFC Editor. https://doi.org/10.17487/RFC2397 + +Antol, S., Agrawal, A., Lu, J., Mitchell, M., Batra, D., Zitnick, C. L., & Parikh, D. (2015). VQA: Visual question answering. In *Proceedings of the IEEE International Conference on Computer Vision* (pp. 2425–2433). https://doi.org/10.1109/ICCV.2015.279 + +## Application + +RFC 2397 defines the `data:` URI and the `base64` encoding used in HTML and reports (Masinter, 1998). Visual question answering shows that image meaning is a separate modality from surrounding words (Antol et al., 2015). TEPP therefore keeps the original URI offset as a span and refuses to treat that payload as topic or lexical evidence (Masinter, 1998; Antol et al., 2015). + +## Verification + +- a PNG data URI between two paragraphs recovers media type `image/png` and the exact URI text; +- `refuse_base64_image_as_lexical_text` denies the full document and allows the surrounding sentence; +- a data URI declaring an implausible image media type (outside the accepted conservative set) fails closed with `ImplausibleImageMediaType`; +- documents without images return `EmptySourceSpan`; +- empty lexical input fails closed. diff --git a/docs/research/standards-and-literature.md b/docs/research/standards-and-literature.md index 70ce245ff..f75ff482d 100644 --- a/docs/research/standards-and-literature.md +++ b/docs/research/standards-and-literature.md @@ -228,6 +228,10 @@ Lebo, T., Sahoo, S., & McGuinness, D. (Eds.). (2013). *PROV-O: The PROV ontology Moreau, L., & Missier, P. (Eds.). (2013). *PROV-DM: The PROV data model*. World Wide Web Consortium. https://www.w3.org/TR/prov-dm/ +Masinter, L. (1998). *The "data" URL scheme* (RFC 2397). RFC Editor. https://doi.org/10.17487/RFC2397 + +Antol, S., Agrawal, A., Lu, J., Mitchell, M., Batra, D., Zitnick, C. L., & Parikh, D. (2015). VQA: Visual question answering. In *Proceedings of the IEEE International Conference on Computer Vision* (pp. 2425–2433). https://doi.org/10.1109/ICCV.2015.279 + TEPP separates stable record identity, content equality, exact text location, wire representation, authorization, and provenance. JSON wire records are explicit versioned DTOs with unknown-field rejection and reconstruct through domain validation. `SHA-256` detects content substitution but is not treated as proof of origin, authority, or chain of custody. A summary, template, or pasted copy is a PROV derivation of the source document, not a state transition and not a reuse of the source identity; documents, serialized records, checkpoints, and LLM outputs remain untrusted until identity, provenance, size, and nesting depth validate (Bray, 2017; Moreau & Missier, 2013). International Organization for Standardization and International Electrotechnical Commission. (2011). *Information technology—Security techniques—Privacy framework* (ISO/IEC Standard No. 29100:2011). Data minimization informs `provider_receipt`; it is not a certification claim. diff --git a/docs/validation/temporal-event-foundation.md b/docs/validation/temporal-event-foundation.md index f3432e399..5b075b429 100644 --- a/docs/validation/temporal-event-foundation.md +++ b/docs/validation/temporal-event-foundation.md @@ -94,6 +94,7 @@ This report tracks exact-head scientific and engineering evidence required befor | Adaptive orchestration router | `tepp_api` | implemented-main | live execution/production ablation | mode selection, document-control denial, ablation, credential-free bind | ADR 0010; `docs/research/adaptive-orchestration-router.md` | | Operational log/source separation | `operational_log` + `persistence_postgres` | active-PR | this PR | replayed action recovery vs collapsed action; inspected `audit_event` insert refuses source text, source identity, and blanket-mask grants | ADR 0009 | | Adaptive orchestration router | `tepp_api` | accepted-target | active PR | mode selection, document-control denial, ablation, credential-free bind | ADR 0010; `docs/research/adaptive-orchestration-router.md` | +| Embedded image source units | `evidence_core` | active-PR | active PR (#58) | exact data-URI span/media-type recovery, empty/incomplete/invalid/recovery cases, lexical refusal | ADR 0008; `docs/research/embedded-image-units.md` | | Derived sensitivity inheritance | `derived_sensitivity` | active-PR | this PR | fixed 3×3 topic/factor/relation × Public/Internal/Restricted truth with kind-and-class recovery and public-collapse rates invariant to reordering; fail-closed unknown kinds, validated public constructors, empty/mismatched payloads, unauthorized derivation-as-public, unauthorized blanket PII masking | ADR 0009; GDPR Art. 4(1)/Recital 26; WP 136 | | Evidence-bounded LLM interpretation | `interpretation_gateway` | active-PR | this PR | span citation + unsupported-claim rate | ADR 0010; live orchestration remaining | | TDT/CHRONOS evidence-status gates | `event_core` | active-PR | PR #50 | admission + first-story rates | known-stream miss/FA; full tracking/calibration/schema extraction remains future; ADR 0016; `docs/research/event-intelligence-status-gates.md` |