Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@ All notable changes to TEPP are documented here. The format follows Keep a Chang

### Added

- `evidence_core` embedded-image units: `data:image/<type>;base64,...` URIs keep their original source spans and media types, and cannot be used as lexical inference text.
- `persistence_postgres` entity/project target SQL now rejects empty, oversized, or hostile type/status labels before insert; interpolated codes are restricted to lowercase ASCII `snake_case` characters so membership foreign keys remain referentially safe (ADR 0003 / ADR 0013).
- `persistence_postgres` live SQLx transport retains one pool-backed PostgreSQL connection per session so tenant binding and the following statement share a session, and closes the connection and owned runtime safely from another Tokio runtime (ADR 0013).
- The authored-line coverage gate now filters LLVM-only literal and expression continuation records while retaining branch coverage for their executable decisions; Rust function signatures and structural branch lines are no longer counted as uncovered statements.
Expand Down
6 changes: 6 additions & 0 deletions crates/evidence_core/src/error.rs
Original file line number Diff line number Diff line change
Expand Up @@ -44,6 +44,10 @@ pub enum EvidenceError {
InvalidLayoutBounds,
/// Layout coordinates exceeded the enclosing page.
LayoutOutOfBounds,
/// A base64 image data URI was treated as lexical inference text.
EmbeddedImageIsNotLexicalText,
/// An embedded image data URI declared an implausible image media type.
ImplausibleImageMediaType,
}

impl fmt::Display for EvidenceError {
Expand All @@ -70,6 +74,8 @@ impl fmt::Display for EvidenceError {
Self::InvalidPageGeometry => "page geometry must be finite and positive",
Self::InvalidLayoutBounds => "layout bounds must be finite, nonnegative, and nonempty",
Self::LayoutOutOfBounds => "layout bounds exceed the page geometry",
Self::EmbeddedImageIsNotLexicalText => "embedded image is not lexical text",
Self::ImplausibleImageMediaType => "embedded image media type is implausible",
Comment thread
seonghobae marked this conversation as resolved.
};
formatter.write_str(message)
}
Expand Down
254 changes: 254 additions & 0 deletions crates/evidence_core/src/image_unit.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,254 @@
//! Embedded `data:image` units that keep their original source location.

use crate::{DocumentRecord, EvidenceError, SourceSpan};

const DATA_IMAGE_PREFIX: &str = "data:image/";
const BASE64_MARK: &str = ";base64,";

/// Image media types accepted as plausible by [`embedded_image_units`].
///
/// The set is deliberately conservative and tracks widely registered or
/// de facto standard image subtypes; anything else fails closed instead of
/// yielding a bogus embedded-image unit.
const PLAUSIBLE_IMAGE_MEDIA_TYPES: [&str; 14] = [
"image/apng",
"image/avif",
"image/bmp",
"image/gif",
"image/heic",
"image/heif",
"image/jpeg",
"image/jpg",
"image/png",
"image/svg+xml",
"image/tiff",
"image/vnd.microsoft.icon",
"image/webp",
"image/x-icon",
];

/// One embedded image located in a document body.
#[derive(Clone, Copy, Debug, PartialEq)]
pub struct EmbeddedImageUnit<'document> {
span: SourceSpan,
media_type: &'document str,
}

impl<'document> EmbeddedImageUnit<'document> {
/// Exact source span of the data URI, including the `data:image/` prefix.
#[must_use]
pub const fn span(self) -> SourceSpan {
self.span
}

/// Declared image media type (`image/png`, `image/jpeg`, …).
#[must_use]
pub const fn media_type(self) -> &'document str {
self.media_type
}
}

/// Locate `data:image/<type>;base64,...` units and retain their original spans.
///
/// Only plausible image media types are accepted: a candidate URI whose
/// declared media type is not in [`PLAUSIBLE_IMAGE_MEDIA_TYPES`] fails the
/// whole parse so malformed bodies cannot produce bogus units.
///
/// # Errors
///
/// Returns [`EvidenceError::EmptySourceSpan`] when the document contains no
/// well-formed embedded image URI, and
/// [`EvidenceError::ImplausibleImageMediaType`] when a candidate URI
/// declares an implausible image media type.
pub fn embedded_image_units(
document: &DocumentRecord,
) -> Result<Vec<EmbeddedImageUnit<'_>>, EvidenceError> {
let text = document.text();
let mut units = Vec::new();
let mut search_from = 0usize;
while let Some(relative) = text[search_from..].find(DATA_IMAGE_PREFIX) {
let start = search_from + relative;
let after_prefix = start + DATA_IMAGE_PREFIX.len();
let Some(mark_rel) = text[after_prefix..].find(BASE64_MARK) else {
search_from = after_prefix;
continue;
};
let media_end = after_prefix + mark_rel;
let payload_start = media_end + BASE64_MARK.len();
let payload_end = payload_start
+ text[payload_start..]
.find(|ch: char| !is_base64_payload_char(ch))
.unwrap_or(text.len() - payload_start);
if payload_end == payload_start {
search_from = payload_start;
continue;
}
let media_type = &text[start + "data:".len()..media_end];
Comment thread
devin-ai-integration[bot] marked this conversation as resolved.
if media_type.contains(DATA_IMAGE_PREFIX) {
search_from = after_prefix;
continue;
}
Comment thread
seonghobae marked this conversation as resolved.
let base_media_type = base_media_type(media_type);
if !is_plausible_image_media_type(base_media_type) {
return Err(EvidenceError::ImplausibleImageMediaType);
}
let scalar_start = text[..start].chars().count();
let scalar_end = scalar_start + text[start..payload_end].chars().count();
let span = SourceSpan::new(document, start, payload_end, scalar_start, scalar_end, None)?;
units.push(EmbeddedImageUnit {
span,
media_type: base_media_type,
});
search_from = payload_end;
}
if units.is_empty() {
return Err(EvidenceError::EmptySourceSpan);
Comment thread
seonghobae marked this conversation as resolved.
}
Ok(units)
}

/// Refuse using a document body that still contains an embedded image as
/// lexical inference text.
///
/// # Errors
///
/// Returns [`EvidenceError::InvalidWirePayload`] for empty input and
/// [`EvidenceError::EmbeddedImageIsNotLexicalText`] when a `data:image`
/// base64 URI is present.
pub fn refuse_base64_image_as_lexical_text(text: &str) -> Result<(), EvidenceError> {
if text.is_empty() {
return Err(EvidenceError::InvalidWirePayload);
}
if contains_base64_image_data_uri(text) {
return Err(EvidenceError::EmbeddedImageIsNotLexicalText);
Comment thread
seonghobae marked this conversation as resolved.
}
Ok(())
}

fn contains_base64_image_data_uri(text: &str) -> bool {
let mut search_from = 0usize;
while let Some(relative) = text[search_from..].find(DATA_IMAGE_PREFIX) {
let start = search_from + relative;
let after_prefix = start + DATA_IMAGE_PREFIX.len();
let Some(mark_rel) = text[after_prefix..].find(BASE64_MARK) else {
search_from = after_prefix;
continue;
};
let media_end = after_prefix + mark_rel;
let media_type = &text[start + "data:".len()..media_end];
if is_image_media_type_token(media_type) {
return true;
}
search_from = after_prefix;
}
false
}

fn is_image_media_type_token(media_type: &str) -> bool {
let Some(subtype) = base_media_type(media_type).strip_prefix("image/") else {
return false;
};
!subtype.is_empty()
&& subtype
.bytes()
.all(|byte| byte.is_ascii_alphanumeric() || matches!(byte, b'.' | b'+' | b'-'))
}
Comment thread
seonghobae marked this conversation as resolved.
Comment on lines +147 to +155

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

📝 Info: Refusal and extraction classify media types differently

refuse_base64_image_as_lexical_text accepts any image/<token> subtype (is_image_media_type_token), while embedded_image_units restricts to the whitelist. The two functions therefore classify the same data:image/...;base64, input differently. Tests confirm this is intentional, refusal being deliberately broader.

Open in Devin Review

Was this helpful? React with 👍 or 👎 to provide feedback.


fn base_media_type(media_type: &str) -> &str {
media_type.split(';').next().unwrap_or("")
}

fn is_base64_payload_char(ch: char) -> bool {
ch.is_ascii_alphanumeric() || matches!(ch, '+' | '/' | '=')
}

/// Report whether a declared media type is a plausible image media type.
fn is_plausible_image_media_type(media_type: &str) -> bool {
PLAUSIBLE_IMAGE_MEDIA_TYPES.contains(&media_type)
}
Comment thread
seonghobae marked this conversation as resolved.

#[cfg(test)]
mod tests {
use super::{embedded_image_units, refuse_base64_image_as_lexical_text};
use crate::{DocumentRecord, EvidenceError, SourceArtifact};

#[test]
fn jpeg_uri_and_incomplete_prefix_are_classified() {
let text = "x data:image/jpeg;base64,/9j/4AA= y data:image/gif y";
let artifact = SourceArtifact::from_bytes(text.as_bytes()).expect("artifact");
let document = DocumentRecord::from_text(artifact.id(), text).expect("document");
let units = embedded_image_units(&document).expect("jpeg");
assert_eq!(units.len(), 1);
assert_eq!(units[0].media_type(), "image/jpeg");
refuse_base64_image_as_lexical_text("plain note").expect("plain");
refuse_base64_image_as_lexical_text("data:image/png").expect("incomplete image");
assert_eq!(
refuse_base64_image_as_lexical_text("data:image/png;base64,AAAA"),
Err(EvidenceError::EmbeddedImageIsNotLexicalText)
);

let empty_text = "data:image/png;base64, following text";
let empty_artifact = SourceArtifact::from_bytes(empty_text.as_bytes()).expect("artifact");
let empty_document =
DocumentRecord::from_text(empty_artifact.id(), empty_text).expect("document");
assert_eq!(
embedded_image_units(&empty_document),
Err(EvidenceError::EmptySourceSpan)
);
}

#[test]
fn implausible_media_types_fail_closed() {
let text = "data:image/not-a-type;base64,AAAA";
let artifact = SourceArtifact::from_bytes(text.as_bytes()).expect("artifact");
let document = DocumentRecord::from_text(artifact.id(), text).expect("document");
assert_eq!(
embedded_image_units(&document),
Err(EvidenceError::ImplausibleImageMediaType)
);
assert_eq!(
refuse_base64_image_as_lexical_text(text),
Err(EvidenceError::EmbeddedImageIsNotLexicalText)
);
}

#[test]
fn common_raster_media_types_are_accepted() {
let text = "a data:image/png;base64,AAAA b data:image/jpeg;base64,BBBB \
c data:image/webp;base64,CCCC d data:image/gif;base64,DDDD e";
let artifact = SourceArtifact::from_bytes(text.as_bytes()).expect("artifact");
let document = DocumentRecord::from_text(artifact.id(), text).expect("document");
let units = embedded_image_units(&document).expect("units");
let media_types: Vec<&str> = units.iter().map(|unit| unit.media_type()).collect();
assert_eq!(
media_types,
vec!["image/png", "image/jpeg", "image/webp", "image/gif"]
);
}

#[test]
fn empty_payload_is_not_an_image_unit() {
let text = "data:image/png;base64,";
let artifact = SourceArtifact::from_bytes(text.as_bytes()).expect("artifact");
let document = DocumentRecord::from_text(artifact.id(), text).expect("document");
assert_eq!(
embedded_image_units(&document),
Err(EvidenceError::EmptySourceSpan)
);
}

#[test]
fn malformed_image_prefix_does_not_swallow_later_valid_image() {
let text = "data:image/gif then data:image/png;base64,AAAA";
let artifact = SourceArtifact::from_bytes(text.as_bytes()).expect("artifact");
let document = DocumentRecord::from_text(artifact.id(), text).expect("document");

let units = embedded_image_units(&document).expect("png");
assert_eq!(units.len(), 1);
assert_eq!(units[0].media_type(), "image/png");
assert_eq!(
units[0].span().byte_start(),
text.find("data:image/png").expect("png start")
);
}
}
10 changes: 9 additions & 1 deletion crates/evidence_core/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -7,13 +7,15 @@
//! records, source spans whose byte, Unicode-scalar, page, and layout
//! coordinates are validated before entering later temporal or psychometric
//! layers, and strict versioned JSON wire contracts that reconstruct records
//! only through the same domain validation boundary.
//! only through the same domain validation boundary. Embedded `data:image`
//! units keep their original offsets and are not lexical inference text.

mod artifact;
mod digest;
mod document;
mod error;
mod identifier;
mod image_unit;
mod span;
mod wire;

Expand All @@ -27,6 +29,12 @@ pub use document::DocumentRecord;
pub use error::EvidenceError;
/// A validated RFC 9562 `UUIDv7` evidence identifier.
pub use identifier::EvidenceId;
/// One embedded image located in a document body.
pub use image_unit::EmbeddedImageUnit;
/// Locate `data:image` base64 units with exact source spans.
pub use image_unit::embedded_image_units;
/// Refuse treating an embedded image URI as lexical inference text.
pub use image_unit::refuse_base64_image_as_lexical_text;
/// A validated page-relative location for source evidence.
pub use span::PageLocation;
/// An exact byte, Unicode-scalar, and optional page/layout span.
Expand Down
Loading
Loading