Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 8 additions & 2 deletions .github/labeler.yml
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
# Scope labels for actions/labeler@v6
# Maps file path globs to scope labels. Multiple labels can apply per PR.
# Labels for actions/labeler@v6
# Maps file path globs to labels. Multiple labels can apply per PR.

"scope: agent":
- changed-files:
Expand Down Expand Up @@ -164,3 +164,9 @@
- any-glob-to-any-file:
- Cargo.toml
- Cargo.lock

"DB MIGRATION":
- changed-files:
- any-glob-to-any-file:
- migrations/**
- src/db/libsql_migrations.rs
3 changes: 3 additions & 0 deletions .github/scripts/create-labels.sh
Original file line number Diff line number Diff line change
Expand Up @@ -62,6 +62,9 @@ create "scope: ci" "546E7A" "CI/CD workflows"
create "scope: docs" "78909C" "Documentation"
create "scope: dependencies" "90A4AE" "Dependency updates"

echo "==> Creating coordination labels..."
create "DB MIGRATION" "C62828" "PR adds or modifies PostgreSQL or libSQL migration definitions"

echo "==> Creating workflow labels..."
create "skip-regression-check" "9E9E9E" "Acknowledged: fix without regression test"

Expand Down
7 changes: 7 additions & 0 deletions .github/workflows/pr-label-scope.yml
Original file line number Diff line number Diff line change
Expand Up @@ -6,12 +6,19 @@ on:

permissions:
contents: read
issues: write
pull-requests: write

jobs:
scope:
runs-on: ubuntu-latest
steps:
- name: Ensure DB MIGRATION label exists
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
REPO: ${{ github.repository }}
run: gh label create "DB MIGRATION" --repo "$REPO" --color C62828 --description "PR adds or modifies PostgreSQL or libSQL migration definitions" --force

- uses: actions/labeler@8558fd74291d67161a8a78ce36a881fa63b766a9 # v5
with:
configuration-path: .github/labeler.yml
Expand Down
71 changes: 58 additions & 13 deletions channels-src/telegram/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -376,6 +376,26 @@ const TELEGRAM_STATUS_MAX_CHARS: usize = 600;
/// Telegram's hard limit for message text length.
const TELEGRAM_MAX_MESSAGE_LEN: usize = 4096;

fn utf16_code_unit_len(text: &str) -> usize {
text.encode_utf16().count()
}

fn prefix_within_utf16_limit(text: &str, max_units: usize) -> usize {
let mut units = 0;
let mut end = 0;

for (byte_idx, ch) in text.char_indices() {
let ch_units = ch.len_utf16();
if units + ch_units > max_units {
break;
}
units += ch_units;
end = byte_idx + ch.len_utf8();
}

end
}

fn truncate_status_message(input: &str, max_chars: usize) -> String {
let mut iter = input.chars();
let truncated: String = iter.by_ref().take(max_chars).collect();
Expand All @@ -386,7 +406,7 @@ fn truncate_status_message(input: &str, max_chars: usize) -> String {
}
}

/// Split a long message into chunks that fit within Telegram's 4096-char limit.
/// Split a long message into chunks that fit within Telegram's 4096 UTF-16-unit limit.
///
/// Tries to split at the most natural boundary available (in priority order):
/// 1. Double newline (paragraph break)
Expand All @@ -395,28 +415,36 @@ fn truncate_status_message(input: &str, max_chars: usize) -> String {
/// 4. Word boundary (space)
/// 5. Hard cut at the limit (last resort for pathological input)
fn split_message(text: &str) -> Vec<String> {
if text.chars().count() <= TELEGRAM_MAX_MESSAGE_LEN {
if utf16_code_unit_len(text) <= TELEGRAM_MAX_MESSAGE_LEN {
return vec![text.to_string()];
}

let mut chunks: Vec<String> = Vec::new();
let mut remaining = text;

while !remaining.is_empty() {
// Count chars to find the byte offset for our window.
let window_bytes = remaining
.char_indices()
.take(TELEGRAM_MAX_MESSAGE_LEN)
.last()
.map(|(byte_idx, ch)| byte_idx + ch.len_utf8())
.unwrap_or(remaining.len());
// Find the longest UTF-8 prefix that fits within Telegram's UTF-16 limit.
let window_bytes = prefix_within_utf16_limit(remaining, TELEGRAM_MAX_MESSAGE_LEN);

if window_bytes >= remaining.len() {
// Remainder fits entirely.
chunks.push(remaining.to_string());
break;
}

if window_bytes == 0 {
// Defensive fallback: make progress even if a future caller uses a
// smaller limit than a single scalar value can fit within.
let first_char_len = remaining
.chars()
.next()
.map(|ch| ch.len_utf8())
.unwrap_or(remaining.len());
chunks.push(remaining[..first_char_len].to_string());
remaining = &remaining[first_char_len..];
continue;
}
Comment on lines +435 to +446

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

medium

The defensive fallback for window_bytes == 0 correctly ensures progress if a single character exceeds the UTF-16 limit. However, it pushes the chunk directly without the trim_end() applied in the main loop logic. While unlikely to be an issue with a 4096-unit limit, for consistency and to avoid sending chunks that might be just whitespace (which Telegram rejects), consider applying the same trimming logic or filtering out empty chunks.


let window = &remaining[..window_bytes];

// 1. Double newline — best paragraph boundary
Expand Down Expand Up @@ -2268,6 +2296,10 @@ export!(TelegramChannel);
mod tests {
use super::*;

fn utf16_len(text: &str) -> usize {
text.encode_utf16().count()
}

#[test]
fn test_split_message_short() {
let text = "Hello, world!";
Expand Down Expand Up @@ -2295,7 +2327,7 @@ mod tests {
let chunks = split_message(&text);
assert!(chunks.len() > 1, "expected multiple chunks");
for chunk in &chunks {
assert!(chunk.chars().count() <= TELEGRAM_MAX_MESSAGE_LEN);
assert!(utf16_len(chunk) <= TELEGRAM_MAX_MESSAGE_LEN);
}
// Rejoined chunks must equal the original text exactly.
let rejoined = chunks.join(" ");
Expand All @@ -2311,7 +2343,7 @@ mod tests {
assert!(text.len() > TELEGRAM_MAX_MESSAGE_LEN);
let chunks = split_message(&text);
for chunk in &chunks {
assert!(chunk.chars().count() <= TELEGRAM_MAX_MESSAGE_LEN);
assert!(utf16_len(chunk) <= TELEGRAM_MAX_MESSAGE_LEN);
}
}

Expand Down Expand Up @@ -2341,7 +2373,7 @@ mod tests {
let chunks = split_message(&text);
assert!(chunks.len() >= 2);
for chunk in &chunks {
assert!(chunk.chars().count() <= TELEGRAM_MAX_MESSAGE_LEN);
assert!(utf16_len(chunk) <= TELEGRAM_MAX_MESSAGE_LEN);
}
// Rejoined must preserve all characters
let rejoined: String = chunks.concat();
Expand All @@ -2358,12 +2390,25 @@ mod tests {
let chunks = split_message(&text);
assert!(chunks.len() >= 2);
for chunk in &chunks {
assert!(chunk.chars().count() <= TELEGRAM_MAX_MESSAGE_LEN);
assert!(utf16_len(chunk) <= TELEGRAM_MAX_MESSAGE_LEN);
// Every char should be a complete emoji
assert!(chunk.chars().all(|c| c == '\u{1F600}'));
}
}

#[test]
fn test_split_message_exact_utf16_limit_for_surrogate_pairs() {
let emoji = "\u{1F600}"; // 😀
let text = emoji.repeat(TELEGRAM_MAX_MESSAGE_LEN);

let chunks = split_message(&text);

assert_eq!(chunks.len(), 2);
assert!(chunks
.iter()
.all(|chunk| utf16_len(chunk) <= TELEGRAM_MAX_MESSAGE_LEN));
}

#[test]
fn test_clean_message_text() {
// Without bot_username: strips any leading @mention
Expand Down
Loading