Fix Telegram UTF-16 message splitting (#1961)

* Fix Telegram UTF-16 message splitting

* fix: bump telegram channel registry version
This commit is contained in:
firat.sertgoz
2026-04-08 11:06:15 +03:00
committed by GitHub
parent 7be3b910f4
commit 6f7575de5c
2 changed files with 76 additions and 20 deletions

View File

@@ -363,6 +363,26 @@ const TELEGRAM_STATUS_MAX_CHARS: usize = 600;
/// Telegram's hard limit for message text length.
const TELEGRAM_MAX_MESSAGE_LEN: usize = 4096;
fn utf16_code_unit_len(text: &str) -> usize {
text.encode_utf16().count()
}
fn prefix_within_utf16_limit(text: &str, max_units: usize) -> usize {
let mut units = 0;
let mut end = 0;
for (byte_idx, ch) in text.char_indices() {
let ch_units = ch.len_utf16();
if units + ch_units > max_units {
break;
}
units += ch_units;
end = byte_idx + ch.len_utf8();
}
end
}
fn truncate_status_message(input: &str, max_chars: usize) -> String {
let mut iter = input.chars();
let truncated: String = iter.by_ref().take(max_chars).collect();
@@ -373,7 +393,7 @@ fn truncate_status_message(input: &str, max_chars: usize) -> String {
}
}
/// Split a long message into chunks that fit within Telegram's 4096-char limit.
/// Split a long message into chunks that fit within Telegram's 4096 UTF-16-unit limit.
///
/// Tries to split at the most natural boundary available (in priority order):
/// 1. Double newline (paragraph break)
@@ -382,7 +402,7 @@ fn truncate_status_message(input: &str, max_chars: usize) -> String {
/// 4. Word boundary (space)
/// 5. Hard cut at the limit (last resort for pathological input)
fn split_message(text: &str) -> Vec<String> {
if text.chars().count() <= TELEGRAM_MAX_MESSAGE_LEN {
if utf16_code_unit_len(text) <= TELEGRAM_MAX_MESSAGE_LEN {
return vec![text.to_string()];
}
@@ -390,13 +410,8 @@ fn split_message(text: &str) -> Vec<String> {
let mut remaining = text;
while !remaining.is_empty() {
// Count chars to find the byte offset for our window.
let window_bytes = remaining
.char_indices()
.take(TELEGRAM_MAX_MESSAGE_LEN)
.last()
.map(|(byte_idx, ch)| byte_idx + ch.len_utf8())
.unwrap_or(remaining.len());
// Find the longest UTF-8 prefix that fits within Telegram's UTF-16 limit.
let window_bytes = prefix_within_utf16_limit(remaining, TELEGRAM_MAX_MESSAGE_LEN);
if window_bytes >= remaining.len() {
// Remainder fits entirely.
@@ -404,10 +419,24 @@ fn split_message(text: &str) -> Vec<String> {
break;
}
if window_bytes == 0 {
// Defensive fallback: make progress even if a future caller uses a
// smaller limit than a single scalar value can fit within.
let first_char_len = remaining
.chars()
.next()
.map(|ch| ch.len_utf8())
.unwrap_or(remaining.len());
chunks.push(remaining[..first_char_len].to_string());
remaining = &remaining[first_char_len..];
continue;
}
let window = &remaining[..window_bytes];
// 1. Double newline — best paragraph boundary
let split_at = window.rfind("\n\n")
let split_at = window
.rfind("\n\n")
// 2. Single newline
.or_else(|| window.rfind('\n'))
// 3. Sentence-ending punctuation followed by space.
@@ -417,9 +446,9 @@ fn split_message(text: &str) -> Vec<String> {
.or_else(|| {
let bytes = window.as_bytes();
// Search backwards for '. ', '! ', '? '
(1..bytes.len()).rev().find(|&i| {
matches!(bytes[i - 1], b'.' | b'!' | b'?') && bytes[i] == b' '
})
(1..bytes.len())
.rev()
.find(|&i| matches!(bytes[i - 1], b'.' | b'!' | b'?') && bytes[i] == b' ')
})
// 4. Word boundary (last space)
.or_else(|| window.rfind(' '))
@@ -427,7 +456,11 @@ fn split_message(text: &str) -> Vec<String> {
.unwrap_or(window_bytes);
// Avoid empty chunks (e.g. text starting with \n\n).
let split_at = if split_at == 0 { window_bytes } else { split_at };
let split_at = if split_at == 0 {
window_bytes
} else {
split_at
};
// Trim whitespace at chunk boundaries for clean Telegram display.
// Note: this drops leading/trailing spaces at split points, which is
@@ -1321,7 +1354,13 @@ fn send_response(
for (i, chunk) in chunks.into_iter().enumerate() {
// Try Markdown, fall back to plain text on parse errors
let result = send_message(chat_id, &chunk, reply_to, Some("Markdown"), message_thread_id);
let result = send_message(
chat_id,
&chunk,
reply_to,
Some("Markdown"),
message_thread_id,
);
let msg_id = match result {
Ok(id) => {
@@ -2150,6 +2189,10 @@ export!(TelegramChannel);
mod tests {
use super::*;
fn utf16_len(text: &str) -> usize {
text.encode_utf16().count()
}
#[test]
fn test_split_message_short() {
let text = "Hello, world!";
@@ -2177,7 +2220,7 @@ mod tests {
let chunks = split_message(&text);
assert!(chunks.len() > 1, "expected multiple chunks");
for chunk in &chunks {
assert!(chunk.chars().count() <= TELEGRAM_MAX_MESSAGE_LEN);
assert!(utf16_len(chunk) <= TELEGRAM_MAX_MESSAGE_LEN);
}
// Rejoined chunks must equal the original text exactly.
let rejoined = chunks.join(" ");
@@ -2193,7 +2236,7 @@ mod tests {
assert!(text.len() > TELEGRAM_MAX_MESSAGE_LEN);
let chunks = split_message(&text);
for chunk in &chunks {
assert!(chunk.chars().count() <= TELEGRAM_MAX_MESSAGE_LEN);
assert!(utf16_len(chunk) <= TELEGRAM_MAX_MESSAGE_LEN);
}
}
@@ -2223,7 +2266,7 @@ mod tests {
let chunks = split_message(&text);
assert!(chunks.len() >= 2);
for chunk in &chunks {
assert!(chunk.chars().count() <= TELEGRAM_MAX_MESSAGE_LEN);
assert!(utf16_len(chunk) <= TELEGRAM_MAX_MESSAGE_LEN);
}
// Rejoined must preserve all characters
let rejoined: String = chunks.concat();
@@ -2240,12 +2283,25 @@ mod tests {
let chunks = split_message(&text);
assert!(chunks.len() >= 2);
for chunk in &chunks {
assert!(chunk.chars().count() <= TELEGRAM_MAX_MESSAGE_LEN);
assert!(utf16_len(chunk) <= TELEGRAM_MAX_MESSAGE_LEN);
// Every char should be a complete emoji
assert!(chunk.chars().all(|c| c == '\u{1F600}'));
}
}
#[test]
fn test_split_message_exact_utf16_limit_for_surrogate_pairs() {
let emoji = "\u{1F600}"; // 😀
let text = emoji.repeat(TELEGRAM_MAX_MESSAGE_LEN);
let chunks = split_message(&text);
assert_eq!(chunks.len(), 2);
assert!(chunks
.iter()
.all(|chunk| utf16_len(chunk) <= TELEGRAM_MAX_MESSAGE_LEN));
}
#[test]
fn test_clean_message_text() {
// Without bot_username: strips any leading @mention

View File

@@ -2,7 +2,7 @@
"name": "telegram",
"display_name": "Telegram Channel",
"kind": "channel",
"version": "0.2.5",
"version": "0.2.6",
"wit_version": "0.3.0",
"description": "Talk to your agent through a Telegram bot",
"keywords": [