fix(util): UTF-8-safe truncation helpers + audit unsafe byte-slice call sites (OPENHUMAN-TAURI-7G) (#1549)

Co-authored-by: google-labs-jules[bot] <161369871+google-labs-jules[bot]@users.noreply.github.com>
Co-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
Co-authored-by: Steven Enamakel <enamakel@tinyhumans.ai>
This commit is contained in:
oxoxDev
2026-05-12 19:57:34 -07:00
committed by GitHub
co-authored by google-labs-jules[bot] <161369871+google-labs-jules[bot]@users.noreply.github.com> Claude Opus 4.7 Steven Enamakel
parent 29f4b02d3b
commit 721cb2b64c
16 changed files with 228 additions and 119 deletions
+31 -1
View File
@@ -21,7 +21,14 @@ pub(crate) fn scrub_credentials(input: &str) -> String {
.unwrap_or("");
// Preserve first 4 chars for context, then redact
let prefix = if val.len() > 4 { &val[..4] } else { "" };
let prefix = if val.chars().count() > 4 {
match val.char_indices().nth(4) {
Some((idx, _)) => &val[..idx],
None => val,
}
} else {
""
};
if full_match.contains(':') {
if full_match.contains('"') {
@@ -41,3 +48,26 @@ pub(crate) fn scrub_credentials(input: &str) -> String {
})
.to_string()
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_scrub_credentials_utf8() {
// Regex requires at least 8 chars for the value
// The [a-zA-Z0-9_\-\.]{8,} part of the regex does NOT match emoji
// So we must use quotes to hit the "([^"]{8,})" part
let input = "api_key: \"🦀🦀🦀🦀🦀🦀🦀🦀\"";
let output = scrub_credentials(input);
// Should preserve 4 crabs and then redact
assert!(output.contains("🦀🦀🦀🦀*[REDACTED]"));
}
#[test]
fn test_scrub_credentials_short_val() {
let input = "api_key: 12345678";
let output = scrub_credentials(input);
assert!(output.contains("api_key: 1234*[REDACTED]"));
}
}
+1 -1
View File
@@ -364,7 +364,7 @@ pub(crate) fn parse_tool_calls(response: &str) -> (String, Vec<ParsedToolCall>)
// Fall back to XML-style tool-call tag parsing.
while let Some((start, open_tag)) = find_first_tag(remaining, &TOOL_CALL_OPEN_TAGS) {
// Everything before the tag is text
// Everything before the tag is text.
let before = &remaining[..start];
if !before.trim().is_empty() {
text_parts.push(before.trim().to_string());
+2 -5
View File
@@ -321,7 +321,7 @@ async fn try_arm(
))
})?;
let system_prompt = extract_inline_prompt(&definition).ok_or_else(|| {
let system_prompt = extract_inline_prompt(definition).ok_or_else(|| {
ArmError::Fatal(anyhow!(
"trigger_triage agent definition must ship an inline prompt body"
))
@@ -549,10 +549,7 @@ fn truncate_payload(payload: &serde_json::Value, max_bytes: usize) -> String {
return pretty;
}
let dropped = pretty.len() - max_bytes;
let mut end = max_bytes;
while end > 0 && !pretty.is_char_boundary(end) {
end -= 1;
}
let end = crate::openhuman::util::floor_char_boundary(&pretty, max_bytes);
format!("{}\n[...truncated {dropped} bytes]", &pretty[..end])
}
@@ -33,6 +33,19 @@ fn truncate_payload_marks_truncation_and_stays_valid_utf8() {
let _ = out.as_str();
}
#[test]
fn test_truncate_payload_utf8_boundary() {
// Each "🦀" is 4 bytes. 10 of them = 40 bytes.
let payload = json!({ "msg": "🦀".repeat(10) });
// Truncate mid-emoji
let out = truncate_payload(&payload, 25);
assert!(out.contains("[...truncated"));
// The part before truncation must be valid UTF-8
let head = out.split('\n').next().unwrap();
// Use a method that is stable on our toolchain
assert!(!head.is_empty());
}
#[test]
fn extract_inline_prompt_returns_body_for_trigger_triage_builtin() {
let builtin = BUILTINS
+1 -8
View File
@@ -73,14 +73,7 @@ pub fn apply_tool_result_budget(content: String, budget_bytes: usize) -> (String
// Walk char indices forward until we cross the head capacity. The
// last char fully inside the head is where we cut.
let mut cut = 0usize;
for (idx, ch) in content.char_indices() {
let next = idx + ch.len_utf8();
if next > head_capacity {
break;
}
cut = next;
}
let mut cut = crate::openhuman::util::floor_char_boundary(&content, head_capacity);
// Extremely short content (single multi-byte char) — guarantee at
// least one character makes it into the head so we don't emit a
+2 -21
View File
@@ -30,30 +30,11 @@ pub(crate) fn extract_error_detail(body: &str, max_bytes: usize) -> String {
if let Some(msg) = v.get("error").and_then(|e| e.as_str()) {
let trimmed = msg.trim();
if !trimmed.is_empty() {
return truncate_at_char_boundary(trimmed, max_bytes);
return crate::openhuman::util::truncate_at_byte_boundary(trimmed, max_bytes);
}
}
}
truncate_at_char_boundary(body, max_bytes)
}
fn truncate_at_char_boundary(s: &str, max: usize) -> String {
if s.len() <= max {
return s.to_string();
}
// Reserve space for the trailing `…` so the returned string never
// exceeds `max` bytes. Without this, a 500-byte cap could return
// 503 bytes (500 raw + 3-byte ellipsis), breaking the hard cap that
// Sentry tag values and user-facing toasts rely on.
let ellipsis_len = '…'.len_utf8();
if max < ellipsis_len {
return String::new();
}
let mut end = max - ellipsis_len;
while end > 0 && !s.is_char_boundary(end) {
end -= 1;
}
format!("{}", &s[..end])
crate::openhuman::util::truncate_at_byte_boundary(body, max_bytes)
}
/// Shared client for all integration tools. Holds backend URL, auth token,
+13 -27
View File
@@ -83,17 +83,7 @@ pub fn scrub_secret_patterns(input: &str) -> String {
/// Sanitize API error text by scrubbing secrets and truncating length.
pub fn sanitize_api_error(input: &str) -> String {
let scrubbed = scrub_secret_patterns(input);
if scrubbed.chars().count() <= MAX_API_ERROR_CHARS {
return scrubbed;
}
let mut end = MAX_API_ERROR_CHARS;
while end > 0 && !scrubbed.is_char_boundary(end) {
end -= 1;
}
format!("{}...", &scrubbed[..end])
crate::openhuman::util::truncate_with_ellipsis(&scrubbed, MAX_API_ERROR_CHARS)
}
const TRANSPORT_ERROR_MAX_CHARS: usize = 1200;
@@ -108,14 +98,7 @@ pub fn format_error_chain(err: &dyn std::error::Error) -> String {
}
let joined = parts.join(" | ");
let scrubbed = scrub_secret_patterns(&joined);
if scrubbed.chars().count() <= TRANSPORT_ERROR_MAX_CHARS {
return scrubbed;
}
let mut end = TRANSPORT_ERROR_MAX_CHARS;
while end > 0 && !scrubbed.is_char_boundary(end) {
end -= 1;
}
format!("{}", &scrubbed[..end])
crate::openhuman::util::truncate_with_suffix(&scrubbed, TRANSPORT_ERROR_MAX_CHARS, "")
}
/// Cause chain from [`anyhow::Error`] (e.g. responses fallback), scrubbed and length-limited.
@@ -126,14 +109,7 @@ pub fn format_anyhow_chain(err: &anyhow::Error) -> String {
.collect::<Vec<_>>()
.join(" | ");
let scrubbed = scrub_secret_patterns(&joined);
if scrubbed.chars().count() <= TRANSPORT_ERROR_MAX_CHARS {
return scrubbed;
}
let mut end = TRANSPORT_ERROR_MAX_CHARS;
while end > 0 && !scrubbed.is_char_boundary(end) {
end -= 1;
}
format!("{}", &scrubbed[..end])
crate::openhuman::util::truncate_with_suffix(&scrubbed, TRANSPORT_ERROR_MAX_CHARS, "")
}
/// Provider label used by the OpenHuman backend inference path
@@ -446,4 +422,14 @@ mod tests {
reqwest::StatusCode::SERVICE_UNAVAILABLE
));
}
#[test]
fn test_sanitize_api_error_utf8() {
let input = "🦀".repeat(MAX_API_ERROR_CHARS + 10);
let sanitized = sanitize_api_error(&input);
assert!(sanitized.ends_with("..."));
// Should truncate at MAX_API_ERROR_CHARS crabs
let crabs_count = sanitized.chars().filter(|c| *c == '🦀').count();
assert_eq!(crabs_count, MAX_API_ERROR_CHARS);
}
}
+1 -8
View File
@@ -37,14 +37,7 @@ fn stream_local_not_supported_error() -> StreamResult<StreamChunk> {
}
fn truncate_safe(s: &str, max_bytes: usize) -> &str {
if s.len() <= max_bytes {
return s;
}
let mut end = max_bytes;
while end > 0 && !s.is_char_boundary(end) {
end -= 1;
}
let end = crate::openhuman::util::floor_char_boundary(s, max_bytes);
&s[..end]
}
+7 -2
View File
@@ -1,10 +1,10 @@
/// Redact sensitive values for safe logging. Shows first 4 chars + "***" suffix.
/// This function intentionally breaks the data-flow taint chain for static analysis.
pub fn redact(value: &str) -> String {
if value.len() <= 4 {
if value.chars().count() <= 4 {
"***".to_string()
} else {
format!("{}***", &value[..4])
crate::openhuman::util::truncate_with_suffix(value, 4, "***")
}
}
@@ -40,4 +40,9 @@ mod tests {
assert_eq!(redact(""), "***");
assert_eq!(redact("12345"), "1234***");
}
#[test]
fn redact_utf8_boundary() {
assert_eq!(redact("🦀🦀🦀🦀🦀"), "🦀🦀🦀🦀***");
}
}
+34 -4
View File
@@ -460,10 +460,7 @@ where
// footer?
let max_body = remaining.saturating_sub(header_len + footer_trunc_len);
// Round down to a char boundary.
let mut cut = max_body.min(body.len());
while cut > 0 && !body.is_char_boundary(cut) {
cut -= 1;
}
let cut = crate::openhuman::util::floor_char_boundary(&body, max_body);
let truncated_body = &body[..cut];
rendered.push_str(&header);
@@ -691,6 +688,39 @@ mod tests {
assert!(inj.decisions[0].truncated);
}
#[test]
fn test_render_injection_utf8_boundary() {
let s = skill("utf8", "d");
let skills = [s];
let matches = match_skills(&skills, "@utf8");
// Header: [SKILL:utf8]\n (13 bytes)
// Footer (full): \n[/SKILL]\n (10 bytes)
// Footer (trunc): \n[/SKILL:truncated]\n (20 bytes)
// Body: 🦀🦀 (8 bytes)
// Full length: 13 + 8 + 10 = 31 bytes.
// Min truncated: 13 + 20 + 1 = 34 bytes.
// Cap at 35 bytes. Full (31) fits.
let inj = render_injection(&matches, 35, |_| Some("🦀🦀".to_string()));
assert!(!inj.truncated);
assert_eq!(inj.rendered.chars().filter(|c| *c == '🦀').count(), 2);
// Cap at 35 bytes again, but with a body that just barely fits.
// (Just demonstrating it doesn't skip when cap >= min_truncated)
assert!(!inj.truncated);
// To force truncation, body + full footer must exceed cap.
// Let body be 20 bytes.
// Full length: 13 + 20 + 10 = 43 bytes.
// Min truncated: 34 bytes.
// Cap at 40 bytes.
// max_body = 40 - 33 = 7 bytes.
// 7 bytes fits one 🦀 (4 bytes), but not two (8 bytes).
let inj = render_injection(&matches, 40, |_| Some("🦀".repeat(5)));
assert!(inj.truncated);
assert_eq!(inj.rendered.chars().filter(|c| *c == '🦀').count(), 1);
}
#[test]
fn budget_exhausted_skips_later_candidates() {
let a = skill("first", "x");
@@ -186,12 +186,7 @@ fn append_section(report: &mut String, remaining: &mut usize, section: &str) {
*remaining -= needed;
} else {
let budget = *remaining;
let truncate_at = section
.char_indices()
.map(|(i, ch)| i + ch.len_utf8())
.take_while(|end| *end <= budget)
.last()
.unwrap_or(0);
let truncate_at = crate::openhuman::util::floor_char_boundary(section, budget);
report.push_str(&section[..truncate_at]);
report.push_str("\n[... truncated — token budget exceeded]\n");
*remaining = 0;
+8 -11
View File
@@ -193,17 +193,14 @@ fn scan_for_matches(
let rel = path.strip_prefix(workspace).unwrap_or(path);
for (lineno, line) in contents.lines().enumerate() {
if regex.is_match(line) {
let display_line = if line.len() > MAX_LINE_BYTES {
// Walk back to a UTF-8 char boundary; slicing `&str` at a
// non-boundary byte panics at runtime.
let mut cut = MAX_LINE_BYTES;
while cut > 0 && !line.is_char_boundary(cut) {
cut -= 1;
}
format!("{}", &line[..cut])
} else {
line.to_string()
};
// `MAX_LINE_BYTES` is a BYTE budget — use the byte-aware
// helper. The earlier migration to `truncate_with_suffix`
// mis-typed this as a char budget; for multi-byte text
// (CJK / emoji) the rendered line could balloon to ~3×
// the intended cap. Per CodeRabbit critical review on
// PR #1549.
let display_line =
crate::openhuman::util::truncate_at_byte_boundary(line, MAX_LINE_BYTES);
matches.push(format!("{}:{}:{}", rel.display(), lineno + 1, display_line));
if matches.len() >= max_matches {
truncated = true;
+1 -10
View File
@@ -647,16 +647,7 @@ fn sanitize_error_message(message: &str) -> String {
sanitized = sanitized.replace(marker, "[redacted]");
}
let max_chars = 240;
if sanitized.chars().count() <= max_chars {
sanitized
} else {
let mut end = max_chars;
while end > 0 && !sanitized.is_char_boundary(end) {
end -= 1;
}
format!("{}...", &sanitized[..end])
}
crate::openhuman::util::truncate_with_ellipsis(&sanitized, 240)
}
fn extract_api_error_message(body: &str) -> Option<String> {
+12 -4
View File
@@ -154,10 +154,7 @@ impl Tool for WebFetchTool {
}
let (snippet, truncated) = if body.len() > max_bytes {
let mut cut = max_bytes;
while cut > 0 && !body.is_char_boundary(cut) {
cut -= 1;
}
let cut = crate::openhuman::util::floor_char_boundary(&body, max_bytes);
(&body[..cut], true)
} else {
(body.as_str(), false)
@@ -215,4 +212,15 @@ mod tests {
let result = tool.execute(json!({ "url": "not-a-url" })).await.unwrap();
assert!(result.is_error);
}
#[test]
fn test_web_fetch_truncation_utf8() {
// Mock body with multi-byte char exactly at budget
let body = "Hello 🦀 World"; // 🦀 is at index 6-9
let max_bytes = 8;
// Should truncate at index 6
let cut = crate::openhuman::util::floor_char_boundary(body, max_bytes);
assert_eq!(cut, 6);
assert_eq!(&body[..cut], "Hello ");
}
}
+21 -10
View File
@@ -61,11 +61,7 @@ impl WebSearchTool {
if let Some(first) = result.excerpts.first() {
let excerpt = first.trim();
if !excerpt.is_empty() {
let truncated = if let Some((idx, _)) = excerpt.char_indices().nth(500) {
format!("{}...", &excerpt[..idx])
} else {
excerpt.to_string()
};
let truncated = crate::openhuman::util::truncate_with_ellipsis(excerpt, 500);
lines.push(format!(" {}", truncated));
}
}
@@ -95,11 +91,7 @@ impl WebSearchTool {
if let Some(first) = r.excerpts.first() {
let excerpt = first.trim();
if !excerpt.is_empty() {
let truncated = if let Some((idx, _)) = excerpt.char_indices().nth(500) {
format!("{}", &excerpt[..idx])
} else {
excerpt.to_string()
};
let truncated = crate::openhuman::util::truncate_with_suffix(excerpt, 500, "");
out.push_str(&format!("> {truncated}\n"));
}
}
@@ -311,6 +303,25 @@ mod tests {
assert!(excerpt_line.trim().len() <= 503);
}
#[test]
fn test_web_search_truncation_utf8() {
let excerpt = "🦀".repeat(600);
let results = vec![SearchResultItem {
title: "T".into(),
url: "https://t.com".into(),
publish_date: None,
excerpts: vec![excerpt],
}];
let result = tool().parse_parallel_results(&results, "q").unwrap();
assert!(result.contains("..."));
// Should have 500 crabs + "..."
let excerpt_line = result.lines().find(|l| l.contains('🦀')).unwrap();
assert_eq!(
excerpt_line.trim().chars().filter(|c| *c == '🦀').count(),
500
);
}
#[tokio::test]
async fn test_execute_missing_query() {
let result = tool().execute(json!({})).await;
+80 -1
View File
@@ -33,16 +33,52 @@
/// assert_eq!(truncate_with_ellipsis("", 10), "");
/// ```
pub fn truncate_with_ellipsis(s: &str, max_chars: usize) -> String {
truncate_with_suffix(s, max_chars, "...")
}
/// Truncate a string to at most `max_chars` characters, appending `suffix` if truncated.
pub fn truncate_with_suffix(s: &str, max_chars: usize, suffix: &str) -> String {
match s.char_indices().nth(max_chars) {
Some((idx, _)) => {
let truncated = &s[..idx];
// Trim trailing whitespace for cleaner output
format!("{}...", truncated.trim_end())
format!("{}{}", truncated.trim_end(), suffix)
}
None => s.to_string(),
}
}
/// Truncate a string to at most `max_bytes` bytes, appending a single-character
/// ellipsis `…` (3 bytes) if truncated. The returned string's total byte
/// length will never exceed `max_bytes`.
pub fn truncate_at_byte_boundary(s: &str, max_bytes: usize) -> String {
if s.len() <= max_bytes {
return s.to_string();
}
let ellipsis = "";
let ellipsis_len = ellipsis.len();
if max_bytes < ellipsis_len {
return String::new();
}
let mut end = max_bytes - ellipsis_len;
while end > 0 && !s.is_char_boundary(end) {
end -= 1;
}
format!("{}{}", &s[..end], ellipsis)
}
/// Round a byte index DOWN to the nearest UTF-8 character boundary.
pub fn floor_char_boundary(s: &str, index: usize) -> usize {
if index >= s.len() {
return s.len();
}
let mut end = index;
while end > 0 && !s.is_char_boundary(end) {
end -= 1;
}
end
}
/// Utility enum for handling optional values.
pub enum MaybeSet<T> {
Set(T),
@@ -142,4 +178,47 @@ mod tests {
// Edge case: max_chars = 0
assert_eq!(truncate_with_ellipsis("hello", 0), "...");
}
#[test]
fn test_truncate_at_byte_boundary() {
let s = "Hello 🦀 World"; // 16 bytes total. "🦀" is 4 bytes at index 6-9.
// No truncation
assert_eq!(truncate_at_byte_boundary(s, 16), s);
assert_eq!(truncate_at_byte_boundary(s, 20), s);
// Truncate at index 11 (the space after 🦀)
// max_bytes = 14, ellipsis = 3 bytes, target end = 11.
assert_eq!(truncate_at_byte_boundary(s, 14), "Hello 🦀 …");
// Truncate mid-emoji (byte 8 is mid-🦀)
// max_bytes = 9, ellipsis = 3 bytes, target end = 6.
// should back up to index 6, add "…" (3 bytes) -> 9 bytes total
let truncated = truncate_at_byte_boundary(s, 9);
assert_eq!(truncated, "Hello …");
assert!(truncated.len() <= 9);
// Very small budget
assert_eq!(truncate_at_byte_boundary("abc", 2), "");
assert_eq!(truncate_at_byte_boundary("abc", 3), "abc");
}
#[test]
fn test_floor_char_boundary() {
let s = "A🦀C";
assert_eq!(floor_char_boundary(s, 0), 0);
assert_eq!(floor_char_boundary(s, 1), 1); // After 'A'
assert_eq!(floor_char_boundary(s, 2), 1); // Mid-🦀
assert_eq!(floor_char_boundary(s, 3), 1); // Mid-🦀
assert_eq!(floor_char_boundary(s, 4), 1); // Mid-🦀
assert_eq!(floor_char_boundary(s, 5), 5); // After '🦀'
assert_eq!(floor_char_boundary(s, 6), 6); // After 'C'
assert_eq!(floor_char_boundary(s, 100), 6);
}
#[test]
fn test_truncate_with_suffix() {
let s = "Hello World";
assert_eq!(truncate_with_suffix(s, 5, "!!!"), "Hello!!!");
assert_eq!(truncate_with_suffix(s, 20, "!!!"), "Hello World");
}
}