From fcfd35a1fe2a323be52d3658584a375f18e0cd9b Mon Sep 17 00:00:00 2001 From: Peter LaFosse Date: Wed, 23 Sep 2026 16:13:56 -0400 Subject: [PATCH 1/3] Preserve escapes and format sequences when wrapping strings Keep complete escape and format atoms, preserve whitespace inside literals, and avoid replaying text at the annotation parsing limit. Fixes Vector35/binaryninja#2063 and Vector35/binaryninja#2074. --- formatter/generic/genericformatter.cpp | 79 ++++++++++++++++++-------- 1 file changed, 56 insertions(+), 23 deletions(-) diff --git a/formatter/generic/genericformatter.cpp b/formatter/generic/genericformatter.cpp index 8b9d565d5b..fe6b7270fc 100644 --- a/formatter/generic/genericformatter.cpp +++ b/formatter/generic/genericformatter.cpp @@ -156,9 +156,12 @@ struct Item if (!tokens.empty()) { InstructionTextToken token = tokens.front(); - string trimmedText = TrimLeadingWhitespace(token.text); - token.width -= token.text.size() - trimmedText.size(); - token.text = trimmedText; + if (token.type != StringToken) + { + string trimmedText = TrimLeadingWhitespace(token.text); + token.width -= token.text.size() - trimmedText.size(); + token.text = trimmedText; + } output.emplace_back(token); output.insert(output.end(), tokens.begin() + 1, tokens.end()); firstTokenOfLine = false; @@ -364,7 +367,29 @@ static vector ParseStringToken( size_t start = curEnd; curEnd++; // consume '\' if (curEnd < tail) - curEnd++; // consume escaped char + { + char escape = src[curEnd++]; + if (escape == 'x') + { + // A hexadecimal escape includes its digits. Splitting + // after the x produces an invalid C string literal. + while (curEnd < tail && isxdigit((unsigned char)src[curEnd])) + curEnd++; + } + else if (escape == 'u' || escape == 'U') + { + size_t digits = escape == 'u' ? 4 : 8; + while (digits-- && curEnd < tail && isxdigit((unsigned char)src[curEnd])) + curEnd++; + } + else if (escape >= '0' && escape <= '7') + { + // The first of at most three octal digits was consumed. + size_t digits = 2; + while (digits-- && curEnd < tail && src[curEnd] >= '0' && src[curEnd] <= '7') + curEnd++; + } + } ConstructToken(start, curEnd); curStart = curEnd; } @@ -399,18 +424,11 @@ static vector ParseStringToken( // Check if we've exceeded max parsing length if (curEnd > maxParsingLength) { - // Never cut in the middle of a grapheme cluster, which would leave both sides holding a piece - // of a character that neither renders nor measures on its own. Walking the clusters gives the - // last boundary that stays within the limit. - size_t splitPos = 0; - while (splitPos < src.size()) - { - size_t next = Unicode::GetNextGraphemeClusterBoundary(unprocessedStringToken.text, splitPos); - // Always consume at least one cluster to guarantee making progress - if (splitPos > 0 && next > maxParsingLength) - break; - splitPos = next; - } + // Finish the current atom/character without moving back into tokens + // already emitted (in particular an escape spanning the limit). + size_t splitPos = curStart; + while (splitPos < curEnd) + splitPos = Unicode::GetNextGraphemeClusterBoundary(unprocessedStringToken.text, splitPos); // Flush any pending token flushToken(curStart, splitPos); @@ -419,7 +437,8 @@ static vector ParseStringToken( InstructionTextToken remainingToken = unprocessedStringToken; remainingToken.text = string(src.substr(splitPos)); remainingToken.width = Unicode::GetDisplayWidth(remainingToken.text); - result.emplace_back(std::move(remainingToken)); + if (!remainingToken.text.empty()) + result.emplace_back(std::move(remainingToken)); return result; } } @@ -1137,10 +1156,14 @@ vector GenericLineFormatter::FormatLines( auto newLine = [&](const bool forString = false) { if (!firstTokenOfLine) { - string lastTokenText = outputLine.tokens.back().text; - string trimmedText = TrimTrailingWhitespace(lastTokenText); - outputLine.tokens.back().width -= lastTokenText.size() - trimmedText.size(); - outputLine.tokens.back().text = trimmedText; + // Whitespace inside a literal is data, including at a wrap point. + if (outputLine.tokens.back().type != StringToken) + { + string lastTokenText = outputLine.tokens.back().text; + string trimmedText = TrimTrailingWhitespace(lastTokenText); + outputLine.tokens.back().width -= lastTokenText.size() - trimmedText.size(); + outputLine.tokens.back().text = trimmedText; + } if (forString && outputLine.tokens.back().type == StringToken) { outputLine.tokens.emplace_back(BraceToken, "\""); @@ -1209,8 +1232,18 @@ vector GenericLineFormatter::FormatLines( if (desiredContinuationWidth < settings.minimumContentLength) desiredContinuationWidth = settings.minimumContentLength; - layoutStack.push({item->items, additionalContinuationIndentation, desiredWidth, - desiredContinuationWidth, desiredStringWidth, false}); + if (item->tokens.empty()) + { + layoutStack.push({item->items, additionalContinuationIndentation, desiredWidth, + desiredContinuationWidth, desiredStringWidth, false}); + } + else + { + // Escape/format atoms store their text in tokens, not + // child items. Keep the atom that caused the wrap. + item->AppendAllTokens(outputLine.tokens, firstTokenOfLine); + currentWidth += item->width; + } break; } From 95396bc62f96db420fd1ce62aecaa7024627aab6 Mon Sep 17 00:00:00 2001 From: Peter LaFosse Date: Wed, 23 Sep 2026 16:13:56 -0400 Subject: [PATCH 2/3] Preserve Unicode prefixes in Pseudo C constant strings Use the detected UTF-16 or UTF-32 literal prefix when rendering constant string data. Fixes Vector35/binaryninja#2064. --- lang/c/pseudoc.cpp | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/lang/c/pseudoc.cpp b/lang/c/pseudoc.cpp index d3aef7fdda..9d3bb6e668 100644 --- a/lang/c/pseudoc.cpp +++ b/lang/c/pseudoc.cpp @@ -1422,7 +1422,8 @@ void PseudoCFunction::GetExprTextInternal(const HighLevelILInstruction& instr, H { if (auto unicode = GetFunction()->GetView()->StringifyUnicodeData(instr.function->GetArchitecture(), db, nullTerminates); unicode.has_value()) { - auto wideStringPrefix = (builtin == BuiltinWcscpy) ? "L" : ""; + auto wideStringPrefix = (builtin == BuiltinWcscpy) ? "L" : + DisassemblyTextRenderer::GetStringLiteralPrefix(unicode.value().second); auto tokenContext = (builtin == BuiltinWcscpy) ? ConstStringDataTokenContext : ConstDataTokenContext; tokens.Append(BraceToken, wideStringPrefix + string("\"")); tokens.Append(StringToken, tokenContext, unicode.value().first, instr.address, data.value); From 4925d41d194b5a1871919d3bcc9d61560c6fc254 Mon Sep 17 00:00:00 2001 From: Peter LaFosse Date: Wed, 23 Sep 2026 16:13:56 -0400 Subject: [PATCH 3/3] Preserve every byte of counted Pseudo C copies Render memcpy constant data as complete byte escapes. Embedded NULs and display annotation limits must not truncate a counted byte copy. Fixes Vector35/binaryninja#2065. --- lang/c/pseudoc.cpp | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/lang/c/pseudoc.cpp b/lang/c/pseudoc.cpp index 9d3bb6e668..f7737bfa25 100644 --- a/lang/c/pseudoc.cpp +++ b/lang/c/pseudoc.cpp @@ -1393,6 +1393,15 @@ void PseudoCFunction::GetExprTextInternal(const HighLevelILInstruction& instr, H bool nullTerminates = true; switch (builtin) { + case BuiltinMemcpy: + { + // A byte copy needs the complete buffer, including embedded NULs. + // Unicode annotations can stop early or abbreviate long data. + tokens.Append(BraceToken, "\""); + tokens.Append(StringToken, ConstDataTokenContext, db.ToEscapedString(false, true), instr.address, data.value); + tokens.Append(BraceToken, "\""); + break; + } case BuiltinStrcpy: case BuiltinStrncpy: {