From d8d7bc8e7fdbedaf3f31b50063abe02075c2d767 Mon Sep 17 00:00:00 2001 From: Sidharth Menon Date: Mon, 5 Oct 2026 01:45:02 -0700 Subject: [PATCH 1/2] Escape U+2028 and U+2029 in NDJSON records JSON allows both characters raw inside strings, so diffr wrote them raw when a diffed file held one. Since Node 24, node:readline also ends a line at them, so a client reading diffr with readline got a record cut in two and failed to parse it (devdotfast/whiteboard#624). Records are now written through protocol::write_record, whose formatter escapes both as \u2028 and \u2029. The decoded text is unchanged. AI assistance: written with Claude Code; reviewed by the author. Agent-Session: 82924cb5-dcfa-42e4-abb0-ef40393249f8 Agent-Session: ea775653-fef8-47be-96ea-16815bdff22c Agent-Session: d72470a3-1f93-4201-8e6d-2730c798e92b --- crates/diffr-core/src/protocol/mod.rs | 59 +++++++++++++++++++++++++++ src/run.rs | 5 +-- 2 files changed, 61 insertions(+), 3 deletions(-) diff --git a/crates/diffr-core/src/protocol/mod.rs b/crates/diffr-core/src/protocol/mod.rs index 774ba5011..49da36de5 100644 --- a/crates/diffr-core/src/protocol/mod.rs +++ b/crates/diffr-core/src/protocol/mod.rs @@ -11,10 +11,16 @@ //! last line. Columns are 0-based byte offsets into the UTF-8 text on the //! wire. All ranges are half-open. //! +//! Records end at `\n` only. U+2028 and U+2029 are written as `\u2028` and +//! `\u2029`, because some line readers also end a line at them. +//! //! Sides are always `lhs` (before) and `rhs` (after). A `Pairing` says which //! sides exist and serializes by presence: `{lhs, rhs}`, `{lhs}`, or `{rhs}`. +use std::io::{self, Write}; + use serde::{Deserialize, Serialize}; +use serde_json::ser::Formatter; use crate::pairing::Pairing; @@ -23,6 +29,39 @@ pub mod project; /// The current wire version. Changes within a version are additive. pub const VERSION: u32 = 3; +/// Write `event` as one record: compact JSON and a `\n`. +pub fn write_record(writer: &mut W, event: &Event) -> serde_json::Result<()> { + event.serialize(&mut serde_json::Serializer::with_formatter( + &mut *writer, + RecordFormatter, + ))?; + writer.write_all(b"\n").map_err(serde_json::Error::io) +} + +/// Compact JSON that also escapes U+2028 and U+2029. +struct RecordFormatter; + +impl Formatter for RecordFormatter { + fn write_string_fragment( + &mut self, + writer: &mut W, + fragment: &str, + ) -> io::Result<()> { + fragment + .split_inclusive(['\u{2028}', '\u{2029}']) + .try_for_each(|piece| { + let mut chars = piece.chars(); + let escape: &[u8] = match chars.next_back() { + Some('\u{2028}') => b"\\u2028", + Some('\u{2029}') => b"\\u2029", + _ => return writer.write_all(piece.as_bytes()), + }; + writer.write_all(chars.as_str().as_bytes())?; + writer.write_all(escape) + }) + } +} + // ── stream ──────────────────────────────────────────────────────────────── #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] @@ -590,6 +629,26 @@ mod tests { } } + #[test] + fn records_escape_line_and_paragraph_separators() { + let event = Event::Complete { + succeeded: 0, + failed: 1, + aborted: Some(Problem { + code: "x".to_owned(), + message: "a\u{2028}b\u{2029}c\u{2028}".to_owned(), + }), + }; + let mut record = Vec::new(); + write_record(&mut record, &event).unwrap(); + let line = String::from_utf8(record).unwrap(); + assert!( + line.ends_with("\"message\":\"a\\u2028b\\u2029c\\u2028\"}}\n"), + "{line}" + ); + assert_eq!(serde_json::from_str::(&line).unwrap(), event); + } + #[test] fn defaults_are_omitted() { let line = serde_json::to_string(&example_file()).unwrap(); diff --git a/src/run.rs b/src/run.rs index d16ebe878..cd427c5d5 100644 --- a/src/run.rs +++ b/src/run.rs @@ -7,7 +7,7 @@ use crate::options::DiffOptions; use crate::plugin::{Classifier, MutationFailed, Pipeline}; use crate::present::present; use crate::protocol::project::{self, Inputs}; -use crate::protocol::{Diff, Event, FileChange, Outcome, Problem, VERSION}; +use crate::protocol::{self, Diff, Event, FileChange, Outcome, Problem, VERSION}; use crate::summary::{DiffResult, FallbackCause, FileContent, FileFormat}; use crate::tags; use std::io::{BufWriter, Write}; @@ -102,8 +102,7 @@ pub(crate) fn stream( ended.failed |= *failed > 0; ended.aborted = aborted.is_some(); } - serde_json::to_writer(&mut output, &event)?; - output.write_all(b"\n")?; + protocol::write_record(&mut output, &event)?; output.flush()?; } Ok(ended) From 3c396add4ad1d8f497fa093d37f5551bcd199043 Mon Sep 17 00:00:00 2001 From: Sidharth Menon Date: Mon, 5 Oct 2026 12:06:38 -0700 Subject: [PATCH 2/2] Escape U+0085 in NDJSON records Python's str.splitlines() also ends a line at U+0085 (NEL), which serde_json writes raw. Every other character it splits on is a control character JSON already escapes, so records now survive splitlines() whole. AI assistance: written with Claude Code; reviewed by the author. Agent-Session: ea775653-fef8-47be-96ea-16815bdff22c Agent-Session: d72470a3-1f93-4201-8e6d-2730c798e92b --- crates/diffr-core/src/protocol/mod.rs | 16 +++++++++------- 1 file changed, 9 insertions(+), 7 deletions(-) diff --git a/crates/diffr-core/src/protocol/mod.rs b/crates/diffr-core/src/protocol/mod.rs index 49da36de5..9515b96a3 100644 --- a/crates/diffr-core/src/protocol/mod.rs +++ b/crates/diffr-core/src/protocol/mod.rs @@ -11,8 +11,9 @@ //! last line. Columns are 0-based byte offsets into the UTF-8 text on the //! wire. All ranges are half-open. //! -//! Records end at `\n` only. U+2028 and U+2029 are written as `\u2028` and -//! `\u2029`, because some line readers also end a line at them. +//! Records end at `\n` only. U+0085, U+2028 and U+2029 are written as +//! `\u0085`, `\u2028` and `\u2029`, because some line readers also end a +//! line at them. //! //! Sides are always `lhs` (before) and `rhs` (after). A `Pairing` says which //! sides exist and serializes by presence: `{lhs, rhs}`, `{lhs}`, or `{rhs}`. @@ -38,7 +39,7 @@ pub fn write_record(writer: &mut W, event: &Event) -> serde_json::Resu writer.write_all(b"\n").map_err(serde_json::Error::io) } -/// Compact JSON that also escapes U+2028 and U+2029. +/// Compact JSON that also escapes U+0085, U+2028 and U+2029. struct RecordFormatter; impl Formatter for RecordFormatter { @@ -48,10 +49,11 @@ impl Formatter for RecordFormatter { fragment: &str, ) -> io::Result<()> { fragment - .split_inclusive(['\u{2028}', '\u{2029}']) + .split_inclusive(['\u{85}', '\u{2028}', '\u{2029}']) .try_for_each(|piece| { let mut chars = piece.chars(); let escape: &[u8] = match chars.next_back() { + Some('\u{85}') => b"\\u0085", Some('\u{2028}') => b"\\u2028", Some('\u{2029}') => b"\\u2029", _ => return writer.write_all(piece.as_bytes()), @@ -630,20 +632,20 @@ mod tests { } #[test] - fn records_escape_line_and_paragraph_separators() { + fn records_escape_line_separators() { let event = Event::Complete { succeeded: 0, failed: 1, aborted: Some(Problem { code: "x".to_owned(), - message: "a\u{2028}b\u{2029}c\u{2028}".to_owned(), + message: "a\u{2028}b\u{2029}c\u{85}d\u{2028}".to_owned(), }), }; let mut record = Vec::new(); write_record(&mut record, &event).unwrap(); let line = String::from_utf8(record).unwrap(); assert!( - line.ends_with("\"message\":\"a\\u2028b\\u2029c\\u2028\"}}\n"), + line.ends_with("\"message\":\"a\\u2028b\\u2029c\\u0085d\\u2028\"}}\n"), "{line}" ); assert_eq!(serde_json::from_str::(&line).unwrap(), event);