From 3bd0a7208a12e2792afff94aca168cb1cd748c2b Mon Sep 17 00:00:00 2001 From: GaoYuYang Date: Thu, 30 Jul 2026 15:54:22 +0800 Subject: [PATCH] fix(cli): decode stdin across chunk boundaries (#127) Co-authored-by: Cursor Co-authored-by: Yaowei Zheng --- .../2026-07-30-paste-filter-utf8.md | 9 ++++ changelog/unreleased/README.md | 2 + packages/cli/src/input.ts | 13 ++++- packages/cli/test/input.test.ts | 51 +++++++++++++++++++ 4 files changed, 74 insertions(+), 1 deletion(-) create mode 100644 changelog/unreleased/2026-07-30-paste-filter-utf8.md diff --git a/changelog/unreleased/2026-07-30-paste-filter-utf8.md b/changelog/unreleased/2026-07-30-paste-filter-utf8.md new file mode 100644 index 0000000..66a75e5 --- /dev/null +++ b/changelog/unreleased/2026-07-30-paste-filter-utf8.md @@ -0,0 +1,9 @@ +# CLI: pasting CJK or emoji into chat no longer corrupts characters at chunk boundaries + +`penguin chat` decoded each stdin chunk on its own, so a multi-byte character that a terminal split across two reads was destroyed in both halves. Pasting a few thousand Chinese characters was enough to lose one. + +`PasteFilter` sits between raw-mode stdin and readline, and a raw-mode chunk ends wherever the terminal's buffer ended — routinely mid-character, since a paste is delivered in fixed-size blocks that have nothing to do with character boundaries. Decoding a chunk in isolation turned its incomplete trailing bytes into U+FFFD immediately, and the leading bytes of the next chunk into more of them: a single 3-byte character came out as three replacement characters rather than one, so the text also silently grew. Pasting about 1400 Chinese characters — roughly one 4KB terminal block — was already enough to hit it, and the loss was invisible until the model answered about text the user never sent. + +Decoding now spans chunks via `StringDecoder`, which holds an incomplete trailing sequence until the following chunk completes it. This is what `setEncoding("utf8")` already does for the output of commands the Agent runs; the input path was the one place decoding was hand-rolled. The existing `leftover` mechanism, which reassembles a bracketed-paste marker split across chunks, is untouched: it works on decoded text and could only ever see characters that were already intact. + +Tests feed byte slices rather than strings, since a string chunk is by definition a whole number of characters and cannot express the bug: a typed CJK character cut in half, a pasted emoji cut two bytes in, a character delivered one byte at a time, and a marker and a character torn across the same chunk boundaries. diff --git a/changelog/unreleased/README.md b/changelog/unreleased/README.md index d5e8644..ae0fdc7 100644 --- a/changelog/unreleased/README.md +++ b/changelog/unreleased/README.md @@ -2,6 +2,8 @@ Changes since v0.1.4. The version number is assigned at release, when this folder is renamed. +- [2026-07-30] CLI: pasting CJK or emoji into `penguin chat` no longer corrupts a character wherever the terminal split its stdin blocks — each chunk was decoded on its own, so a character torn across two reads became three replacement characters instead of one, and about 1400 Chinese characters was enough to trigger it. ([details](2026-07-30-paste-filter-utf8.md)) + - [2026-07-30] Docs: three reference blocks catch up with the code they document — `run_subagent`'s argument block lists the `provider` that `model_id` must be paired with, the provider credential table covers the three gateway groups it had been missing, and the Project model entry table documents `max_tokens`. ([details](2026-07-30-docs-tools-and-configuration-reference.md)) - [2026-07-29] Core: every LLM failure except a rejected credential now retries inside the run — the classifier separating transient from permanent is an allowlist, so a gateway wording a transient fault its own way used to kill the turn — with the retry visible in both frontends, compaction on the same set under its own shorter budget, and a recovered failure no longer reported to the operator as an incident. Separately, pressing Stop mid-request can no longer leave a Session running forever when the provider's stream neither yields nor rejects after the abort. ([details](2026-07-29-llm-request-lifecycle.md)) diff --git a/packages/cli/src/input.ts b/packages/cli/src/input.ts index c609b9a..7a1fad0 100644 --- a/packages/cli/src/input.ts +++ b/packages/cli/src/input.ts @@ -12,6 +12,7 @@ * pending buffer as a whole and is sent on Enter. */ import { Transform, type TransformCallback } from "node:stream"; +import { StringDecoder } from "node:string_decoder"; const PASTE_START = "\x1b[200~"; const PASTE_END = "\x1b[201~"; @@ -35,9 +36,19 @@ export class PasteFilter extends Transform { private inPaste = false; private pasteBuf = ""; private leftover = ""; + /** + * Decodes across chunk boundaries: a raw-mode stdin chunk ends wherever the terminal's + * buffer did, so a multi-byte character can be torn in half. Decoding each chunk on its own + * would turn both halves into U+FFFD; the decoder holds an incomplete trailing sequence + * until the next chunk completes it. `leftover` below is the same idea one layer up, for a + * paste marker split across chunks, and it can only work on already-intact characters. + */ + private decoder = new StringDecoder("utf8"); override _transform(chunk: Buffer | string, _enc: BufferEncoding, cb: TransformCallback): void { - let data = this.leftover + chunk.toString("utf8"); + // stdin always delivers Buffers here (Writable decodes strings before _transform), but the + // signature allows a string, which is already-decoded text and needs no decoder. + let data = this.leftover + (typeof chunk === "string" ? chunk : this.decoder.write(chunk)); this.leftover = ""; while (data.length > 0) { diff --git a/packages/cli/test/input.test.ts b/packages/cli/test/input.test.ts index a5c056d..e7efe50 100644 --- a/packages/cli/test/input.test.ts +++ b/packages/cli/test/input.test.ts @@ -22,6 +22,24 @@ async function runFilter(chunks: string[]): Promise<{ forwarded: string; pastes: return { forwarded, pastes }; } +/** + * Feeds raw byte slices, so a multi-byte character can be torn across chunks the way a + * terminal's buffer tears one during a large paste. Output Buffers are concatenated before + * decoding: decoding each one on its own would introduce the very corruption under test. + */ +async function runFilterBytes(parts: Buffer[]): Promise<{ forwarded: string; pastes: string[] }> { + const filter = new PasteFilter(); + const out: Buffer[] = []; + const pastes: string[] = []; + filter.on("data", (d: Buffer) => out.push(d)); + filter.on("paste", (t: string) => pastes.push(t)); + for (const p of parts) filter.write(p); + await new Promise((resolve) => { + filter.end(() => resolve()); + }); + return { forwarded: Buffer.concat(out).toString("utf8"), pastes }; +} + describe("splitTrailingPartial", () => { it("holds a trailing partial-marker prefix", () => { expect(splitTrailingPartial("abc\x1b[200", "\x1b[200~")).toEqual({ @@ -61,6 +79,39 @@ describe("PasteFilter", () => { expect(forwarded).toBe("xy\r"); expect(pastes).toEqual(["mid"]); }); + + it("keeps a typed CJK character split across chunks intact", async () => { + const b = Buffer.from("你好世界\r", "utf8"); + // Cut inside 「好」, between its first and second byte. + const { forwarded } = await runFilterBytes([b.subarray(0, 4), b.subarray(4)]); + expect(forwarded).toBe("你好世界\r"); + }); + + it("keeps a pasted emoji split across chunks intact", async () => { + const b = Buffer.from("\x1b[200~ok🐧done\x1b[201~", "utf8"); + // Cut two bytes into the 4-byte emoji. + const cut = b.indexOf(Buffer.from("🐧", "utf8")) + 2; + const { forwarded, pastes } = await runFilterBytes([b.subarray(0, cut), b.subarray(cut)]); + expect(pastes).toEqual(["ok🐧done"]); + expect(forwarded).toBe(""); + }); + + it("keeps CJK intact when fed one byte at a time", async () => { + const b = Buffer.from("中\r", "utf8"); + const { forwarded } = await runFilterBytes([...b].map((x) => Buffer.from([x]))); + expect(forwarded).toBe("中\r"); + }); + + it("handles markers and multi-byte characters both split across the same chunks", async () => { + // 前(0-2) START(3-8) 文(9-11) 字(12-14) END(15-20) 後(21-23) \r(24); every cut below + // falls inside either a marker or a character. + const b = Buffer.from("前\x1b[200~文字\x1b[201~後\r", "utf8"); + const cuts = [2, 5, 10, 17, 22]; + const parts = [0, ...cuts].map((from, i) => b.subarray(from, cuts[i] ?? b.length)); + const { forwarded, pastes } = await runFilterBytes(parts); + expect(pastes).toEqual(["文字"]); + expect(forwarded).toBe("前後\r"); + }); }); describe("endsWithContinuation", () => {