From 7ca49a10420587452ffd63c818dc10d60387fc33 Mon Sep 17 00:00:00 2001 From: Zihan Dai Date: Fri, 2 Oct 2026 18:02:33 +1000 Subject: [PATCH] fix(csv-parse): preserve data when trimming single-byte encodings --- packages/csv-parse/lib/api/init_state.js | 14 +++-- packages/csv-parse/test/option.trim.ts | 65 ++++++++++++++++++++++++ 2 files changed, 71 insertions(+), 8 deletions(-) diff --git a/packages/csv-parse/lib/api/init_state.js b/packages/csv-parse/lib/api/init_state.js index 50b9e025c..f53d41874 100644 --- a/packages/csv-parse/lib/api/init_state.js +++ b/packages/csv-parse/lib/api/init_state.js @@ -6,9 +6,9 @@ const init_state = function (options) { // https://tc39.es/ecma262/#sec-white-space // https://tc39.es/ecma262/#sec-line-terminators // - // Codepoints unrepresentable in the target encoding are dropped: Node's - // Buffer substitutes them with `?` (0x3F), and including those would cause - // literal `?` bytes in the input to be trimmed under `latin1`/`ascii`. + // Drop codepoints that do not round-trip through the target encoding. + // Single-byte encodings truncate them and could otherwise trim ordinary + // input bytes, e.g. U+205F becomes `_` under `latin1` and `ascii`. const timchars = [ // Basic Latin 0x0020, // [Space](https://www.fileformat.info/info/unicode/char/0020/index.htm) @@ -40,11 +40,9 @@ const init_state = function (options) { 0x3000, // [IDEOGRAPHIC SPACE](https://www.fileformat.info/info/unicode/char/3000/index.htm) 0xfeff, // [ZERO WIDTH NO-BREAK SPACE (BOM)](https://www.fileformat.info/info/unicode/char/feff/index.htm) ].reduce((acc, codepoint) => { - const encoded = Buffer.from( - String.fromCharCode(codepoint), - options.encoding, - ); - if (codepoint !== 0x3f && encoded.length === 1 && encoded[0] === 0x3f) { + const character = String.fromCharCode(codepoint); + const encoded = Buffer.from(character, options.encoding); + if (encoded.toString(options.encoding || "utf8") !== character) { return acc; } acc.push(encoded); diff --git a/packages/csv-parse/test/option.trim.ts b/packages/csv-parse/test/option.trim.ts index 8975f8100..a04727d09 100644 --- a/packages/csv-parse/test/option.trim.ts +++ b/packages/csv-parse/test/option.trim.ts @@ -1,6 +1,7 @@ import "should"; import dedent from "dedent"; import { parse } from "../lib/index.js"; +import { parse as parseSync } from "../lib/sync.js"; describe("Option `trim`", function () { it("validation", function () { @@ -385,4 +386,68 @@ describe("Option `trim`", function () { parser.end(); }); }); + describe("single-byte encodings", function () { + for (const encoding of ["ascii", "latin1", "binary"] as const) { + for (const option of ["trim", "ltrim"] as const) { + it(`preserves field data with ${option} and ${encoding}`, function () { + const input = " _alice_, /home/user, (draft), ?value?"; + parseSync(Buffer.from(input, encoding), { + encoding, + [option]: true, + }).should.eql([["_alice_", "/home/user", "(draft)", "?value?"]]); + }); + + it(`preserves column names with ${option} and ${encoding}`, function () { + const input = " _id, /path, (status)\n _alice, /home, (active)"; + parseSync(Buffer.from(input, encoding), { + encoding, + columns: true, + [option]: true, + }).should.eql([ + { _id: "_alice", "/path": "/home", "(status)": "(active)" }, + ]); + }); + + it(`preserves data across writes with ${option} and ${encoding}`, function (next) { + const input = Buffer.from(" _alice_, /home, (draft)", encoding); + const parser = parse({ encoding, [option]: true }, (err, records) => { + if (err) return next(err); + records.should.eql([["_alice_", "/home", "(draft)"]]); + next(); + }); + for (let i = 0; i < input.length; i++) { + parser.write(input.subarray(i, i + 1)); + } + parser.end(); + }); + } + + it(`trims representable whitespace with ${encoding}`, function () { + const whitespace = + encoding === "ascii" ? " \t\r\n\v\f" : " \t\r\n\v\f\u00a0"; + const input = `${whitespace}_alice_${whitespace},${whitespace}/home${whitespace}`; + parseSync(Buffer.from(input, encoding), { + encoding, + trim: true, + record_delimiter: "|", + }).should.eql([["_alice_", "/home"]]); + }); + + it(`preserves quoted whitespace and data with ${encoding}`, function () { + parseSync(Buffer.from(' " _alice_ ", " /home "', encoding), { + encoding, + trim: true, + }).should.eql([[" _alice_ ", " /home "]]); + }); + + it(`rejects non-whitespace after closing quotes with ${encoding}`, function () { + (() => { + parseSync(Buffer.from('"value" _', encoding), { + encoding, + trim: true, + }); + }).should.throw({ code: "CSV_NON_TRIMABLE_CHAR_AFTER_CLOSING_QUOTE" }); + }); + } + }); });