Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 6 additions & 8 deletions packages/csv-parse/lib/api/init_state.js
Original file line number Diff line number Diff line change
Expand Up @@ -6,9 +6,9 @@ const init_state = function (options) {
// https://tc39.es/ecma262/#sec-white-space
// https://tc39.es/ecma262/#sec-line-terminators
//
// Codepoints unrepresentable in the target encoding are dropped: Node's
// Buffer substitutes them with `?` (0x3F), and including those would cause
// literal `?` bytes in the input to be trimmed under `latin1`/`ascii`.
// Drop codepoints that do not round-trip through the target encoding.
// Single-byte encodings truncate them and could otherwise trim ordinary
// input bytes, e.g. U+205F becomes `_` under `latin1` and `ascii`.
const timchars = [
// Basic Latin
0x0020, // [Space](https://www.fileformat.info/info/unicode/char/0020/index.htm)
Expand Down Expand Up @@ -40,11 +40,9 @@ const init_state = function (options) {
0x3000, // [IDEOGRAPHIC SPACE](https://www.fileformat.info/info/unicode/char/3000/index.htm)
0xfeff, // [ZERO WIDTH NO-BREAK SPACE (BOM)](https://www.fileformat.info/info/unicode/char/feff/index.htm)
].reduce((acc, codepoint) => {
const encoded = Buffer.from(
String.fromCharCode(codepoint),
options.encoding,
);
if (codepoint !== 0x3f && encoded.length === 1 && encoded[0] === 0x3f) {
const character = String.fromCharCode(codepoint);
const encoded = Buffer.from(character, options.encoding);
if (encoded.toString(options.encoding || "utf8") !== character) {
return acc;
}
acc.push(encoded);
Expand Down
65 changes: 65 additions & 0 deletions packages/csv-parse/test/option.trim.ts
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
import "should";
import dedent from "dedent";
import { parse } from "../lib/index.js";
import { parse as parseSync } from "../lib/sync.js";

describe("Option `trim`", function () {
it("validation", function () {
Expand Down Expand Up @@ -385,4 +386,68 @@ describe("Option `trim`", function () {
parser.end();
});
});
describe("single-byte encodings", function () {
for (const encoding of ["ascii", "latin1", "binary"] as const) {
for (const option of ["trim", "ltrim"] as const) {
it(`preserves field data with ${option} and ${encoding}`, function () {
const input = " _alice_, /home/user, (draft), ?value?";
parseSync(Buffer.from(input, encoding), {
encoding,
[option]: true,
}).should.eql([["_alice_", "/home/user", "(draft)", "?value?"]]);
});

it(`preserves column names with ${option} and ${encoding}`, function () {
const input = " _id, /path, (status)\n _alice, /home, (active)";
parseSync(Buffer.from(input, encoding), {
encoding,
columns: true,
[option]: true,
}).should.eql([
{ _id: "_alice", "/path": "/home", "(status)": "(active)" },
]);
});

it(`preserves data across writes with ${option} and ${encoding}`, function (next) {
const input = Buffer.from(" _alice_, /home, (draft)", encoding);
const parser = parse({ encoding, [option]: true }, (err, records) => {
if (err) return next(err);
records.should.eql([["_alice_", "/home", "(draft)"]]);
next();
});
for (let i = 0; i < input.length; i++) {
parser.write(input.subarray(i, i + 1));
}
parser.end();
});
}

it(`trims representable whitespace with ${encoding}`, function () {
const whitespace =
encoding === "ascii" ? " \t\r\n\v\f" : " \t\r\n\v\f\u00a0";
const input = `${whitespace}_alice_${whitespace},${whitespace}/home${whitespace}`;
parseSync(Buffer.from(input, encoding), {
encoding,
trim: true,
record_delimiter: "|",
}).should.eql([["_alice_", "/home"]]);
});

it(`preserves quoted whitespace and data with ${encoding}`, function () {
parseSync(Buffer.from(' " _alice_ ", " /home "', encoding), {
encoding,
trim: true,
}).should.eql([[" _alice_ ", " /home "]]);
});

it(`rejects non-whitespace after closing quotes with ${encoding}`, function () {
(() => {
parseSync(Buffer.from('"value" _', encoding), {
encoding,
trim: true,
});
}).should.throw({ code: "CSV_NON_TRIMABLE_CHAR_AFTER_CLOSING_QUOTE" });
});
}
});
});