From d1e0ca6be363a94c7b2b1626296bfea83de404bc Mon Sep 17 00:00:00 2001 From: Benedikt Franke Date: Sat, 3 Oct 2026 18:23:35 +0200 Subject: [PATCH 01/10] Add Printer::stripIgnoredCharacters() MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Port of graphql-js stripIgnoredCharacters, including its test and fuzz suites. BlockString::print() gains a $minimize flag. 🤖 Generated with Claude Code --- docs/class-reference.md | 42 +++ src/Language/BlockString.php | 17 +- src/Language/Printer.php | 91 ++++++ tests/Language/BlockStringTest.php | 9 + .../StripIgnoredCharactersFuzzTest.php | 303 ++++++++++++++++++ tests/Language/StripIgnoredCharactersTest.php | 292 +++++++++++++++++ 6 files changed, 747 insertions(+), 7 deletions(-) create mode 100644 tests/Language/StripIgnoredCharactersFuzzTest.php create mode 100644 tests/Language/StripIgnoredCharactersTest.php diff --git a/docs/class-reference.md b/docs/class-reference.md index 81cd48e81..c886b99b2 100644 --- a/docs/class-reference.md +++ b/docs/class-reference.md @@ -1386,6 +1386,48 @@ $printed = GraphQL\Language\Printer::doPrint($ast); static function doPrint(GraphQL\Language\AST\Node $ast): string ``` +````php +/** + * Strips characters that are not significant to the validity or execution of a GraphQL document: + * - UnicodeBOM + * - WhiteSpace + * - LineTerminator + * - Comment + * - Comma + * - BlockString indentation + * + * Neighboring non-punctuator tokens are always delimited by a single space. + * Parsing input and output yields the same AST, apart from node locations. + * The output is stable, but may change between releases. + * + * ```graphql + * query SomeQuery($foo: String!, $bar: String) { + * someField(foo: $foo, bar: $bar) { + * a + * b { + * c + * d + * } + * } + * } + * ``` + * + * becomes + * + * ```graphql + * query SomeQuery($foo:String!$bar:String){someField(foo:$foo bar:$bar){a b{c d}}} + * ``` + * + * @param Source|string $source + * + * @throws \JsonException + * @throws SyntaxError + * + * @api + */ +static function stripIgnoredCharacters($source): string +```` + ## GraphQL\Language\Visitor Utility for efficient AST traversal and modification. diff --git a/src/Language/BlockString.php b/src/Language/BlockString.php index 76a9bd0e7..5cecdcbdb 100644 --- a/src/Language/BlockString.php +++ b/src/Language/BlockString.php @@ -102,7 +102,7 @@ public static function getIndentation(string $value): int * trailing blank line. However, if a block string starts with whitespace and is * a single-line, adding a leading blank line would strip that whitespace. */ - public static function print(string $value): string + public static function print(string $value, bool $minimize = false): string { $escapedValue = str_replace('"""', '\\"""', $value); @@ -131,11 +131,14 @@ public static function print(string $value): string $forceTrailingNewline = $hasTrailingQuote || $hasTrailingSlash; // add leading and trailing new lines only if it improves readability - $printAsMultipleLines = ! $isSingleLine - || mb_strlen($value) > 70 - || $forceTrailingNewline - || $forceLeadingNewLine - || $hasTrailingTripleQuotes; + $printAsMultipleLines = ! $minimize + && ( + ! $isSingleLine + || mb_strlen($value) > 70 + || $forceTrailingNewline + || $forceLeadingNewLine + || $hasTrailingTripleQuotes + ); $result = ''; @@ -146,7 +149,7 @@ public static function print(string $value): string } $result .= $escapedValue; - if ($printAsMultipleLines) { + if ($printAsMultipleLines || $forceTrailingNewline) { $result .= "\n"; } diff --git a/src/Language/Printer.php b/src/Language/Printer.php index bc061ad0f..889ef4936 100644 --- a/src/Language/Printer.php +++ b/src/Language/Printer.php @@ -2,6 +2,7 @@ namespace GraphQL\Language; +use GraphQL\Error\SyntaxError; use GraphQL\Language\AST\ArgumentNode; use GraphQL\Language\AST\BooleanValueNode; use GraphQL\Language\AST\DirectiveDefinitionNode; @@ -78,6 +79,96 @@ public static function doPrint(Node $ast): string return static::p($ast); } + /** + * Strips characters that are not significant to the validity or execution of a GraphQL document: + * - UnicodeBOM + * - WhiteSpace + * - LineTerminator + * - Comment + * - Comma + * - BlockString indentation + * + * Neighboring non-punctuator tokens are always delimited by a single space. + * Parsing input and output yields the same AST, apart from node locations. + * The output is stable, but may change between releases. + * + * ```graphql + * query SomeQuery($foo: String!, $bar: String) { + * someField(foo: $foo, bar: $bar) { + * a + * b { + * c + * d + * } + * } + * } + * ``` + * + * becomes + * + * ```graphql + * query SomeQuery($foo:String!$bar:String){someField(foo:$foo bar:$bar){a b{c d}}} + * ``` + * + * @param Source|string $source + * + * @throws \JsonException + * @throws SyntaxError + * + * @api + */ + public static function stripIgnoredCharacters($source): string + { + $sourceObj = $source instanceof Source + ? $source + : new Source($source); + $lexer = new Lexer($sourceObj); + $utf32Body = mb_convert_encoding($sourceObj->body, 'UTF-32', 'UTF-8'); + + $stripped = ''; + $wasLastAddedTokenNonPunctuator = false; + while (($token = $lexer->advance())->kind !== Token::EOF) { + $isNonPunctuator = ! static::isPunctuatorTokenKind($token->kind); + + // `1...` would lex as an invalid float + if ($wasLastAddedTokenNonPunctuator && ($isNonPunctuator || $token->kind === Token::SPREAD)) { + $stripped .= ' '; + } + + $stripped .= static::tokenSource($utf32Body, $token); + $wasLastAddedTokenNonPunctuator = $isNonPunctuator; + } + + return $stripped; + } + + protected static function isPunctuatorTokenKind(string $kind): bool + { + return ! in_array($kind, [Token::NAME, Token::INT, Token::FLOAT, Token::STRING, Token::BLOCK_STRING], true); + } + + protected static function tokenSource(string $utf32Body, Token $token): string + { + switch ($token->kind) { + case Token::BLOCK_STRING: + assert(is_string($token->value)); + + return BlockString::print($token->value, true); + case Token::STRING: + $utf32Token = substr($utf32Body, $token->start * 4, ($token->end - $token->start) * 4); + + return mb_convert_encoding($utf32Token, 'UTF-8', 'UTF-32'); + case Token::NAME: + case Token::INT: + case Token::FLOAT: + assert(is_string($token->value)); + + return $token->value; + default: + return $token->kind; + } + } + /** @throws \JsonException */ protected static function p(?Node $node): string { diff --git a/tests/Language/BlockStringTest.php b/tests/Language/BlockStringTest.php index b1d9d74df..ccef6464f 100644 --- a/tests/Language/BlockStringTest.php +++ b/tests/Language/BlockStringTest.php @@ -331,6 +331,7 @@ public function testDoNotEscapeCharacters(): void EOF, BlockString::print($str) ); + self::assertSame("\"\"\"\n{$str}\"\"\"", BlockString::print($str, true)); } /** @see it('by default print block strings as single line', () => { */ @@ -338,6 +339,7 @@ public function testByDefaultPrintBlockStringsAsSingleLine(): void { $str = 'one liner'; self::assertSame('"""one liner"""', BlockString::print($str)); + self::assertSame('"""one liner"""', BlockString::print($str, true)); } /** @see it('by default print block strings ending with triple quotation as multi-line', () => { */ @@ -352,6 +354,7 @@ public function testByDefaultPrintBlockStringsEndingWithTripleQuotationAsMultiLi EOF, BlockString::print($str) ); + self::assertSame('"""triple quotation \\""""""', BlockString::print($str, true)); } /** @see it('correctly prints single-line with leading space') */ @@ -359,6 +362,7 @@ public function testCorrectlyPrintsSingleLineWithLeadingSpace(): void { $str = ' space-led string'; self::assertSame('""" space-led string"""', BlockString::print($str)); + self::assertSame('""" space-led string"""', BlockString::print($str, true)); } /** @see it('correctly prints single-line with leading space and trailing quotation', () => { */ @@ -372,6 +376,7 @@ public function testCorrectlyPrintsSingleLineWithLeadingSpaceAndQuotation(): voi EOF, BlockString::print($str) ); + self::assertSame("\"\"\" space-led value \"quoted string\"\n\"\"\"", BlockString::print($str, true)); } /** @see it('correctly prints single-line with trailing backslash') */ @@ -386,6 +391,7 @@ public function testCorrectlyPrintsSingleLineWithTrailingBackslash(): void EOF, BlockString::print($str) ); + self::assertSame("\"\"\"backslash \\\n\"\"\"", BlockString::print($str, true)); } /** @see it('correctly prints multi-line with internal indent', () => { */ @@ -404,6 +410,7 @@ public function testCorrectlyPrintsMultiLineWithInternalIndent(): void EOF, BlockString::print($str) ); + self::assertSame("\"\"\"\nno indent\n with indent\"\"\"", BlockString::print($str, true)); } /** @see it('correctly prints string with a first line indentation') */ @@ -426,11 +433,13 @@ public function testCorrectlyPrintsStringWithAFirstLineIndentation(): void EOF, BlockString::print($str) ); + self::assertSame("\"\"\"{$str}\"\"\"", BlockString::print($str, true)); } public function testCorrectlyPrintsEmptyString(): void { $str = ''; self::assertSame('""""""', BlockString::print($str)); + self::assertSame('""""""', BlockString::print($str, true)); } } diff --git a/tests/Language/StripIgnoredCharactersFuzzTest.php b/tests/Language/StripIgnoredCharactersFuzzTest.php new file mode 100644 index 000000000..cc4498799 --- /dev/null +++ b/tests/Language/StripIgnoredCharactersFuzzTest.php @@ -0,0 +1,303 @@ + { + */ +final class StripIgnoredCharactersFuzzTest extends TestCase +{ + private const IGNORED_TOKENS = [ + "\u{FEFF}", + "\t", + ' ', + "\n", + "\r", + "\r\n", + "# \"Comment\" string\n", + ',', + ]; + + private const PUNCTUATOR_TOKENS = [ + '!', + '$', + '(', + ')', + '...', + ':', + '=', + '@', + '[', + ']', + '{', + '|', + '}', + ]; + + private const NON_PUNCTUATOR_TOKENS = [ + 'name_token', + '1', + '3.14', + '"some string value"', + "\"\"\"block\nstring\nvalue\"\"\"", + ]; + + /** + * @throws \JsonException + * @throws SyntaxError + */ + private static function lexValue(string $str): ?string + { + $lexer = new Lexer(new Source($str)); + $value = $lexer->advance()->value; + + self::assertSame(Token::EOF, $lexer->advance()->kind, 'Expected EOF'); + + return $value; + } + + /** + * @throws \JsonException + * @throws SyntaxError + */ + private static function assertStripped(string $expected, string $docString): void + { + $stripped = Printer::stripIgnoredCharacters($docString); + self::assertSame($expected, $stripped, 'Stripping ' . json_encode($docString)); + + $strippedTwice = Printer::stripIgnoredCharacters($stripped); + self::assertSame($stripped, $strippedTwice, 'Stripping twice ' . json_encode($stripped)); + } + + /** + * @throws \JsonException + * @throws SyntaxError + */ + private static function assertStaysTheSame(string $docString): void + { + self::assertStripped($docString, $docString); + } + + /** @throws \JsonException */ + private static function isSingleToken(string $str): bool + { + $lexer = new Lexer(new Source($str)); + try { + $lexer->advance(); + + return $lexer->advance()->kind === Token::EOF; + } catch (SyntaxError $invalidToken) { + return false; + } + } + + /** + * @param list $allowedChars + * + * @return \Generator + */ + private static function genFuzzStrings(array $allowedChars, int $maxLength): \Generator + { + $numAllowedChars = count($allowedChars); + + $numCombinations = 0; + for ($length = 1; $length <= $maxLength; ++$length) { + $numCombinations += $numAllowedChars ** $length; + } + + yield ''; + for ($combination = 0; $combination < $numCombinations; ++$combination) { + $permutation = ''; + + $leftOver = $combination; + while ($leftOver >= 0) { + $reminder = $leftOver % $numAllowedChars; + $permutation = $allowedChars[$reminder] . $permutation; + $leftOver = intdiv($leftOver - $reminder, $numAllowedChars) - 1; + } + + yield $permutation; + } + } + + /** @see it('strips documents with random combination of ignored characters', () => { */ + public function testStripsDocumentsWithRandomCombinationOfIgnoredCharacters(): void + { + foreach (self::IGNORED_TOKENS as $ignored) { + self::assertStripped('', $ignored); + + foreach (self::IGNORED_TOKENS as $anotherIgnored) { + self::assertStripped('', $ignored . $anotherIgnored); + } + } + self::assertStripped('', implode('', self::IGNORED_TOKENS)); + } + + /** @see it('strips random leading and trailing ignored tokens', () => { */ + public function testStripsRandomLeadingAndTrailingIgnoredTokens(): void + { + foreach ([...self::PUNCTUATOR_TOKENS, ...self::NON_PUNCTUATOR_TOKENS] as $token) { + foreach (self::IGNORED_TOKENS as $ignored) { + self::assertStripped($token, $ignored . $token); + self::assertStripped($token, $token . $ignored); + + foreach (self::IGNORED_TOKENS as $anotherIgnored) { + self::assertStripped($token, $token . $ignored . $anotherIgnored); + self::assertStripped($token, $ignored . $anotherIgnored . $token); + } + } + + self::assertStripped($token, implode('', self::IGNORED_TOKENS) . $token); + self::assertStripped($token, $token . implode('', self::IGNORED_TOKENS)); + } + } + + /** @see it('strips random ignored tokens between punctuator tokens', () => { */ + public function testStripsRandomIgnoredTokensBetweenPunctuatorTokens(): void + { + foreach (self::PUNCTUATOR_TOKENS as $left) { + foreach (self::PUNCTUATOR_TOKENS as $right) { + foreach (self::IGNORED_TOKENS as $ignored) { + self::assertStripped($left . $right, $left . $ignored . $right); + + foreach (self::IGNORED_TOKENS as $anotherIgnored) { + self::assertStripped($left . $right, $left . $ignored . $anotherIgnored . $right); + } + } + + self::assertStripped($left . $right, $left . implode('', self::IGNORED_TOKENS) . $right); + } + } + } + + /** @see it('strips random ignored tokens between punctuator and non-punctuator tokens', () => { */ + public function testStripsRandomIgnoredTokensBetweenPunctuatorAndNonPunctuatorTokens(): void + { + foreach (self::NON_PUNCTUATOR_TOKENS as $nonPunctuator) { + foreach (self::PUNCTUATOR_TOKENS as $punctuator) { + foreach (self::IGNORED_TOKENS as $ignored) { + self::assertStripped($punctuator . $nonPunctuator, $punctuator . $ignored . $nonPunctuator); + + foreach (self::IGNORED_TOKENS as $anotherIgnored) { + self::assertStripped($punctuator . $nonPunctuator, $punctuator . $ignored . $anotherIgnored . $nonPunctuator); + } + } + + self::assertStripped($punctuator . $nonPunctuator, $punctuator . implode('', self::IGNORED_TOKENS) . $nonPunctuator); + } + } + } + + /** @see it('strips random ignored tokens between non-punctuator and punctuator tokens', () => { */ + public function testStripsRandomIgnoredTokensBetweenNonPunctuatorAndPunctuatorTokens(): void + { + foreach (self::NON_PUNCTUATOR_TOKENS as $nonPunctuator) { + foreach (self::PUNCTUATOR_TOKENS as $punctuator) { + // Covered by testReplaceRandomIgnoredTokensBetweenNonPunctuatorTokensAndSpreadWithSpace + if ($punctuator === '...') { + continue; + } + + foreach (self::IGNORED_TOKENS as $ignored) { + self::assertStripped($nonPunctuator . $punctuator, $nonPunctuator . $ignored . $punctuator); + + foreach (self::IGNORED_TOKENS as $anotherIgnored) { + self::assertStripped($nonPunctuator . $punctuator, $nonPunctuator . $ignored . $anotherIgnored . $punctuator); + } + } + + self::assertStripped($nonPunctuator . $punctuator, $nonPunctuator . implode('', self::IGNORED_TOKENS) . $punctuator); + } + } + } + + /** @see it('replace random ignored tokens between non-punctuator tokens and spread with space', () => { */ + public function testReplaceRandomIgnoredTokensBetweenNonPunctuatorTokensAndSpreadWithSpace(): void + { + foreach (self::NON_PUNCTUATOR_TOKENS as $nonPunctuator) { + foreach (self::IGNORED_TOKENS as $ignored) { + self::assertStripped($nonPunctuator . ' ...', $nonPunctuator . $ignored . '...'); + + foreach (self::IGNORED_TOKENS as $anotherIgnored) { + self::assertStripped($nonPunctuator . ' ...', $nonPunctuator . $ignored . $anotherIgnored . ' ...'); + } + } + + self::assertStripped($nonPunctuator . ' ...', $nonPunctuator . implode('', self::IGNORED_TOKENS) . '...'); + } + } + + /** @see it('replace random ignored tokens between non-punctuator tokens with space', () => { */ + public function testReplaceRandomIgnoredTokensBetweenNonPunctuatorTokensWithSpace(): void + { + foreach (self::NON_PUNCTUATOR_TOKENS as $left) { + foreach (self::NON_PUNCTUATOR_TOKENS as $right) { + foreach (self::IGNORED_TOKENS as $ignored) { + self::assertStripped($left . ' ' . $right, $left . $ignored . $right); + + foreach (self::IGNORED_TOKENS as $anotherIgnored) { + self::assertStripped($left . ' ' . $right, $left . $ignored . $anotherIgnored . $right); + } + } + + self::assertStripped($left . ' ' . $right, $left . implode('', self::IGNORED_TOKENS) . $right); + } + } + } + + /** @see it('does not strip random ignored tokens embedded in the string', () => { */ + public function testDoesNotStripRandomIgnoredTokensEmbeddedInTheString(): void + { + foreach (self::IGNORED_TOKENS as $ignored) { + self::assertStaysTheSame(json_encode($ignored, JSON_UNESCAPED_UNICODE)); + + foreach (self::IGNORED_TOKENS as $anotherIgnored) { + self::assertStaysTheSame(json_encode($ignored . $anotherIgnored, JSON_UNESCAPED_UNICODE)); + } + } + + self::assertStaysTheSame(json_encode(implode('', self::IGNORED_TOKENS), JSON_UNESCAPED_UNICODE)); + } + + /** @see it('does not strip random ignored tokens embedded in the block string', () => { */ + public function testDoesNotStripRandomIgnoredTokensEmbeddedInTheBlockString(): void + { + $ignoredTokensWithoutFormatting = array_diff(self::IGNORED_TOKENS, ["\n", "\r", "\r\n", "\t", ' ']); + foreach ($ignoredTokensWithoutFormatting as $ignored) { + self::assertStaysTheSame('"""|' . $ignored . '|"""'); + + foreach ($ignoredTokensWithoutFormatting as $anotherIgnored) { + self::assertStaysTheSame('"""|' . $ignored . $anotherIgnored . '|"""'); + } + } + + self::assertStaysTheSame('"""|' . implode('', $ignoredTokensWithoutFormatting) . '|"""'); + } + + /** @see it('strips ignored characters inside random block strings', () => { */ + public function testStripsIgnoredCharactersInsideRandomBlockStrings(): void + { + // Lengths above 7 take exponentially longer, but test with them when changing the implementation + foreach (self::genFuzzStrings(["\n", "\t", ' ', '"', 'a', '\\'], 7) as $fuzzStr) { + $testStr = '"""' . $fuzzStr . '"""'; + + if (! self::isSingleToken($testStr)) { + continue; + } + + $testValue = self::lexValue($testStr); + + $strippedValue = self::lexValue(Printer::stripIgnoredCharacters($testStr)); + self::assertSame($testValue, $strippedValue, 'Stripping ' . json_encode($testStr)); + } + } +} diff --git a/tests/Language/StripIgnoredCharactersTest.php b/tests/Language/StripIgnoredCharactersTest.php new file mode 100644 index 000000000..24255ec17 --- /dev/null +++ b/tests/Language/StripIgnoredCharactersTest.php @@ -0,0 +1,292 @@ + { + */ +final class StripIgnoredCharactersTest extends TestCase +{ + /** + * @throws \JsonException + * @throws SyntaxError + */ + private static function lexValue(string $str): ?string + { + $lexer = new Lexer(new Source($str)); + $value = $lexer->advance()->value; + + self::assertSame(Token::EOF, $lexer->advance()->kind, 'Expected EOF'); + + return $value; + } + + /** + * @throws \JsonException + * @throws SyntaxError + */ + private static function assertStripped(string $expected, string $docString): void + { + $stripped = Printer::stripIgnoredCharacters($docString); + self::assertSame($expected, $stripped); + + $strippedTwice = Printer::stripIgnoredCharacters($stripped); + self::assertSame($expected, $strippedTwice); + } + + /** + * @throws \JsonException + * @throws SyntaxError + */ + private static function assertStaysTheSame(string $docString): void + { + self::assertStripped($docString, $docString); + } + + /** + * @throws \JsonException + * @throws SyntaxError + */ + private static function assertStrippedString(string $expected, string $blockStr): void + { + $originalValue = self::lexValue($blockStr); + $strippedValue = self::lexValue(Printer::stripIgnoredCharacters($blockStr)); + self::assertSame($originalValue, $strippedValue); + + self::assertStripped($expected, $blockStr); + } + + /** @see it('strips ignored characters from GraphQL query document', () => { */ + public function testStripsIgnoredCharactersFromGraphQLQueryDocument(): void + { + $query = <<<'GRAPHQL' + query SomeQuery($foo: String!, $bar: String) { + someField(foo: $foo, bar: $bar) { + a + b { + c + d + } + } + } + GRAPHQL; + + self::assertSame( + 'query SomeQuery($foo:String!$bar:String){someField(foo:$foo bar:$bar){a b{c d}}}', + Printer::stripIgnoredCharacters($query) + ); + } + + /** @see it('accepts Source object', () => { */ + public function testAcceptsSourceObject(): void + { + self::assertSame('{a}', Printer::stripIgnoredCharacters(new Source('{ a }'))); + } + + /** @see it('strips ignored characters from GraphQL SDL document', () => { */ + public function testStripsIgnoredCharactersFromGraphQLSDLDocument(): void + { + $sdl = <<<'GRAPHQL' + """ + Type description + """ + type Foo { + """ + Field description + """ + bar: String + } + GRAPHQL; + + self::assertSame( + '"""Type description""" type Foo{"""Field description""" bar:String}', + Printer::stripIgnoredCharacters($sdl) + ); + } + + /** @see it('report document with invalid token', () => { */ + public function testReportDocumentWithInvalidToken(): void + { + try { + Printer::stripIgnoredCharacters("{ foo(arg: \"\n\""); + self::fail('Expected SyntaxError'); + } catch (SyntaxError $error) { + self::assertSame( + <<<'EOF' + Syntax Error: Unterminated string. + + GraphQL request (1:13) + 1: { foo(arg: " + ^ + 2: " + + EOF, + FormattedError::printError($error) + ); + } + } + + /** @see it('strips non-parsable document', () => { */ + public function testStripsNonParsableDocument(): void + { + self::assertStripped('{foo(arg:"str"', '{ foo(arg: "str"'); + } + + /** @see it('strips documents with only ignored characters', () => { */ + public function testStripsDocumentsWithOnlyIgnoredCharacters(): void + { + self::assertStripped('', "\n"); + self::assertStripped('', ','); + self::assertStripped('', ',,'); + self::assertStripped('', "#comment\n, \n"); + } + + /** @see it('strips leading and trailing ignored tokens', () => { */ + public function testStripsLeadingAndTrailingIgnoredTokens(): void + { + self::assertStripped('1', "\n1"); + self::assertStripped('1', ',1'); + self::assertStripped('1', ',,1'); + self::assertStripped('1', "#comment\n, \n1"); + + self::assertStripped('1', "1\n"); + self::assertStripped('1', '1,'); + self::assertStripped('1', '1,,'); + self::assertStripped('1', "1#comment\n, \n"); + } + + /** @see it('strips ignored tokens between punctuator tokens', () => { */ + public function testStripsIgnoredTokensBetweenPunctuatorTokens(): void + { + self::assertStripped('[)', '[,)'); + self::assertStripped('[)', "[\r)"); + self::assertStripped('[)', "[\r\r)"); + self::assertStripped('[)', "[\r,)"); + self::assertStripped('[)', "[,\n)"); + } + + /** @see it('strips ignored tokens between punctuator and non-punctuator tokens', () => { */ + public function testStripsIgnoredTokensBetweenPunctuatorAndNonPunctuatorTokens(): void + { + self::assertStripped('[1', '[,1'); + self::assertStripped('[1', "[\r1"); + self::assertStripped('[1', "[\r\r1"); + self::assertStripped('[1', "[\r,1"); + self::assertStripped('[1', "[,\n1"); + } + + /** @see it('strips ignored tokens between non-punctuator and punctuator tokens', () => { */ + public function testStripsIgnoredTokensBetweenNonPunctuatorAndPunctuatorTokens(): void + { + self::assertStripped('1[', '1,['); + self::assertStripped('1[', "1\r["); + self::assertStripped('1[', "1\r\r["); + self::assertStripped('1[', "1\r,["); + self::assertStripped('1[', "1,\n["); + } + + /** @see it('replace ignored tokens between non-punctuator tokens and spread with space', () => { */ + public function testReplaceIgnoredTokensBetweenNonPunctuatorTokensAndSpreadWithSpace(): void + { + self::assertStripped('a ...', 'a ...'); + self::assertStripped('1 ...', '1 ...'); + self::assertStripped('1 ......', '1 ... ...'); + } + + /** @see it('replace ignored tokens between non-punctuator tokens with space', () => { */ + public function testReplaceIgnoredTokensBetweenNonPunctuatorTokensWithSpace(): void + { + self::assertStaysTheSame('1 2'); + self::assertStaysTheSame('"" ""'); + self::assertStaysTheSame('a b'); + + self::assertStripped('a 1', 'a,1'); + self::assertStripped('a 1', 'a,,1'); + self::assertStripped('a 1', 'a 1'); + self::assertStripped('a 1', "a \t 1"); + } + + /** @see it('does not strip ignored tokens embedded in the string', () => { */ + public function testDoesNotStripIgnoredTokensEmbeddedInTheString(): void + { + self::assertStaysTheSame('" "'); + self::assertStaysTheSame('","'); + self::assertStaysTheSame('",,"'); + self::assertStaysTheSame('",|"'); + } + + /** @see it('does not strip ignored tokens embedded in the block string', () => { */ + public function testDoesNotStripIgnoredTokensEmbeddedInTheBlockString(): void + { + self::assertStaysTheSame('""","""'); + self::assertStaysTheSame('""",,"""'); + self::assertStaysTheSame('""",|"""'); + } + + /** @see it('strips ignored characters inside block strings', () => { */ + public function testStripsIgnoredCharactersInsideBlockStrings(): void + { + self::assertStrippedString('""""""', '""""""'); + self::assertStrippedString('""""""', '""" """'); + + self::assertStrippedString('"""a"""', '"""a"""'); + self::assertStrippedString('""" a"""', '""" a"""'); + self::assertStrippedString('""" a """', '""" a """'); + + self::assertStrippedString('""""""', "\"\"\"\n\"\"\""); + self::assertStrippedString("\"\"\"a\nb\"\"\"", "\"\"\"a\nb\"\"\""); + self::assertStrippedString("\"\"\"a\nb\"\"\"", "\"\"\"a\rb\"\"\""); + self::assertStrippedString("\"\"\"a\nb\"\"\"", "\"\"\"a\r\nb\"\"\""); + self::assertStrippedString("\"\"\"a\n\nb\"\"\"", "\"\"\"a\r\n\nb\"\"\""); + + self::assertStrippedString("\"\"\"\\\n\"\"\"", "\"\"\"\\\n\"\"\""); + self::assertStrippedString("\"\"\"\"\n\"\"\"", "\"\"\"\"\n\"\"\""); + self::assertStrippedString('"""\\""""""', "\"\"\"\\\"\"\"\n\"\"\""); + + self::assertStrippedString("\"\"\"\na\n b\"\"\"", "\"\"\"\na\n b\"\"\""); + self::assertStrippedString("\"\"\"a\nb\"\"\"", "\"\"\"\n a\n b\"\"\""); + self::assertStrippedString("\"\"\"a\n b\nc\"\"\"", "\"\"\"\na\n b\nc\"\"\""); + } + + public function testStripsNonASCIICharactersInStrings(): void + { + self::assertStripped('{a(b:"ä ö" c:""" ü """)}', '{ a(b: "ä ö", c: """ ü """) }'); + } + + /** @see it('strips kitchen sink query but maintains the exact same AST', () => { */ + public function testStripsKitchenSinkQueryButMaintainsTheExactSameAST(): void + { + $kitchenSinkQuery = file_get_contents(__DIR__ . '/kitchen-sink.graphql'); + + $strippedQuery = Printer::stripIgnoredCharacters($kitchenSinkQuery); + self::assertSame($strippedQuery, Printer::stripIgnoredCharacters($strippedQuery)); + + $queryAST = Parser::parse($kitchenSinkQuery, ['noLocation' => true]); + $strippedAST = Parser::parse($strippedQuery, ['noLocation' => true]); + self::assertEquals($queryAST, $strippedAST); + } + + /** @see it('strips kitchen sink SDL but maintains the exact same AST', () => { */ + public function testStripsKitchenSinkSDLButMaintainsTheExactSameAST(): void + { + $kitchenSinkSDL = file_get_contents(__DIR__ . '/schema-kitchen-sink.graphql'); + + $strippedSDL = Printer::stripIgnoredCharacters($kitchenSinkSDL); + self::assertSame($strippedSDL, Printer::stripIgnoredCharacters($strippedSDL)); + + $sdlAST = Parser::parse($kitchenSinkSDL, ['noLocation' => true]); + $strippedAST = Parser::parse($strippedSDL, ['noLocation' => true]); + self::assertEquals($sdlAST, $strippedAST); + } +} From 91bbca72716b421af6066729d220912e3b672642 Mon Sep 17 00:00:00 2001 From: Benedikt Franke Date: Sat, 3 Oct 2026 18:24:55 +0200 Subject: [PATCH 02/10] Add CHANGELOG entry MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 🤖 Generated with Claude Code --- CHANGELOG.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 6f37c1f21..86a229d09 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,10 @@ You can find and compare releases at the [GitHub release page](https://github.co ## Unreleased +### Added + +- Add `Printer::stripIgnoredCharacters()` to remove characters that do not change a document's meaning https://github.com/webonyx/graphql-php/pull/1985 + ### Fixed - Strip tab-only leading and trailing lines from block string values https://github.com/webonyx/graphql-php/pull/1984 From 0cb2d877cce4db6941e419bcf7600b1ae649e14c Mon Sep 17 00:00:00 2001 From: Benedikt Franke Date: Sat, 3 Oct 2026 18:29:11 +0200 Subject: [PATCH 03/10] Fix PHPStan on PHP 7.4 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 🤖 Generated with Claude Code --- phpstan/php-below-8.0.neon | 5 +++++ tests/Language/StripIgnoredCharactersFuzzTest.php | 14 ++++++-------- 2 files changed, 11 insertions(+), 8 deletions(-) diff --git a/phpstan/php-below-8.0.neon b/phpstan/php-below-8.0.neon index 133dfdd7b..eee1a2969 100644 --- a/phpstan/php-below-8.0.neon +++ b/phpstan/php-below-8.0.neon @@ -5,6 +5,11 @@ parameters: message: "#::chr#" count: 1 + - path: ../src/Language/Printer.php + identifier: argument.type + message: "#tokenSource\(\) expects string, string\|false given#" + count: 1 + # Native enums require PHP 8.1, but checking if a value is of an unknown class still works - path: ../src/Type/Definition/EnumType.php identifier: class.notFound diff --git a/tests/Language/StripIgnoredCharactersFuzzTest.php b/tests/Language/StripIgnoredCharactersFuzzTest.php index cc4498799..86bfcd716 100644 --- a/tests/Language/StripIgnoredCharactersFuzzTest.php +++ b/tests/Language/StripIgnoredCharactersFuzzTest.php @@ -9,8 +9,6 @@ use GraphQL\Language\Token; use PHPUnit\Framework\TestCase; -use function Safe\json_encode; - /** * @see describe('stripIgnoredCharacters', () => { */ @@ -72,10 +70,10 @@ private static function lexValue(string $str): ?string private static function assertStripped(string $expected, string $docString): void { $stripped = Printer::stripIgnoredCharacters($docString); - self::assertSame($expected, $stripped, 'Stripping ' . json_encode($docString)); + self::assertSame($expected, $stripped, 'Stripping ' . json_encode($docString, JSON_THROW_ON_ERROR)); $strippedTwice = Printer::stripIgnoredCharacters($stripped); - self::assertSame($stripped, $strippedTwice, 'Stripping twice ' . json_encode($stripped)); + self::assertSame($stripped, $strippedTwice, 'Stripping twice ' . json_encode($stripped, JSON_THROW_ON_ERROR)); } /** @@ -258,14 +256,14 @@ public function testReplaceRandomIgnoredTokensBetweenNonPunctuatorTokensWithSpac public function testDoesNotStripRandomIgnoredTokensEmbeddedInTheString(): void { foreach (self::IGNORED_TOKENS as $ignored) { - self::assertStaysTheSame(json_encode($ignored, JSON_UNESCAPED_UNICODE)); + self::assertStaysTheSame(json_encode($ignored, JSON_UNESCAPED_UNICODE | JSON_THROW_ON_ERROR)); foreach (self::IGNORED_TOKENS as $anotherIgnored) { - self::assertStaysTheSame(json_encode($ignored . $anotherIgnored, JSON_UNESCAPED_UNICODE)); + self::assertStaysTheSame(json_encode($ignored . $anotherIgnored, JSON_UNESCAPED_UNICODE | JSON_THROW_ON_ERROR)); } } - self::assertStaysTheSame(json_encode(implode('', self::IGNORED_TOKENS), JSON_UNESCAPED_UNICODE)); + self::assertStaysTheSame(json_encode(implode('', self::IGNORED_TOKENS), JSON_UNESCAPED_UNICODE | JSON_THROW_ON_ERROR)); } /** @see it('does not strip random ignored tokens embedded in the block string', () => { */ @@ -297,7 +295,7 @@ public function testStripsIgnoredCharactersInsideRandomBlockStrings(): void $testValue = self::lexValue($testStr); $strippedValue = self::lexValue(Printer::stripIgnoredCharacters($testStr)); - self::assertSame($testValue, $strippedValue, 'Stripping ' . json_encode($testStr)); + self::assertSame($testValue, $strippedValue, 'Stripping ' . json_encode($testStr, JSON_THROW_ON_ERROR)); } } } From eb1843fcff0b2b65a8a5fb1684ea8882058da37b Mon Sep 17 00:00:00 2001 From: Benedikt Franke Date: Sat, 3 Oct 2026 18:30:28 +0200 Subject: [PATCH 04/10] Fix NEON quoting in PHP 7.4 PHPStan config MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 🤖 Generated with Claude Code --- phpstan/php-below-8.0.neon | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/phpstan/php-below-8.0.neon b/phpstan/php-below-8.0.neon index eee1a2969..5e508d0b2 100644 --- a/phpstan/php-below-8.0.neon +++ b/phpstan/php-below-8.0.neon @@ -7,7 +7,7 @@ parameters: - path: ../src/Language/Printer.php identifier: argument.type - message: "#tokenSource\(\) expects string, string\|false given#" + message: '#tokenSource\(\) expects string, string\|false given#' count: 1 # Native enums require PHP 8.1, but checking if a value is of an unknown class still works From d965d70abbc913263b456e495b17d12bda3ff0e4 Mon Sep 17 00:00:00 2001 From: Benedikt Franke Date: Sat, 3 Oct 2026 18:52:45 +0200 Subject: [PATCH 05/10] Explain split between example and fuzz tests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 🤖 Generated with Claude Code --- tests/Language/StripIgnoredCharactersFuzzTest.php | 2 ++ tests/Language/StripIgnoredCharactersTest.php | 2 ++ 2 files changed, 4 insertions(+) diff --git a/tests/Language/StripIgnoredCharactersFuzzTest.php b/tests/Language/StripIgnoredCharactersFuzzTest.php index 86bfcd716..a63983081 100644 --- a/tests/Language/StripIgnoredCharactersFuzzTest.php +++ b/tests/Language/StripIgnoredCharactersFuzzTest.php @@ -10,6 +10,8 @@ use PHPUnit\Framework\TestCase; /** + * Generated input combinations, split out like stripIgnoredCharacters-fuzz.ts in graphql-js. + * * @see describe('stripIgnoredCharacters', () => { */ final class StripIgnoredCharactersFuzzTest extends TestCase diff --git a/tests/Language/StripIgnoredCharactersTest.php b/tests/Language/StripIgnoredCharactersTest.php index 24255ec17..d4b1e396b 100644 --- a/tests/Language/StripIgnoredCharactersTest.php +++ b/tests/Language/StripIgnoredCharactersTest.php @@ -14,6 +14,8 @@ use function Safe\file_get_contents; /** + * Hand-picked cases, generated ones are in StripIgnoredCharactersFuzzTest. + * * @see describe('stripIgnoredCharacters', () => { */ final class StripIgnoredCharactersTest extends TestCase From a0d6d9bd2c7254f92b037c78c527016e1fd9536a Mon Sep 17 00:00:00 2001 From: Benedikt Franke Date: Sat, 3 Oct 2026 19:16:48 +0200 Subject: [PATCH 06/10] Reject invalid UTF-8 in documents MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Since the bulk string scan from https://github.com/webonyx/graphql-php/pull/1948, the Lexer accepts invalid UTF-8 inside strings and keeps the raw bytes. Comments and block strings swallow up to three bytes after an invalid lead byte. The spec only allows Unicode scalar values as SourceCharacter. 🤖 Generated with Claude Code --- src/Language/Lexer.php | 22 ++++++++++++++++++++++ tests/Language/LexerTest.php | 20 ++++++++++++++++++++ 2 files changed, 42 insertions(+) diff --git a/src/Language/Lexer.php b/src/Language/Lexer.php index c5b8711a3..90b0bc928 100644 --- a/src/Language/Lexer.php +++ b/src/Language/Lexer.php @@ -46,6 +46,9 @@ class Lexer /** Ignored single-byte characters that never start or end a line, so runs of them can be skipped in bulk. */ private const HORIZONTAL_WHITESPACE_BYTES = "\t ,"; + /** Matches the longest valid UTF-8 prefix, see https://www.rfc-editor.org/rfc/rfc3629#section-4. */ + private const VALID_UTF8_PREFIX = '/\A(?:[\x00-\x7F]|[\xC2-\xDF][\x80-\xBF]|\xE0[\xA0-\xBF][\x80-\xBF]|[\xE1-\xEC\xEE\xEF][\x80-\xBF]{2}|\xED[\x80-\x9F][\x80-\xBF]|\xF0[\x90-\xBF][\x80-\xBF]{2}|[\xF1-\xF3][\x80-\xBF]{3}|\xF4[\x80-\x8F][\x80-\xBF]{2})*+/'; + public Source $source; /** @phpstan-var ParserOptions */ @@ -113,6 +116,10 @@ public function lookahead(): Token */ private function readToken(Token $prev): Token { + if ($prev->kind === Token::SOF) { + $this->assertValidUTF8(); + } + $bodyLength = $this->source->length; $this->positionAfterWhitespace(); @@ -259,6 +266,21 @@ private function readToken(Token $prev): Token throw new SyntaxError($this->source, $position, $this->unexpectedCharacterMessage($code)); } + /** @throws SyntaxError */ + private function assertValidUTF8(): void + { + $body = $this->source->body; + if (mb_check_encoding($body, 'UTF-8')) { + return; + } + + preg_match(self::VALID_UTF8_PREFIX, $body, $matches); + $validPrefix = $matches[0] ?? ''; + $invalidByte = strtoupper(bin2hex($body[strlen($validPrefix)])); + + throw new SyntaxError($this->source, mb_strlen($validPrefix, 'UTF-8'), "Invalid UTF-8 byte: 0x{$invalidByte}"); + } + /** @throws \JsonException */ private function unexpectedCharacterMessage(?int $code): string { diff --git a/tests/Language/LexerTest.php b/tests/Language/LexerTest.php index 313abdf45..a483c253b 100644 --- a/tests/Language/LexerTest.php +++ b/tests/Language/LexerTest.php @@ -674,6 +674,26 @@ public function testReportsUsefulUnknownCharErrors(string $str, string $expected $this->expectSyntaxError($str, $expectedMessage, $location); } + /** @return iterable */ + public static function reportsInvalidUTF8(): iterable + { + yield 'in name position' => ["\xFF", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 1)]; + yield 'in string' => ["\"a\xFFbcd\"", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 3)]; + yield 'after multibyte character' => ["\"ä\xC3x\"", 'Invalid UTF-8 byte: 0xC3', self::loc(1, 3)]; + yield 'truncated at end' => ["\"ab\xE2\x82", 'Invalid UTF-8 byte: 0xE2', self::loc(1, 4)]; + yield 'encoded surrogate' => ["\"\xED\xA0\x80\"", 'Invalid UTF-8 byte: 0xED', self::loc(1, 2)]; + yield 'overlong encoding' => ["\"\xC0\xAF\"", 'Invalid UTF-8 byte: 0xC0', self::loc(1, 2)]; + yield 'in block string' => ["\"\"\"\xFF\"\"\"", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 4)]; + yield 'in comment' => ["# \xFF\nfoo", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 3)]; + yield 'after the first token' => ["foo\n \"\xFF\"", 'Invalid UTF-8 byte: 0xFF', self::loc(2, 4)]; + } + + /** @dataProvider reportsInvalidUTF8 */ + public function testReportsInvalidUTF8(string $str, string $expectedMessage, SourceLocation $location): void + { + $this->expectSyntaxError($str, $expectedMessage, $location); + } + /** @see it('lex reports useful information for dashes in names') */ public function testReportsUsefulDashesInfo(): void { From 90df90b3755044a374409bd56e8eec7447192d76 Mon Sep 17 00:00:00 2001 From: Benedikt Franke Date: Sat, 3 Oct 2026 19:17:00 +0200 Subject: [PATCH 07/10] Add CHANGELOG entry MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 🤖 Generated with Claude Code --- CHANGELOG.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 74ccea04c..1973fcc35 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,10 @@ You can find and compare releases at the [GitHub release page](https://github.co ## Unreleased +### Fixed + +- Reject invalid UTF-8 in documents https://github.com/webonyx/graphql-php/pull/1987 + ## v15.37.3 ### Fixed From 5f05d32c7b53c22c29ff65c7d6cd9cd71e0d2e2c Mon Sep 17 00:00:00 2001 From: Benedikt Franke Date: Sat, 3 Oct 2026 19:17:15 +0200 Subject: [PATCH 08/10] Slice string tokens by byte position instead of a UTF-32 copy MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Token positions count characters the way Lexer::readChar() does, so stepping by lead byte finds the byte offsets without converting the whole body. Relies on https://github.com/webonyx/graphql-php/pull/1987 rejecting invalid UTF-8. 🤖 Generated with Claude Code --- phpstan/php-below-8.0.neon | 5 --- src/Language/Printer.php | 41 +++++++++++++++---- tests/Language/StripIgnoredCharactersTest.php | 7 ++++ 3 files changed, 41 insertions(+), 12 deletions(-) diff --git a/phpstan/php-below-8.0.neon b/phpstan/php-below-8.0.neon index 5e508d0b2..133dfdd7b 100644 --- a/phpstan/php-below-8.0.neon +++ b/phpstan/php-below-8.0.neon @@ -5,11 +5,6 @@ parameters: message: "#::chr#" count: 1 - - path: ../src/Language/Printer.php - identifier: argument.type - message: '#tokenSource\(\) expects string, string\|false given#' - count: 1 - # Native enums require PHP 8.1, but checking if a value is of an unknown class still works - path: ../src/Type/Definition/EnumType.php identifier: class.notFound diff --git a/src/Language/Printer.php b/src/Language/Printer.php index 889ef4936..0b71e3fd6 100644 --- a/src/Language/Printer.php +++ b/src/Language/Printer.php @@ -122,11 +122,13 @@ public static function stripIgnoredCharacters($source): string $sourceObj = $source instanceof Source ? $source : new Source($source); + $body = $sourceObj->body; $lexer = new Lexer($sourceObj); - $utf32Body = mb_convert_encoding($sourceObj->body, 'UTF-32', 'UTF-8'); $stripped = ''; $wasLastAddedTokenNonPunctuator = false; + $charPosition = 0; + $bytePosition = 0; while (($token = $lexer->advance())->kind !== Token::EOF) { $isNonPunctuator = ! static::isPunctuatorTokenKind($token->kind); @@ -135,7 +137,17 @@ public static function stripIgnoredCharacters($source): string $stripped .= ' '; } - $stripped .= static::tokenSource($utf32Body, $token); + if ($token->kind === Token::STRING) { + $bytePosition = static::advanceBytePosition($body, $bytePosition, $token->start - $charPosition); + $tokenBytePosition = static::advanceBytePosition($body, $bytePosition, $token->end - $token->start); + $stripped .= substr($body, $bytePosition, $tokenBytePosition - $bytePosition); + + $charPosition = $token->end; + $bytePosition = $tokenBytePosition; + } else { + $stripped .= static::tokenSource($token); + } + $wasLastAddedTokenNonPunctuator = $isNonPunctuator; } @@ -147,17 +159,32 @@ protected static function isPunctuatorTokenKind(string $kind): bool return ! in_array($kind, [Token::NAME, Token::INT, Token::FLOAT, Token::STRING, Token::BLOCK_STRING], true); } - protected static function tokenSource(string $utf32Body, Token $token): string + /** Steps through characters the same way as Lexer::readChar(), since token positions count them. */ + protected static function advanceBytePosition(string $body, int $bytePosition, int $charCount): int + { + for ($i = 0; $i < $charCount; ++$i) { + $leadByte = ord($body[$bytePosition]); + if ($leadByte < 128) { + ++$bytePosition; + } elseif ($leadByte < 224) { + $bytePosition += 2; + } elseif ($leadByte < 240) { + $bytePosition += 3; + } else { + $bytePosition += 4; + } + } + + return $bytePosition; + } + + protected static function tokenSource(Token $token): string { switch ($token->kind) { case Token::BLOCK_STRING: assert(is_string($token->value)); return BlockString::print($token->value, true); - case Token::STRING: - $utf32Token = substr($utf32Body, $token->start * 4, ($token->end - $token->start) * 4); - - return mb_convert_encoding($utf32Token, 'UTF-8', 'UTF-32'); case Token::NAME: case Token::INT: case Token::FLOAT: diff --git a/tests/Language/StripIgnoredCharactersTest.php b/tests/Language/StripIgnoredCharactersTest.php index d4b1e396b..0e9318936 100644 --- a/tests/Language/StripIgnoredCharactersTest.php +++ b/tests/Language/StripIgnoredCharactersTest.php @@ -266,6 +266,13 @@ public function testStripsNonASCIICharactersInStrings(): void self::assertStripped('{a(b:"ä ö" c:""" ü """)}', '{ a(b: "ä ö", c: """ ü """) }'); } + public function testRejectsInvalidUTF8(): void + { + $this->expectException(SyntaxError::class); + $this->expectExceptionMessage('Invalid UTF-8 byte: 0xFF'); + Printer::stripIgnoredCharacters("{ a(b: \"\xFFabc\") }"); + } + /** @see it('strips kitchen sink query but maintains the exact same AST', () => { */ public function testStripsKitchenSinkQueryButMaintainsTheExactSameAST(): void { From 85a1dce1d970bddc8cfa01382f98ac843e217f94 Mon Sep 17 00:00:00 2001 From: Benedikt Franke Date: Sat, 3 Oct 2026 19:33:27 +0200 Subject: [PATCH 09/10] Find the invalid byte without a regex MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The prefix regex hit pcre.backtrack_limit on about 1 million 3- or 4-byte characters and then reported the first byte of the document. mb_scrub() replaces only invalid bytes, so the first differing byte is the invalid one. 🤖 Generated with Claude Code --- src/Language/Lexer.php | 10 ++++------ tests/Language/LexerTest.php | 3 +++ 2 files changed, 7 insertions(+), 6 deletions(-) diff --git a/src/Language/Lexer.php b/src/Language/Lexer.php index 90b0bc928..841132271 100644 --- a/src/Language/Lexer.php +++ b/src/Language/Lexer.php @@ -46,9 +46,6 @@ class Lexer /** Ignored single-byte characters that never start or end a line, so runs of them can be skipped in bulk. */ private const HORIZONTAL_WHITESPACE_BYTES = "\t ,"; - /** Matches the longest valid UTF-8 prefix, see https://www.rfc-editor.org/rfc/rfc3629#section-4. */ - private const VALID_UTF8_PREFIX = '/\A(?:[\x00-\x7F]|[\xC2-\xDF][\x80-\xBF]|\xE0[\xA0-\xBF][\x80-\xBF]|[\xE1-\xEC\xEE\xEF][\x80-\xBF]{2}|\xED[\x80-\x9F][\x80-\xBF]|\xF0[\x90-\xBF][\x80-\xBF]{2}|[\xF1-\xF3][\x80-\xBF]{3}|\xF4[\x80-\x8F][\x80-\xBF]{2})*+/'; - public Source $source; /** @phpstan-var ParserOptions */ @@ -274,9 +271,10 @@ private function assertValidUTF8(): void return; } - preg_match(self::VALID_UTF8_PREFIX, $body, $matches); - $validPrefix = $matches[0] ?? ''; - $invalidByte = strtoupper(bin2hex($body[strlen($validPrefix)])); + $scrubbedBody = mb_scrub($body, 'UTF-8'); + $invalidByteOffset = strspn($body ^ $scrubbedBody, "\0"); + $invalidByte = strtoupper(bin2hex($body[$invalidByteOffset])); + $validPrefix = substr($body, 0, $invalidByteOffset); throw new SyntaxError($this->source, mb_strlen($validPrefix, 'UTF-8'), "Invalid UTF-8 byte: 0x{$invalidByte}"); } diff --git a/tests/Language/LexerTest.php b/tests/Language/LexerTest.php index a483c253b..b1fc0ddc6 100644 --- a/tests/Language/LexerTest.php +++ b/tests/Language/LexerTest.php @@ -686,6 +686,9 @@ public static function reportsInvalidUTF8(): iterable yield 'in block string' => ["\"\"\"\xFF\"\"\"", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 4)]; yield 'in comment' => ["# \xFF\nfoo", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 3)]; yield 'after the first token' => ["foo\n \"\xFF\"", 'Invalid UTF-8 byte: 0xFF', self::loc(2, 4)]; + yield 'after BOM' => ["\u{FEFF}\xFF", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 2)]; + yield 'first of two' => ["\"a\xFEb\xFF\"", 'Invalid UTF-8 byte: 0xFE', self::loc(1, 3)]; + yield 'after long multibyte prefix' => ['"' . str_repeat('€', 2000) . "\xFF\"", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 2002)]; } /** @dataProvider reportsInvalidUTF8 */ From 8726026d64e223959c9b51e22033ca7267e34414 Mon Sep 17 00:00:00 2001 From: Benedikt Franke Date: Sat, 3 Oct 2026 19:35:49 +0200 Subject: [PATCH 10/10] Address detail review MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Test multibyte characters before string tokens, including a 4-byte character followed by a 3-byte one, which is the shortest input that catches a wrong 4-byte step. Add assert messages, fix the remainder typo carried over from graphql-js and tighten the fuzz length comment. 🤖 Generated with Claude Code --- src/Language/Printer.php | 4 ++-- tests/Language/StripIgnoredCharactersFuzzTest.php | 8 ++++---- tests/Language/StripIgnoredCharactersTest.php | 3 +++ 3 files changed, 9 insertions(+), 6 deletions(-) diff --git a/src/Language/Printer.php b/src/Language/Printer.php index 0b71e3fd6..dba2da19a 100644 --- a/src/Language/Printer.php +++ b/src/Language/Printer.php @@ -182,13 +182,13 @@ protected static function tokenSource(Token $token): string { switch ($token->kind) { case Token::BLOCK_STRING: - assert(is_string($token->value)); + assert(is_string($token->value), 'Lexer sets the dedented value of block strings'); return BlockString::print($token->value, true); case Token::NAME: case Token::INT: case Token::FLOAT: - assert(is_string($token->value)); + assert(is_string($token->value), 'Lexer sets the raw value of names and numbers'); return $token->value; default: diff --git a/tests/Language/StripIgnoredCharactersFuzzTest.php b/tests/Language/StripIgnoredCharactersFuzzTest.php index a63983081..52ac688d9 100644 --- a/tests/Language/StripIgnoredCharactersFuzzTest.php +++ b/tests/Language/StripIgnoredCharactersFuzzTest.php @@ -120,9 +120,9 @@ private static function genFuzzStrings(array $allowedChars, int $maxLength): \Ge $leftOver = $combination; while ($leftOver >= 0) { - $reminder = $leftOver % $numAllowedChars; - $permutation = $allowedChars[$reminder] . $permutation; - $leftOver = intdiv($leftOver - $reminder, $numAllowedChars) - 1; + $remainder = $leftOver % $numAllowedChars; + $permutation = $allowedChars[$remainder] . $permutation; + $leftOver = intdiv($leftOver - $remainder, $numAllowedChars) - 1; } yield $permutation; @@ -286,7 +286,7 @@ public function testDoesNotStripRandomIgnoredTokensEmbeddedInTheBlockString(): v /** @see it('strips ignored characters inside random block strings', () => { */ public function testStripsIgnoredCharactersInsideRandomBlockStrings(): void { - // Lengths above 7 take exponentially longer, but test with them when changing the implementation + // Increase when changing the implementation, lengths above 7 are exponentially slower foreach (self::genFuzzStrings(["\n", "\t", ' ', '"', 'a', '\\'], 7) as $fuzzStr) { $testStr = '"""' . $fuzzStr . '"""'; diff --git a/tests/Language/StripIgnoredCharactersTest.php b/tests/Language/StripIgnoredCharactersTest.php index 0e9318936..040c20879 100644 --- a/tests/Language/StripIgnoredCharactersTest.php +++ b/tests/Language/StripIgnoredCharactersTest.php @@ -264,6 +264,9 @@ public function testStripsIgnoredCharactersInsideBlockStrings(): void public function testStripsNonASCIICharactersInStrings(): void { self::assertStripped('{a(b:"ä ö" c:""" ü """)}', '{ a(b: "ä ö", c: """ ü """) }'); + self::assertStripped('{a(b:"😀€" c:"x")}', '{ a(b: "😀€", c: "x") }'); + self::assertStripped('{a(b:"😀x" c:"€y")}', "# €😀\n{ a(b: \"😀x\", c: \"€y\") }"); + self::assertStripped('"ö" "ä"', "\u{FEFF}# ü\n\"ö\" \"ä\""); } public function testRejectsInvalidUTF8(): void