diff --git a/CHANGELOG.md b/CHANGELOG.md index 74ccea04c..d3f63cee6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,14 @@ You can find and compare releases at the [GitHub release page](https://github.co ## Unreleased +### Added + +- Add `Printer::stripIgnoredCharacters()` to remove characters that do not change a document's meaning https://github.com/webonyx/graphql-php/pull/1985 + +### Fixed + +- Reject invalid UTF-8 in documents https://github.com/webonyx/graphql-php/pull/1987 + ## v15.37.3 ### Fixed diff --git a/docs/class-reference.md b/docs/class-reference.md index 81cd48e81..c886b99b2 100644 --- a/docs/class-reference.md +++ b/docs/class-reference.md @@ -1386,6 +1386,48 @@ $printed = GraphQL\Language\Printer::doPrint($ast); static function doPrint(GraphQL\Language\AST\Node $ast): string ``` +````php +/** + * Strips characters that are not significant to the validity or execution of a GraphQL document: + * - UnicodeBOM + * - WhiteSpace + * - LineTerminator + * - Comment + * - Comma + * - BlockString indentation + * + * Neighboring non-punctuator tokens are always delimited by a single space. + * Parsing input and output yields the same AST, apart from node locations. + * The output is stable, but may change between releases. + * + * ```graphql + * query SomeQuery($foo: String!, $bar: String) { + * someField(foo: $foo, bar: $bar) { + * a + * b { + * c + * d + * } + * } + * } + * ``` + * + * becomes + * + * ```graphql + * query SomeQuery($foo:String!$bar:String){someField(foo:$foo bar:$bar){a b{c d}}} + * ``` + * + * @param Source|string $source + * + * @throws \JsonException + * @throws SyntaxError + * + * @api + */ +static function stripIgnoredCharacters($source): string +```` + ## GraphQL\Language\Visitor Utility for efficient AST traversal and modification. diff --git a/src/Language/BlockString.php b/src/Language/BlockString.php index 76a9bd0e7..5cecdcbdb 100644 --- a/src/Language/BlockString.php +++ b/src/Language/BlockString.php @@ -102,7 +102,7 @@ public static function getIndentation(string $value): int * trailing blank line. However, if a block string starts with whitespace and is * a single-line, adding a leading blank line would strip that whitespace. */ - public static function print(string $value): string + public static function print(string $value, bool $minimize = false): string { $escapedValue = str_replace('"""', '\\"""', $value); @@ -131,11 +131,14 @@ public static function print(string $value): string $forceTrailingNewline = $hasTrailingQuote || $hasTrailingSlash; // add leading and trailing new lines only if it improves readability - $printAsMultipleLines = ! $isSingleLine - || mb_strlen($value) > 70 - || $forceTrailingNewline - || $forceLeadingNewLine - || $hasTrailingTripleQuotes; + $printAsMultipleLines = ! $minimize + && ( + ! $isSingleLine + || mb_strlen($value) > 70 + || $forceTrailingNewline + || $forceLeadingNewLine + || $hasTrailingTripleQuotes + ); $result = ''; @@ -146,7 +149,7 @@ public static function print(string $value): string } $result .= $escapedValue; - if ($printAsMultipleLines) { + if ($printAsMultipleLines || $forceTrailingNewline) { $result .= "\n"; } diff --git a/src/Language/Lexer.php b/src/Language/Lexer.php index c5b8711a3..841132271 100644 --- a/src/Language/Lexer.php +++ b/src/Language/Lexer.php @@ -113,6 +113,10 @@ public function lookahead(): Token */ private function readToken(Token $prev): Token { + if ($prev->kind === Token::SOF) { + $this->assertValidUTF8(); + } + $bodyLength = $this->source->length; $this->positionAfterWhitespace(); @@ -259,6 +263,22 @@ private function readToken(Token $prev): Token throw new SyntaxError($this->source, $position, $this->unexpectedCharacterMessage($code)); } + /** @throws SyntaxError */ + private function assertValidUTF8(): void + { + $body = $this->source->body; + if (mb_check_encoding($body, 'UTF-8')) { + return; + } + + $scrubbedBody = mb_scrub($body, 'UTF-8'); + $invalidByteOffset = strspn($body ^ $scrubbedBody, "\0"); + $invalidByte = strtoupper(bin2hex($body[$invalidByteOffset])); + $validPrefix = substr($body, 0, $invalidByteOffset); + + throw new SyntaxError($this->source, mb_strlen($validPrefix, 'UTF-8'), "Invalid UTF-8 byte: 0x{$invalidByte}"); + } + /** @throws \JsonException */ private function unexpectedCharacterMessage(?int $code): string { diff --git a/src/Language/Printer.php b/src/Language/Printer.php index bc061ad0f..dba2da19a 100644 --- a/src/Language/Printer.php +++ b/src/Language/Printer.php @@ -2,6 +2,7 @@ namespace GraphQL\Language; +use GraphQL\Error\SyntaxError; use GraphQL\Language\AST\ArgumentNode; use GraphQL\Language\AST\BooleanValueNode; use GraphQL\Language\AST\DirectiveDefinitionNode; @@ -78,6 +79,123 @@ public static function doPrint(Node $ast): string return static::p($ast); } + /** + * Strips characters that are not significant to the validity or execution of a GraphQL document: + * - UnicodeBOM + * - WhiteSpace + * - LineTerminator + * - Comment + * - Comma + * - BlockString indentation + * + * Neighboring non-punctuator tokens are always delimited by a single space. + * Parsing input and output yields the same AST, apart from node locations. + * The output is stable, but may change between releases. + * + * ```graphql + * query SomeQuery($foo: String!, $bar: String) { + * someField(foo: $foo, bar: $bar) { + * a + * b { + * c + * d + * } + * } + * } + * ``` + * + * becomes + * + * ```graphql + * query SomeQuery($foo:String!$bar:String){someField(foo:$foo bar:$bar){a b{c d}}} + * ``` + * + * @param Source|string $source + * + * @throws \JsonException + * @throws SyntaxError + * + * @api + */ + public static function stripIgnoredCharacters($source): string + { + $sourceObj = $source instanceof Source + ? $source + : new Source($source); + $body = $sourceObj->body; + $lexer = new Lexer($sourceObj); + + $stripped = ''; + $wasLastAddedTokenNonPunctuator = false; + $charPosition = 0; + $bytePosition = 0; + while (($token = $lexer->advance())->kind !== Token::EOF) { + $isNonPunctuator = ! static::isPunctuatorTokenKind($token->kind); + + // `1...` would lex as an invalid float + if ($wasLastAddedTokenNonPunctuator && ($isNonPunctuator || $token->kind === Token::SPREAD)) { + $stripped .= ' '; + } + + if ($token->kind === Token::STRING) { + $bytePosition = static::advanceBytePosition($body, $bytePosition, $token->start - $charPosition); + $tokenBytePosition = static::advanceBytePosition($body, $bytePosition, $token->end - $token->start); + $stripped .= substr($body, $bytePosition, $tokenBytePosition - $bytePosition); + + $charPosition = $token->end; + $bytePosition = $tokenBytePosition; + } else { + $stripped .= static::tokenSource($token); + } + + $wasLastAddedTokenNonPunctuator = $isNonPunctuator; + } + + return $stripped; + } + + protected static function isPunctuatorTokenKind(string $kind): bool + { + return ! in_array($kind, [Token::NAME, Token::INT, Token::FLOAT, Token::STRING, Token::BLOCK_STRING], true); + } + + /** Steps through characters the same way as Lexer::readChar(), since token positions count them. */ + protected static function advanceBytePosition(string $body, int $bytePosition, int $charCount): int + { + for ($i = 0; $i < $charCount; ++$i) { + $leadByte = ord($body[$bytePosition]); + if ($leadByte < 128) { + ++$bytePosition; + } elseif ($leadByte < 224) { + $bytePosition += 2; + } elseif ($leadByte < 240) { + $bytePosition += 3; + } else { + $bytePosition += 4; + } + } + + return $bytePosition; + } + + protected static function tokenSource(Token $token): string + { + switch ($token->kind) { + case Token::BLOCK_STRING: + assert(is_string($token->value), 'Lexer sets the dedented value of block strings'); + + return BlockString::print($token->value, true); + case Token::NAME: + case Token::INT: + case Token::FLOAT: + assert(is_string($token->value), 'Lexer sets the raw value of names and numbers'); + + return $token->value; + default: + return $token->kind; + } + } + /** @throws \JsonException */ protected static function p(?Node $node): string { diff --git a/tests/Language/BlockStringTest.php b/tests/Language/BlockStringTest.php index b1d9d74df..ccef6464f 100644 --- a/tests/Language/BlockStringTest.php +++ b/tests/Language/BlockStringTest.php @@ -331,6 +331,7 @@ public function testDoNotEscapeCharacters(): void EOF, BlockString::print($str) ); + self::assertSame("\"\"\"\n{$str}\"\"\"", BlockString::print($str, true)); } /** @see it('by default print block strings as single line', () => { */ @@ -338,6 +339,7 @@ public function testByDefaultPrintBlockStringsAsSingleLine(): void { $str = 'one liner'; self::assertSame('"""one liner"""', BlockString::print($str)); + self::assertSame('"""one liner"""', BlockString::print($str, true)); } /** @see it('by default print block strings ending with triple quotation as multi-line', () => { */ @@ -352,6 +354,7 @@ public function testByDefaultPrintBlockStringsEndingWithTripleQuotationAsMultiLi EOF, BlockString::print($str) ); + self::assertSame('"""triple quotation \\""""""', BlockString::print($str, true)); } /** @see it('correctly prints single-line with leading space') */ @@ -359,6 +362,7 @@ public function testCorrectlyPrintsSingleLineWithLeadingSpace(): void { $str = ' space-led string'; self::assertSame('""" space-led string"""', BlockString::print($str)); + self::assertSame('""" space-led string"""', BlockString::print($str, true)); } /** @see it('correctly prints single-line with leading space and trailing quotation', () => { */ @@ -372,6 +376,7 @@ public function testCorrectlyPrintsSingleLineWithLeadingSpaceAndQuotation(): voi EOF, BlockString::print($str) ); + self::assertSame("\"\"\" space-led value \"quoted string\"\n\"\"\"", BlockString::print($str, true)); } /** @see it('correctly prints single-line with trailing backslash') */ @@ -386,6 +391,7 @@ public function testCorrectlyPrintsSingleLineWithTrailingBackslash(): void EOF, BlockString::print($str) ); + self::assertSame("\"\"\"backslash \\\n\"\"\"", BlockString::print($str, true)); } /** @see it('correctly prints multi-line with internal indent', () => { */ @@ -404,6 +410,7 @@ public function testCorrectlyPrintsMultiLineWithInternalIndent(): void EOF, BlockString::print($str) ); + self::assertSame("\"\"\"\nno indent\n with indent\"\"\"", BlockString::print($str, true)); } /** @see it('correctly prints string with a first line indentation') */ @@ -426,11 +433,13 @@ public function testCorrectlyPrintsStringWithAFirstLineIndentation(): void EOF, BlockString::print($str) ); + self::assertSame("\"\"\"{$str}\"\"\"", BlockString::print($str, true)); } public function testCorrectlyPrintsEmptyString(): void { $str = ''; self::assertSame('""""""', BlockString::print($str)); + self::assertSame('""""""', BlockString::print($str, true)); } } diff --git a/tests/Language/LexerTest.php b/tests/Language/LexerTest.php index 313abdf45..b1fc0ddc6 100644 --- a/tests/Language/LexerTest.php +++ b/tests/Language/LexerTest.php @@ -674,6 +674,29 @@ public function testReportsUsefulUnknownCharErrors(string $str, string $expected $this->expectSyntaxError($str, $expectedMessage, $location); } + /** @return iterable */ + public static function reportsInvalidUTF8(): iterable + { + yield 'in name position' => ["\xFF", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 1)]; + yield 'in string' => ["\"a\xFFbcd\"", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 3)]; + yield 'after multibyte character' => ["\"ä\xC3x\"", 'Invalid UTF-8 byte: 0xC3', self::loc(1, 3)]; + yield 'truncated at end' => ["\"ab\xE2\x82", 'Invalid UTF-8 byte: 0xE2', self::loc(1, 4)]; + yield 'encoded surrogate' => ["\"\xED\xA0\x80\"", 'Invalid UTF-8 byte: 0xED', self::loc(1, 2)]; + yield 'overlong encoding' => ["\"\xC0\xAF\"", 'Invalid UTF-8 byte: 0xC0', self::loc(1, 2)]; + yield 'in block string' => ["\"\"\"\xFF\"\"\"", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 4)]; + yield 'in comment' => ["# \xFF\nfoo", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 3)]; + yield 'after the first token' => ["foo\n \"\xFF\"", 'Invalid UTF-8 byte: 0xFF', self::loc(2, 4)]; + yield 'after BOM' => ["\u{FEFF}\xFF", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 2)]; + yield 'first of two' => ["\"a\xFEb\xFF\"", 'Invalid UTF-8 byte: 0xFE', self::loc(1, 3)]; + yield 'after long multibyte prefix' => ['"' . str_repeat('€', 2000) . "\xFF\"", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 2002)]; + } + + /** @dataProvider reportsInvalidUTF8 */ + public function testReportsInvalidUTF8(string $str, string $expectedMessage, SourceLocation $location): void + { + $this->expectSyntaxError($str, $expectedMessage, $location); + } + /** @see it('lex reports useful information for dashes in names') */ public function testReportsUsefulDashesInfo(): void { diff --git a/tests/Language/StripIgnoredCharactersFuzzTest.php b/tests/Language/StripIgnoredCharactersFuzzTest.php new file mode 100644 index 000000000..52ac688d9 --- /dev/null +++ b/tests/Language/StripIgnoredCharactersFuzzTest.php @@ -0,0 +1,303 @@ + { + */ +final class StripIgnoredCharactersFuzzTest extends TestCase +{ + private const IGNORED_TOKENS = [ + "\u{FEFF}", + "\t", + ' ', + "\n", + "\r", + "\r\n", + "# \"Comment\" string\n", + ',', + ]; + + private const PUNCTUATOR_TOKENS = [ + '!', + '$', + '(', + ')', + '...', + ':', + '=', + '@', + '[', + ']', + '{', + '|', + '}', + ]; + + private const NON_PUNCTUATOR_TOKENS = [ + 'name_token', + '1', + '3.14', + '"some string value"', + "\"\"\"block\nstring\nvalue\"\"\"", + ]; + + /** + * @throws \JsonException + * @throws SyntaxError + */ + private static function lexValue(string $str): ?string + { + $lexer = new Lexer(new Source($str)); + $value = $lexer->advance()->value; + + self::assertSame(Token::EOF, $lexer->advance()->kind, 'Expected EOF'); + + return $value; + } + + /** + * @throws \JsonException + * @throws SyntaxError + */ + private static function assertStripped(string $expected, string $docString): void + { + $stripped = Printer::stripIgnoredCharacters($docString); + self::assertSame($expected, $stripped, 'Stripping ' . json_encode($docString, JSON_THROW_ON_ERROR)); + + $strippedTwice = Printer::stripIgnoredCharacters($stripped); + self::assertSame($stripped, $strippedTwice, 'Stripping twice ' . json_encode($stripped, JSON_THROW_ON_ERROR)); + } + + /** + * @throws \JsonException + * @throws SyntaxError + */ + private static function assertStaysTheSame(string $docString): void + { + self::assertStripped($docString, $docString); + } + + /** @throws \JsonException */ + private static function isSingleToken(string $str): bool + { + $lexer = new Lexer(new Source($str)); + try { + $lexer->advance(); + + return $lexer->advance()->kind === Token::EOF; + } catch (SyntaxError $invalidToken) { + return false; + } + } + + /** + * @param list $allowedChars + * + * @return \Generator + */ + private static function genFuzzStrings(array $allowedChars, int $maxLength): \Generator + { + $numAllowedChars = count($allowedChars); + + $numCombinations = 0; + for ($length = 1; $length <= $maxLength; ++$length) { + $numCombinations += $numAllowedChars ** $length; + } + + yield ''; + for ($combination = 0; $combination < $numCombinations; ++$combination) { + $permutation = ''; + + $leftOver = $combination; + while ($leftOver >= 0) { + $remainder = $leftOver % $numAllowedChars; + $permutation = $allowedChars[$remainder] . $permutation; + $leftOver = intdiv($leftOver - $remainder, $numAllowedChars) - 1; + } + + yield $permutation; + } + } + + /** @see it('strips documents with random combination of ignored characters', () => { */ + public function testStripsDocumentsWithRandomCombinationOfIgnoredCharacters(): void + { + foreach (self::IGNORED_TOKENS as $ignored) { + self::assertStripped('', $ignored); + + foreach (self::IGNORED_TOKENS as $anotherIgnored) { + self::assertStripped('', $ignored . $anotherIgnored); + } + } + self::assertStripped('', implode('', self::IGNORED_TOKENS)); + } + + /** @see it('strips random leading and trailing ignored tokens', () => { */ + public function testStripsRandomLeadingAndTrailingIgnoredTokens(): void + { + foreach ([...self::PUNCTUATOR_TOKENS, ...self::NON_PUNCTUATOR_TOKENS] as $token) { + foreach (self::IGNORED_TOKENS as $ignored) { + self::assertStripped($token, $ignored . $token); + self::assertStripped($token, $token . $ignored); + + foreach (self::IGNORED_TOKENS as $anotherIgnored) { + self::assertStripped($token, $token . $ignored . $anotherIgnored); + self::assertStripped($token, $ignored . $anotherIgnored . $token); + } + } + + self::assertStripped($token, implode('', self::IGNORED_TOKENS) . $token); + self::assertStripped($token, $token . implode('', self::IGNORED_TOKENS)); + } + } + + /** @see it('strips random ignored tokens between punctuator tokens', () => { */ + public function testStripsRandomIgnoredTokensBetweenPunctuatorTokens(): void + { + foreach (self::PUNCTUATOR_TOKENS as $left) { + foreach (self::PUNCTUATOR_TOKENS as $right) { + foreach (self::IGNORED_TOKENS as $ignored) { + self::assertStripped($left . $right, $left . $ignored . $right); + + foreach (self::IGNORED_TOKENS as $anotherIgnored) { + self::assertStripped($left . $right, $left . $ignored . $anotherIgnored . $right); + } + } + + self::assertStripped($left . $right, $left . implode('', self::IGNORED_TOKENS) . $right); + } + } + } + + /** @see it('strips random ignored tokens between punctuator and non-punctuator tokens', () => { */ + public function testStripsRandomIgnoredTokensBetweenPunctuatorAndNonPunctuatorTokens(): void + { + foreach (self::NON_PUNCTUATOR_TOKENS as $nonPunctuator) { + foreach (self::PUNCTUATOR_TOKENS as $punctuator) { + foreach (self::IGNORED_TOKENS as $ignored) { + self::assertStripped($punctuator . $nonPunctuator, $punctuator . $ignored . $nonPunctuator); + + foreach (self::IGNORED_TOKENS as $anotherIgnored) { + self::assertStripped($punctuator . $nonPunctuator, $punctuator . $ignored . $anotherIgnored . $nonPunctuator); + } + } + + self::assertStripped($punctuator . $nonPunctuator, $punctuator . implode('', self::IGNORED_TOKENS) . $nonPunctuator); + } + } + } + + /** @see it('strips random ignored tokens between non-punctuator and punctuator tokens', () => { */ + public function testStripsRandomIgnoredTokensBetweenNonPunctuatorAndPunctuatorTokens(): void + { + foreach (self::NON_PUNCTUATOR_TOKENS as $nonPunctuator) { + foreach (self::PUNCTUATOR_TOKENS as $punctuator) { + // Covered by testReplaceRandomIgnoredTokensBetweenNonPunctuatorTokensAndSpreadWithSpace + if ($punctuator === '...') { + continue; + } + + foreach (self::IGNORED_TOKENS as $ignored) { + self::assertStripped($nonPunctuator . $punctuator, $nonPunctuator . $ignored . $punctuator); + + foreach (self::IGNORED_TOKENS as $anotherIgnored) { + self::assertStripped($nonPunctuator . $punctuator, $nonPunctuator . $ignored . $anotherIgnored . $punctuator); + } + } + + self::assertStripped($nonPunctuator . $punctuator, $nonPunctuator . implode('', self::IGNORED_TOKENS) . $punctuator); + } + } + } + + /** @see it('replace random ignored tokens between non-punctuator tokens and spread with space', () => { */ + public function testReplaceRandomIgnoredTokensBetweenNonPunctuatorTokensAndSpreadWithSpace(): void + { + foreach (self::NON_PUNCTUATOR_TOKENS as $nonPunctuator) { + foreach (self::IGNORED_TOKENS as $ignored) { + self::assertStripped($nonPunctuator . ' ...', $nonPunctuator . $ignored . '...'); + + foreach (self::IGNORED_TOKENS as $anotherIgnored) { + self::assertStripped($nonPunctuator . ' ...', $nonPunctuator . $ignored . $anotherIgnored . ' ...'); + } + } + + self::assertStripped($nonPunctuator . ' ...', $nonPunctuator . implode('', self::IGNORED_TOKENS) . '...'); + } + } + + /** @see it('replace random ignored tokens between non-punctuator tokens with space', () => { */ + public function testReplaceRandomIgnoredTokensBetweenNonPunctuatorTokensWithSpace(): void + { + foreach (self::NON_PUNCTUATOR_TOKENS as $left) { + foreach (self::NON_PUNCTUATOR_TOKENS as $right) { + foreach (self::IGNORED_TOKENS as $ignored) { + self::assertStripped($left . ' ' . $right, $left . $ignored . $right); + + foreach (self::IGNORED_TOKENS as $anotherIgnored) { + self::assertStripped($left . ' ' . $right, $left . $ignored . $anotherIgnored . $right); + } + } + + self::assertStripped($left . ' ' . $right, $left . implode('', self::IGNORED_TOKENS) . $right); + } + } + } + + /** @see it('does not strip random ignored tokens embedded in the string', () => { */ + public function testDoesNotStripRandomIgnoredTokensEmbeddedInTheString(): void + { + foreach (self::IGNORED_TOKENS as $ignored) { + self::assertStaysTheSame(json_encode($ignored, JSON_UNESCAPED_UNICODE | JSON_THROW_ON_ERROR)); + + foreach (self::IGNORED_TOKENS as $anotherIgnored) { + self::assertStaysTheSame(json_encode($ignored . $anotherIgnored, JSON_UNESCAPED_UNICODE | JSON_THROW_ON_ERROR)); + } + } + + self::assertStaysTheSame(json_encode(implode('', self::IGNORED_TOKENS), JSON_UNESCAPED_UNICODE | JSON_THROW_ON_ERROR)); + } + + /** @see it('does not strip random ignored tokens embedded in the block string', () => { */ + public function testDoesNotStripRandomIgnoredTokensEmbeddedInTheBlockString(): void + { + $ignoredTokensWithoutFormatting = array_diff(self::IGNORED_TOKENS, ["\n", "\r", "\r\n", "\t", ' ']); + foreach ($ignoredTokensWithoutFormatting as $ignored) { + self::assertStaysTheSame('"""|' . $ignored . '|"""'); + + foreach ($ignoredTokensWithoutFormatting as $anotherIgnored) { + self::assertStaysTheSame('"""|' . $ignored . $anotherIgnored . '|"""'); + } + } + + self::assertStaysTheSame('"""|' . implode('', $ignoredTokensWithoutFormatting) . '|"""'); + } + + /** @see it('strips ignored characters inside random block strings', () => { */ + public function testStripsIgnoredCharactersInsideRandomBlockStrings(): void + { + // Increase when changing the implementation, lengths above 7 are exponentially slower + foreach (self::genFuzzStrings(["\n", "\t", ' ', '"', 'a', '\\'], 7) as $fuzzStr) { + $testStr = '"""' . $fuzzStr . '"""'; + + if (! self::isSingleToken($testStr)) { + continue; + } + + $testValue = self::lexValue($testStr); + + $strippedValue = self::lexValue(Printer::stripIgnoredCharacters($testStr)); + self::assertSame($testValue, $strippedValue, 'Stripping ' . json_encode($testStr, JSON_THROW_ON_ERROR)); + } + } +} diff --git a/tests/Language/StripIgnoredCharactersTest.php b/tests/Language/StripIgnoredCharactersTest.php new file mode 100644 index 000000000..040c20879 --- /dev/null +++ b/tests/Language/StripIgnoredCharactersTest.php @@ -0,0 +1,304 @@ + { + */ +final class StripIgnoredCharactersTest extends TestCase +{ + /** + * @throws \JsonException + * @throws SyntaxError + */ + private static function lexValue(string $str): ?string + { + $lexer = new Lexer(new Source($str)); + $value = $lexer->advance()->value; + + self::assertSame(Token::EOF, $lexer->advance()->kind, 'Expected EOF'); + + return $value; + } + + /** + * @throws \JsonException + * @throws SyntaxError + */ + private static function assertStripped(string $expected, string $docString): void + { + $stripped = Printer::stripIgnoredCharacters($docString); + self::assertSame($expected, $stripped); + + $strippedTwice = Printer::stripIgnoredCharacters($stripped); + self::assertSame($expected, $strippedTwice); + } + + /** + * @throws \JsonException + * @throws SyntaxError + */ + private static function assertStaysTheSame(string $docString): void + { + self::assertStripped($docString, $docString); + } + + /** + * @throws \JsonException + * @throws SyntaxError + */ + private static function assertStrippedString(string $expected, string $blockStr): void + { + $originalValue = self::lexValue($blockStr); + $strippedValue = self::lexValue(Printer::stripIgnoredCharacters($blockStr)); + self::assertSame($originalValue, $strippedValue); + + self::assertStripped($expected, $blockStr); + } + + /** @see it('strips ignored characters from GraphQL query document', () => { */ + public function testStripsIgnoredCharactersFromGraphQLQueryDocument(): void + { + $query = <<<'GRAPHQL' + query SomeQuery($foo: String!, $bar: String) { + someField(foo: $foo, bar: $bar) { + a + b { + c + d + } + } + } + GRAPHQL; + + self::assertSame( + 'query SomeQuery($foo:String!$bar:String){someField(foo:$foo bar:$bar){a b{c d}}}', + Printer::stripIgnoredCharacters($query) + ); + } + + /** @see it('accepts Source object', () => { */ + public function testAcceptsSourceObject(): void + { + self::assertSame('{a}', Printer::stripIgnoredCharacters(new Source('{ a }'))); + } + + /** @see it('strips ignored characters from GraphQL SDL document', () => { */ + public function testStripsIgnoredCharactersFromGraphQLSDLDocument(): void + { + $sdl = <<<'GRAPHQL' + """ + Type description + """ + type Foo { + """ + Field description + """ + bar: String + } + GRAPHQL; + + self::assertSame( + '"""Type description""" type Foo{"""Field description""" bar:String}', + Printer::stripIgnoredCharacters($sdl) + ); + } + + /** @see it('report document with invalid token', () => { */ + public function testReportDocumentWithInvalidToken(): void + { + try { + Printer::stripIgnoredCharacters("{ foo(arg: \"\n\""); + self::fail('Expected SyntaxError'); + } catch (SyntaxError $error) { + self::assertSame( + <<<'EOF' + Syntax Error: Unterminated string. + + GraphQL request (1:13) + 1: { foo(arg: " + ^ + 2: " + + EOF, + FormattedError::printError($error) + ); + } + } + + /** @see it('strips non-parsable document', () => { */ + public function testStripsNonParsableDocument(): void + { + self::assertStripped('{foo(arg:"str"', '{ foo(arg: "str"'); + } + + /** @see it('strips documents with only ignored characters', () => { */ + public function testStripsDocumentsWithOnlyIgnoredCharacters(): void + { + self::assertStripped('', "\n"); + self::assertStripped('', ','); + self::assertStripped('', ',,'); + self::assertStripped('', "#comment\n, \n"); + } + + /** @see it('strips leading and trailing ignored tokens', () => { */ + public function testStripsLeadingAndTrailingIgnoredTokens(): void + { + self::assertStripped('1', "\n1"); + self::assertStripped('1', ',1'); + self::assertStripped('1', ',,1'); + self::assertStripped('1', "#comment\n, \n1"); + + self::assertStripped('1', "1\n"); + self::assertStripped('1', '1,'); + self::assertStripped('1', '1,,'); + self::assertStripped('1', "1#comment\n, \n"); + } + + /** @see it('strips ignored tokens between punctuator tokens', () => { */ + public function testStripsIgnoredTokensBetweenPunctuatorTokens(): void + { + self::assertStripped('[)', '[,)'); + self::assertStripped('[)', "[\r)"); + self::assertStripped('[)', "[\r\r)"); + self::assertStripped('[)', "[\r,)"); + self::assertStripped('[)', "[,\n)"); + } + + /** @see it('strips ignored tokens between punctuator and non-punctuator tokens', () => { */ + public function testStripsIgnoredTokensBetweenPunctuatorAndNonPunctuatorTokens(): void + { + self::assertStripped('[1', '[,1'); + self::assertStripped('[1', "[\r1"); + self::assertStripped('[1', "[\r\r1"); + self::assertStripped('[1', "[\r,1"); + self::assertStripped('[1', "[,\n1"); + } + + /** @see it('strips ignored tokens between non-punctuator and punctuator tokens', () => { */ + public function testStripsIgnoredTokensBetweenNonPunctuatorAndPunctuatorTokens(): void + { + self::assertStripped('1[', '1,['); + self::assertStripped('1[', "1\r["); + self::assertStripped('1[', "1\r\r["); + self::assertStripped('1[', "1\r,["); + self::assertStripped('1[', "1,\n["); + } + + /** @see it('replace ignored tokens between non-punctuator tokens and spread with space', () => { */ + public function testReplaceIgnoredTokensBetweenNonPunctuatorTokensAndSpreadWithSpace(): void + { + self::assertStripped('a ...', 'a ...'); + self::assertStripped('1 ...', '1 ...'); + self::assertStripped('1 ......', '1 ... ...'); + } + + /** @see it('replace ignored tokens between non-punctuator tokens with space', () => { */ + public function testReplaceIgnoredTokensBetweenNonPunctuatorTokensWithSpace(): void + { + self::assertStaysTheSame('1 2'); + self::assertStaysTheSame('"" ""'); + self::assertStaysTheSame('a b'); + + self::assertStripped('a 1', 'a,1'); + self::assertStripped('a 1', 'a,,1'); + self::assertStripped('a 1', 'a 1'); + self::assertStripped('a 1', "a \t 1"); + } + + /** @see it('does not strip ignored tokens embedded in the string', () => { */ + public function testDoesNotStripIgnoredTokensEmbeddedInTheString(): void + { + self::assertStaysTheSame('" "'); + self::assertStaysTheSame('","'); + self::assertStaysTheSame('",,"'); + self::assertStaysTheSame('",|"'); + } + + /** @see it('does not strip ignored tokens embedded in the block string', () => { */ + public function testDoesNotStripIgnoredTokensEmbeddedInTheBlockString(): void + { + self::assertStaysTheSame('""","""'); + self::assertStaysTheSame('""",,"""'); + self::assertStaysTheSame('""",|"""'); + } + + /** @see it('strips ignored characters inside block strings', () => { */ + public function testStripsIgnoredCharactersInsideBlockStrings(): void + { + self::assertStrippedString('""""""', '""""""'); + self::assertStrippedString('""""""', '""" """'); + + self::assertStrippedString('"""a"""', '"""a"""'); + self::assertStrippedString('""" a"""', '""" a"""'); + self::assertStrippedString('""" a """', '""" a """'); + + self::assertStrippedString('""""""', "\"\"\"\n\"\"\""); + self::assertStrippedString("\"\"\"a\nb\"\"\"", "\"\"\"a\nb\"\"\""); + self::assertStrippedString("\"\"\"a\nb\"\"\"", "\"\"\"a\rb\"\"\""); + self::assertStrippedString("\"\"\"a\nb\"\"\"", "\"\"\"a\r\nb\"\"\""); + self::assertStrippedString("\"\"\"a\n\nb\"\"\"", "\"\"\"a\r\n\nb\"\"\""); + + self::assertStrippedString("\"\"\"\\\n\"\"\"", "\"\"\"\\\n\"\"\""); + self::assertStrippedString("\"\"\"\"\n\"\"\"", "\"\"\"\"\n\"\"\""); + self::assertStrippedString('"""\\""""""', "\"\"\"\\\"\"\"\n\"\"\""); + + self::assertStrippedString("\"\"\"\na\n b\"\"\"", "\"\"\"\na\n b\"\"\""); + self::assertStrippedString("\"\"\"a\nb\"\"\"", "\"\"\"\n a\n b\"\"\""); + self::assertStrippedString("\"\"\"a\n b\nc\"\"\"", "\"\"\"\na\n b\nc\"\"\""); + } + + public function testStripsNonASCIICharactersInStrings(): void + { + self::assertStripped('{a(b:"ä ö" c:""" ü """)}', '{ a(b: "ä ö", c: """ ü """) }'); + self::assertStripped('{a(b:"😀€" c:"x")}', '{ a(b: "😀€", c: "x") }'); + self::assertStripped('{a(b:"😀x" c:"€y")}', "# €😀\n{ a(b: \"😀x\", c: \"€y\") }"); + self::assertStripped('"ö" "ä"', "\u{FEFF}# ü\n\"ö\" \"ä\""); + } + + public function testRejectsInvalidUTF8(): void + { + $this->expectException(SyntaxError::class); + $this->expectExceptionMessage('Invalid UTF-8 byte: 0xFF'); + Printer::stripIgnoredCharacters("{ a(b: \"\xFFabc\") }"); + } + + /** @see it('strips kitchen sink query but maintains the exact same AST', () => { */ + public function testStripsKitchenSinkQueryButMaintainsTheExactSameAST(): void + { + $kitchenSinkQuery = file_get_contents(__DIR__ . '/kitchen-sink.graphql'); + + $strippedQuery = Printer::stripIgnoredCharacters($kitchenSinkQuery); + self::assertSame($strippedQuery, Printer::stripIgnoredCharacters($strippedQuery)); + + $queryAST = Parser::parse($kitchenSinkQuery, ['noLocation' => true]); + $strippedAST = Parser::parse($strippedQuery, ['noLocation' => true]); + self::assertEquals($queryAST, $strippedAST); + } + + /** @see it('strips kitchen sink SDL but maintains the exact same AST', () => { */ + public function testStripsKitchenSinkSDLButMaintainsTheExactSameAST(): void + { + $kitchenSinkSDL = file_get_contents(__DIR__ . '/schema-kitchen-sink.graphql'); + + $strippedSDL = Printer::stripIgnoredCharacters($kitchenSinkSDL); + self::assertSame($strippedSDL, Printer::stripIgnoredCharacters($strippedSDL)); + + $sdlAST = Parser::parse($kitchenSinkSDL, ['noLocation' => true]); + $strippedAST = Parser::parse($strippedSDL, ['noLocation' => true]); + self::assertEquals($sdlAST, $strippedAST); + } +}