Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,14 @@ You can find and compare releases at the [GitHub release page](https://github.co

## Unreleased

### Added

- Add `Printer::stripIgnoredCharacters()` to remove characters that do not change a document's meaning https://github.com/webonyx/graphql-php/pull/1985

### Fixed

- Reject invalid UTF-8 in documents https://github.com/webonyx/graphql-php/pull/1987

## v15.37.3

### Fixed
Expand Down
42 changes: 42 additions & 0 deletions docs/class-reference.md
Original file line number Diff line number Diff line change
Expand Up @@ -1386,6 +1386,48 @@ $printed = GraphQL\Language\Printer::doPrint($ast);
static function doPrint(GraphQL\Language\AST\Node $ast): string
```

````php
/**
* Strips characters that are not significant to the validity or execution of a GraphQL document:
* - UnicodeBOM
* - WhiteSpace
* - LineTerminator
* - Comment
* - Comma
* - BlockString indentation
*
* Neighboring non-punctuator tokens are always delimited by a single space.
* Parsing input and output yields the same AST, apart from node locations.
* The output is stable, but may change between releases.
*
* ```graphql
* query SomeQuery($foo: String!, $bar: String) {
* someField(foo: $foo, bar: $bar) {
* a
* b {
* c
* d
* }
* }
* }
* ```
*
* becomes
*
* ```graphql
* query SomeQuery($foo:String!$bar:String){someField(foo:$foo bar:$bar){a b{c d}}}
* ```
*
* @param Source|string $source
*
* @throws \JsonException
* @throws SyntaxError
*
* @api
*/
static function stripIgnoredCharacters($source): string
````

## GraphQL\Language\Visitor

Utility for efficient AST traversal and modification.
Expand Down
17 changes: 10 additions & 7 deletions src/Language/BlockString.php
Original file line number Diff line number Diff line change
Expand Up @@ -102,7 +102,7 @@ public static function getIndentation(string $value): int
* trailing blank line. However, if a block string starts with whitespace and is
* a single-line, adding a leading blank line would strip that whitespace.
*/
public static function print(string $value): string
public static function print(string $value, bool $minimize = false): string
{
$escapedValue = str_replace('"""', '\\"""', $value);

Expand Down Expand Up @@ -131,11 +131,14 @@ public static function print(string $value): string
$forceTrailingNewline = $hasTrailingQuote || $hasTrailingSlash;

// add leading and trailing new lines only if it improves readability
$printAsMultipleLines = ! $isSingleLine
|| mb_strlen($value) > 70
|| $forceTrailingNewline
|| $forceLeadingNewLine
|| $hasTrailingTripleQuotes;
$printAsMultipleLines = ! $minimize
&& (
! $isSingleLine
|| mb_strlen($value) > 70
|| $forceTrailingNewline
|| $forceLeadingNewLine
|| $hasTrailingTripleQuotes
);

$result = '';

Expand All @@ -146,7 +149,7 @@ public static function print(string $value): string
}

$result .= $escapedValue;
if ($printAsMultipleLines) {
if ($printAsMultipleLines || $forceTrailingNewline) {
$result .= "\n";
}

Expand Down
20 changes: 20 additions & 0 deletions src/Language/Lexer.php
Original file line number Diff line number Diff line change
Expand Up @@ -113,6 +113,10 @@ public function lookahead(): Token
*/
private function readToken(Token $prev): Token
{
if ($prev->kind === Token::SOF) {
$this->assertValidUTF8();
}

$bodyLength = $this->source->length;

$this->positionAfterWhitespace();
Expand Down Expand Up @@ -259,6 +263,22 @@ private function readToken(Token $prev): Token
throw new SyntaxError($this->source, $position, $this->unexpectedCharacterMessage($code));
}

/** @throws SyntaxError */
private function assertValidUTF8(): void
{
$body = $this->source->body;
if (mb_check_encoding($body, 'UTF-8')) {
return;
}

$scrubbedBody = mb_scrub($body, 'UTF-8');
$invalidByteOffset = strspn($body ^ $scrubbedBody, "\0");
$invalidByte = strtoupper(bin2hex($body[$invalidByteOffset]));
$validPrefix = substr($body, 0, $invalidByteOffset);

throw new SyntaxError($this->source, mb_strlen($validPrefix, 'UTF-8'), "Invalid UTF-8 byte: 0x{$invalidByte}");
}

/** @throws \JsonException */
private function unexpectedCharacterMessage(?int $code): string
{
Expand Down
118 changes: 118 additions & 0 deletions src/Language/Printer.php
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@

namespace GraphQL\Language;

use GraphQL\Error\SyntaxError;
use GraphQL\Language\AST\ArgumentNode;
use GraphQL\Language\AST\BooleanValueNode;
use GraphQL\Language\AST\DirectiveDefinitionNode;
Expand Down Expand Up @@ -78,6 +79,123 @@ public static function doPrint(Node $ast): string
return static::p($ast);
}

/**
* Strips characters that are not significant to the validity or execution of a GraphQL document:
* - UnicodeBOM
* - WhiteSpace
* - LineTerminator
* - Comment
* - Comma
* - BlockString indentation
*
* Neighboring non-punctuator tokens are always delimited by a single space.
* Parsing input and output yields the same AST, apart from node locations.
* The output is stable, but may change between releases.
*
* ```graphql
* query SomeQuery($foo: String!, $bar: String) {
* someField(foo: $foo, bar: $bar) {
* a
* b {
* c
* d
* }
* }
* }
* ```
*
* becomes
*
* ```graphql
* query SomeQuery($foo:String!$bar:String){someField(foo:$foo bar:$bar){a b{c d}}}
* ```
*
* @param Source|string $source
*
* @throws \JsonException
* @throws SyntaxError
*
* @api
*/
public static function stripIgnoredCharacters($source): string
{
$sourceObj = $source instanceof Source
? $source
: new Source($source);
$body = $sourceObj->body;
$lexer = new Lexer($sourceObj);

$stripped = '';
$wasLastAddedTokenNonPunctuator = false;
$charPosition = 0;
$bytePosition = 0;
while (($token = $lexer->advance())->kind !== Token::EOF) {
$isNonPunctuator = ! static::isPunctuatorTokenKind($token->kind);

// `1...` would lex as an invalid float
if ($wasLastAddedTokenNonPunctuator && ($isNonPunctuator || $token->kind === Token::SPREAD)) {
$stripped .= ' ';
}

if ($token->kind === Token::STRING) {
$bytePosition = static::advanceBytePosition($body, $bytePosition, $token->start - $charPosition);
$tokenBytePosition = static::advanceBytePosition($body, $bytePosition, $token->end - $token->start);
$stripped .= substr($body, $bytePosition, $tokenBytePosition - $bytePosition);

$charPosition = $token->end;
$bytePosition = $tokenBytePosition;
} else {
$stripped .= static::tokenSource($token);
}

$wasLastAddedTokenNonPunctuator = $isNonPunctuator;
}

return $stripped;
}

protected static function isPunctuatorTokenKind(string $kind): bool
{
return ! in_array($kind, [Token::NAME, Token::INT, Token::FLOAT, Token::STRING, Token::BLOCK_STRING], true);
}

/** Steps through characters the same way as Lexer::readChar(), since token positions count them. */
protected static function advanceBytePosition(string $body, int $bytePosition, int $charCount): int
{
for ($i = 0; $i < $charCount; ++$i) {
$leadByte = ord($body[$bytePosition]);
if ($leadByte < 128) {
++$bytePosition;
} elseif ($leadByte < 224) {
$bytePosition += 2;
} elseif ($leadByte < 240) {
$bytePosition += 3;
} else {
$bytePosition += 4;
}
}

return $bytePosition;
}

protected static function tokenSource(Token $token): string
{
switch ($token->kind) {
case Token::BLOCK_STRING:
assert(is_string($token->value), 'Lexer sets the dedented value of block strings');

return BlockString::print($token->value, true);
case Token::NAME:
case Token::INT:
case Token::FLOAT:
assert(is_string($token->value), 'Lexer sets the raw value of names and numbers');

return $token->value;
default:
return $token->kind;
}
}

/** @throws \JsonException */
protected static function p(?Node $node): string
{
Expand Down
9 changes: 9 additions & 0 deletions tests/Language/BlockStringTest.php
Original file line number Diff line number Diff line change
Expand Up @@ -331,13 +331,15 @@ public function testDoNotEscapeCharacters(): void
EOF,
BlockString::print($str)
);
self::assertSame("\"\"\"\n{$str}\"\"\"", BlockString::print($str, true));
}

/** @see it('by default print block strings as single line', () => { */
public function testByDefaultPrintBlockStringsAsSingleLine(): void
{
$str = 'one liner';
self::assertSame('"""one liner"""', BlockString::print($str));
self::assertSame('"""one liner"""', BlockString::print($str, true));
}

/** @see it('by default print block strings ending with triple quotation as multi-line', () => { */
Expand All @@ -352,13 +354,15 @@ public function testByDefaultPrintBlockStringsEndingWithTripleQuotationAsMultiLi
EOF,
BlockString::print($str)
);
self::assertSame('"""triple quotation \\""""""', BlockString::print($str, true));
}

/** @see it('correctly prints single-line with leading space') */
public function testCorrectlyPrintsSingleLineWithLeadingSpace(): void
{
$str = ' space-led string';
self::assertSame('""" space-led string"""', BlockString::print($str));
self::assertSame('""" space-led string"""', BlockString::print($str, true));
}

/** @see it('correctly prints single-line with leading space and trailing quotation', () => { */
Expand All @@ -372,6 +376,7 @@ public function testCorrectlyPrintsSingleLineWithLeadingSpaceAndQuotation(): voi
EOF,
BlockString::print($str)
);
self::assertSame("\"\"\" space-led value \"quoted string\"\n\"\"\"", BlockString::print($str, true));
}

/** @see it('correctly prints single-line with trailing backslash') */
Expand All @@ -386,6 +391,7 @@ public function testCorrectlyPrintsSingleLineWithTrailingBackslash(): void
EOF,
BlockString::print($str)
);
self::assertSame("\"\"\"backslash \\\n\"\"\"", BlockString::print($str, true));
}

/** @see it('correctly prints multi-line with internal indent', () => { */
Expand All @@ -404,6 +410,7 @@ public function testCorrectlyPrintsMultiLineWithInternalIndent(): void
EOF,
BlockString::print($str)
);
self::assertSame("\"\"\"\nno indent\n with indent\"\"\"", BlockString::print($str, true));
}

/** @see it('correctly prints string with a first line indentation') */
Expand All @@ -426,11 +433,13 @@ public function testCorrectlyPrintsStringWithAFirstLineIndentation(): void
EOF,
BlockString::print($str)
);
self::assertSame("\"\"\"{$str}\"\"\"", BlockString::print($str, true));
}

public function testCorrectlyPrintsEmptyString(): void
{
$str = '';
self::assertSame('""""""', BlockString::print($str));
self::assertSame('""""""', BlockString::print($str, true));
}
}
23 changes: 23 additions & 0 deletions tests/Language/LexerTest.php
Original file line number Diff line number Diff line change
Expand Up @@ -674,6 +674,29 @@ public function testReportsUsefulUnknownCharErrors(string $str, string $expected
$this->expectSyntaxError($str, $expectedMessage, $location);
}

/** @return iterable<array{string, string, SourceLocation}> */
public static function reportsInvalidUTF8(): iterable
{
yield 'in name position' => ["\xFF", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 1)];
yield 'in string' => ["\"a\xFFbcd\"", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 3)];
yield 'after multibyte character' => ["\"ä\xC3x\"", 'Invalid UTF-8 byte: 0xC3', self::loc(1, 3)];
yield 'truncated at end' => ["\"ab\xE2\x82", 'Invalid UTF-8 byte: 0xE2', self::loc(1, 4)];
yield 'encoded surrogate' => ["\"\xED\xA0\x80\"", 'Invalid UTF-8 byte: 0xED', self::loc(1, 2)];
yield 'overlong encoding' => ["\"\xC0\xAF\"", 'Invalid UTF-8 byte: 0xC0', self::loc(1, 2)];
yield 'in block string' => ["\"\"\"\xFF\"\"\"", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 4)];
yield 'in comment' => ["# \xFF\nfoo", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 3)];
yield 'after the first token' => ["foo\n \"\xFF\"", 'Invalid UTF-8 byte: 0xFF', self::loc(2, 4)];
yield 'after BOM' => ["\u{FEFF}\xFF", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 2)];
yield 'first of two' => ["\"a\xFEb\xFF\"", 'Invalid UTF-8 byte: 0xFE', self::loc(1, 3)];
yield 'after long multibyte prefix' => ['"' . str_repeat('€', 2000) . "\xFF\"", 'Invalid UTF-8 byte: 0xFF', self::loc(1, 2002)];
}

/** @dataProvider reportsInvalidUTF8 */
public function testReportsInvalidUTF8(string $str, string $expectedMessage, SourceLocation $location): void
{
$this->expectSyntaxError($str, $expectedMessage, $location);
}

/** @see it('lex reports useful information for dashes in names') */
public function testReportsUsefulDashesInfo(): void
{
Expand Down
Loading
Loading