diff --git a/CHANGELOG.md b/CHANGELOG.md index 74ccea04c..fdd843295 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,10 @@ You can find and compare releases at the [GitHub release page](https://github.co ## Unreleased +### Added + +- Add the `maxTokens` parser option to limit the number of tokens a document may contain https://github.com/webonyx/graphql-php/issues/1905 + ## v15.37.3 ### Fixed diff --git a/docs/class-reference.md b/docs/class-reference.md index 81cd48e81..30bd7b20a 100644 --- a/docs/class-reference.md +++ b/docs/class-reference.md @@ -1180,7 +1180,8 @@ Parses string containing GraphQL query language or [schema definition language]( allowLegacySDLEmptyFields?: bool, allowLegacySDLImplementsInterfaces?: bool, experimentalFragmentVariables?: bool, - recursionLimit?: int<0, max> + recursionLimit?: int<0, max>, + maxTokens?: int<1, max>|null } ``` @@ -1216,6 +1217,13 @@ Parses string containing GraphQL query language or [schema definition language]( The counter is shared across `parseSelectionSet`, `parseValueLiteral`, and `parseTypeReference`. Defaults to 256. Set to 0 to disable the limit. +- **maxTokens**: + Parser CPU and memory usage scales linearly with the number of tokens in a document, + and parsing happens before validation, so even an invalid document can use a lot of resources. + Set this option to limit the number of tokens a document may contain. + Parsing then fails with a syntax error. + There is no limit by default. + Those magic functions allow partial parsing: @method static NameNode name(Source|string $source, ParserOptions $options = []) diff --git a/src/Language/Parser.php b/src/Language/Parser.php index e91e24830..f7627668a 100644 --- a/src/Language/Parser.php +++ b/src/Language/Parser.php @@ -66,7 +66,8 @@ * allowLegacySDLEmptyFields?: bool, * allowLegacySDLImplementsInterfaces?: bool, * experimentalFragmentVariables?: bool, - * recursionLimit?: int<0, max> + * recursionLimit?: int<0, max>, + * maxTokens?: int<1, max>|null * } * * - **noLocation**: @@ -101,6 +102,13 @@ * The counter is shared across `parseSelectionSet`, `parseValueLiteral`, and `parseTypeReference`. * Defaults to 256. Set to 0 to disable the limit. * + * - **maxTokens**: + * Parser CPU and memory usage scales linearly with the number of tokens in a document, + * and parsing happens before validation, so even an invalid document can use a lot of resources. + * Set this option to limit the number of tokens a document may contain. + * Parsing then fails with a syntax error. + * There is no limit by default. + * * Those magic functions allow partial parsing: * * @method static NameNode name(Source|string $source, ParserOptions $options = []) @@ -330,6 +338,10 @@ public static function __callStatic(string $name, array $arguments) private int $recursionLimit; + private ?int $maxTokens; + + private int $tokenCount = 0; + /** * @param Source|string $source * @@ -342,6 +354,7 @@ public function __construct($source, array $options = []) : new Source($source); $this->lexer = new Lexer($sourceObj, $options); $this->recursionLimit = $options['recursionLimit'] ?? self::DEFAULT_RECURSION_LIMIT; + $this->maxTokens = $options['maxTokens'] ?? null; } /** @@ -367,6 +380,25 @@ private function increaseRecursionDepth(): void ++$this->recursionDepth; } + /** + * Advances the lexer, counting the tokens of the document to enforce the maxTokens option. + * + * @throws \JsonException + * @throws SyntaxError + */ + private function advanceLexer(): void + { + $token = $this->lexer->advance(); + + if ($token->kind !== Token::EOF) { + ++$this->tokenCount; + + if ($this->maxTokens !== null && $this->tokenCount > $this->maxTokens) { + throw new SyntaxError($this->lexer->source, $token->start, "Document contains more than {$this->maxTokens} tokens. Parsing aborted."); + } + } + } + /** Determines if the next token is of a given kind. */ private function peek(string $kind): bool { @@ -385,7 +417,7 @@ private function skip(string $kind): bool $match = $this->lexer->token->kind === $kind; if ($match) { - $this->lexer->advance(); + $this->advanceLexer(); } return $match; @@ -403,7 +435,7 @@ private function expect(string $kind): Token $token = $this->lexer->token; if ($token->kind === $kind) { - $this->lexer->advance(); + $this->advanceLexer(); return $token; } @@ -425,7 +457,7 @@ private function expectKeyword(string $value): void throw new SyntaxError($this->lexer->source, $token->start, "Expected \"{$value}\", found {$token->getDescription()}"); } - $this->lexer->advance(); + $this->advanceLexer(); } /** @@ -439,7 +471,7 @@ private function expectOptionalKeyword(string $value): bool { $token = $this->lexer->token; if ($token->kind === Token::NAME && $token->value === $value) { - $this->lexer->advance(); + $this->advanceLexer(); return true; } @@ -953,7 +985,7 @@ private function parseValueLiteral(bool $isConst): ValueNode return $this->parseObject($isConst); case Token::INT: - $this->lexer->advance(); + $this->advanceLexer(); return new IntValueNode([ 'value' => $token->value, @@ -961,7 +993,7 @@ private function parseValueLiteral(bool $isConst): ValueNode ]); case Token::FLOAT: - $this->lexer->advance(); + $this->advanceLexer(); return new FloatValueNode([ 'value' => $token->value, @@ -974,7 +1006,7 @@ private function parseValueLiteral(bool $isConst): ValueNode case Token::NAME: if ($token->value === 'true' || $token->value === 'false') { - $this->lexer->advance(); + $this->advanceLexer(); return new BooleanValueNode([ 'value' => $token->value === 'true', @@ -983,13 +1015,13 @@ private function parseValueLiteral(bool $isConst): ValueNode } if ($token->value === 'null') { - $this->lexer->advance(); + $this->advanceLexer(); return new NullValueNode([ 'loc' => $this->loc($token), ]); } - $this->lexer->advance(); + $this->advanceLexer(); return new EnumValueNode([ 'value' => $token->value, @@ -1017,7 +1049,7 @@ private function parseValueLiteral(bool $isConst): ValueNode private function parseStringLiteral(): StringValueNode { $token = $this->lexer->token; - $this->lexer->advance(); + $this->advanceLexer(); return new StringValueNode([ 'value' => $token->value, @@ -1376,8 +1408,8 @@ private function parseFieldsDefinition(): NodeList && $this->peek(Token::BRACE_L) && $this->lexer->lookahead()->kind === Token::BRACE_R ) { - $this->lexer->advance(); - $this->lexer->advance(); + $this->advanceLexer(); + $this->advanceLexer(); return new NodeList([]); } diff --git a/tests/Language/ParserTest.php b/tests/Language/ParserTest.php index 229ff0988..8f9ade6e8 100644 --- a/tests/Language/ParserTest.php +++ b/tests/Language/ParserTest.php @@ -809,4 +809,35 @@ public function testSiblingBranchesDontAccumulate(): void Parser::parse($query, ['recursionLimit' => $limit]); $this->expectNotToPerformAssertions(); } + + /** @see it('limit maximum number of tokens') */ + public function testLimitMaximumNumberOfTokens(): void + { + Parser::parse('{ foo }', ['maxTokens' => 3]); + $this->assertSyntaxErrorMessage('{ foo }', 2); + + Parser::parse('{ foo(bar: "baz") }', ['maxTokens' => 8]); + $this->assertSyntaxErrorMessage('{ foo(bar: "baz") }', 7); + } + + public function testNoTokenLimitByDefault(): void + { + Parser::parse('{ ' . str_repeat('a ', 10000) . '}'); + $this->expectNotToPerformAssertions(); + } + + /** + * @param int<1, max> $maxTokens + * + * @throws \JsonException + */ + private function assertSyntaxErrorMessage(string $source, int $maxTokens): void + { + try { + Parser::parse($source, ['maxTokens' => $maxTokens]); + self::fail('Expected a SyntaxError'); + } catch (SyntaxError $error) { + self::assertSame("Syntax Error: Document contains more than {$maxTokens} tokens. Parsing aborted.", $error->getMessage()); + } + } }