From 90309927bdf77380793d713cb9fb7a3d07363677 Mon Sep 17 00:00:00 2001 From: Pascal CESCON - Amoifr Date: Sun, 4 Oct 2026 19:55:27 +0200 Subject: [PATCH 1/3] Add the maxTokens parser option --- CHANGELOG.md | 4 +++ src/Language/Parser.php | 57 +++++++++++++++++++++++++++-------- tests/Language/ParserTest.php | 31 +++++++++++++++++++ 3 files changed, 79 insertions(+), 13 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 74ccea04c..fdd843295 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,10 @@ You can find and compare releases at the [GitHub release page](https://github.co ## Unreleased +### Added + +- Add the `maxTokens` parser option to limit the number of tokens a document may contain https://github.com/webonyx/graphql-php/issues/1905 + ## v15.37.3 ### Fixed diff --git a/src/Language/Parser.php b/src/Language/Parser.php index e91e24830..5f5ba6328 100644 --- a/src/Language/Parser.php +++ b/src/Language/Parser.php @@ -66,7 +66,8 @@ * allowLegacySDLEmptyFields?: bool, * allowLegacySDLImplementsInterfaces?: bool, * experimentalFragmentVariables?: bool, - * recursionLimit?: int<0, max> + * recursionLimit?: int<0, max>, + * maxTokens?: int<1, max>|null * } * * - **noLocation**: @@ -101,6 +102,12 @@ * The counter is shared across `parseSelectionSet`, `parseValueLiteral`, and `parseTypeReference`. * Defaults to 256. Set to 0 to disable the limit. * + * - **maxTokens**: + * Parser CPU and memory usage is linear to the number of tokens in a document, + * and parsing happens before validation, so even an invalid document can use a lot of resources. + * Set this option to limit the number of tokens a document may contain, parsing then fails with a syntax error. + * There is no limit by default. + * * Those magic functions allow partial parsing: * * @method static NameNode name(Source|string $source, ParserOptions $options = []) @@ -330,6 +337,10 @@ public static function __callStatic(string $name, array $arguments) private int $recursionLimit; + private ?int $maxTokens; + + private int $tokenCount = 0; + /** * @param Source|string $source * @@ -342,6 +353,7 @@ public function __construct($source, array $options = []) : new Source($source); $this->lexer = new Lexer($sourceObj, $options); $this->recursionLimit = $options['recursionLimit'] ?? self::DEFAULT_RECURSION_LIMIT; + $this->maxTokens = $options['maxTokens'] ?? null; } /** @@ -367,6 +379,25 @@ private function increaseRecursionDepth(): void ++$this->recursionDepth; } + /** + * Advances the lexer, counting the tokens of the document to enforce the maxTokens option. + * + * @throws \JsonException + * @throws SyntaxError + */ + private function advanceLexer(): void + { + $token = $this->lexer->advance(); + + if ($token->kind !== Token::EOF) { + ++$this->tokenCount; + + if ($this->maxTokens !== null && $this->tokenCount > $this->maxTokens) { + throw new SyntaxError($this->lexer->source, $token->start, "Document contains more than {$this->maxTokens} tokens. Parsing aborted."); + } + } + } + /** Determines if the next token is of a given kind. */ private function peek(string $kind): bool { @@ -385,7 +416,7 @@ private function skip(string $kind): bool $match = $this->lexer->token->kind === $kind; if ($match) { - $this->lexer->advance(); + $this->advanceLexer(); } return $match; @@ -403,7 +434,7 @@ private function expect(string $kind): Token $token = $this->lexer->token; if ($token->kind === $kind) { - $this->lexer->advance(); + $this->advanceLexer(); return $token; } @@ -425,7 +456,7 @@ private function expectKeyword(string $value): void throw new SyntaxError($this->lexer->source, $token->start, "Expected \"{$value}\", found {$token->getDescription()}"); } - $this->lexer->advance(); + $this->advanceLexer(); } /** @@ -439,7 +470,7 @@ private function expectOptionalKeyword(string $value): bool { $token = $this->lexer->token; if ($token->kind === Token::NAME && $token->value === $value) { - $this->lexer->advance(); + $this->advanceLexer(); return true; } @@ -953,7 +984,7 @@ private function parseValueLiteral(bool $isConst): ValueNode return $this->parseObject($isConst); case Token::INT: - $this->lexer->advance(); + $this->advanceLexer(); return new IntValueNode([ 'value' => $token->value, @@ -961,7 +992,7 @@ private function parseValueLiteral(bool $isConst): ValueNode ]); case Token::FLOAT: - $this->lexer->advance(); + $this->advanceLexer(); return new FloatValueNode([ 'value' => $token->value, @@ -974,7 +1005,7 @@ private function parseValueLiteral(bool $isConst): ValueNode case Token::NAME: if ($token->value === 'true' || $token->value === 'false') { - $this->lexer->advance(); + $this->advanceLexer(); return new BooleanValueNode([ 'value' => $token->value === 'true', @@ -983,13 +1014,13 @@ private function parseValueLiteral(bool $isConst): ValueNode } if ($token->value === 'null') { - $this->lexer->advance(); + $this->advanceLexer(); return new NullValueNode([ 'loc' => $this->loc($token), ]); } - $this->lexer->advance(); + $this->advanceLexer(); return new EnumValueNode([ 'value' => $token->value, @@ -1017,7 +1048,7 @@ private function parseValueLiteral(bool $isConst): ValueNode private function parseStringLiteral(): StringValueNode { $token = $this->lexer->token; - $this->lexer->advance(); + $this->advanceLexer(); return new StringValueNode([ 'value' => $token->value, @@ -1376,8 +1407,8 @@ private function parseFieldsDefinition(): NodeList && $this->peek(Token::BRACE_L) && $this->lexer->lookahead()->kind === Token::BRACE_R ) { - $this->lexer->advance(); - $this->lexer->advance(); + $this->advanceLexer(); + $this->advanceLexer(); return new NodeList([]); } diff --git a/tests/Language/ParserTest.php b/tests/Language/ParserTest.php index 229ff0988..8f9ade6e8 100644 --- a/tests/Language/ParserTest.php +++ b/tests/Language/ParserTest.php @@ -809,4 +809,35 @@ public function testSiblingBranchesDontAccumulate(): void Parser::parse($query, ['recursionLimit' => $limit]); $this->expectNotToPerformAssertions(); } + + /** @see it('limit maximum number of tokens') */ + public function testLimitMaximumNumberOfTokens(): void + { + Parser::parse('{ foo }', ['maxTokens' => 3]); + $this->assertSyntaxErrorMessage('{ foo }', 2); + + Parser::parse('{ foo(bar: "baz") }', ['maxTokens' => 8]); + $this->assertSyntaxErrorMessage('{ foo(bar: "baz") }', 7); + } + + public function testNoTokenLimitByDefault(): void + { + Parser::parse('{ ' . str_repeat('a ', 10000) . '}'); + $this->expectNotToPerformAssertions(); + } + + /** + * @param int<1, max> $maxTokens + * + * @throws \JsonException + */ + private function assertSyntaxErrorMessage(string $source, int $maxTokens): void + { + try { + Parser::parse($source, ['maxTokens' => $maxTokens]); + self::fail('Expected a SyntaxError'); + } catch (SyntaxError $error) { + self::assertSame("Syntax Error: Document contains more than {$maxTokens} tokens. Parsing aborted.", $error->getMessage()); + } + } } From af6de7fc275e2bdfb7c8212895aefc6458417e4b Mon Sep 17 00:00:00 2001 From: "autofix-ci[bot]" <114827586+autofix-ci[bot]@users.noreply.github.com> Date: Tue, 6 Oct 2026 18:50:10 +0000 Subject: [PATCH 2/3] Autofix --- docs/class-reference.md | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/docs/class-reference.md b/docs/class-reference.md index 81cd48e81..07c515f09 100644 --- a/docs/class-reference.md +++ b/docs/class-reference.md @@ -1180,7 +1180,8 @@ Parses string containing GraphQL query language or [schema definition language]( allowLegacySDLEmptyFields?: bool, allowLegacySDLImplementsInterfaces?: bool, experimentalFragmentVariables?: bool, - recursionLimit?: int<0, max> + recursionLimit?: int<0, max>, + maxTokens?: int<1, max>|null } ``` @@ -1216,6 +1217,12 @@ Parses string containing GraphQL query language or [schema definition language]( The counter is shared across `parseSelectionSet`, `parseValueLiteral`, and `parseTypeReference`. Defaults to 256. Set to 0 to disable the limit. +- **maxTokens**: + Parser CPU and memory usage is linear to the number of tokens in a document, + and parsing happens before validation, so even an invalid document can use a lot of resources. + Set this option to limit the number of tokens a document may contain, parsing then fails with a syntax error. + There is no limit by default. + Those magic functions allow partial parsing: @method static NameNode name(Source|string $source, ParserOptions $options = []) From 96cfc71e69aac4dc8ae1a9cf6a14615b9b7e30c1 Mon Sep 17 00:00:00 2001 From: Pascal CESCON - Amoifr Date: Wed, 7 Oct 2026 08:48:02 +0200 Subject: [PATCH 3/3] Reword the maxTokens option description --- docs/class-reference.md | 5 +++-- src/Language/Parser.php | 5 +++-- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/docs/class-reference.md b/docs/class-reference.md index 07c515f09..30bd7b20a 100644 --- a/docs/class-reference.md +++ b/docs/class-reference.md @@ -1218,9 +1218,10 @@ Parses string containing GraphQL query language or [schema definition language]( Defaults to 256. Set to 0 to disable the limit. - **maxTokens**: - Parser CPU and memory usage is linear to the number of tokens in a document, + Parser CPU and memory usage scales linearly with the number of tokens in a document, and parsing happens before validation, so even an invalid document can use a lot of resources. - Set this option to limit the number of tokens a document may contain, parsing then fails with a syntax error. + Set this option to limit the number of tokens a document may contain. + Parsing then fails with a syntax error. There is no limit by default. Those magic functions allow partial parsing: diff --git a/src/Language/Parser.php b/src/Language/Parser.php index 5f5ba6328..f7627668a 100644 --- a/src/Language/Parser.php +++ b/src/Language/Parser.php @@ -103,9 +103,10 @@ * Defaults to 256. Set to 0 to disable the limit. * * - **maxTokens**: - * Parser CPU and memory usage is linear to the number of tokens in a document, + * Parser CPU and memory usage scales linearly with the number of tokens in a document, * and parsing happens before validation, so even an invalid document can use a lot of resources. - * Set this option to limit the number of tokens a document may contain, parsing then fails with a syntax error. + * Set this option to limit the number of tokens a document may contain. + * Parsing then fails with a syntax error. * There is no limit by default. * * Those magic functions allow partial parsing: