From 4943bb405cd4880dc14883331da8c05bbed4a8f2 Mon Sep 17 00:00:00 2001 From: Fabien Potencier Date: Fri, 5 Jun 2026 20:18:56 +0200 Subject: [PATCH] Track the source offset of each token --- CHANGELOG | 3 +- src/Environment.php | 8 ++-- src/Lexer.php | 78 ++++++++++++++++++++++++++------------- src/Token.php | 17 +++++++++ tests/LexerTest.php | 89 +++++++++++++++++++++++++++++++++++++++++++++ 5 files changed, 165 insertions(+), 30 deletions(-) diff --git a/CHANGELOG b/CHANGELOG index 443f4f010..a45bdfb62 100644 --- a/CHANGELOG +++ b/CHANGELOG @@ -1,5 +1,6 @@ -# 3.27.2 (2026-XX-XX) +# 3.28.0 (2026-XX-XX) + * Track the source offset of each token and expose it via `Token::getOffset()` * Fix nested `block()` calls to resolve against the overriding template when a block rendered through `block(name, template)` calls `parent()` * Stop reporting a skipped test in `IntegrationTestCase` when there is no legacy test to run * Make the `IntegrationTestCase` and `NodeTestCase` test helpers compatible with PHPUnit 11 diff --git a/src/Environment.php b/src/Environment.php index ba9e8a18d..c6b1d8964 100644 --- a/src/Environment.php +++ b/src/Environment.php @@ -43,11 +43,11 @@ use Twig\TokenParser\TokenParserInterface; */ class Environment { - public const VERSION = '3.27.2-DEV'; - public const VERSION_ID = 32702; + public const VERSION = '3.28.0-DEV'; + public const VERSION_ID = 32800; public const MAJOR_VERSION = 3; - public const MINOR_VERSION = 27; - public const RELEASE_VERSION = 2; + public const MINOR_VERSION = 28; + public const RELEASE_VERSION = 0; public const EXTRA_VERSION = 'DEV'; private $charset; diff --git a/src/Lexer.php b/src/Lexer.php index e65f5bedc..60c4815ba 100644 --- a/src/Lexer.php +++ b/src/Lexer.php @@ -62,6 +62,8 @@ class Lexer public const REGEX_INLINE_COMMENT = '/#[^\n]*/A'; public const PUNCTUATION = '()[]{}?:.,|'; + private const REGEX_RAW_INLINE_COMMENT = '/#[^\r\n]*/A'; + private const SPECIAL_CHARS = [ 'f' => "\f", 'n' => "\n", @@ -113,7 +115,7 @@ class Lexer '|'. preg_quote($this->options['whitespace_line_trim'].$this->options['tag_block'][1], '#').'['.$this->options['whitespace_line_chars'].']*'. // ~%}[ \t\0\x0B]* '|'. - preg_quote($this->options['tag_block'][1], '#').'\n?'. // %}\n? + preg_quote($this->options['tag_block'][1], '#').'(?:\r\n?|\n)?'. // %}(?:\r\n?|\n)? ') }Ax', @@ -143,7 +145,7 @@ class Lexer '|'. preg_quote($this->options['whitespace_line_trim'].$this->options['tag_comment'][1], '#').'['.$this->options['whitespace_line_chars'].']*'. // ~#}[ \t\0\x0B]* '|'. - preg_quote($this->options['tag_comment'][1], '#').'\n?'. // #}\n? + preg_quote($this->options['tag_comment'][1], '#').'(?:\r\n?|\n)?'. // #}(?:\r\n?|\n)? ') }sx', @@ -187,7 +189,7 @@ class Lexer $this->initialize(); $this->source = $source; - $this->code = str_replace(["\r\n", "\r"], "\n", $source->getCode()); + $this->code = $source->getCode(); $this->cursor = 0; $this->lineno = 1; $this->end = \strlen($this->code); @@ -241,8 +243,9 @@ class Lexer { // if no matches are left we return the rest of the template as simple text token if ($this->position == \count($this->positions[0]) - 1) { - $this->pushToken(Token::TEXT_TYPE, substr($this->code, $this->cursor)); - $this->cursor = $this->end; + $text = substr($this->code, $this->cursor); + $this->pushToken(Token::TEXT_TYPE, $this->normalizeNewlines($text)); + $this->moveCursor($text); return; } @@ -270,15 +273,20 @@ class Lexer $text = rtrim($text, " \t\0\x0B"); } } - $this->pushToken(Token::TEXT_TYPE, $text); - $this->moveCursor($textContent.$position[0]); + $this->pushToken(Token::TEXT_TYPE, $this->normalizeNewlines($text)); + $this->moveCursor($textContent); switch ($this->positions[1][$this->position][0]) { case $this->options['tag_comment'][0]: + $this->moveCursor($position[0]); $this->lexComment(); break; case $this->options['tag_block'][0]: + $lineno = $this->lineno; + $cursor = $this->cursor; + $this->moveCursor($position[0]); + // raw data? if (preg_match($this->regexes['lex_block_raw'], $this->code, $match, 0, $this->cursor)) { $this->moveCursor($match[0]); @@ -288,14 +296,17 @@ class Lexer $this->moveCursor($match[0]); $this->lineno = (int) $match[1]; } else { - $this->pushToken(Token::BLOCK_START_TYPE); + $this->pushToken(Token::BLOCK_START_TYPE, '', $cursor, $lineno); $this->pushState(self::STATE_BLOCK); $this->currentVarBlockLine = $this->lineno; } break; case $this->options['tag_variable'][0]: - $this->pushToken(Token::VAR_START_TYPE); + $lineno = $this->lineno; + $cursor = $this->cursor; + $this->moveCursor($position[0]); + $this->pushToken(Token::VAR_START_TYPE, '', $cursor, $lineno); $this->pushState(self::STATE_VAR); $this->currentVarBlockLine = $this->lineno; break; @@ -305,8 +316,7 @@ class Lexer private function lexBlock(): void { if (!$this->brackets && preg_match($this->regexes['lex_block'], $this->code, $match, 0, $this->cursor)) { - $this->pushToken(Token::BLOCK_END_TYPE); - $this->moveCursor($match[0]); + $this->pushClosingToken(Token::BLOCK_END_TYPE, $match[0]); $this->popState(); } else { $this->lexExpression(); @@ -316,8 +326,7 @@ class Lexer private function lexVar(): void { if (!$this->brackets && preg_match($this->regexes['lex_var'], $this->code, $match, 0, $this->cursor)) { - $this->pushToken(Token::VAR_END_TYPE); - $this->moveCursor($match[0]); + $this->pushClosingToken(Token::VAR_END_TYPE, $match[0]); $this->popState(); } else { $this->lexExpression(); @@ -358,11 +367,11 @@ class Lexer elseif (str_contains(self::PUNCTUATION, $this->code[$this->cursor])) { $this->checkBrackets($this->code[$this->cursor]); $this->pushToken(Token::PUNCTUATION_TYPE, $this->code[$this->cursor]); - ++$this->cursor; + $this->moveCursor($this->code[$this->cursor]); } // strings elseif (preg_match(self::REGEX_STRING, $this->code, $match, 0, $this->cursor)) { - $this->pushToken(Token::STRING_TYPE, $this->stripcslashes(substr($match[0], 1, -1), substr($match[0], 0, 1))); + $this->pushToken(Token::STRING_TYPE, $this->stripcslashes($this->normalizeNewlines(substr($match[0], 1, -1)), substr($match[0], 0, 1))); $this->moveCursor($match[0]); } // opening double quoted string @@ -372,7 +381,7 @@ class Lexer $this->moveCursor($match[0]); } // inline comment - elseif (preg_match(self::REGEX_INLINE_COMMENT, $this->code, $match, 0, $this->cursor)) { + elseif (preg_match(self::REGEX_RAW_INLINE_COMMENT, $this->code, $match, 0, $this->cursor)) { $this->moveCursor($match[0]); } // unlexable @@ -444,6 +453,7 @@ class Lexer throw new SyntaxError('Unexpected end of file: Unclosed "verbatim" block.', $this->lineno, $this->source); } + $offset = $this->cursor; $text = substr($this->code, $this->cursor, $match[0][1] - $this->cursor); $this->moveCursor($text.$match[0][0]); @@ -459,7 +469,7 @@ class Lexer } } - $this->pushToken(Token::TEXT_TYPE, $text); + $this->pushToken(Token::TEXT_TYPE, $this->normalizeNewlines($text), $offset); } private function lexComment(): void @@ -479,7 +489,7 @@ class Lexer $this->moveCursor($match[0]); $this->pushState(self::STATE_INTERPOLATION); } elseif (preg_match(self::REGEX_DQ_STRING_PART, $this->code, $match, 0, $this->cursor) && '' !== $match[0]) { - $this->pushToken(Token::STRING_TYPE, $this->stripcslashes($match[0], '"')); + $this->pushToken(Token::STRING_TYPE, $this->stripcslashes($this->normalizeNewlines($match[0]), '"')); $this->moveCursor($match[0]); } elseif (preg_match(self::REGEX_DQ_STRING_DELIM, $this->code, $match, 0, $this->cursor)) { [$expect, $lineno] = array_pop($this->brackets); @@ -488,7 +498,7 @@ class Lexer } $this->popState(); - ++$this->cursor; + $this->moveCursor($match[0]); } else { // unlexable throw new SyntaxError(\sprintf('Unexpected character "%s".', $this->code[$this->cursor]), $this->lineno, $this->source); @@ -500,28 +510,46 @@ class Lexer $bracket = end($this->brackets); if ($this->options['interpolation'][0] === $bracket[0] && preg_match($this->regexes['interpolation_end'], $this->code, $match, 0, $this->cursor)) { array_pop($this->brackets); - $this->pushToken(Token::INTERPOLATION_END_TYPE); - $this->moveCursor($match[0]); + $this->pushClosingToken(Token::INTERPOLATION_END_TYPE, $match[0]); $this->popState(); } else { $this->lexExpression(); } } - private function pushToken($type, $value = ''): void + private function pushToken($type, $value = '', ?int $offset = null, ?int $lineno = null): void { // do not push empty text tokens if (Token::TEXT_TYPE === $type && '' === $value) { return; } - $this->tokens[] = new Token($type, $value, $this->lineno); + // by default the token starts at the current cursor; callers that + // emit a token after consuming it must pass an explicit offset + $this->tokens[] = new Token($type, $value, $lineno ?? $this->lineno, $offset ?? $this->cursor); } private function moveCursor($text): void { - $this->cursor += \strlen($text); - $this->lineno += substr_count($text, "\n"); + $length = \strlen($text); + $this->cursor += $length; + $this->lineno += substr_count($this->normalizeNewlines($text), "\n"); + } + + private function normalizeNewlines(string $text): string + { + return str_replace(["\r\n", "\r"], "\n", $text); + } + + private function pushClosingToken(int $type, string $match): void + { + $leadingWhitespaceLength = \strlen($match) - \strlen(ltrim($match)); + if ($leadingWhitespaceLength) { + $this->moveCursor(substr($match, 0, $leadingWhitespaceLength)); + } + + $this->pushToken($type); + $this->moveCursor(substr($match, $leadingWhitespaceLength)); } private function getOperatorRegex(): string diff --git a/src/Token.php b/src/Token.php index 823c77387..0c0e46a81 100644 --- a/src/Token.php +++ b/src/Token.php @@ -39,10 +39,14 @@ final class Token */ public const SPREAD_TYPE = 13; + /** + * @param non-negative-int|null $offset + */ public function __construct( private int $type, private $value, private int $lineno, + private ?int $offset = null, ) { if (self::ARROW_TYPE === $type) { trigger_deprecation('twig/twig', '3.21', 'The "%s" token type is deprecated, "arrow" is now an operator.', self::ARROW_TYPE); @@ -124,6 +128,19 @@ final class Token return $this->lineno; } + /** + * Returns the 0-based byte offset of the token in the source code. + * + * Returns null for tokens that are not tied to a source position (e.g. + * tokens synthesized by a token parser). + * + * @return non-negative-int|null + */ + public function getOffset(): ?int + { + return $this->offset; + } + /** * @deprecated since Twig 3.19 */ diff --git a/tests/LexerTest.php b/tests/LexerTest.php index 0b2ae755a..579aa7da4 100644 --- a/tests/LexerTest.php +++ b/tests/LexerTest.php @@ -734,4 +734,93 @@ bar yield ['{{ { a: 1 }}', '}']; yield ['{{ ([1] + 3)) }}', ')']; } + + public function testTokensCarryTheirSourceOffset() + { + $template = 'Hello {{ name }}!'; + + $lexer = new Lexer(new Environment(new ArrayLoader())); + $stream = $lexer->tokenize(new Source($template, 'index')); + + // [type, offset] pairs; offsets point at the start of each lexeme in the source + $expected = [ + [Token::TEXT_TYPE, 0], // "Hello " + [Token::VAR_START_TYPE, 6], // "{{" + [Token::NAME_TYPE, 9], // "name" + [Token::VAR_END_TYPE, 14], // "}}" + [Token::TEXT_TYPE, 16], // "!" + [Token::EOF_TYPE, 17], + ]; + + foreach ($expected as [$type, $offset]) { + $token = $stream->getCurrent(); + $this->assertTrue($token->test($type), \sprintf('Expected token "%s".', Token::typeToEnglish($type))); + $this->assertSame($offset, $token->getOffset()); + + if (!$stream->isEOF()) { + $stream->next(); + } + } + } + + public function testOffsetsAllowRecoveringTheRawExpressionSource() + { + $template = "Hello {{ name|upper ~ '!' }}"; + + $lexer = new Lexer(new Environment(new ArrayLoader())); + $stream = $lexer->tokenize(new Source($template, 'index')); + + $stream->expect(Token::TEXT_TYPE); + $start = $stream->expect(Token::VAR_START_TYPE)->getOffset(); + while (!$stream->test(Token::VAR_END_TYPE)) { + $stream->next(); + } + $end = $stream->getCurrent()->getOffset(); + + // slice the raw expression out of the original source, between "{{" and "}}" + $raw = trim(substr($template, $start + 2, $end - $start - 2)); + + $this->assertSame("name|upper ~ '!'", $raw); + } + + public function testOffsetsReferToTheOriginalSourceWhenLineEndingsAreNormalized() + { + $template = "Hello\r\n{{ name }}"; + + $lexer = new Lexer(new Environment(new ArrayLoader())); + $stream = $lexer->tokenize(new Source($template, 'index')); + + $stream->expect(Token::TEXT_TYPE); + $start = $stream->expect(Token::VAR_START_TYPE)->getOffset(); + $this->assertSame('{{', substr($template, $start, 2)); + + $name = $stream->expect(Token::NAME_TYPE); + $this->assertSame('name', substr($template, $name->getOffset(), 4)); + + $end = $stream->expect(Token::VAR_END_TYPE)->getOffset(); + $this->assertSame('name', trim(substr($template, $start + 2, $end - $start - 2))); + } + + public function testBlockTagDelimitersPointAtTheMarkers() + { + $template = '{% set x = 1 %}'; + + $lexer = new Lexer(new Environment(new ArrayLoader())); + $stream = $lexer->tokenize(new Source($template, 'index')); + + // the opening "{%" and the closing "%}" both point at the marker itself, + // not at the whitespace the closing regex also consumes + $this->assertSame(0, $stream->expect(Token::BLOCK_START_TYPE)->getOffset()); + $this->assertSame('{%', substr($template, 0, 2)); + while (!$stream->test(Token::BLOCK_END_TYPE)) { + $stream->next(); + } + $end = $stream->getCurrent()->getOffset(); + $this->assertSame('%}', substr($template, $end, 2)); + } + + public function testSyntheticTokensHaveNoOffset() + { + $this->assertNull((new Token(Token::NAME_TYPE, 'foo', 1))->getOffset()); + } }