mirror of
https://github.com/twigphp/Twig.git
synced 2026-10-03 10:27:22 +00:00
Track the source offset of each token
This commit is contained in:
@@ -1,5 +1,6 @@
|
||||
# 3.27.2 (2026-XX-XX)
|
||||
# 3.28.0 (2026-XX-XX)
|
||||
|
||||
* Track the source offset of each token and expose it via `Token::getOffset()`
|
||||
* Fix nested `block()` calls to resolve against the overriding template when a block rendered through `block(name, template)` calls `parent()`
|
||||
* Stop reporting a skipped test in `IntegrationTestCase` when there is no legacy test to run
|
||||
* Make the `IntegrationTestCase` and `NodeTestCase` test helpers compatible with PHPUnit 11
|
||||
|
||||
+4
-4
@@ -43,11 +43,11 @@ use Twig\TokenParser\TokenParserInterface;
|
||||
*/
|
||||
class Environment
|
||||
{
|
||||
public const VERSION = '3.27.2-DEV';
|
||||
public const VERSION_ID = 32702;
|
||||
public const VERSION = '3.28.0-DEV';
|
||||
public const VERSION_ID = 32800;
|
||||
public const MAJOR_VERSION = 3;
|
||||
public const MINOR_VERSION = 27;
|
||||
public const RELEASE_VERSION = 2;
|
||||
public const MINOR_VERSION = 28;
|
||||
public const RELEASE_VERSION = 0;
|
||||
public const EXTRA_VERSION = 'DEV';
|
||||
|
||||
private $charset;
|
||||
|
||||
+53
-25
@@ -62,6 +62,8 @@ class Lexer
|
||||
public const REGEX_INLINE_COMMENT = '/#[^\n]*/A';
|
||||
public const PUNCTUATION = '()[]{}?:.,|';
|
||||
|
||||
private const REGEX_RAW_INLINE_COMMENT = '/#[^\r\n]*/A';
|
||||
|
||||
private const SPECIAL_CHARS = [
|
||||
'f' => "\f",
|
||||
'n' => "\n",
|
||||
@@ -113,7 +115,7 @@ class Lexer
|
||||
'|'.
|
||||
preg_quote($this->options['whitespace_line_trim'].$this->options['tag_block'][1], '#').'['.$this->options['whitespace_line_chars'].']*'. // ~%}[ \t\0\x0B]*
|
||||
'|'.
|
||||
preg_quote($this->options['tag_block'][1], '#').'\n?'. // %}\n?
|
||||
preg_quote($this->options['tag_block'][1], '#').'(?:\r\n?|\n)?'. // %}(?:\r\n?|\n)?
|
||||
')
|
||||
}Ax',
|
||||
|
||||
@@ -143,7 +145,7 @@ class Lexer
|
||||
'|'.
|
||||
preg_quote($this->options['whitespace_line_trim'].$this->options['tag_comment'][1], '#').'['.$this->options['whitespace_line_chars'].']*'. // ~#}[ \t\0\x0B]*
|
||||
'|'.
|
||||
preg_quote($this->options['tag_comment'][1], '#').'\n?'. // #}\n?
|
||||
preg_quote($this->options['tag_comment'][1], '#').'(?:\r\n?|\n)?'. // #}(?:\r\n?|\n)?
|
||||
')
|
||||
}sx',
|
||||
|
||||
@@ -187,7 +189,7 @@ class Lexer
|
||||
$this->initialize();
|
||||
|
||||
$this->source = $source;
|
||||
$this->code = str_replace(["\r\n", "\r"], "\n", $source->getCode());
|
||||
$this->code = $source->getCode();
|
||||
$this->cursor = 0;
|
||||
$this->lineno = 1;
|
||||
$this->end = \strlen($this->code);
|
||||
@@ -241,8 +243,9 @@ class Lexer
|
||||
{
|
||||
// if no matches are left we return the rest of the template as simple text token
|
||||
if ($this->position == \count($this->positions[0]) - 1) {
|
||||
$this->pushToken(Token::TEXT_TYPE, substr($this->code, $this->cursor));
|
||||
$this->cursor = $this->end;
|
||||
$text = substr($this->code, $this->cursor);
|
||||
$this->pushToken(Token::TEXT_TYPE, $this->normalizeNewlines($text));
|
||||
$this->moveCursor($text);
|
||||
|
||||
return;
|
||||
}
|
||||
@@ -270,15 +273,20 @@ class Lexer
|
||||
$text = rtrim($text, " \t\0\x0B");
|
||||
}
|
||||
}
|
||||
$this->pushToken(Token::TEXT_TYPE, $text);
|
||||
$this->moveCursor($textContent.$position[0]);
|
||||
$this->pushToken(Token::TEXT_TYPE, $this->normalizeNewlines($text));
|
||||
$this->moveCursor($textContent);
|
||||
|
||||
switch ($this->positions[1][$this->position][0]) {
|
||||
case $this->options['tag_comment'][0]:
|
||||
$this->moveCursor($position[0]);
|
||||
$this->lexComment();
|
||||
break;
|
||||
|
||||
case $this->options['tag_block'][0]:
|
||||
$lineno = $this->lineno;
|
||||
$cursor = $this->cursor;
|
||||
$this->moveCursor($position[0]);
|
||||
|
||||
// raw data?
|
||||
if (preg_match($this->regexes['lex_block_raw'], $this->code, $match, 0, $this->cursor)) {
|
||||
$this->moveCursor($match[0]);
|
||||
@@ -288,14 +296,17 @@ class Lexer
|
||||
$this->moveCursor($match[0]);
|
||||
$this->lineno = (int) $match[1];
|
||||
} else {
|
||||
$this->pushToken(Token::BLOCK_START_TYPE);
|
||||
$this->pushToken(Token::BLOCK_START_TYPE, '', $cursor, $lineno);
|
||||
$this->pushState(self::STATE_BLOCK);
|
||||
$this->currentVarBlockLine = $this->lineno;
|
||||
}
|
||||
break;
|
||||
|
||||
case $this->options['tag_variable'][0]:
|
||||
$this->pushToken(Token::VAR_START_TYPE);
|
||||
$lineno = $this->lineno;
|
||||
$cursor = $this->cursor;
|
||||
$this->moveCursor($position[0]);
|
||||
$this->pushToken(Token::VAR_START_TYPE, '', $cursor, $lineno);
|
||||
$this->pushState(self::STATE_VAR);
|
||||
$this->currentVarBlockLine = $this->lineno;
|
||||
break;
|
||||
@@ -305,8 +316,7 @@ class Lexer
|
||||
private function lexBlock(): void
|
||||
{
|
||||
if (!$this->brackets && preg_match($this->regexes['lex_block'], $this->code, $match, 0, $this->cursor)) {
|
||||
$this->pushToken(Token::BLOCK_END_TYPE);
|
||||
$this->moveCursor($match[0]);
|
||||
$this->pushClosingToken(Token::BLOCK_END_TYPE, $match[0]);
|
||||
$this->popState();
|
||||
} else {
|
||||
$this->lexExpression();
|
||||
@@ -316,8 +326,7 @@ class Lexer
|
||||
private function lexVar(): void
|
||||
{
|
||||
if (!$this->brackets && preg_match($this->regexes['lex_var'], $this->code, $match, 0, $this->cursor)) {
|
||||
$this->pushToken(Token::VAR_END_TYPE);
|
||||
$this->moveCursor($match[0]);
|
||||
$this->pushClosingToken(Token::VAR_END_TYPE, $match[0]);
|
||||
$this->popState();
|
||||
} else {
|
||||
$this->lexExpression();
|
||||
@@ -358,11 +367,11 @@ class Lexer
|
||||
elseif (str_contains(self::PUNCTUATION, $this->code[$this->cursor])) {
|
||||
$this->checkBrackets($this->code[$this->cursor]);
|
||||
$this->pushToken(Token::PUNCTUATION_TYPE, $this->code[$this->cursor]);
|
||||
++$this->cursor;
|
||||
$this->moveCursor($this->code[$this->cursor]);
|
||||
}
|
||||
// strings
|
||||
elseif (preg_match(self::REGEX_STRING, $this->code, $match, 0, $this->cursor)) {
|
||||
$this->pushToken(Token::STRING_TYPE, $this->stripcslashes(substr($match[0], 1, -1), substr($match[0], 0, 1)));
|
||||
$this->pushToken(Token::STRING_TYPE, $this->stripcslashes($this->normalizeNewlines(substr($match[0], 1, -1)), substr($match[0], 0, 1)));
|
||||
$this->moveCursor($match[0]);
|
||||
}
|
||||
// opening double quoted string
|
||||
@@ -372,7 +381,7 @@ class Lexer
|
||||
$this->moveCursor($match[0]);
|
||||
}
|
||||
// inline comment
|
||||
elseif (preg_match(self::REGEX_INLINE_COMMENT, $this->code, $match, 0, $this->cursor)) {
|
||||
elseif (preg_match(self::REGEX_RAW_INLINE_COMMENT, $this->code, $match, 0, $this->cursor)) {
|
||||
$this->moveCursor($match[0]);
|
||||
}
|
||||
// unlexable
|
||||
@@ -444,6 +453,7 @@ class Lexer
|
||||
throw new SyntaxError('Unexpected end of file: Unclosed "verbatim" block.', $this->lineno, $this->source);
|
||||
}
|
||||
|
||||
$offset = $this->cursor;
|
||||
$text = substr($this->code, $this->cursor, $match[0][1] - $this->cursor);
|
||||
$this->moveCursor($text.$match[0][0]);
|
||||
|
||||
@@ -459,7 +469,7 @@ class Lexer
|
||||
}
|
||||
}
|
||||
|
||||
$this->pushToken(Token::TEXT_TYPE, $text);
|
||||
$this->pushToken(Token::TEXT_TYPE, $this->normalizeNewlines($text), $offset);
|
||||
}
|
||||
|
||||
private function lexComment(): void
|
||||
@@ -479,7 +489,7 @@ class Lexer
|
||||
$this->moveCursor($match[0]);
|
||||
$this->pushState(self::STATE_INTERPOLATION);
|
||||
} elseif (preg_match(self::REGEX_DQ_STRING_PART, $this->code, $match, 0, $this->cursor) && '' !== $match[0]) {
|
||||
$this->pushToken(Token::STRING_TYPE, $this->stripcslashes($match[0], '"'));
|
||||
$this->pushToken(Token::STRING_TYPE, $this->stripcslashes($this->normalizeNewlines($match[0]), '"'));
|
||||
$this->moveCursor($match[0]);
|
||||
} elseif (preg_match(self::REGEX_DQ_STRING_DELIM, $this->code, $match, 0, $this->cursor)) {
|
||||
[$expect, $lineno] = array_pop($this->brackets);
|
||||
@@ -488,7 +498,7 @@ class Lexer
|
||||
}
|
||||
|
||||
$this->popState();
|
||||
++$this->cursor;
|
||||
$this->moveCursor($match[0]);
|
||||
} else {
|
||||
// unlexable
|
||||
throw new SyntaxError(\sprintf('Unexpected character "%s".', $this->code[$this->cursor]), $this->lineno, $this->source);
|
||||
@@ -500,28 +510,46 @@ class Lexer
|
||||
$bracket = end($this->brackets);
|
||||
if ($this->options['interpolation'][0] === $bracket[0] && preg_match($this->regexes['interpolation_end'], $this->code, $match, 0, $this->cursor)) {
|
||||
array_pop($this->brackets);
|
||||
$this->pushToken(Token::INTERPOLATION_END_TYPE);
|
||||
$this->moveCursor($match[0]);
|
||||
$this->pushClosingToken(Token::INTERPOLATION_END_TYPE, $match[0]);
|
||||
$this->popState();
|
||||
} else {
|
||||
$this->lexExpression();
|
||||
}
|
||||
}
|
||||
|
||||
private function pushToken($type, $value = ''): void
|
||||
private function pushToken($type, $value = '', ?int $offset = null, ?int $lineno = null): void
|
||||
{
|
||||
// do not push empty text tokens
|
||||
if (Token::TEXT_TYPE === $type && '' === $value) {
|
||||
return;
|
||||
}
|
||||
|
||||
$this->tokens[] = new Token($type, $value, $this->lineno);
|
||||
// by default the token starts at the current cursor; callers that
|
||||
// emit a token after consuming it must pass an explicit offset
|
||||
$this->tokens[] = new Token($type, $value, $lineno ?? $this->lineno, $offset ?? $this->cursor);
|
||||
}
|
||||
|
||||
private function moveCursor($text): void
|
||||
{
|
||||
$this->cursor += \strlen($text);
|
||||
$this->lineno += substr_count($text, "\n");
|
||||
$length = \strlen($text);
|
||||
$this->cursor += $length;
|
||||
$this->lineno += substr_count($this->normalizeNewlines($text), "\n");
|
||||
}
|
||||
|
||||
private function normalizeNewlines(string $text): string
|
||||
{
|
||||
return str_replace(["\r\n", "\r"], "\n", $text);
|
||||
}
|
||||
|
||||
private function pushClosingToken(int $type, string $match): void
|
||||
{
|
||||
$leadingWhitespaceLength = \strlen($match) - \strlen(ltrim($match));
|
||||
if ($leadingWhitespaceLength) {
|
||||
$this->moveCursor(substr($match, 0, $leadingWhitespaceLength));
|
||||
}
|
||||
|
||||
$this->pushToken($type);
|
||||
$this->moveCursor(substr($match, $leadingWhitespaceLength));
|
||||
}
|
||||
|
||||
private function getOperatorRegex(): string
|
||||
|
||||
@@ -39,10 +39,14 @@ final class Token
|
||||
*/
|
||||
public const SPREAD_TYPE = 13;
|
||||
|
||||
/**
|
||||
* @param non-negative-int|null $offset
|
||||
*/
|
||||
public function __construct(
|
||||
private int $type,
|
||||
private $value,
|
||||
private int $lineno,
|
||||
private ?int $offset = null,
|
||||
) {
|
||||
if (self::ARROW_TYPE === $type) {
|
||||
trigger_deprecation('twig/twig', '3.21', 'The "%s" token type is deprecated, "arrow" is now an operator.', self::ARROW_TYPE);
|
||||
@@ -124,6 +128,19 @@ final class Token
|
||||
return $this->lineno;
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns the 0-based byte offset of the token in the source code.
|
||||
*
|
||||
* Returns null for tokens that are not tied to a source position (e.g.
|
||||
* tokens synthesized by a token parser).
|
||||
*
|
||||
* @return non-negative-int|null
|
||||
*/
|
||||
public function getOffset(): ?int
|
||||
{
|
||||
return $this->offset;
|
||||
}
|
||||
|
||||
/**
|
||||
* @deprecated since Twig 3.19
|
||||
*/
|
||||
|
||||
@@ -734,4 +734,93 @@ bar
|
||||
yield ['{{ { a: 1 }}', '}'];
|
||||
yield ['{{ ([1] + 3)) }}', ')'];
|
||||
}
|
||||
|
||||
public function testTokensCarryTheirSourceOffset()
|
||||
{
|
||||
$template = 'Hello {{ name }}!';
|
||||
|
||||
$lexer = new Lexer(new Environment(new ArrayLoader()));
|
||||
$stream = $lexer->tokenize(new Source($template, 'index'));
|
||||
|
||||
// [type, offset] pairs; offsets point at the start of each lexeme in the source
|
||||
$expected = [
|
||||
[Token::TEXT_TYPE, 0], // "Hello "
|
||||
[Token::VAR_START_TYPE, 6], // "{{"
|
||||
[Token::NAME_TYPE, 9], // "name"
|
||||
[Token::VAR_END_TYPE, 14], // "}}"
|
||||
[Token::TEXT_TYPE, 16], // "!"
|
||||
[Token::EOF_TYPE, 17],
|
||||
];
|
||||
|
||||
foreach ($expected as [$type, $offset]) {
|
||||
$token = $stream->getCurrent();
|
||||
$this->assertTrue($token->test($type), \sprintf('Expected token "%s".', Token::typeToEnglish($type)));
|
||||
$this->assertSame($offset, $token->getOffset());
|
||||
|
||||
if (!$stream->isEOF()) {
|
||||
$stream->next();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
public function testOffsetsAllowRecoveringTheRawExpressionSource()
|
||||
{
|
||||
$template = "Hello {{ name|upper ~ '!' }}";
|
||||
|
||||
$lexer = new Lexer(new Environment(new ArrayLoader()));
|
||||
$stream = $lexer->tokenize(new Source($template, 'index'));
|
||||
|
||||
$stream->expect(Token::TEXT_TYPE);
|
||||
$start = $stream->expect(Token::VAR_START_TYPE)->getOffset();
|
||||
while (!$stream->test(Token::VAR_END_TYPE)) {
|
||||
$stream->next();
|
||||
}
|
||||
$end = $stream->getCurrent()->getOffset();
|
||||
|
||||
// slice the raw expression out of the original source, between "{{" and "}}"
|
||||
$raw = trim(substr($template, $start + 2, $end - $start - 2));
|
||||
|
||||
$this->assertSame("name|upper ~ '!'", $raw);
|
||||
}
|
||||
|
||||
public function testOffsetsReferToTheOriginalSourceWhenLineEndingsAreNormalized()
|
||||
{
|
||||
$template = "Hello\r\n{{ name }}";
|
||||
|
||||
$lexer = new Lexer(new Environment(new ArrayLoader()));
|
||||
$stream = $lexer->tokenize(new Source($template, 'index'));
|
||||
|
||||
$stream->expect(Token::TEXT_TYPE);
|
||||
$start = $stream->expect(Token::VAR_START_TYPE)->getOffset();
|
||||
$this->assertSame('{{', substr($template, $start, 2));
|
||||
|
||||
$name = $stream->expect(Token::NAME_TYPE);
|
||||
$this->assertSame('name', substr($template, $name->getOffset(), 4));
|
||||
|
||||
$end = $stream->expect(Token::VAR_END_TYPE)->getOffset();
|
||||
$this->assertSame('name', trim(substr($template, $start + 2, $end - $start - 2)));
|
||||
}
|
||||
|
||||
public function testBlockTagDelimitersPointAtTheMarkers()
|
||||
{
|
||||
$template = '{% set x = 1 %}';
|
||||
|
||||
$lexer = new Lexer(new Environment(new ArrayLoader()));
|
||||
$stream = $lexer->tokenize(new Source($template, 'index'));
|
||||
|
||||
// the opening "{%" and the closing "%}" both point at the marker itself,
|
||||
// not at the whitespace the closing regex also consumes
|
||||
$this->assertSame(0, $stream->expect(Token::BLOCK_START_TYPE)->getOffset());
|
||||
$this->assertSame('{%', substr($template, 0, 2));
|
||||
while (!$stream->test(Token::BLOCK_END_TYPE)) {
|
||||
$stream->next();
|
||||
}
|
||||
$end = $stream->getCurrent()->getOffset();
|
||||
$this->assertSame('%}', substr($template, $end, 2));
|
||||
}
|
||||
|
||||
public function testSyntheticTokensHaveNoOffset()
|
||||
{
|
||||
$this->assertNull((new Token(Token::NAME_TYPE, 'foo', 1))->getOffset());
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user