packages/bison-parser/src/Scanner/Scanner.php

1<?php
2
3declare(strict_types=1);
4
5namespace BisonParser\Scanner;
6
7use BisonParser\Ast\Location;
8use BisonParser\SyntaxException;
9
10/**
11 * Splits a grammar file into the tokens Bison's own scanner produces.
12 *
13 * This follows `scan-gram.l`: an identifier is looked past to see whether
14 * a colon or a bracketed name follows, a `%name-prefix` may be followed by
15 * `=`, a stray comma is whitespace, and everything after the second `%%`
16 * is one epilogue token.
17 *
18 * @visibility root
19 */
20final class Scanner
21{
22    /**
23     * Pattern of an identifier: a letter, underscore or dot, then also digits and hyphens.
24     */
25    public const IDENTIFIER = '[.A-Za-z_][-.A-Za-z0-9_]*';
26
27    /**
28     * @param CodeReader $code Reads host code
29     * @param Escapes $escapes Decodes literals
30     */
31    public function __construct(
32        private readonly CodeReader $code = new CodeReader(),
33        private readonly Escapes $escapes = new Escapes(),
34    ) {
35    }
36
37    /**
38     * Tokenizes a grammar file.
39     *
40     * @param string $source Contents of the file
41     *
42     * @return list<Token> The tokens, ending with the end-of-file token
43     *
44     * @throws SyntaxException When the file holds something the grammar language does not allow
45     */
46    public function scan(string $source): array
47    {
48        $cursor = new Cursor($source);
49        $tokens = [];
50        $sections = 0;
51        while (true) {
52            $this->skipTrivia($cursor);
53            if ($cursor->eof()) {
54                $tokens[] = new Token(TokenKind::End, '', $cursor->location());
55
56                return $tokens;
57            }
58            foreach ($this->next($cursor) as $token) {
59                $tokens[] = $token;
60            }
61            $last = $tokens[count($tokens) - 1];
62            if ($last->is(TokenKind::Section) && ++$sections === 2) {
63                $location = $cursor->location();
64                $tokens[] = new Token(TokenKind::Epilogue, $cursor->take(strlen($source)), $location);
65                $tokens[] = new Token(TokenKind::End, '', $cursor->location());
66
67                return $tokens;
68            }
69        }
70    }
71
72    /**
73     * Skips whitespace, comments, stray commas and `#line` directives.
74     *
75     * @param Cursor $cursor Cursor to advance
76     *
77     * @throws SyntaxException When a comment never closes
78     */
79    public function skipTrivia(Cursor $cursor): void
80    {
81        while (!$cursor->eof()) {
82            if ($cursor->match('[ \f\t\v\r\n,]+') !== null || $cursor->match('//[^\n]*') !== null) {
83                continue;
84            }
85            if ($cursor->startsWith('/*')) {
86                $location = $cursor->location();
87                if ($cursor->takeUntil('*/') === null) {
88                    throw SyntaxException::unterminated('comment', '*/', $location);
89                }
90                continue;
91            }
92
93            return;
94        }
95    }
96
97    /**
98     * Reads the token, or the tokens, that begin at the cursor.
99     *
100     * @param Cursor $cursor Cursor positioned on a token
101     *
102     * @return list<Token> One token, or an identifier and its bracketed name
103     *
104     * @throws SyntaxException When no token begins here or it never closes
105     */
106    public function next(Cursor $cursor): array
107    {
108        $location = $cursor->location();
109        $byte = $cursor->peek();
110        if ($byte === '#' && $location->column === 1) {
111            $directive = $cursor->match('#line [0-9]+(?: "[^"\n]*")?(?=\r?\n)');
112            if ($directive !== null) {
113                return [new Token(TokenKind::Line, $directive, $location)];
114            }
115        }
116        if ($byte === '%') {
117            return [$this->percent($cursor, $location)];
118        }
119        if ($byte === '{') {
120            return [new Token(TokenKind::Code, $this->code->braced($cursor), $location)];
121        }
122        if (preg_match('/[.A-Za-z_]/', $byte) === 1 && !$cursor->startsWith('_("')) {
123            return $this->identifier($cursor, $location);
124        }
125        if (ctype_digit($byte)) {
126            return [$this->integer($cursor, $location)];
127        }
128        $token = $this->literal($cursor, $location) ?? $this->punctuation($cursor, $location);
129        if ($token === null) {
130            throw SyntaxException::invalid("invalid character: '{$byte}'", $location);
131        }
132
133        return [$token];
134    }
135
136    /**
137     * Reads what begins with a percent sign: a prologue, a section marker, a predicate or a directive.
138     *
139     * @param Cursor $cursor Cursor positioned on the percent sign
140     * @param Location $location Where it is
141     *
142     * @return Token The token
143     *
144     * @throws SyntaxException When the directive is unknown or the code never closes
145     */
146    public function percent(Cursor $cursor, Location $location): Token
147    {
148        if ($cursor->startsWith('%{')) {
149            return new Token(TokenKind::Prologue, $this->code->prologue($cursor), $location);
150        }
151        if ($cursor->startsWith('%%')) {
152            $cursor->take(2);
153
154            return new Token(TokenKind::Section, '%%', $location);
155        }
156        if ($cursor->match('%\?[ \f\t\v]*(?:\r?\n[ \f\t\v]*)*(?=\{)') !== null) {
157            return new Token(TokenKind::Predicate, $this->code->braced($cursor), $location);
158        }
159        $raw = $cursor->match('%' . self::IDENTIFIER);
160        $name = $raw === null ? null : Directives::canonical(substr($raw, 1));
161        if ($raw === null || $name === null) {
162            throw SyntaxException::invalid('invalid directive: ' . ($raw ?? '%'), $location);
163        }
164        if (in_array($name, Directives::EQUAL_OPTIONAL, true)) {
165            $cursor->match('\s*=\s*');
166        }
167
168        return new Token(TokenKind::Directive, $name, $location, $raw);
169    }
170
171    /**
172     * Reads an identifier and looks past it for a bracketed name and a colon.
173     *
174     * @param Cursor $cursor Cursor positioned on the identifier
175     * @param Location $location Where it is
176     *
177     * @return list<Token> The identifier, followed by its bracketed name when one follows it
178     *
179     * @throws SyntaxException When a bracketed name is malformed
180     */
181    public function identifier(Cursor $cursor, Location $location): array
182    {
183        $name = $cursor->match(self::IDENTIFIER) ?? '';
184        $this->skipTrivia($cursor);
185        $bracketed = null;
186        if ($cursor->peek() === '[') {
187            $bracketed = $this->bracketed($cursor);
188            $this->skipTrivia($cursor);
189        }
190        $kind = $cursor->peek() === ':' ? TokenKind::IdentifierColon : TokenKind::Identifier;
191        $tokens = [new Token($kind, $name, $location)];
192        if ($bracketed !== null) {
193            $tokens[] = $bracketed;
194        }
195
196        return $tokens;
197    }
198
199    /**
200     * Reads a bracketed name such as `[left]`.
201     *
202     * @param Cursor $cursor Cursor positioned on the opening bracket
203     *
204     * @return Token The bracketed identifier
205     *
206     * @throws SyntaxException When the brackets hold anything but one identifier
207     */
208    public function bracketed(Cursor $cursor): Token
209    {
210        $location = $cursor->location();
211        $cursor->take(1);
212        $this->skipTrivia($cursor);
213        $name = $cursor->match(self::IDENTIFIER);
214        if ($name === null) {
215            throw SyntaxException::invalid('an identifier expected', $cursor->location());
216        }
217        $this->skipTrivia($cursor);
218        if ($cursor->peek() !== ']') {
219            throw SyntaxException::unexpected("']'", $cursor->eof() ? 'end of file' : "'{$cursor->peek()}'", $cursor->location());
220        }
221        $cursor->take(1);
222
223        return new Token(TokenKind::BracketedIdentifier, $name, $location);
224    }
225
226    /**
227     * Reads a decimal or hexadecimal integer.
228     *
229     * @param Cursor $cursor Cursor positioned on the first digit
230     * @param Location $location Where it is
231     *
232     * @return Token The integer, its text in decimal
233     *
234     * @throws SyntaxException When letters follow the digits
235     */
236    public function integer(Cursor $cursor, Location $location): Token
237    {
238        $text = $cursor->match('0[xX][0-9A-Fa-f]+|[0-9]+') ?? '';
239        if ($cursor->lookingAt('[-.A-Za-z0-9_]')) {
240            throw SyntaxException::invalid('invalid identifier: ' . $text . $cursor->match('[-.A-Za-z0-9_]*'), $location);
241        }
242        $hex = str_starts_with(strtolower($text), '0x');
243        $digits = ltrim($hex ? substr($text, 2) : $text, '0');
244        if (strlen($digits) > ($hex ? 8 : 10) || ($hex ? hexdec($digits) : (int) $digits) > 0x7FFFFFFF) {
245            throw SyntaxException::invalid("integer out of range: '{$text}'", $location);
246        }
247        $value = $hex ? hexdec($digits) : (int) $digits;
248
249        return new Token(TokenKind::Integer, (string) $value, $location, $text);
250    }
251
252    /**
253     * Reads a character literal, a string, a translatable string, or a tag.
254     *
255     * @param Cursor $cursor Cursor positioned on the literal
256     * @param Location $location Where it is
257     *
258     * @return Token|null The token, or null when no literal begins here
259     *
260     * @throws SyntaxException When the literal never closes or a character literal is not one byte
261     */
262    public function literal(Cursor $cursor, Location $location): ?Token
263    {
264        $byte = $cursor->peek();
265        if ($byte === "'") {
266            $literal = $cursor->match("'(?:\\\\.|[^\\\\'\\n])*'");
267            if ($literal === null) {
268                throw SyntaxException::unterminated('character literal', "'", $location);
269            }
270            $decoded = $this->escapes->decode(substr($literal, 1, -1), $location);
271            if ($decoded === '') {
272                throw SyntaxException::invalid('empty character literal', $location);
273            }
274            if (strlen($decoded) !== 1) {
275                throw SyntaxException::invalid('extra characters in character literal', $location);
276            }
277
278            return new Token(TokenKind::CharLiteral, $decoded, $location, $literal);
279        }
280        if ($byte === '"' || $cursor->startsWith('_("')) {
281            $translatable = $byte === '_';
282            $literal = $cursor->match($translatable ? '_\("(?:\\\\.|[^\\\\"\n])*"\)' : '"(?:\\\\.|[^\\\\"\n])*"');
283            if ($literal === null) {
284                throw SyntaxException::unterminated('string', '"', $location);
285            }
286            $body = $translatable ? substr($literal, 3, -2) : substr($literal, 1, -1);
287
288            return new Token($translatable ? TokenKind::TranslatableString : TokenKind::String, $this->escapes->decode($body, $location), $location, $literal);
289        }
290        if ($byte === '<') {
291            return $this->tag($cursor, $location);
292        }
293
294        return null;
295    }
296
297    /**
298     * Reads a tag, which may nest angle brackets as a C++ template does.
299     *
300     * @param Cursor $cursor Cursor positioned on the opening angle bracket
301     * @param Location $location Where it is
302     *
303     * @return Token The tag
304     *
305     * @throws SyntaxException When the tag never closes
306     */
307    public function tag(Cursor $cursor, Location $location): Token
308    {
309        if ($cursor->startsWith('<*>')) {
310            $cursor->take(3);
311
312            return new Token(TokenKind::TagAny, '*', $location);
313        }
314        if ($cursor->startsWith('<>')) {
315            $cursor->take(2);
316
317            return new Token(TokenKind::TagNone, '', $location);
318        }
319        $cursor->take(1);
320        $nesting = 0;
321        $text = '';
322        while (!$cursor->eof()) {
323            $unit = $cursor->match('(?:->|[^<>])+|<+') ?? $cursor->take(1);
324            if ($unit === '>') {
325                if ($nesting === 0) {
326                    return new Token(TokenKind::Tag, $text, $location);
327                }
328                $nesting--;
329            } elseif ($unit[0] === '<') {
330                $nesting += strlen($unit);
331            }
332            $text .= $unit;
333        }
334
335        throw SyntaxException::unterminated('tag', '>', $location);
336    }
337
338    /**
339     * Reads one punctuation byte.
340     *
341     * @param Cursor $cursor Cursor positioned on the byte
342     * @param Location $location Where it is
343     *
344     * @return Token|null The token, or null when the byte is not punctuation the language has
345     *
346     * @throws SyntaxException When a bracketed name is malformed
347     */
348    public function punctuation(Cursor $cursor, Location $location): ?Token
349    {
350        $byte = $cursor->peek();
351        if ($byte === '[') {
352            return $this->bracketed($cursor);
353        }
354        $kind = match ($byte) {
355            ':' => TokenKind::Colon,
356            '=' => TokenKind::Equal,
357            '|' => TokenKind::Pipe,
358            ';' => TokenKind::Semicolon,
359            default => null,
360        };
361        if ($kind === null) {
362            return null;
363        }
364        $cursor->take(1);
365
366        return new Token($kind, $byte, $location);
367    }
368}
369