packages/bison-parser/src/Scanner/Scanner.php
1<?php
2
3declare(strict_types=1);
4
5namespace BisonParser\Scanner;
6
7use BisonParser\Ast\Location;
8use BisonParser\SyntaxException;
9
10/**
11 * Splits a grammar file into the tokens Bison's own scanner produces.
12 *
13 * This follows `scan-gram.l`: an identifier is looked past to see whether
14 * a colon or a bracketed name follows, a `%name-prefix` may be followed by
15 * `=`, a stray comma is whitespace, and everything after the second `%%`
16 * is one epilogue token.
17 *
18 * @visibility root
19 */
20final class Scanner
21{
22 /**
23 * Pattern of an identifier: a letter, underscore or dot, then also digits and hyphens.
24 */
25 public const IDENTIFIER = '[.A-Za-z_][-.A-Za-z0-9_]*';
26
27 /**
28 * @param CodeReader $code Reads host code
29 * @param Escapes $escapes Decodes literals
30 */
31 public function __construct(
32 private readonly CodeReader $code = new CodeReader(),
33 private readonly Escapes $escapes = new Escapes(),
34 ) {
35 }
36
37 /**
38 * Tokenizes a grammar file.
39 *
40 * @param string $source Contents of the file
41 *
42 * @return list<Token> The tokens, ending with the end-of-file token
43 *
44 * @throws SyntaxException When the file holds something the grammar language does not allow
45 */
46 public function scan(string $source): array
47 {
48 $cursor = new Cursor($source);
49 $tokens = [];
50 $sections = 0;
51 while (true) {
52 $this->skipTrivia($cursor);
53 if ($cursor->eof()) {
54 $tokens[] = new Token(TokenKind::End, '', $cursor->location());
55
56 return $tokens;
57 }
58 foreach ($this->next($cursor) as $token) {
59 $tokens[] = $token;
60 }
61 $last = $tokens[count($tokens) - 1];
62 if ($last->is(TokenKind::Section) && ++$sections === 2) {
63 $location = $cursor->location();
64 $tokens[] = new Token(TokenKind::Epilogue, $cursor->take(strlen($source)), $location);
65 $tokens[] = new Token(TokenKind::End, '', $cursor->location());
66
67 return $tokens;
68 }
69 }
70 }
71
72 /**
73 * Skips whitespace, comments, stray commas and `#line` directives.
74 *
75 * @param Cursor $cursor Cursor to advance
76 *
77 * @throws SyntaxException When a comment never closes
78 */
79 public function skipTrivia(Cursor $cursor): void
80 {
81 while (!$cursor->eof()) {
82 if ($cursor->match('[ \f\t\v\r\n,]+') !== null || $cursor->match('//[^\n]*') !== null) {
83 continue;
84 }
85 if ($cursor->startsWith('/*')) {
86 $location = $cursor->location();
87 if ($cursor->takeUntil('*/') === null) {
88 throw SyntaxException::unterminated('comment', '*/', $location);
89 }
90 continue;
91 }
92
93 return;
94 }
95 }
96
97 /**
98 * Reads the token, or the tokens, that begin at the cursor.
99 *
100 * @param Cursor $cursor Cursor positioned on a token
101 *
102 * @return list<Token> One token, or an identifier and its bracketed name
103 *
104 * @throws SyntaxException When no token begins here or it never closes
105 */
106 public function next(Cursor $cursor): array
107 {
108 $location = $cursor->location();
109 $byte = $cursor->peek();
110 if ($byte === '#' && $location->column === 1) {
111 $directive = $cursor->match('#line [0-9]+(?: "[^"\n]*")?(?=\r?\n)');
112 if ($directive !== null) {
113 return [new Token(TokenKind::Line, $directive, $location)];
114 }
115 }
116 if ($byte === '%') {
117 return [$this->percent($cursor, $location)];
118 }
119 if ($byte === '{') {
120 return [new Token(TokenKind::Code, $this->code->braced($cursor), $location)];
121 }
122 if (preg_match('/[.A-Za-z_]/', $byte) === 1 && !$cursor->startsWith('_("')) {
123 return $this->identifier($cursor, $location);
124 }
125 if (ctype_digit($byte)) {
126 return [$this->integer($cursor, $location)];
127 }
128 $token = $this->literal($cursor, $location) ?? $this->punctuation($cursor, $location);
129 if ($token === null) {
130 throw SyntaxException::invalid("invalid character: '{$byte}'", $location);
131 }
132
133 return [$token];
134 }
135
136 /**
137 * Reads what begins with a percent sign: a prologue, a section marker, a predicate or a directive.
138 *
139 * @param Cursor $cursor Cursor positioned on the percent sign
140 * @param Location $location Where it is
141 *
142 * @return Token The token
143 *
144 * @throws SyntaxException When the directive is unknown or the code never closes
145 */
146 public function percent(Cursor $cursor, Location $location): Token
147 {
148 if ($cursor->startsWith('%{')) {
149 return new Token(TokenKind::Prologue, $this->code->prologue($cursor), $location);
150 }
151 if ($cursor->startsWith('%%')) {
152 $cursor->take(2);
153
154 return new Token(TokenKind::Section, '%%', $location);
155 }
156 if ($cursor->match('%\?[ \f\t\v]*(?:\r?\n[ \f\t\v]*)*(?=\{)') !== null) {
157 return new Token(TokenKind::Predicate, $this->code->braced($cursor), $location);
158 }
159 $raw = $cursor->match('%' . self::IDENTIFIER);
160 $name = $raw === null ? null : Directives::canonical(substr($raw, 1));
161 if ($raw === null || $name === null) {
162 throw SyntaxException::invalid('invalid directive: ' . ($raw ?? '%'), $location);
163 }
164 if (in_array($name, Directives::EQUAL_OPTIONAL, true)) {
165 $cursor->match('\s*=\s*');
166 }
167
168 return new Token(TokenKind::Directive, $name, $location, $raw);
169 }
170
171 /**
172 * Reads an identifier and looks past it for a bracketed name and a colon.
173 *
174 * @param Cursor $cursor Cursor positioned on the identifier
175 * @param Location $location Where it is
176 *
177 * @return list<Token> The identifier, followed by its bracketed name when one follows it
178 *
179 * @throws SyntaxException When a bracketed name is malformed
180 */
181 public function identifier(Cursor $cursor, Location $location): array
182 {
183 $name = $cursor->match(self::IDENTIFIER) ?? '';
184 $this->skipTrivia($cursor);
185 $bracketed = null;
186 if ($cursor->peek() === '[') {
187 $bracketed = $this->bracketed($cursor);
188 $this->skipTrivia($cursor);
189 }
190 $kind = $cursor->peek() === ':' ? TokenKind::IdentifierColon : TokenKind::Identifier;
191 $tokens = [new Token($kind, $name, $location)];
192 if ($bracketed !== null) {
193 $tokens[] = $bracketed;
194 }
195
196 return $tokens;
197 }
198
199 /**
200 * Reads a bracketed name such as `[left]`.
201 *
202 * @param Cursor $cursor Cursor positioned on the opening bracket
203 *
204 * @return Token The bracketed identifier
205 *
206 * @throws SyntaxException When the brackets hold anything but one identifier
207 */
208 public function bracketed(Cursor $cursor): Token
209 {
210 $location = $cursor->location();
211 $cursor->take(1);
212 $this->skipTrivia($cursor);
213 $name = $cursor->match(self::IDENTIFIER);
214 if ($name === null) {
215 throw SyntaxException::invalid('an identifier expected', $cursor->location());
216 }
217 $this->skipTrivia($cursor);
218 if ($cursor->peek() !== ']') {
219 throw SyntaxException::unexpected("']'", $cursor->eof() ? 'end of file' : "'{$cursor->peek()}'", $cursor->location());
220 }
221 $cursor->take(1);
222
223 return new Token(TokenKind::BracketedIdentifier, $name, $location);
224 }
225
226 /**
227 * Reads a decimal or hexadecimal integer.
228 *
229 * @param Cursor $cursor Cursor positioned on the first digit
230 * @param Location $location Where it is
231 *
232 * @return Token The integer, its text in decimal
233 *
234 * @throws SyntaxException When letters follow the digits
235 */
236 public function integer(Cursor $cursor, Location $location): Token
237 {
238 $text = $cursor->match('0[xX][0-9A-Fa-f]+|[0-9]+') ?? '';
239 if ($cursor->lookingAt('[-.A-Za-z0-9_]')) {
240 throw SyntaxException::invalid('invalid identifier: ' . $text . $cursor->match('[-.A-Za-z0-9_]*'), $location);
241 }
242 $hex = str_starts_with(strtolower($text), '0x');
243 $digits = ltrim($hex ? substr($text, 2) : $text, '0');
244 if (strlen($digits) > ($hex ? 8 : 10) || ($hex ? hexdec($digits) : (int) $digits) > 0x7FFFFFFF) {
245 throw SyntaxException::invalid("integer out of range: '{$text}'", $location);
246 }
247 $value = $hex ? hexdec($digits) : (int) $digits;
248
249 return new Token(TokenKind::Integer, (string) $value, $location, $text);
250 }
251
252 /**
253 * Reads a character literal, a string, a translatable string, or a tag.
254 *
255 * @param Cursor $cursor Cursor positioned on the literal
256 * @param Location $location Where it is
257 *
258 * @return Token|null The token, or null when no literal begins here
259 *
260 * @throws SyntaxException When the literal never closes or a character literal is not one byte
261 */
262 public function literal(Cursor $cursor, Location $location): ?Token
263 {
264 $byte = $cursor->peek();
265 if ($byte === "'") {
266 $literal = $cursor->match("'(?:\\\\.|[^\\\\'\\n])*'");
267 if ($literal === null) {
268 throw SyntaxException::unterminated('character literal', "'", $location);
269 }
270 $decoded = $this->escapes->decode(substr($literal, 1, -1), $location);
271 if ($decoded === '') {
272 throw SyntaxException::invalid('empty character literal', $location);
273 }
274 if (strlen($decoded) !== 1) {
275 throw SyntaxException::invalid('extra characters in character literal', $location);
276 }
277
278 return new Token(TokenKind::CharLiteral, $decoded, $location, $literal);
279 }
280 if ($byte === '"' || $cursor->startsWith('_("')) {
281 $translatable = $byte === '_';
282 $literal = $cursor->match($translatable ? '_\("(?:\\\\.|[^\\\\"\n])*"\)' : '"(?:\\\\.|[^\\\\"\n])*"');
283 if ($literal === null) {
284 throw SyntaxException::unterminated('string', '"', $location);
285 }
286 $body = $translatable ? substr($literal, 3, -2) : substr($literal, 1, -1);
287
288 return new Token($translatable ? TokenKind::TranslatableString : TokenKind::String, $this->escapes->decode($body, $location), $location, $literal);
289 }
290 if ($byte === '<') {
291 return $this->tag($cursor, $location);
292 }
293
294 return null;
295 }
296
297 /**
298 * Reads a tag, which may nest angle brackets as a C++ template does.
299 *
300 * @param Cursor $cursor Cursor positioned on the opening angle bracket
301 * @param Location $location Where it is
302 *
303 * @return Token The tag
304 *
305 * @throws SyntaxException When the tag never closes
306 */
307 public function tag(Cursor $cursor, Location $location): Token
308 {
309 if ($cursor->startsWith('<*>')) {
310 $cursor->take(3);
311
312 return new Token(TokenKind::TagAny, '*', $location);
313 }
314 if ($cursor->startsWith('<>')) {
315 $cursor->take(2);
316
317 return new Token(TokenKind::TagNone, '', $location);
318 }
319 $cursor->take(1);
320 $nesting = 0;
321 $text = '';
322 while (!$cursor->eof()) {
323 $unit = $cursor->match('(?:->|[^<>])+|<+') ?? $cursor->take(1);
324 if ($unit === '>') {
325 if ($nesting === 0) {
326 return new Token(TokenKind::Tag, $text, $location);
327 }
328 $nesting--;
329 } elseif ($unit[0] === '<') {
330 $nesting += strlen($unit);
331 }
332 $text .= $unit;
333 }
334
335 throw SyntaxException::unterminated('tag', '>', $location);
336 }
337
338 /**
339 * Reads one punctuation byte.
340 *
341 * @param Cursor $cursor Cursor positioned on the byte
342 * @param Location $location Where it is
343 *
344 * @return Token|null The token, or null when the byte is not punctuation the language has
345 *
346 * @throws SyntaxException When a bracketed name is malformed
347 */
348 public function punctuation(Cursor $cursor, Location $location): ?Token
349 {
350 $byte = $cursor->peek();
351 if ($byte === '[') {
352 return $this->bracketed($cursor);
353 }
354 $kind = match ($byte) {
355 ':' => TokenKind::Colon,
356 '=' => TokenKind::Equal,
357 '|' => TokenKind::Pipe,
358 ';' => TokenKind::Semicolon,
359 default => null,
360 };
361 if ($kind === null) {
362 return null;
363 }
364 $cursor->take(1);
365
366 return new Token($kind, $byte, $location);
367 }
368}
369