Aleph-w 3.0
A C++ Library for Data Structures and Algorithms
Loading...
Searching...
No Matches
Compiler_Lexer.H
Go to the documentation of this file.
1/*
2 Aleph_w
3
4 Data structures & Algorithms
5 version 2.0.0b
6 https://github.com/lrleon/Aleph-w
7
8 This file is part of Aleph-w library
9
10 Copyright (c) 2002-2026 Leandro Rabindranath Leon
11
12 Permission is hereby granted, free of charge, to any person obtaining a copy
13 of this software and associated documentation files (the "Software"), to deal
14 in the Software without restriction, including without limitation the rights
15 to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
16 copies of the Software, and to permit persons to whom the Software is
17 furnished to do so, subject to the following conditions:
18
19 The above copyright notice and this permission notice shall be included in all
20 copies or substantial portions of the Software.
21
22 THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
23 IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
24 FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
25 AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
26 LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
27 OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
28 SOFTWARE.
29*/
30
56#ifndef COMPILER_LEXER_H
57#define COMPILER_LEXER_H
58
59#include <cctype>
60#include <string>
61
62#include <Compiler_Token.H>
63#include <ah-diagnostics.H>
64
65namespace Aleph {
66
76{
77 bool keep_comments = false;
79};
80
90{
91 const Source_Manager *sources_ = nullptr;
93 const std::string *input_ = nullptr;
97 bool has_lookahead_ = false;
99
101 static bool is_ident_start(const unsigned char ch) noexcept
102 {
103 return std::isalpha(ch) or ch == '_';
104 }
105
107 static bool is_ident_continue(const unsigned char ch) noexcept
108 {
109 return std::isalnum(ch) or ch == '_';
110 }
111
114 {
115 return cursor_ >= input_->size();
116 }
117
119 [[nodiscard]] char char_at(const Source_Offset off) const noexcept
120 {
121 return off < input_->size() ? (*input_)[off] : '\0';
122 }
123
126 {
127 return char_at(cursor_);
128 }
129
132 {
133 return char_at(cursor_ + 1);
134 }
135
138 {
139 return sources_->span(file_id_, begin, end);
140 }
141
144 const Source_Offset begin,
145 const Source_Offset end) const
146 {
147 const auto sp = make_span(begin, end);
148 return {kind, sources_->slice(sp), sp};
149 }
150
153 {
154 const auto sp = make_span(cursor_, cursor_);
156 }
157
159 void emit_error(const Source_Span &span,
160 const std::string &message,
161 const std::string &code,
162 const std::string &note = "",
163 const std::string &help = "") const
164 {
165 if (diagnostics_ == nullptr)
166 return;
167
168 auto builder = diagnostics_->error(span, message).code(code);
169 if (not note.empty())
170 builder.note(note);
171 if (not help.empty())
172 builder.help(help);
173 builder.emit();
174 }
175
178 {
179 while (not at_end() and std::isspace(static_cast<unsigned char>(current_char())))
180 ++cursor_;
181 }
182
185 {
186 const Source_Offset begin = cursor_;
187 cursor_ += 2;
188 while (not at_end() and current_char() != '\n')
189 ++cursor_;
191 }
192
195 {
196 const Source_Offset begin = cursor_;
197 cursor_ += 2;
198
199 while (not at_end())
200 {
201 if (current_char() == '*' and next_char() == '/')
202 {
203 cursor_ += 2;
205 }
206 ++cursor_;
207 }
208
209 const auto sp = make_span(begin, cursor_);
210 emit_error(sp, "unterminated block comment", "LEX004", "block comments must end with '*/'");
212 }
213
216 {
217 const Source_Offset begin = cursor_;
218 ++cursor_;
219 while (not at_end() and is_ident_continue(static_cast<unsigned char>(current_char())))
220 ++cursor_;
221
222 const auto sp = make_span(begin, cursor_);
223 const auto lexeme = sources_->slice(sp);
224 return {classify_compiler_keyword(lexeme), lexeme, sp};
225 }
226
229 {
230 const Source_Offset begin = cursor_;
231 ++cursor_;
232 while (not at_end())
233 {
234 const auto ch = static_cast<unsigned char>(current_char());
235 if (std::isdigit(ch))
236 {
237 ++cursor_;
238 continue;
239 }
240
241 if (ch == '_' and std::isdigit(static_cast<unsigned char>(next_char())))
242 {
243 ++cursor_;
244 continue;
245 }
246
247 break;
248 }
249
251 }
252
255 {
256 const Source_Offset begin = cursor_;
257 ++cursor_;
258
259 while (not at_end())
260 {
261 const char ch = current_char();
262 if (ch == '"')
263 {
264 ++cursor_;
266 }
267
268 if (ch == '\\')
269 {
270 ++cursor_;
271 if (at_end())
272 break;
273 ++cursor_;
274 continue;
275 }
276
277 if (ch == '\n' or ch == '\r')
278 break;
279
280 ++cursor_;
281 }
282
283 const auto sp = make_span(begin, cursor_);
285 "unterminated string literal",
286 "LEX002",
287 "string literal started here",
288 "close the string with a double quote");
290 }
291
294 {
295 const Source_Offset begin = cursor_;
296 ++cursor_;
297
298 if (at_end() or current_char() == '\n' or current_char() == '\r')
299 {
300 const auto sp = make_span(begin, cursor_);
301 emit_error(sp, "unterminated character literal", "LEX003", "character literal started here");
303 }
304
305 if (current_char() == '\\')
306 {
307 ++cursor_;
308 if (at_end() or current_char() == '\n' or current_char() == '\r')
309 {
310 const auto sp = make_span(begin, cursor_);
311 emit_error(sp, "unterminated escape sequence in character literal", "LEX003");
313 }
314 ++cursor_;
315 }
316 else
317 ++cursor_;
318
319 if (not at_end() and current_char() == '\'')
320 {
321 ++cursor_;
323 }
324
325 while (not at_end() and current_char() != '\'' and current_char() != '\n'
326 and current_char() != '\r')
327 ++cursor_;
328
329 if (not at_end() and current_char() == '\'')
330 ++cursor_;
331
332 const auto sp = make_span(begin, cursor_);
334 "malformed character literal",
335 "LEX003",
336 "character literals must contain exactly one character or escape");
338 }
339
342 {
343 const Source_Offset begin = cursor_;
344 const char a = current_char();
345 const char b = next_char();
346
347 // Multicharacter operators/punctuation
348 if (a == ':' and b == ':')
349 {
350 cursor_ += 2;
352 }
353 if (a == '-' and b == '>')
354 {
355 cursor_ += 2;
357 }
358 if (a == '=' and b == '=')
359 {
360 cursor_ += 2;
362 }
363 if (a == '!' and b == '=')
364 {
365 cursor_ += 2;
367 }
368 if (a == '<' and b == '=')
369 {
370 cursor_ += 2;
372 }
373 if (a == '>' and b == '=')
374 {
375 cursor_ += 2;
377 }
378 if (a == '&' and b == '&')
379 {
380 cursor_ += 2;
382 }
383 if (a == '|' and b == '|')
384 {
385 cursor_ += 2;
387 }
388 if (a == '+' and b == '=')
389 {
390 cursor_ += 2;
392 }
393 if (a == '-' and b == '=')
394 {
395 cursor_ += 2;
397 }
398 if (a == '*' and b == '=')
399 {
400 cursor_ += 2;
402 }
403 if (a == '/' and b == '=')
404 {
405 cursor_ += 2;
407 }
408 if (a == '%' and b == '=')
409 {
410 cursor_ += 2;
412 }
413 if (a == '+' and b == '+')
414 {
415 cursor_ += 2;
417 }
418 if (a == '-' and b == '-')
419 {
420 cursor_ += 2;
422 }
423
424 ++cursor_;
425 switch (a)
426 {
427 case '(':
429 case ')':
431 case '{':
433 case '}':
435 case '[':
437 case ']':
439 case ',':
441 case '.':
443 case ':':
445 case ';':
447 case '?':
449 case '=':
451 case '+':
453 case '-':
455 case '*':
457 case '/':
459 case '%':
461 case '&':
463 case '|':
465 case '^':
467 case '!':
469 case '~':
471 case '<':
473 case '>':
475 default:
476 {
477 const auto sp = make_span(begin, cursor_);
478 emit_error(sp, std::string("unexpected character '") + a + "'", "LEX001");
480 }
481 }
482 }
483
486 {
487 while (true)
488 {
490 if (at_end())
491 return make_eof_token();
492
493 if (current_char() == '/' and next_char() == '/')
494 {
495 if (auto tok = lex_line_comment(); options_.keep_comments or tok.is_invalid())
496 return tok;
497 continue;
498 }
499
501 {
502 if (auto tok = lex_block_comment(); options_.keep_comments or tok.is_invalid())
503 return tok;
504 continue;
505 }
506
507 break;
508 }
509
510 const auto ch = static_cast<unsigned char>(current_char());
511 if (is_ident_start(ch))
513 if (std::isdigit(ch))
514 return lex_number();
515 if (current_char() == '"')
516 return lex_string();
517 if (current_char() == '\'')
518 return lex_char_literal();
520 }
521
522public:
535 const Source_File_Id id,
536 Diagnostic_Engine *dx = nullptr,
537 const Compiler_Lexer_Options &opts = {})
538 : sources_(&sm), file_id_(id), input_(&sm.file_text(id)), diagnostics_(dx), options_(opts)
539 {}
540
551
564
575 void reset(const Source_Offset offset = 0)
576 {
578 << "Compiler_Lexer::reset(): invalid offset " << offset;
579 cursor_ = offset;
580 has_lookahead_ = false;
582 }
583
594 {
596 {
598 has_lookahead_ = true;
599 }
600 return lookahead_;
601 }
602
613 {
614 if (has_lookahead_)
615 {
616 has_lookahead_ = false;
617 return lookahead_;
618 }
619 return lex_token();
620 }
621
628 bool eof()
629 {
630 return peek().is_eof();
631 }
632};
633} // namespace Aleph
634
635#endif // COMPILER_LEXER_H
Token model and parser-facing metadata for compiler front-ends.
Plain-text diagnostic engine for compiler-style tooling.
#define ah_out_of_range_error_unless(C)
Throws std::out_of_range if condition does NOT hold.
Definition ah-errors.H:600
Token generator for a single source file.
Compiler_Token lex_identifier_or_keyword()
Lexes an identifier or classifies it as a keyword.
Source_Offset current_offset() const noexcept
Returns the current byte offset in the file.
Source_File_Id file_id_
Compiler_Lexer_Options options_
static bool is_ident_start(const unsigned char ch) noexcept
Checks if a character can start an identifier.
Source_Span make_span(const Source_Offset begin, const Source_Offset end) const
Creates a source span for the given range in the current file.
Compiler_Token lex_punctuation_or_operator()
Lexes punctuation or operator symbols.
Compiler_Token lex_char_literal()
Lexes a character literal.
void emit_error(const Source_Span &span, const std::string &message, const std::string &code, const std::string &note="", const std::string &help="") const
Emits a diagnostic error through the engine.
Compiler_Token lex_number()
Lexes a numeric literal.
const Source_Manager * sources_
Compiler_Token lex_line_comment()
Lexes a single-line comment.
Compiler_Token make_eof_token() const
Creates an End_Of_File token at the current position.
char current_char() const noexcept
Returns the character at the current cursor position.
Compiler_Token make_token(const Compiler_Token_Kind kind, const Source_Offset begin, const Source_Offset end) const
Creates a token with the given kind and range.
void skip_whitespace()
Consumes whitespace characters.
Compiler_Token next()
Consumes and returns the next token from the stream.
bool eof()
Indicates whether the end of the file has been reached.
char char_at(const Source_Offset off) const noexcept
Safe character access at a given offset.
const Compiler_Token & peek()
Peeks at the next token without advancing the cursor.
Compiler_Token lex_token()
Dispatches the next token from the input stream.
Source_File_Id source_file_id() const noexcept
Returns the file ID currently being analyzed.
Diagnostic_Engine * diagnostics_
void reset(const Source_Offset offset=0)
Repositions the reading cursor.
bool at_end() const noexcept
Returns true if the cursor has reached the end of the input.
Compiler_Token lex_block_comment()
Lexes a multi-line block comment.
static bool is_ident_continue(const unsigned char ch) noexcept
Checks if a character can continue an identifier.
const std::string * input_
Compiler_Token lex_string()
Lexes a string literal.
Compiler_Token lookahead_
Compiler_Lexer(const Source_Manager &sm, const Source_File_Id id, Diagnostic_Engine *dx=nullptr, const Compiler_Lexer_Options &opts={})
Initializes a lexer for a registered source file.
char next_char() const noexcept
Returns the character following the current cursor position.
Diagnostic_Builder & code(const std::string &value)
Sets the stable diagnostic code.
size_t emit() const noexcept
Finalizes the builder and returns the diagnostic index.
Diagnostic_Builder & help(const std::string &msg)
Appends a help line.
Diagnostic_Builder & note(const std::string &msg)
Appends a note line.
Diagnostic accumulator and renderer.
Diagnostic_Builder error(const Source_Span &span, const std::string &msg)
Starts an error diagnostic.
Stores source files and resolves offsets into human-readable data.
Definition ah-source.H:184
std::string slice(const Source_Span &span) const
Returns the exact text covered by span.
Definition ah-source.H:363
Source_Span span(const Source_File_Id id, const Source_Offset begin, const Source_Offset end) const
Creates a validated span in file id.
Definition ah-source.H:325
size_t blossom_maximum_cardinality_matching(const GT &g, DynDlist< typename GT::Arc * > &matching, SA sa=SA())
Alias of compute_maximum_cardinality_general_matching().
Definition Blossom.H:466
const long double offset[]
Offset values indexed by symbol string length (bounded by MAX_OFFSET_INDEX)
Main namespace for Aleph-w library functions.
Definition ah-arena.H:89
Compiler_Token_Kind classify_compiler_keyword(std::string_view lexeme) noexcept
Classifies a lexeme as a keyword or Identifier.
void message(const char *file, int line, const char *format,...)
Print an informational message with file and line info.
Definition ahDefs.C:95
size_t size(Node *root) noexcept
and
Check uniqueness with explicit hash + equality functors.
std::string code(Node *root)
Compute a string with the Lukasiewicz`s word of a tree.
Compiler_Token_Kind
Token kinds supported by the compiler MVP.
@ Line_Comment
Single-line comment (// ...).
@ Integer_Literal
Integer constant.
@ End_Of_File
End of the source stream.
@ Invalid
Sentinel for lexer errors or uninitialized tokens.
@ Char_Literal
Single character constant.
@ Block_Comment
Multi-line comment (/* ... *&zwj;/).
@ String_Literal
String constant (with escape sequences).
size_t Source_File_Id
Definition ah-source.H:61
size_t Source_Offset
Definition ah-source.H:62
Configuration options for lexer behavior.
bool allow_block_comments
If false, /* is treated as / followed by *.
bool keep_comments
If true, comments are returned as tokens.
Lexical token representation with its value and location.
bool is_eof() const noexcept
Returns whether the token marks the end of the file.
Half-open byte range inside a source file.
Definition ah-source.H:100