载入中...
搜索中...
未找到
DnutLexer.cpp
浏览该文件的文档.
2
3#include <cctype>
4#include <cstddef>
5#include <string>
6#include <utility>
7#include <vector>
8
9namespace eve::dnut {
10
11namespace {
12
13bool isIdentifierStart(char c) {
14 return std::isalpha(static_cast<unsigned char>(c)) != 0 || c == '_';
15}
16
17bool isIdentifierBody(char c) {
18 return std::isalnum(static_cast<unsigned char>(c)) != 0 || c == '_' || c == '.';
19}
20
21bool isDigit(char c) { return c >= '0' && c <= '9'; }
22
23eve::Result<std::vector<DnutToken>> lexFailure(const std::string& path, int line, int column,
24 std::string message) {
27 {{"line", std::to_string(line)}, {"column", std::to_string(column)}},
28 "dnut.lexer"));
29}
30
32std::size_t twoCharacterPunctuatorLength(std::string_view source, std::size_t i) {
33 if (i + 1 >= source.size()) return 0;
34 const std::string_view pair = source.substr(i, 2);
35 for (const std::string_view candidate : {"==", "!=", ">=", "<=", "&&", "||", "->"}) {
36 if (pair == candidate) return 2;
37 }
38 return 0;
39}
40
41bool isSingleCharacterPunctuator(char c) {
42 switch (c) {
43 case '{':
44 case '}':
45 case '(':
46 case ')':
47 case '[':
48 case ']':
49 case ':':
50 case ',':
51 case '=':
52 case '>':
53 case '<':
54 case '!':
55 case '-':
56 case '+':
57 case '*':
58 case '/':
59 case '|': return true;
60 default: return false;
61 }
62}
63
65bool appendEscape(char escape, std::string& out) {
66 switch (escape) {
67 case 'n': out += '\n'; return true;
68 case 't': out += '\t'; return true;
69 case 'r': out += '\r'; return true;
70 case '"': out += '"'; return true;
71 case '\'': out += '\''; return true;
72 case '\\': out += '\\'; return true;
73 case '{': out += '{'; return true;
74 case '}': out += '}'; return true;
75 default:
76 // The original dialogue lexer treated an unknown escape as the
77 // escaped character itself. Keep that accepted source set while
78 // still normalizing the known control escapes above.
79 out += escape;
80 return true;
81 }
82}
83
84} // namespace
85
86const char* dnutTokenKindName(DnutTokenKind kind) noexcept {
87 switch (kind) {
88 case DnutTokenKind::Identifier: return "identifier";
89 case DnutTokenKind::String: return "string";
90 case DnutTokenKind::Number: return "number";
91 case DnutTokenKind::Punctuator: return "punctuator";
92 case DnutTokenKind::EndOfFile: return "eof";
93 }
94 return "unknown";
95}
96
97eve::Result<std::vector<DnutToken>> lexDnut(std::string_view source, const std::string& path) {
98 std::vector<DnutToken> tokens;
99 std::size_t i = 0;
100 int line = 1;
101 int column = 1;
102 const std::size_t size = source.size();
103
104 const auto advanceOne = [&]() {
105 if (source[i] == '\n') {
106 ++line;
107 column = 1;
108 } else {
109 ++column;
110 }
111 ++i;
112 };
113
114 while (i < size) {
115 const char c = source[i];
116
117 if (c == '\n' || c == ' ' || c == '\t' || c == '\r') {
118 advanceOne();
119 continue;
120 }
121 if (c == '/' && i + 1 < size && source[i + 1] == '/') {
122 while (i < size && source[i] != '\n') ++i;
123 continue;
124 }
125 if (c == '/' && i + 1 < size && source[i + 1] == '*') {
126 const int startLine = line;
127 const int startColumn = column;
128 advanceOne();
129 advanceOne();
130 bool closed = false;
131 while (i < size) {
132 if (source[i] == '*' && i + 1 < size && source[i + 1] == '/') {
133 advanceOne();
134 advanceOne();
135 closed = true;
136 break;
137 }
138 advanceOne();
139 }
140 if (!closed)
141 return lexFailure(path, startLine, startColumn, "unterminated block comment");
142 continue;
143 }
144 if (c == '"' || c == '\'') {
145 const char quote = c;
146 const int startLine = line;
147 const int startColumn = column;
148 advanceOne();
149 std::string text;
150 bool terminated = false;
151 while (i < size) {
152 if (source[i] == quote) {
153 advanceOne();
154 terminated = true;
155 break;
156 }
157 if (source[i] == '\\') {
158 advanceOne();
159 if (i >= size) break;
160 if (!appendEscape(source[i], text))
161 return lexFailure(path, line, column,
162 std::string("unknown escape sequence '\\") + source[i] + "'");
163 advanceOne();
164 continue;
165 }
166 text += source[i];
167 advanceOne();
168 }
169 if (!terminated) return lexFailure(path, startLine, startColumn, "unterminated string literal");
170 tokens.push_back({DnutTokenKind::String, std::move(text), startLine, startColumn});
171 continue;
172 }
173 if (isDigit(c) || (c == '.' && i + 1 < size && isDigit(source[i + 1]))) {
174 const int startLine = line;
175 const int startColumn = column;
176 const std::size_t start = i;
177 while (i < size) {
178 const char d = source[i];
179 if (isDigit(d) || d == '.') {
180 advanceOne();
181 } else if ((d == 'e' || d == 'E') && i + 1 < size &&
182 (isDigit(source[i + 1]) || source[i + 1] == '+' || source[i + 1] == '-')) {
183 advanceOne();
184 advanceOne();
185 } else {
186 break;
187 }
188 }
189 tokens.push_back({DnutTokenKind::Number, std::string(source.substr(start, i - start)), startLine,
190 startColumn});
191 continue;
192 }
193 if (isIdentifierStart(c)) {
194 const int startLine = line;
195 const int startColumn = column;
196 const std::size_t start = i;
197 // A '-' joins an identifier only when a word character follows it, so
198 // dashed content ids (`iron-sword`) stay one token while arrow and
199 // minus punctuation (`->`, `-3`) are unaffected.
200 const auto continuesIdentifier = [&source, size](std::size_t position) {
201 if (isIdentifierBody(source[position])) return true;
202 return source[position] == '-' && position + 1 < size && isIdentifierBody(source[position + 1]);
203 };
204 while (i < size && continuesIdentifier(i)) advanceOne();
205 tokens.push_back(
206 {DnutTokenKind::Identifier, std::string(source.substr(start, i - start)), startLine, startColumn});
207 continue;
208 }
209 if (const std::size_t pairLength = twoCharacterPunctuatorLength(source, i); pairLength != 0) {
210 const int startLine = line;
211 const int startColumn = column;
212 const std::string text(source.substr(i, pairLength));
213 advanceOne();
214 advanceOne();
215 tokens.push_back({DnutTokenKind::Punctuator, text, startLine, startColumn});
216 continue;
217 }
218 if (isSingleCharacterPunctuator(c)) {
219 const int startLine = line;
220 const int startColumn = column;
221 const std::string text(1, c);
222 advanceOne();
223 tokens.push_back({DnutTokenKind::Punctuator, std::move(text), startLine, startColumn});
224 continue;
225 }
226 return lexFailure(path, line, column, std::string("unexpected character '") + c + "'");
227 }
228
229 tokens.push_back({DnutTokenKind::EndOfFile, std::string{}, line, column});
230 return eve::Result<std::vector<DnutToken>>::success(std::move(tokens));
231}
232
233bool hasErrors(const std::vector<DnutDiagnostic>& diagnostics) {
234 for (const auto& diagnostic : diagnostics) {
235 if (diagnostic.severity == DnutSeverity::Error) return true;
236 }
237 return false;
238}
239
240} // namespace eve::dnut
Duration start
Shared lexer for every `.dnut` dialect.
int column
std::string message
std::unordered_map< const Graphics *, std::shared_ptr< Lifetime > > tokens
std::int32_t c
std::string text
TokenKind kind
char quote
std::array< float, 3 > position
std::string path
Definition PlayHost.cpp:110
float d
float size
Definition TreeMesh.cpp:156
const UnitySourceAsset & source
static Diagnostic error(DiagnosticCode code, std::string message, std::string path={}, DiagnosticDetails details={}, std::string source={})
Construct an error diagnostic with the standard error severity.
Definition Diagnostic.h:125
Move-only operation result carrying either a value or Status.
Definition Result.h:155
EVENGINE_API_PLATFORM bool hasErrors(const std::vector< DnutDiagnostic > &diagnostics)
Return whether any diagnostic is an error.
eve::Result< std::vector< DnutToken > > lexDnut(std::string_view source, const std::string &path)
Lex one .dnut source buffer into owned tokens.
Definition DnutLexer.cpp:97
const char * dnutTokenKindName(DnutTokenKind kind) noexcept
Return the stable lowercase spelling of a token kind.
Definition DnutLexer.cpp:86
DnutTokenKind
Token classes produced by the shared .dnut lexer.
Definition DnutLexer.h:17
@ String
Quoted string with escapes already resolved.
@ Identifier
Bare word: keyword, field name, identifier, or bare enum value.
@ Punctuator
One- or two-character punctuator such as {, =, ==, ->.
@ EndOfFile
Synthetic terminal token; always the last element.
@ Number
Unsigned numeric literal; sign handling belongs to the parser.