Repository navigation
Expand file tree
/
Copy pathLexer.js
More file actions
257 lines (245 loc) · 8.13 KB
/
Copy pathLexer.js
File metadata and controls
257 lines (245 loc) · 8.13 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
/*
* Jexl
* Copyright 2020 Tom Shawver
*/
const numericRegex = /^-?(?:(?:[0-9]*\.[0-9]+)|[0-9]+)$/
const identRegex = /^[a-zA-Zа-яА-Я_\u00C0-\u00D6\u00D8-\u00F6\u00F8-\u00FF$][a-zA-Zа-яА-Я0-9_\u00C0-\u00D6\u00D8-\u00F6\u00F8-\u00FF$]*$/
const escEscRegex = /\\\\/
const whitespaceRegex = /^\s*$/
const preOpRegexElems = [
// Strings
"'(?:(?:\\\\')|[^'])*'",
'"(?:(?:\\\\")|[^"])*"',
// Whitespace
'\\s+',
// Booleans
'\\btrue\\b',
'\\bfalse\\b'
]
const postOpRegexElems = [
// Identifiers
'[a-zA-Zа-яА-Я_\u00C0-\u00D6\u00D8-\u00F6\u00F8-\u00FF\\$][a-zA-Z0-9а-яА-Я_\u00C0-\u00D6\u00D8-\u00F6\u00F8-\u00FF\\$]*',
// Numerics (without negative symbol)
'(?:(?:[0-9]*\\.[0-9]+)|[0-9]+)'
]
const minusNegatesAfter = [
'binaryOp',
'unaryOp',
'openParen',
'openBracket',
'question',
'colon'
]
/**
* Lexer is a collection of stateless, statically-accessed functions for the
* lexical parsing of a Jexl string. Its responsibility is to identify the
* "parts of speech" of a Jexl expression, and tokenize and label each, but
* to do only the most minimal syntax checking; the only errors the Lexer
* should be concerned with are if it's unable to identify the utility of
* any of its tokens. Errors stemming from these tokens not being in a
* sensible configuration should be left for the Parser to handle.
* @type {{}}
*/
class Lexer {
constructor(grammar) {
this._grammar = grammar
}
/**
* Splits a Jexl expression string into an array of expression elements.
* @param {string} str A Jexl expression string
* @returns {Array<string>} An array of substrings defining the functional
* elements of the expression.
*/
getElements(str) {
const regex = this._getSplitRegex()
return str.split(regex).filter((elem) => {
// Remove empty strings
return elem
})
}
/**
* Converts an array of expression elements into an array of tokens. Note that
* the resulting array may not equal the element array in length, as any
* elements that consist only of whitespace get appended to the previous
* token's "raw" property. For the structure of a token object, please see
* {@link Lexer#tokenize}.
* @param {Array<string>} elements An array of Jexl expression elements to be
* converted to tokens
* @returns {Array<{type, value, raw}>} an array of token objects.
*/
getTokens(elements) {
const tokens = []
let negate = false
for (let i = 0; i < elements.length; i++) {
if (this._isWhitespace(elements[i])) {
if (tokens.length) {
tokens[tokens.length - 1].raw += elements[i]
}
} else if (elements[i] === '-' && this._isNegative(tokens)) {
negate = true
} else {
if (negate) {
elements[i] = '-' + elements[i]
negate = false
}
tokens.push(this._createToken(elements[i]))
}
}
// Catch a - at the end of the string. Let the parser handle that issue.
if (negate) {
tokens.push(this._createToken('-'))
}
return tokens
}
/**
* Converts a Jexl string into an array of tokens. Each token is an object
* in the following format:
*
* {
* type: <string>,
* [name]: <string>,
* value: <boolean|number|string>,
* raw: <string>
* }
*
* Type is one of the following:
*
* literal, identifier, binaryOp, unaryOp
*
* OR, if the token is a control character its type is the name of the element
* defined in the Grammar.
*
* Name appears only if the token is a control string found in
* {@link grammar#elements}, and is set to the name of the element.
*
* Value is the value of the token in the correct type (boolean or numeric as
* appropriate). Raw is the string representation of this value taken directly
* from the expression string, including any trailing spaces.
* @param {string} str The Jexl string to be tokenized
* @returns {Array<{type, value, raw}>} an array of token objects.
* @throws {Error} if the provided string contains an invalid token.
*/
tokenize(str) {
const elements = this.getElements(str)
return this.getTokens(elements)
}
/**
* Creates a new token object from an element of a Jexl string. See
* {@link Lexer#tokenize} for a description of the token object.
* @param {string} element The element from which a token should be made
* @returns {{value: number|boolean|string, [name]: string, type: string,
* raw: string}} a token object describing the provided element.
* @throws {Error} if the provided string is not a valid expression element.
* @private
*/
_createToken(element) {
const token = {
type: 'literal',
value: element,
raw: element
}
if (element[0] === '"' || element[0] === "'") {
token.value = this._unquote(element)
} else if (element.match(numericRegex)) {
token.value = parseFloat(element)
} else if (element === 'true' || element === 'false') {
token.value = element === 'true'
} else if (this._grammar.elements[element]) {
token.type = this._grammar.elements[element].type
} else if (element.match(identRegex)) {
token.type = 'identifier'
} else {
throw new Error(`Invalid expression token: ${element}`)
}
return token
}
/**
* Escapes a string so that it can be treated as a string literal within a
* regular expression.
* @param {string} str The string to be escaped
* @returns {string} the RegExp-escaped string.
* @see https://developer.mozilla.org/en/docs/Web/JavaScript/Guide/Regular_Expressions
* @private
*/
_escapeRegExp(str) {
str = str.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')
if (str.match(identRegex)) {
str = '\\b' + str + '\\b'
}
return str
}
/**
* Gets a RegEx object appropriate for splitting a Jexl string into its core
* elements.
* @returns {RegExp} An element-splitting RegExp object
* @private
*/
_getSplitRegex() {
if (!this._splitRegex) {
// Sort by most characters to least, then regex escape each
const elemArray = Object.keys(this._grammar.elements)
.sort((a, b) => {
return b.length - a.length
})
.map((elem) => {
return this._escapeRegExp(elem)
}, this)
this._splitRegex = new RegExp(
'(' +
[
preOpRegexElems.join('|'),
elemArray.join('|'),
postOpRegexElems.join('|')
].join('|') +
')'
)
}
return this._splitRegex
}
/**
* Determines whether the addition of a '-' token should be interpreted as a
* negative symbol for an upcoming number, given an array of tokens already
* processed.
* @param {Array<Object>} tokens An array of tokens already processed
* @returns {boolean} true if adding a '-' should be considered a negative
* symbol; false otherwise
* @private
*/
_isNegative(tokens) {
if (!tokens.length) return true
return minusNegatesAfter.some(
(type) => type === tokens[tokens.length - 1].type
)
}
/**
* A utility function to determine if a string consists of only space
* characters.
* @param {string} str A string to be tested
* @returns {boolean} true if the string is empty or consists of only spaces;
* false otherwise.
* @private
*/
_isWhitespace(str) {
return !!str.match(whitespaceRegex)
}
/**
* Removes the beginning and trailing quotes from a string, unescapes any
* escaped quotes on its interior, and unescapes any escaped escape
* characters. Note that this function is not defensive; it assumes that the
* provided string is not empty, and that its first and last characters are
* actually quotes.
* @param {string} str A string whose first and last characters are quotes
* @returns {string} a string with the surrounding quotes stripped and escapes
* properly processed.
* @private
*/
_unquote(str) {
const quote = str[0]
const escQuoteRegex = new RegExp('\\\\' + quote, 'g')
return str
.substr(1, str.length - 2)
.replace(escQuoteRegex, quote)
.replace(escEscRegex, '\\')
}
}
module.exports = Lexer