This commit is contained in:
2026-08-18 18:11:28 -04:00
parent 0e9a650fa2
commit bbaf47028f
259 changed files with 0 additions and 0 deletions
@@ -0,0 +1,342 @@
/**
* Shell lexical analysis (operators + words) shared by {@link ./shell.js} and diagnostics.
*/
/**
* @param {string} s
* @param {number} start
*/
export function findArithmeticClose(s, start) {
let depth = 0
for (let i = start; i < s.length - 1; i++) {
const ch = s[i]
if (ch === '(') depth++
else if (ch === ')') {
if (depth > 0) depth--
else if (s[i + 1] === ')') return i
}
}
return -1
}
/**
* POSIX-style ANSI strings for `$'…'` (lexer-only; emitted as single-quoted parts).
* @param {string} raw bytes between quotes after `$'`, still containing backslashes
*/
export function decodeBareOsDollarQuote(raw) {
let out = ''
let j = 0
while (j < raw.length) {
if (raw[j] !== '\\') {
out += raw[j++]
continue
}
j++
if (j >= raw.length) break
const e = raw[j++]
switch (e) {
case 'n':
out += '\n'
break
case 'r':
out += '\r'
break
case 't':
out += '\t'
break
case 'a':
out += '\x07'
break
case 'b':
out += '\b'
break
case 'f':
out += '\f'
break
case 'v':
out += '\v'
break
case '\\':
out += '\\'
break
case "'":
out += "'"
break
case 'x': {
let hex = ''
while (j < raw.length && /[0-9a-fA-F]/.test(raw[j]) && hex.length < 2)
hex += raw[j++]
if (hex.length)
out += String.fromCharCode(Number.parseInt(hex, 16) || 0)
break
}
default:
out += e
}
}
return out
}
/**
* @typedef {{ q: 'u' | 's' | 'd', t: string }} ShellWordPart
*/
/**
* @typedef {{
* type: 'word',
* value: string,
* parts: ShellWordPart[],
* start?: number,
* end?: number
* } | {
* type: 'op',
* value: string,
* start?: number,
* end?: number
* }} ShellLexToken
*/
/**
* Tokenize one shell line into words and operators (POSIX-ish).
* @param {string} line
* @returns {ShellLexToken[]}
*/
export function lexShellLine(line) {
/** @type {ShellLexToken[]} */
const tokens = []
let i = 0
const skipWs = () => {
while (i < line.length && /\s/.test(line[i])) i++
}
while (i < line.length) {
skipWs()
if (i >= line.length) break
const c = line[i]
const opStart = i
if (c === ';') {
if (line[i + 1] === ';') {
tokens.push({ type: 'op', value: ';;', start: opStart, end: i + 2 })
i += 2
} else if (line[i + 1] === '&') {
tokens.push({ type: 'op', value: ';&', start: opStart, end: i + 2 })
i += 2
} else {
tokens.push({ type: 'op', value: ';', start: opStart, end: i + 1 })
i++
}
continue
}
if (c === '&') {
if (line[i + 1] === '&') {
tokens.push({ type: 'op', value: '&&', start: opStart, end: i + 2 })
i += 2
} else {
tokens.push({ type: 'op', value: '&', start: opStart, end: i + 1 })
i++
}
continue
}
if (c === '|') {
if (line[i + 1] === '|') {
tokens.push({ type: 'op', value: '||', start: opStart, end: i + 2 })
i += 2
} else if (line[i + 1] === '&') {
tokens.push({ type: 'op', value: '|&', start: opStart, end: i + 2 })
i += 2
} else {
tokens.push({ type: 'op', value: '|', start: opStart, end: i + 1 })
i++
}
continue
}
if (c === '2' && line[i + 1] === '>') {
if (line[i + 2] === '>') {
tokens.push({ type: 'op', value: '2>>', start: opStart, end: i + 3 })
i += 3
continue
}
if (line[i + 2] === '&' && line[i + 3] === '1') {
tokens.push({ type: 'op', value: '2>&1', start: opStart, end: i + 4 })
i += 4
continue
}
tokens.push({ type: 'op', value: '2>', start: opStart, end: i + 2 })
i += 2
continue
}
if (c === '>') {
if (line[i + 1] === '>') {
tokens.push({ type: 'op', value: '>>', start: opStart, end: i + 2 })
i += 2
} else if (line[i + 1] === '&') {
tokens.push({ type: 'op', value: '>&', start: opStart, end: i + 2 })
i += 2
} else {
tokens.push({ type: 'op', value: '>', start: opStart, end: i + 1 })
i++
}
continue
}
if (c === '(') {
tokens.push({ type: 'op', value: '(', start: opStart, end: i + 1 })
i++
continue
}
if (c === '{') {
tokens.push({ type: 'op', value: '{', start: opStart, end: i + 1 })
i++
continue
}
if (c === ')') {
tokens.push({ type: 'op', value: ')', start: opStart, end: i + 1 })
i++
continue
}
if (c === '}') {
tokens.push({ type: 'op', value: '}', start: opStart, end: i + 1 })
i++
continue
}
if (c === '<') {
if (line[i + 1] === '<' && line[i + 2] === '<') {
tokens.push({ type: 'op', value: '<<<', start: opStart, end: i + 3 })
i += 3
continue
}
if (line[i + 1] === '<' && line[i + 2] === '-') {
tokens.push({ type: 'op', value: '<<-', start: opStart, end: i + 3 })
i += 3
continue
}
if (line[i + 1] === '<') {
tokens.push({ type: 'op', value: '<<', start: opStart, end: i + 2 })
i += 2
continue
}
tokens.push({ type: 'op', value: '<', start: opStart, end: i + 1 })
i++
continue
}
const wordStart = i
/** @type {ShellWordPart[]} */
const parts = []
/** @type {{ q: 'u', t: string }} */
let cur = { q: 'u', t: '' }
const flushU = () => {
if (cur.t.length) {
parts.push(cur)
cur = { q: 'u', t: '' }
}
}
while (i < line.length) {
const ch = line[i]
if (ch === '$' && line[i + 1] === "'") {
flushU()
i += 2
let raw = ''
while (i < line.length) {
if (line[i] === '\\') {
raw += '\\'
i++
if (i < line.length) raw += line[i++]
continue
}
if (line[i] === "'") break
raw += line[i++]
}
if (i < line.length) i++
parts.push({ q: 's', t: decodeBareOsDollarQuote(raw) })
continue
}
if (ch === '\\') {
i++
if (i < line.length) cur.t += line[i++]
continue
}
if (ch === "'") {
flushU()
i++
let inner = ''
while (i < line.length && line[i] !== "'") inner += line[i++]
if (i < line.length) i++
parts.push({ q: 's', t: inner })
continue
}
if (ch === '"') {
flushU()
i++
let inner = ''
while (i < line.length && line[i] !== '"') {
if (
line[i] === '$' &&
line[i + 1] === '(' &&
line[i + 2] === '('
) {
const close = findArithmeticClose(line, i + 3)
if (close < 0) {
inner += line.slice(i)
i = line.length
break
}
inner += line.slice(i, close + 2)
i = close + 2
continue
}
if (line[i] === '\\' && i + 1 < line.length) {
i++
inner += line[i++]
continue
}
inner += line[i++]
}
if (i < line.length) i++
parts.push({ q: 'd', t: inner })
continue
}
if (ch === '$' && line[i + 1] === '(' && line[i + 2] === '(') {
const close = findArithmeticClose(line, i + 3)
if (close < 0) {
cur.t += line.slice(i)
i = line.length
break
}
cur.t += line.slice(i, close + 2)
i = close + 2
continue
}
if (
/\s/.test(ch) ||
ch === '|' ||
ch === '>' ||
ch === '<' ||
ch === ';' ||
ch === '&' ||
ch === '(' ||
ch === ')' ||
ch === '{' ||
ch === '}'
)
break
cur.t += ch
i++
}
flushU()
if (parts.length) {
const value = parts.map((p) => p.t).join('')
tokens.push({
type: 'word',
value,
parts,
start: wordStart,
end: i
})
}
}
return tokens
}