294 lines
8.1 KiB
Go
294 lines
8.1 KiB
Go
package bundler
|
|
|
|
// Top-level source segmentation. segmentTopLevel splits a module into contiguous
|
|
// chunks at top-level declaration boundaries; the Go Solid compiler uses it to
|
|
// find component declarations for solid-refresh instrumentation (wrapComponents
|
|
// in compile_solid_gen.go).
|
|
//
|
|
// Correctness contract: join(chunks) == src, always. A chunk always begins at a
|
|
// top-level declaration keyword and contains only whole top-level statements. The
|
|
// splitter is deliberately conservative — when the lexer is unsure it simply
|
|
// doesn't cut, producing fewer/larger chunks (still correct).
|
|
|
|
import "strings"
|
|
|
|
// declKeywords begin a top-level declaration. A line that starts (at brace depth
|
|
// 0, outside any string/comment/template/regex) with one of these — as a whole
|
|
// word — is a chunk boundary. `export` covers `export default`, `export const`,
|
|
// `export function`, and re-exports; `async` covers `async function`.
|
|
var declKeywords = []string{
|
|
"import", "export", "const", "let", "var", "function",
|
|
"async", "class", "type", "interface", "enum", "declare", "abstract",
|
|
}
|
|
|
|
// regexPrefixKeywords are the identifiers after which a `/` begins a regex
|
|
// literal rather than a division (e.g. `return /x/`), needed so the lexer keeps
|
|
// an accurate brace depth through regexes that contain braces or quotes.
|
|
var regexPrefixKeywords = map[string]bool{
|
|
"return": true, "typeof": true, "instanceof": true, "in": true, "of": true,
|
|
"new": true, "delete": true, "void": true, "do": true, "else": true,
|
|
"yield": true, "await": true, "case": true,
|
|
}
|
|
|
|
// segmentTopLevel splits src into chunks whose concatenation is exactly src.
|
|
// Returns a single chunk (the whole source) when there is nothing safe to split.
|
|
func segmentTopLevel(src string) []string {
|
|
cuts := topLevelCuts(src)
|
|
if len(cuts) <= 1 {
|
|
return []string{src}
|
|
}
|
|
chunks := make([]string, 0, len(cuts))
|
|
for i := range cuts {
|
|
end := len(src)
|
|
if i+1 < len(cuts) {
|
|
end = cuts[i+1]
|
|
}
|
|
chunks = append(chunks, src[cuts[i]:end])
|
|
}
|
|
// Defensive: the construction above is lossless, but never return a
|
|
// non-lossless split — a single chunk is always safe.
|
|
if strings.Join(chunks, "") != src {
|
|
return []string{src}
|
|
}
|
|
return chunks
|
|
}
|
|
|
|
// lexical states
|
|
const (
|
|
stNormal = iota
|
|
stLineComment
|
|
stBlockComment
|
|
stSingle // '...'
|
|
stDouble // "..."
|
|
stTemplate
|
|
stRegex
|
|
)
|
|
|
|
// topLevelCuts returns the sorted byte offsets at which chunks begin. Always
|
|
// includes 0. A cut is placed at the start of any line that begins in column 0
|
|
// (no leading whitespace) with a declaration keyword, provided the lexer is in a
|
|
// clean state there — i.e. not inside a block comment, template body, or `${}`
|
|
// interpolation that spans into this line.
|
|
//
|
|
// Column 0 is the top-level signal: these files indent everything inside a
|
|
// function/JSX, so a keyword in column 0 is a top-level declaration. That lets
|
|
// the lexer ignore brace depth and JSX entirely (JSX and nested statements are
|
|
// always indented) — it need only track string/comment/template state so a
|
|
// keyword *inside* a multi-line string or comment isn't mistaken for a boundary.
|
|
// Regexes and single/double strings can't span lines, so any mis-lex of them
|
|
// self-heals at the newline before the next candidate line.
|
|
func topLevelCuts(src string) []int {
|
|
cuts := []int{0}
|
|
n := len(src)
|
|
|
|
state := stNormal
|
|
depth := 0 // only tracked to match `${ ... }` interpolation braces
|
|
// tmplStack holds the interpolation brace depth captured at each `${` so the
|
|
// matching `}` resumes the template body instead of being counted as a plain
|
|
// brace. Non-empty ⇒ we're inside an interpolation (line not a clean start).
|
|
var tmplStack []int
|
|
var prevSig byte // last significant byte, for regex-vs-division
|
|
|
|
addCut := func(off int) {
|
|
if off > cuts[len(cuts)-1] {
|
|
cuts = append(cuts, off)
|
|
}
|
|
}
|
|
// A newline just moved us to lineStart; if the lexer is clean there and the
|
|
// line begins in column 0 with a declaration keyword, it's a chunk boundary.
|
|
checkCut := func(lineStart int) {
|
|
if state == stNormal && len(tmplStack) == 0 && startsDeclKeyword(src, lineStart) {
|
|
addCut(lineStart)
|
|
}
|
|
}
|
|
|
|
checkCut(0)
|
|
|
|
for i := 0; i < n; i++ {
|
|
c := src[i]
|
|
switch state {
|
|
case stNormal:
|
|
switch c {
|
|
case '/':
|
|
if i+1 < n && src[i+1] == '/' {
|
|
state = stLineComment
|
|
i++
|
|
continue
|
|
}
|
|
if i+1 < n && src[i+1] == '*' {
|
|
state = stBlockComment
|
|
i++
|
|
continue
|
|
}
|
|
if regexAllowed(src, i, prevSig) {
|
|
state = stRegex
|
|
prevSig = c
|
|
continue
|
|
}
|
|
prevSig = c
|
|
case '\'':
|
|
state = stSingle
|
|
prevSig = c
|
|
case '"':
|
|
state = stDouble
|
|
prevSig = c
|
|
case '`':
|
|
state = stTemplate
|
|
prevSig = c
|
|
case '{', '(', '[':
|
|
depth++
|
|
prevSig = c
|
|
case '}':
|
|
if len(tmplStack) > 0 && depth == tmplStack[len(tmplStack)-1] {
|
|
tmplStack = tmplStack[:len(tmplStack)-1]
|
|
depth--
|
|
state = stTemplate
|
|
} else {
|
|
if depth > 0 {
|
|
depth--
|
|
}
|
|
prevSig = c
|
|
}
|
|
case ')', ']':
|
|
if depth > 0 {
|
|
depth--
|
|
}
|
|
prevSig = c
|
|
case '\n':
|
|
checkCut(i + 1)
|
|
case ' ', '\t', '\r':
|
|
// insignificant; leave prevSig
|
|
default:
|
|
prevSig = c
|
|
}
|
|
|
|
case stLineComment:
|
|
if c == '\n' {
|
|
state = stNormal
|
|
checkCut(i + 1)
|
|
}
|
|
|
|
case stBlockComment:
|
|
if c == '*' && i+1 < n && src[i+1] == '/' {
|
|
state = stNormal
|
|
i++
|
|
}
|
|
// a newline inside a block comment is not a clean start: no checkCut
|
|
|
|
case stSingle:
|
|
if c == '\\' {
|
|
i++
|
|
} else if c == '\'' {
|
|
state = stNormal
|
|
prevSig = c
|
|
} else if c == '\n' {
|
|
state = stNormal // strings can't span lines; recover
|
|
checkCut(i + 1)
|
|
}
|
|
|
|
case stDouble:
|
|
if c == '\\' {
|
|
i++
|
|
} else if c == '"' {
|
|
state = stNormal
|
|
prevSig = c
|
|
} else if c == '\n' {
|
|
state = stNormal
|
|
checkCut(i + 1)
|
|
}
|
|
|
|
case stTemplate:
|
|
if c == '\\' {
|
|
i++
|
|
} else if c == '`' {
|
|
state = stNormal
|
|
prevSig = c
|
|
} else if c == '$' && i+1 < n && src[i+1] == '{' {
|
|
depth++
|
|
tmplStack = append(tmplStack, depth)
|
|
state = stNormal
|
|
i++
|
|
}
|
|
// templates may span lines; the continuation is not a clean start
|
|
|
|
case stRegex:
|
|
if c == '\\' {
|
|
i++
|
|
} else if c == '[' {
|
|
for i++; i < n; i++ { // character class: skip to `]`
|
|
if src[i] == '\\' {
|
|
i++
|
|
continue
|
|
}
|
|
if src[i] == ']' {
|
|
break
|
|
}
|
|
}
|
|
} else if c == '/' {
|
|
state = stNormal
|
|
prevSig = c
|
|
} else if c == '\n' {
|
|
state = stNormal // regexes can't span lines; recover
|
|
checkCut(i + 1)
|
|
}
|
|
}
|
|
}
|
|
|
|
return cuts
|
|
}
|
|
|
|
// startsDeclKeyword reports whether src[i:] begins with a declaration keyword as
|
|
// a whole word (the next character is not part of an identifier).
|
|
func startsDeclKeyword(src string, i int) bool {
|
|
for _, kw := range declKeywords {
|
|
if strings.HasPrefix(src[i:], kw) {
|
|
j := i + len(kw)
|
|
if j >= len(src) || !isIdentPart(src[j]) {
|
|
return true
|
|
}
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// regexAllowed reports whether a `/` at position i begins a regex literal (as
|
|
// opposed to a division operator), from the preceding significant byte and, when
|
|
// that byte ends an identifier, whether the identifier is a regex-prefix keyword.
|
|
func regexAllowed(src string, i int, prevSig byte) bool {
|
|
if prevSig == 0 {
|
|
return true // start of input
|
|
}
|
|
if isIdentPart(prevSig) {
|
|
// value context (identifier/number) unless the word is a keyword like
|
|
// `return` after which a regex is expected.
|
|
word := trailingWord(src, i)
|
|
return regexPrefixKeywords[word]
|
|
}
|
|
switch prevSig {
|
|
case ')', ']', '}':
|
|
return false // end of a value/call/index
|
|
default:
|
|
// after operators, punctuation, `(`, `,`, `=`, etc. → regex expected
|
|
return true
|
|
}
|
|
}
|
|
|
|
// trailingWord returns the identifier word ending just before the run of
|
|
// whitespace that precedes position i (used to classify the token before a `/`).
|
|
func trailingWord(src string, i int) string {
|
|
j := i
|
|
for j > 0 && (src[j-1] == ' ' || src[j-1] == '\t' || src[j-1] == '\r' || src[j-1] == '\n') {
|
|
j--
|
|
}
|
|
end := j
|
|
for j > 0 && isIdentPart(src[j-1]) {
|
|
j--
|
|
}
|
|
return src[j:end]
|
|
}
|
|
|
|
func isIdentPart(b byte) bool {
|
|
return b == '_' || b == '$' ||
|
|
(b >= 'a' && b <= 'z') || (b >= 'A' && b <= 'Z') || (b >= '0' && b <= '9')
|
|
}
|