diff --git a/.github/workflows/blink_parity_mautic.yaml b/.github/workflows/blink_parity_mautic.yaml index 0620383f8f..fa63c01bf1 100644 --- a/.github/workflows/blink_parity_mautic.yaml +++ b/.github/workflows/blink_parity_mautic.yaml @@ -95,7 +95,7 @@ jobs: - name: Compare the two trees run: | # ratchet gate: fail once the differing share crosses this; lower it as parity improves - MAX_DIFF_PERCENT=0.1 + MAX_DIFF_PERCENT=0.07 total=0 differ=0 differing_files="" diff --git a/blink/internal/fixer/rules/explicit_string_variable.go b/blink/internal/fixer/rules/explicit_string_variable.go index 5b86446d24..26e34c24d9 100644 --- a/blink/internal/fixer/rules/explicit_string_variable.go +++ b/blink/internal/fixer/rules/explicit_string_variable.go @@ -36,7 +36,10 @@ func (ExplicitStringVariable) Fix(s *tokens.Stream) bool { if end < start { continue } - body, hit := wrapInterpolations(v[start:end]) + // a chunk of a split interpolated string begins with "}" (right after a + // "{$...}" close), so a variable adjacent to it is a continuation PHP leaves + // alone + body, hit := wrapInterpolations(v[start:end], v[0] == '}') if hit { s.SetValue(i, v[:start]+body+v[end:]) changed = true @@ -52,6 +55,11 @@ func interpolatedBody(v string) (int, bool) { if len(v) > 0 && v[0] == '"' { return 1, true } + // a continuation chunk of a split interpolated string starts just after the + // "}" that closed the previous "{$...}" + if len(v) > 0 && v[0] == '}' { + return 1, true + } if len(v) >= 3 && v[0] == '<' && v[1] == '<' && v[2] == '<' { p := 3 for p < len(v) && (v[p] == ' ' || v[p] == '\t') { @@ -74,7 +82,9 @@ func interpolatedBody(v string) (int, bool) { // suffixLen is the length of the trailing delimiter after the body: 1 for the // closing quote of a double-quoted string, or the closing heredoc label line. func suffixLen(v string, bodyStart int) int { - if v[0] == '"' { + // a double-quoted string ends with the quote; a split-string chunk ends with + // the closing quote or with the "{" that opens the next "{$...}" - all length 1 + if v[0] == '"' || v[0] == '}' { return 1 } // heredoc: the closing label sits on the last line; the body ends at the @@ -91,11 +101,11 @@ func suffixLen(v string, bodyStart int) int { return len(v) - last // include the newline and the closing label } -func wrapInterpolations(b string) (string, bool) { +func wrapInterpolations(b string, initialAfterCurly bool) (string, bool) { out := make([]byte, 0, len(b)+8) changed := false i := 0 - afterCurly := false + afterCurly := initialAfterCurly for i < len(b) { c := b[i] // A variable directly following a "{...}" interpolation is skipped by diff --git a/blink/internal/lexer/lexer.go b/blink/internal/lexer/lexer.go index bcc294fa18..46fd0693a1 100644 --- a/blink/internal/lexer/lexer.go +++ b/blink/internal/lexer/lexer.go @@ -238,6 +238,14 @@ func (l *lexer) lexBlockComment(start int) { } func (l *lexer) lexString(start int, quote byte) { + // double-quoted strings may carry a complex "{$...}" interpolation whose inner + // expression is split out as real tokens (so array fixers can reflow it); the + // literal chunks keep the quote, the "{" and the "}" so Render stays lossless. + if quote == '"' { + if l.lexInterpolatedString(start) { + return + } + } l.pos++ // opening quote for l.pos < len(l.src) { c := l.src[l.pos] @@ -254,6 +262,81 @@ func (l *lexer) lexString(start int, quote byte) { l.emit(token.String, start) } +// lexInterpolatedString splits a double-quoted string that contains a complex +// "{$...}" interpolation into: a literal chunk ending with "{", the inner +// expression's real tokens, then a chunk starting with "}", repeated, then the +// closing chunk. It returns false (emitting nothing) when the string has no "{$" +// so the caller lexes it as one opaque token. +func (l *lexer) lexInterpolatedString(start int) bool { + // first pass: does the string contain a top-level "{$" before its close? + p := l.pos + 1 + hasInterp := false + for p < len(l.src) { + c := l.src[p] + if c == '\\' && p+1 < len(l.src) { + p += 2 + continue + } + if c == '"' { + break + } + if c == '{' && p+1 < len(l.src) && l.src[p+1] == '$' { + hasInterp = true + break + } + p++ + } + if !hasInterp { + return false + } + + l.pos++ // opening quote + chunkStart := start + for l.pos < len(l.src) { + c := l.src[l.pos] + if c == '\\' && l.pos+1 < len(l.src) { + l.pos += 2 + continue + } + if c == '"' { + l.pos++ // closing quote + break + } + if c == '{' && l.pos+1 < len(l.src) && l.src[l.pos+1] == '$' { + l.pos++ // consume "{", keeping it in this chunk + l.emit(token.String, chunkStart) + // lex the inner expression as real tokens up to the matching "}" + depth := 1 + for l.pos < len(l.src) { + if l.src[l.pos] == '}' && depth == 1 { + break // interpolation close; starts the next chunk + } + before := len(l.toks) + l.lexPHP() + if len(l.toks) == before { + break // defensive: no progress + } + tk := l.toks[len(l.toks)-1] + if tk.Kind == token.Punct { + switch tk.Value { + case "{": + depth++ + case "}": + depth-- + } + } + } + chunkStart = l.pos // the "}" begins the next literal chunk + continue + } + l.pos++ + } + if l.pos > chunkStart { + l.emit(token.String, chunkStart) + } + return true +} + func (l *lexer) hasPrefix(s string) bool { return strings.HasPrefix(l.src[l.pos:], s) }