#!/usr/bin/env bash # The lost-quote class (5 October 2026, twice in one night): a PowerShell job script carries a bash body inside a # string, a quote is lost on the way through PowerShell, bash refuses the whole body, and the job either reports # exit 0 having done nothing (tools/amd-prove/pc1-cpu-prove.ps1, first version: an apostrophe inside a single-quoted # awk program) or fails in 4 s (the 0.3.10 installer job). Rule (the 0.3.6 cut): a bash body in a PowerShell job is # written to a file and run with `bash `, never inline. This check reads every *.ps1 under relay/playbooks/ and # tools/, finds each bash body however it is handed over (`wsl ... bash -c "..."`, `bash -lc '...'`, `bash -c $var`, # a `+` concatenation in parentheses, a here-string written to a file that is later run with bash), unescapes it the # way PowerShell would (backtick escapes and "" in double-quoted strings, '' in single-quoted strings, here-strings # verbatim; `$var` interpolation left as-is, a PowerShell `$(...)` subexpression replaced by `${PS_SUBEXPR}`; in a `+` # concatenation a variable becomes `${name}`), and runs `bash -n` on it. One line per body: file, line, ok or the # bash -n error. A body the extractor sees but cannot read (`bash -c` with an argument shape it does not parse, or a # variable with no literal assignment above) is "unextractable body" and FAILS: a skip would be a hole in the check. # Exit 1 on any failure. # # Usage: tools/ci/bash-body-check.sh # the tree (tools/ci/fixtures/ left out) # tools/ci/bash-body-check.sh ... # named files # tools/ci/bash-body-check.sh --self-test # must fire on the lost-quote and unextractable fixtures under # # tools/ci/fixtures/ and stay quiet on the correct one # Needs python3 (the extractor) and bash (the parser). Runs on this Mac's bash 3.2. set -uo pipefail HERE="$(cd "$(dirname "$0")" && pwd)" REPO="$(cd "$HERE/../.." && pwd)" # ---- the extractor: python3 reads the ps1 files, writes one body per file under $1, prints an index to stdout ---- # index line: idx file line kind offset status detail # offset: file line = offset + body line (bash -n reports body lines); status: body | unextractable extract() { python3 - "$@" <<'PY' import sys, os, re out_dir = sys.argv[1] files = sys.argv[2:] SUBEXPR = '${PS_SUBEXPR}' BT = chr(96) BT_MAP = {'n': '\n', 't': '\t', 'r': '\r', '0': '\0', 'a': '\a', 'b': '\b', 'f': '\f', 'v': '\v', 'e': '\x1b'} def skip_subexpr(s, i): """s[i] == '(' of a PowerShell $( ... ); returns the index after the matching ')'.""" depth = 0 n = len(s) while i < n: c = s[i] if c == '(': depth += 1; i += 1 elif c == ')': depth -= 1; i += 1 if depth == 0: return i elif c == "'": i += 1 while i < n: if s[i] == "'": if i + 1 < n and s[i+1] == "'": i += 2; continue i += 1; break i += 1 elif c == '"': i = skip_dq(s, i + 1) else: i += 1 return n def skip_dq(s, i): """s[i] is the first char inside a double-quoted string; returns the index after the closing quote.""" n = len(s) while i < n: c = s[i] if c == BT: i += 2; continue if c == '"': if i + 1 < n and s[i+1] == '"': i += 2; continue return i + 1 if c == '$' and i + 1 < n and s[i+1] == '(': i = skip_subexpr(s, i + 1); continue i += 1 return n def decode_dq(raw, here): """A double-quoted string or here-string the way PowerShell reads it, with $var left as-is.""" o = []; i = 0; n = len(raw) while i < n: c = raw[i] if c == BT and i + 1 < n: o.append(BT_MAP.get(raw[i+1], raw[i+1])); i += 2; continue if c == '"' and not here and i + 1 < n and raw[i+1] == '"': o.append('"'); i += 2; continue if c == '$' and i + 1 < n and raw[i+1] == '(': o.append(SUBEXPR); i = skip_subexpr(raw, i + 1); continue o.append(c); i += 1 return ''.join(o) def tokenize(text): """PowerShell tokens: ('str', value, line, sub) | ('var', name, line, None) | ('word', text, line, None) | ('op', char, line, None) | ('nl', '', line, None). Comments are dropped, a trailing backtick joins lines.""" toks = []; i = 0; n = len(text); line = 1 SPECIAL = set(' \t\r\n()[]{},;|=+\'"') while i < n: c = text[i] if c == '\n': toks.append(('nl', '', line, None)); line += 1; i += 1; continue if c in ' \t\r': i += 1; continue if c == BT and i + 1 < n and text[i+1] in '\r\n': i += 1 while i < n and text[i] == '\r': i += 1 if i < n and text[i] == '\n': line += 1; i += 1 continue if text.startswith('<#', i): j = text.find('#>', i + 2) if j < 0: j = n line += text.count('\n', i, j); i = j + 2; continue if c == '#': j = text.find('\n', i) i = n if j < 0 else j; continue if text.startswith('@"', i) or text.startswith("@'", i): q = text[i+1] j = i + 2 while j < n and text[j] == '\r': j += 1 if j < n and text[j] == '\n': start = j + 1 m = re.compile(r'(?:^|\n)' + re.escape(q + '@')).search(text, start - 1) if m: end = m.start() if text[m.start()] == '\n' else m.start() raw = text[start:end] sub = 'hdq' if q == '"' else 'hsq' val = decode_dq(raw, True) if q == '"' else raw toks.append(('str', val, line, sub)) line += text.count('\n', i, m.end()); i = m.end(); continue # not a here-string after all: fall through as an operator toks.append(('op', '@', line, None)); i += 1; continue if c == "'": j = i + 1; o = [] while j < n: if text[j] == "'": if j + 1 < n and text[j+1] == "'": o.append("'"); j += 2; continue break o.append(text[j]); j += 1 toks.append(('str', ''.join(o), line, 'sq')) line += text.count('\n', i, j); i = j + 1; continue if c == '"': j = skip_dq(text, i + 1) raw = text[i+1:j-1] if j <= n and text[j-1] == '"' else text[i+1:j] toks.append(('str', decode_dq(raw, False), line, 'dq')) line += text.count('\n', i, j); i = j; continue if c == '$': if i + 1 < n and text[i+1] == '(': j = skip_subexpr(text, i + 1) toks.append(('subexpr', text[i:j], line, None)) line += text.count('\n', i, j); i = j; continue if i + 1 < n and text[i+1] == '{': j = text.find('}', i) if j < 0: j = n - 1 toks.append(('var', text[i+2:j], line, None)); i = j + 1; continue m = re.compile(r'\$([A-Za-z_][A-Za-z0-9_]*(?::[A-Za-z_][A-Za-z0-9_]*)?)').match(text, i) if m: toks.append(('var', m.group(1), line, None)); i = m.end(); continue toks.append(('op', '$', line, None)); i += 1; continue if c in SPECIAL: toks.append(('op', c, line, None)); i += 1; continue j = i while j < n and text[j] not in SPECIAL: j += 1 toks.append(('word', text[i:j], line, None)); i = j toks.append(('nl', '', line, None)) return toks def is_bash_word(t): v = t[1] return t[0] in ('word', 'str') and (v in ('bash', 'bash.exe') or v.endswith('/bin/bash') or v.endswith('\\bash.exe')) STOP_OPS = set('|;{})') def parse_concat(toks, i, assigns, stop_at_nl): """Reads str|var (+ str|var)* from toks[i]; returns (value, kind, next_index, problem). kind: lit | here | concat.""" parts = []; kinds = []; problem = None want_operand = True while i < len(toks): t = toks[i] if t[0] == 'nl': if stop_at_nl and not want_operand: break if want_operand and parts: i += 1; continue # a newline after '+' continues the expression if not parts: break i += 1; continue if want_operand: if t[0] == 'str': parts.append(t[1]); kinds.append('here' if t[3] in ('hdq', 'hsq') else 'lit'); want_operand = False; i += 1; continue if t[0] == 'var': parts.append('${' + t[1] + '}'); kinds.append('var'); want_operand = False; i += 1; continue problem = 'expression token %r at line %d' % (t[1] or t[0], t[2]); break else: if t[0] == 'op' and t[1] == '+': want_operand = True; i += 1; continue if t[0] == 'op' and t[1] in STOP_OPS: break if t[0] == 'word' and t[1].startswith('-'): problem = 'operator %s at line %d' % (t[1], t[2]); break if t[0] in ('word', 'var', 'str', 'subexpr', 'op'): problem = 'expression token %r at line %d' % (t[1] or t[0], t[2]); break break if want_operand and parts and not problem: problem = 'expression ends after +' if not parts and not problem: problem = 'no string' if problem: return None, None, i, problem if len(parts) == 1: kind = kinds[0] else: kind = 'concat' return ''.join(parts), kind, i, None def collect_assignments(toks): """$name = at statement start. {name: [(line, value|None, kind, problem)]}""" a = {} for i, t in enumerate(toks): if t[0] != 'var': continue prev = toks[i-1] if i > 0 else None if prev is not None and not (prev[0] == 'nl' or (prev[0] == 'op' and prev[1] in ';{')): continue if i + 1 >= len(toks) or toks[i+1] != ('op', '=', toks[i+1][2], None): continue val, kind, _, problem = parse_concat(toks, i + 2, a, True) a.setdefault(t[1], []).append((t[2], val, kind, problem)) return a def lookup(assigns, name, line): best = None for (l, val, kind, problem) in assigns.get(name, []): if l < line: best = (l, val, kind, problem) return best def bash_args(toks, i): """Arguments after a bash word: list of (token-or-expr, index). An expr is ('expr', value, line, kind, problem).""" args = [] while i < len(toks): t = toks[i] if t[0] == 'nl': break if t[0] == 'op' and t[1] in STOP_OPS: break if t[0] == 'op' and t[1] == ',': i += 1; continue if t[0] == 'word' and re.match(r'^[0-9]*>', t[1]): break if t[0] == 'op' and t[1] == '(': val, kind, j, problem = parse_concat(toks, i + 1, None, False) # skip to the matching ')' depth = 1; k = i + 1 while k < len(toks) and depth > 0: if toks[k][0] == 'op' and toks[k][1] == '(': depth += 1 elif toks[k][0] == 'op' and toks[k][1] == ')': depth -= 1 k += 1 if problem is None and j < k - 1: problem = 'more than strings and + inside the parentheses' names = set(x[1] for x in toks[i+1:k] if x[0] == 'var') args.append((('expr', val, t[2], kind, problem, names), i)); i = k; continue if t[0] == 'op' and t[1] == ')': break args.append((t, i)); i += 1 return args def value_of(arg, assigns): """(value, def_line, kind, problem) for an argument token.""" t = arg if t[0] == 'str': return t[1], t[2], ('here' if t[3] in ('hdq', 'hsq') else 'lit'), None if t[0] == 'expr': return t[1], t[2], t[3], t[4] if t[0] == 'var': a = lookup(assigns, t[1], t[2]) if a is None: return None, t[2], 'var', '$%s has no literal assignment above line %d' % (t[1], t[2]) l, val, kind, problem = a if problem: return None, l, 'var', '$%s = (line %d) is not a string: %s' % (t[1], l, problem) return val, l, kind, None if t[0] == 'subexpr': return None, t[2], 'subexpr', 'a $(...) subexpression' return None, t[2], t[0], 'token %r' % t[1] def option_value(arg): if arg[0] in ('word', 'str'): return arg[1] return None def find_bodies(path, toks): assigns = collect_assignments(toks) bodies = [] # (line, kind, value, offset, problem, def_line) seen = set() inv = [] # (index, args) of every bash invocation for i, t in enumerate(toks): if not is_bash_word(t): continue args = bash_args(toks, i + 1) inv.append((i, args)) want_body = False; body_done = False; skip_next = False for (a, ai) in args: if body_done: break if skip_next: skip_next = False; continue if not want_body: ov = option_value(a) if ov is not None and re.match(r'^-[A-Za-z]+$', ov): if 'c' in ov: want_body = True if ov[-1] in ('o', 'O'): skip_next = True continue if ov is not None and ov.startswith('--'): continue if ov is not None and re.match(r'^\+[A-Za-z]+$', ov): if ov[-1] == 'O': skip_next = True continue break # a script file: not an inline body val, dline, kind, problem = value_of(a, assigns) body_done = True how = 'bash -c ' + ('$' + a[1] if a[0] == 'var' else ('(...)' if a[0] == 'expr' else 'literal')) if problem: bodies.append((t[2], how, None, 0, problem, dline)) else: offset = dline if kind == 'here' else dline - 1 key = (dline, how) if key in seen: continue seen.add(key); bodies.append((t[2], how, val, offset, None, dline)) if want_body and not body_done: bodies.append((t[2], 'bash -c', None, 0, 'no body argument after -c', t[2])) # here-strings (and other string variables) written to a file that is later run with bash WRITERS = ('WriteAllText', 'WriteAllLines', 'Set-Content', 'Out-File', 'Add-Content') lines_tokens = {} for t in toks: lines_tokens.setdefault(t[2], []).append(t) for name, lst in assigns.items(): for (l, val, kind, problem) in lst: if val is None: continue for wl, lt in lines_tokens.items(): if wl <= l: continue if not any(x[0] == 'word' and any(w in x[1] for w in WRITERS) for x in lt): continue if not any(x[0] == 'var' and x[1] == name for x in lt): continue paths = set(x[1] for x in lt if x[0] == 'var' and x[1] != name) sh_literal = any(x[0] == 'str' and x[1].endswith('.sh') for x in lt) # a variable derived from the path ($w = WslPath $bashFile) names the same file for al in sorted(lines_tokens): if al <= wl: continue alt = lines_tokens[al] if len(alt) > 1 and alt[0][0] == 'var' and alt[1] == ('op', '=', al, None): if any(x[0] == 'var' and x[1] in paths for x in alt[1:]): paths.add(alt[0][1]) run = False for (bi, args) in inv: if toks[bi][2] <= wl: continue for (a, ai) in args: if a[0] == 'var' and a[1] in paths: run = True if a[0] == 'expr' and (a[5] & paths): run = True if a[0] == 'str' and a[1].endswith('.sh') and sh_literal: run = True if run: break if not run: continue how = '$%s written at line %d and run with bash' % (name, wl) key = (l, how) if key in seen: continue seen.add(key) bodies.append((l, how, val, l if kind == 'here' else l - 1, None, l)) break bodies.sort(key=lambda b: (b[5], b[0])) return bodies idx = 0 for path in files: try: text = open(path, encoding='utf-8', errors='replace').read() except OSError as e: print('\t'.join(['0', path, '0', 'read', '0', 'unextractable', str(e)])); continue toks = tokenize(text) for (line, how, val, offset, problem, dline) in find_bodies(path, toks): idx += 1 if problem: print('\t'.join([str(idx), path, str(line), how, '0', 'unextractable', problem])) else: with open(os.path.join(out_dir, '%d.body' % idx), 'w', encoding='utf-8') as f: f.write(val + '\n') print('\t'.join([str(idx), path, str(dline), how, str(offset), 'body', ''])) PY } # ---- the check over a list of files; prints one line per body, returns 1 on any failure ---- check_files() { local T; T="$(mktemp -d)" local rc=0 n=0 index index="$T/index.tsv" if ! extract "$T" "$@" > "$index"; then echo "FAIL extractor error"; rm -rf "$T"; return 1; fi while IFS="$(printf '\t')" read -r idx file line how offset status detail; do [ -n "$idx" ] || continue n=$((n + 1)) if [ "$status" = "unextractable" ]; then echo "FAIL $file:$line unextractable body ($how): $detail"; rc=1; continue fi local err if err="$(bash -n "$T/$idx.body" 2>&1)"; then echo "ok $file:$line ($how)" else # bash prints ": line N: message"; map N to the file line err="$(printf '%s\n' "$err" | sed -e "s#^$T/$idx.body: ##" | awk -v off="$offset" '{ if (match($0, /^line [0-9]+:/)) { n = substr($0, 6, RLENGTH - 6) + 0; sub(/^line [0-9]+:/, "file line " (n + off) " (body line " n "):") } print }' | tr '\n' ' ')" echo "FAIL $file:$line ($how): $err"; rc=1 fi done < "$index" rm -rf "$T" echo "bash-body-check: $n bodies in $# files, $( [ $rc = 0 ] && echo 'all parse' || echo 'FAILURES above')" return $rc } if [ "${1:-}" = "--self-test" ]; then F="$HERE/fixtures" ok_out="$(check_files "$F/bash-body-ok.ps1" 2>&1)"; ok_rc=$? bad_out="$(check_files "$F/bash-body-lost-quote.ps1" 2>&1)"; bad_rc=$? unx_out="$(check_files "$F/bash-body-unextractable.ps1" 2>&1)"; unx_rc=$? echo "self-test correct fixture: rc $ok_rc"; printf '%s\n' "$ok_out" | sed 's/^/ /' echo "self-test lost-quote fixture: rc $bad_rc"; printf '%s\n' "$bad_out" | sed 's/^/ /' echo "self-test unextractable fixture: rc $unx_rc"; printf '%s\n' "$unx_out" | sed 's/^/ /' [ $ok_rc -eq 0 ] || { echo "SELF-TEST FAILED: the correct fixture was flagged"; exit 1; } [ "$(printf '%s\n' "$ok_out" | grep -c '^ok ')" -eq 8 ] || { echo "SELF-TEST FAILED: the correct fixture has 8 bodies, a different number was found"; exit 1; } [ $bad_rc -ne 0 ] || { echo "SELF-TEST FAILED: the lost-quote fixture passed"; exit 1; } for want in 'bash-body-lost-quote.ps1:9 ' 'bash-body-lost-quote.ps1:18 ' 'bash-body-lost-quote.ps1:20 '; do printf '%s\n' "$bad_out" | grep -q "^FAIL .*$want" || { echo "SELF-TEST FAILED: the lost quote at $want was not reported"; exit 1; } done [ "$(printf '%s\n' "$bad_out" | grep -c '^ok ')" -eq 1 ] || { echo "SELF-TEST FAILED: the one correct body in the lost-quote fixture was not reported ok"; exit 1; } [ $unx_rc -ne 0 ] || { echo "SELF-TEST FAILED: the unextractable fixture passed"; exit 1; } [ "$(printf '%s\n' "$unx_out" | grep -c 'unextractable body')" -eq 3 ] || { echo "SELF-TEST FAILED: 3 unextractable bodies expected"; exit 1; } echo "self-test passed: bash -n fires on the lost quotes, the unreadable bodies fail, the correct bodies pass" exit 0 fi if [ $# -gt 0 ]; then check_files "$@"; exit $? fi cd "$REPO" files=$(git ls-files 'relay/playbooks/**' 'tools/**' | grep -E '\.ps1$' | grep -v '^tools/ci/fixtures/' || true) if [ -z "$files" ]; then echo "bash-body-check: no .ps1 files under relay/playbooks/ or tools/"; exit 0; fi # shellcheck disable=SC2086 check_files $files