fresh-repo.sh: two filter-repo passes (the text pass first; filter-repo skips --replace-text over blobs under a --file-info-callback: two dry runs on 7 October 2026 rewrote identities and dates and no text), the founder's names and second login from the encoded list as well as the history, the target address never personal, the drop list reduced to docs/review (the ledger and its fixes file are published, decision of 5 October 2026)

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
igneum-labs 2026-10-07 19:32:56 +00:00
parent cc3a5b334f
commit a8ec7b2fde

View file

@ -85,12 +85,16 @@ TARGET_EMAIL="$STANDING_ID+$TARGET_LOGIN@users.noreply.github.com"
TARGET_IDENT="$TARGET_LOGIN <$TARGET_EMAIL>"
# every other identity: "name|email" pairs (author and committer)
PERSONAL_PAIRS="$(git log --all --format='%an|%ae%n%cn|%ce' | grep -v "|$STANDING_EMAIL$" | sort -u || true)"
PERSONAL_PAIRS="$(git log --all --format='%an|%ae%n%cn|%ce' | grep -v -E "\|($STANDING_EMAIL|$TARGET_EMAIL)$" | sort -u || true)" # never the standing or the target address (7 Oct 2026: the target's own commits read as personal and every rewritten author line then counted as a hit)
PERSONAL_EMAILS="$(printf '%s\n' "$PERSONAL_PAIRS" | awk -F'|' 'NF==2{print $2}' | sort -u)"
PERSONAL_NAMES="$(printf '%s\n' "$PERSONAL_PAIRS" | awk -F'|' 'NF==2{print $1}' | grep -v "^$STANDING_LOGIN$" | sort -u || true)"
FIRST_NAMES="$(printf '%s\n' "$PERSONAL_NAMES" | awk 'NF>=1{print $1}' | sort -u)"
LAST_NAMES="$(printf '%s\n' "$PERSONAL_NAMES" | awk 'NF>=2{print $NF}' | sort -u)"
SECOND_LOGINS="$(printf '%s\n' "$PERSONAL_EMAILS" | sed -nE 's/^[0-9]+\+([A-Za-z0-9-]+)@users\.noreply\.github\.com$/\1/p' | grep -v "^$STANDING_LOGIN$" | sort -u || true)"
PERSONAL_NAMES="$(printf '%s\n' "$PERSONAL_PAIRS" | awk -F'|' 'NF==2 && $1 ~ / /{print $1}' | grep -v -E "^($STANDING_LOGIN|$TARGET_LOGIN)$" | sort -u || true)" # a name has a space; a login is not a name
# the founder's first name, surname and second login from the encoded list (rows 1 to 3, sample column), so the rules never depend on
# which commits a given mirror carries (7 Oct 2026: build-1's mirror holds none of the 40 personal-identity commits)
LIST_ROWS="$(base64 -d < "$ROOT/tools/ci/founder-strings.b64" | grep -vE '^#')"
LIST_FIRST="$(printf '%s\n' "$LIST_ROWS" | sed -n '1p' | cut -f2 | awk '{print $1}')"; LIST_LAST="$(printf '%s\n' "$LIST_ROWS" | sed -n '2p' | cut -f2)"; LIST_SECOND="$(printf '%s\n' "$LIST_ROWS" | sed -n '3p' | cut -f2)"
FIRST_NAMES="$( { printf '%s\n' "$PERSONAL_NAMES" | awk 'NF>=2{print $1}'; printf '%s\n' "$LIST_FIRST"; } | grep . | sort -u)"
LAST_NAMES="$( { printf '%s\n' "$PERSONAL_NAMES" | awk 'NF>=2{print $NF}'; printf '%s\n' "$LIST_LAST"; } | grep . | sort -u)"
SECOND_LOGINS="$( { printf '%s\n' "$PERSONAL_EMAILS" | sed -nE 's/^[0-9]+\+([A-Za-z0-9-]+)@users\.noreply\.github\.com$/\1/p'; printf '%s\n' "$LIST_SECOND"; } | grep . | grep -v -E "^($STANDING_LOGIN|$TARGET_LOGIN)$" | sort -u || true)"
# the other businesses named in the plan (brand names, not people), from the encoded list tools/ci/founder-strings.b64 (rows 4 to 8;
# no tracked file spells them: the founder-strings check reads every tracked file)
OTHER_BUSINESSES="$(base64 -d < "$ROOT/tools/ci/founder-strings.b64" | grep -vE '^#' | sed -n '4,8p' | cut -f1 | sed -E 's/\\b//g' | paste -sd'|' -)"
@ -165,7 +169,7 @@ scan_blobs() {
| perl -ne 'BEGIN { open(P, "<", shift) or die; @p = map { chomp; qr/$_/i } grep { /\S/ } <P>; $n = 0 } for my $p (@p) { if ($_ =~ $p) { $n++; last } } END { print "$n\n" }' "$1"
}
scan_meta() { git log --all --format='%an%n%ae%n%cn%n%ce%n%s%n%b' | perl -ne 'BEGIN { open(P, "<", shift) or die; @p = map { chomp; qr/$_/i } grep { /\S/ } <P>; $n = 0 } for my $p (@p) { if ($_ =~ $p) { $n++; last } } END { print "$n\n" }' "$1"; }
DROPPED=(docs/fud-ledger.md docs/fud-fixes.md docs/review site/ledger.html)
DROPPED=(docs/review) # the ledger, its fixes file and the ledger page are published with the repository (decision of 5 October 2026, docs/fud-fixes.md section 5); the review folder stays internal
counts() { # <label>
local label="$1"
say ""
@ -198,6 +202,9 @@ PY
CALLBACK_ARGS+=(--file-info-callback "$RULES/file-info.py")
fi
T0=$(date +%s)
# pass 1: the text, the messages, the identities and the dates. Pass 2 (below): the public CLAUDE.md in every commit. Two passes
# because filter-repo does not run --replace-text over blobs when a --file-info-callback is present (7 October 2026, 20:3x UK:
# two dry runs rewrote identities and dates and not one line of text).
"${FILTER[@]}" --force --quiet \
--invert-paths "${PATH_ARGS[@]}" \
--replace-text "$REPLACE" \
@ -208,8 +215,11 @@ for attr in ("author_date", "committer_date"):
d = getattr(commit, attr); parts = d.split(b" ")
if len(parts) == 2 and parts[1] != b"+0000":
setattr(commit, attr, parts[0] + b" +0000")
' ${CALLBACK_ARGS[@]+"${CALLBACK_ARGS[@]}"}
say "pass done in $(( $(date +%s) - T0 )) s"
'
say "pass 1 (text, messages, identities, dates) done in $(( $(date +%s) - T0 )) s"
if [ ${#CALLBACK_ARGS[@]} -gt 0 ]; then
T1=$(date +%s); "${FILTER[@]}" --force --quiet "${CALLBACK_ARGS[@]}"; say "pass 2 (the public CLAUDE.md in every commit) done in $(( $(date +%s) - T1 )) s"
fi
[ -f .git/filter-repo/commit-map ] && cp .git/filter-repo/commit-map "$WORK/commit-map" && say "commit-map: $WORK/commit-map ($(grep -c . "$WORK/commit-map") lines; keep it with the private notes)"
[ -f filter-repo/commit-map ] && cp filter-repo/commit-map "$WORK/commit-map" && say "commit-map: $WORK/commit-map ($(grep -c . "$WORK/commit-map") lines; keep it with the private notes)"