1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
|
#!/bin/sh
# One full wash/rinse/repeat cycle of the eval-tuning pipeline:
#
# 1. Dump the currently-compiled-in DNA as this cycle's baseline.
# 2. Tune a candidate DNA against the TWIC position pool.
# 3. Gate: play baseline vs. candidate head-to-head (match_play.py).
# 4. If candidate scores >= 0.5, bake it into eval.c as the new
# baseline, rebuild, and leave eval.c MODIFIED BUT UNCOMMITTED
# for manual review (per project preference -- this script never
# runs `git commit`). If it loses, eval.c is untouched and the
# losing candidate.dna is kept only for the record.
#
# Run this again after you've reviewed/committed a win, or right away
# after a loss -- tune_eval_dna.py resamples a fresh random batch from
# the pool each time regardless.
#
# Usage: cycle.sh [pgn_pool] [n_games] [workers] [max_positions] [max_passes] \
# [sd_depth] [batch_size] [holdout_frac]
# The 4th/5th args are optional tune_eval_dna.py budget overrides --
# leave them unset for a real cycle (full historical budget); pass
# small values (e.g. 2000 1) for a quick dry run of the whole
# pipeline. sd_depth (default 10) is the gate match's fixed search
# depth -- lower it (e.g. 4-5) for a fast pipeline-wiring smoke test;
# sd 10 games can run long on a loaded box (observed: single game
# still going after 15+ min in one dry run) so don't use the default
# for anything time-boxed. batch_size (default: unset, meaning
# tune_eval_dna.py's own default of 1/10th the training pool) and
# holdout_frac (default: unset, meaning tune_eval_dna.py's own 0.1)
# are further tune_eval_dna.py overrides -- see its --help/docstring.
set -e
cd "$(dirname "$0")/.." # repo src/ root
SRC="$(pwd)"
EVAL_TUNE="$SRC/eval_tune"
PGN="${1:-/usr/home/scott/typhoon/pgn/twic_filtered.pgn}"
N_GAMES="${2:-500}"
WORKERS="${3:-4}"
MAX_POSITIONS="${4:-}"
MAX_PASSES="${5:-}"
SD_DEPTH="${6:-10}"
BATCH_SIZE="${7:-}"
HOLDOUT_FRAC="${8:-}"
ts=$(date -u +%Y%m%dT%H%M%SZ)
cycle_dir="$EVAL_TUNE/cycles/$ts"
mkdir -p "$cycle_dir"
echo "=== cycle $ts ==="
echo "pgn pool: $PGN"
echo "cycle dir: $cycle_dir"
echo "--- building current baseline ---"
gmake -j5 GENETIC=1 PERF_COUNTERS=1 MP=1 SIXTYFOUR=1 >"$cycle_dir/build_baseline.log" 2>&1
echo "--- dumping baseline DNA ---"
./typhoon --batch --command "evaldna write $cycle_dir/baseline.dna" >/dev/null 2>&1 || true
if [ ! -s "$cycle_dir/baseline.dna" ]; then
echo "ERROR: evaldna write did not produce $cycle_dir/baseline.dna" >&2
exit 1
fi
cp eval.c "$cycle_dir/eval.c.baseline"
echo "--- tuning candidate DNA (this can take hours) ---"
python3 "$EVAL_TUNE/tune_eval_dna.py" "$SRC/typhoon" "$PGN" $MAX_POSITIONS $MAX_PASSES \
$BATCH_SIZE $HOLDOUT_FRAC \
> "$cycle_dir/tune.log" 2>&1 || {
echo "tune_eval_dna.py failed -- see $cycle_dir/tune.log" >&2
exit 1
}
# tune_eval_dna.py's __main__ currently hardcodes out_path="tuned.dna"
# in its own cwd; capture wherever it actually landed.
if [ -f "$EVAL_TUNE/tuned.dna" ]; then
mv "$EVAL_TUNE/tuned.dna" "$cycle_dir/candidate.dna"
elif [ -f "tuned.dna" ]; then
mv "tuned.dna" "$cycle_dir/candidate.dna"
else
echo "couldn't find tuned.dna output -- see $cycle_dir/tune.log" >&2
exit 1
fi
echo "--- gate: baseline vs candidate, $N_GAMES games @ sd $SD_DEPTH ---"
match_out="$cycle_dir/match_result.txt"
python3 "$EVAL_TUNE/match_play.py" "$SRC/typhoon" \
"$cycle_dir/baseline.dna" "$cycle_dir/candidate.dna" \
--pgn "$PGN" --games "$N_GAMES" --workers "$WORKERS" --sd "$SD_DEPTH" \
--log "$cycle_dir/match_games.log" --pgn-out "$cycle_dir/match_games.pgn" \
2> "$cycle_dir/match_stderr.log" | tee "$match_out"
score=$(grep -o 'CANDIDATE_SCORE=[0-9.]*' "$match_out" | cut -d= -f2)
lower95=$(grep -o 'LOWER95=[0-9.-]*' "$match_out" | cut -d= -f2)
if [ -z "$score" ] || [ -z "$lower95" ]; then
echo "couldn't parse CANDIDATE_SCORE/LOWER95 from match_play.py output" >&2
exit 1
fi
# Gate on the 95% confidence LOWER bound, not the raw point estimate:
# search is non-deterministic (MP=1 multithreaded), so a bare score
# >= 0.5 is not evidence the candidate is actually better -- only a
# lower bound that still clears break-even is.
pass=$(awk -v s="$lower95" 'BEGIN { print (s >= 0.5) ? "1" : "0" }')
summary="$cycle_dir/summary.txt"
{
echo "cycle: $ts"
echo "pgn pool: $PGN"
echo "games: $N_GAMES @ sd $SD_DEPTH"
cat "$match_out"
echo
} > "$summary"
if [ "$pass" = "1" ]; then
echo "=== candidate lower95 $lower95 (score $score) >= 0.5: baking in as new baseline ==="
python3 "$EVAL_TUNE/bake_dna.py" "$cycle_dir/candidate.dna" \
--eval-c "$SRC/eval.c" --out "$SRC/eval.c"
echo "--- rebuilding with new baseline ---"
gmake clean >"$cycle_dir/build_candidate.log" 2>&1
gmake -j5 GENETIC=1 PERF_COUNTERS=1 MP=1 SIXTYFOUR=1 >>"$cycle_dir/build_candidate.log" 2>&1
python3 "$EVAL_TUNE/dna_diff.py" "$cycle_dir/eval.c.baseline" "$cycle_dir/candidate.dna" \
> "$cycle_dir/dna_diff.txt" 2>&1 || true
{
echo "RESULT: KEPT -- eval.c modified, NOT committed."
echo "Review: git diff eval.c"
echo "Then commit yourself, e.g.:"
echo " git add eval.c"
echo " git commit -m 'Eval DNA tune cycle $ts: score $score vs prior baseline (pgn=$PGN, $N_GAMES games)'"
} >> "$summary"
else
echo "=== candidate lower95 $lower95 (score $score) < 0.5: discarding, eval.c untouched ==="
{
echo "RESULT: DISCARDED -- candidate underperformed baseline."
echo "candidate.dna and match games kept in $cycle_dir for the record."
} >> "$summary"
fi
cat "$summary"
|