Repository navigation
Expand file tree
/
Copy pathEXPERIMENTS.dot
More file actions
218 lines (218 loc) · 58.6 KB
/
Copy pathEXPERIMENTS.dot
File metadata and controls
218 lines (218 loc) · 58.6 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
digraph experiments {
rankdir=TB; bgcolor="white";
graph [fontname="Helvetica", ranksep=1.2, nodesep=0.65, fontsize=13, concentrate=true, splines=spline, pad=0.4];
node [shape=plaintext, fontname="Helvetica"];
edge [fontname="Helvetica", fontsize=8, color="#888888", penwidth=1.1, arrowsize=0.7];
subgraph cluster_found { label=<<B>FOUNDATIONS — SFT scale ladder (contaminated lineage)</B>>; fontsize=13; color="#cccccc"; style="rounded"; bgcolor="#fbfbfb";
sft_3m [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#cfcfcf"><B>char model, plain SFT</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈3M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">writes a guess letter-by-letter; spell warm-up, then imitates a near-optimal solver</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.205 · valid 0.54</B></FONT></TD></TR></TABLE>>];
sft_5m [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>deeper SFT + cosine</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈5M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">deeper net, deeper warm-up, more teacher data</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.300 · valid 0.61</B></FONT></TD></TR></TABLE>>];
sft_25m [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>strong-teacher SFT</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈25M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">5x bigger, 80% near-optimal teacher imitation</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.391 · valid 0.66</B></FONT></TD></TR></TABLE>>];
sft_25m_div [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>diverse-secret SFT</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈25M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">train on full valid-word secrets (not just answers) to stop memorizing</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.220 (gap collapsed)</B></FONT></TD></TR></TABLE>>];
scale_99m [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>pure-scale test</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>99M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">the 25M recipe scaled up, old answer-only data</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>plateaued / under-converged</B></FONT></TD></TR></TABLE>>];
codesign_50m [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>co-design + curriculum</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">redesigned net + difficulty-ordered diverse curriculum</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.188 (rare-word dilution)</B></FONT></TD></TR></TABLE>>];
sft_deep [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>deep SFT, converged</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">redesigned net on strong-teacher answers, trained to convergence; no CoT/aux</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.402 · valid 0.664</B></FONT></TD></TR></TABLE>>];
sft_aux [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>+ spelling helper (aux)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">training-only loss pushing each letter toward real-word paths; no dict at play</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.436 · valid 0.675</B></FONT></TD></TR></TABLE>>];
sft_aux_xl [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>scale + aux</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>98M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">scale and aux stacked, wall-clock capped</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>superseded (no milestone)</B></FONT></TD></TR></TABLE>>];
}
subgraph cluster_cot { label=<<B>FOUNDATIONS — chain-of-thought thread</B>>; fontsize=13; color="#cccccc"; style="rounded"; bgcolor="#fbfbfb";
cot_14m [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>CoT prototype</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈14M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">writes a short <think> candidate list, then commits</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>no-CoT 0.155 vs CoT 0.415</B></FONT></TD></TR></TABLE>>];
cot_50m [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>CoT scaled</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">scale the winning CoT with teacher reasoning traces</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.456 (later → ~0.192)</B></FONT></TD></TR></TABLE>>];
cot_50m_aux [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>CoT + aux</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">stack spelling helper on CoT</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>incomplete / superseded</B></FONT></TD></TR></TABLE>>];
cot_show [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#e8c98a"><B>CoT integrity teardown</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">A/B: teacher-context (rebuilds past think) vs self-context</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.450 vs 0.192 — LEAK found</B></FONT></TD></TR></TABLE>>];
cot_eph [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>ephemeral CoT</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">throwaway scratchpad: think regenerated each turn, discarded at play; board-only history</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.430 · valid 0.671</B></FONT></TD></TR></TABLE>>];
cot_eph_aux [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>ephemeral CoT + aux</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">stack search (CoT) + spelling (aux); long schedule</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.616 · valid 0.788 (era best)</B></FONT></TD></TR></TABLE>>];
}
subgraph cluster_val { label=<<B>TOKENIZER / REPRESENTATION PROBES</B>>; fontsize=13; color="#cccccc"; style="rounded"; bgcolor="#fbfbfb";
bpe_12m [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>BPE tokenizer</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈12M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">from-scratch subword tokenizer on the word list; guesses as letter-chunks</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.212 · valid 0.66→0.85</B></FONT></TD></TR></TABLE>>];
bpe_50m [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>BPE tokenizer, scaled</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">same BPE recipe, bigger</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.188 · valid 0.806 (win flat)</B></FONT></TD></TR></TABLE>>];
oreo_11m [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>real-text pretrain</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈11M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">TinyStories byte-BPE pretrain, then SFT on game transcripts</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.257 / seen 0.87</B></FONT></TD></TR></TABLE>>];
oreo_50m [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>real-text pretrain, scaled</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">same recipe, bigger, more passes</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.190 (overfits)</B></FONT></TD></TR></TABLE>>];
structured_ctx [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>structured-context A/B</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">raw board vs board + explicit greens/present/absent block</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>raw 0.260 vs +state 0.170</B></FONT></TD></TR></TABLE>>];
}
subgraph cluster_rl { label=<<B>REINFORCEMENT LEARNING (×10 GRPO + DPO + DAgger, on contaminated bases)</B>>; fontsize=13; color="#cccccc"; style="rounded"; bgcolor="#fbfbfb";
grpo_5m [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>GRPO #1</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>4.8M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">RL over the train set, loose KL, to move greedy play</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>~0.29 null (memorizes train)</B></FONT></TD></TR></TABLE>>];
grpo_5m_div [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>GRPO #2 diverse</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈5M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">GRPO over 14k non-memorizable secrets</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>~0.27, reward negative</B></FONT></TD></TR></TABLE>>];
self_distill_25m [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>self-distill beam+dict→greedy</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈25M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">SFT on its own always-valid beam+dict games</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.384 (spelling up, win flat)</B></FONT></TD></TR></TABLE>>];
rl_consistency [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>GRPO #3 legality reward</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈25M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">flat reward for legal/consistent guesses</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.384 null</B></FONT></TD></TR></TABLE>>];
rl_perguess [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>GRPO #4 per-guess</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈25M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">episode = one guess, clean per-guess credit</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.384 null</B></FONT></TD></TR></TABLE>>];
rl_dict [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>GRPO #5 dict-in-loop</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈25M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">behavior policy samples trie-valid words, push free-gen toward high-adv</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.384 null</B></FONT></TD></TR></TABLE>>];
rl_constrained [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>GRPO #6 consistency (decisive)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈25M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">behavior samples the still-consistent set (answer surfaced ~22%)</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.389 null — barrier is generalization</B></FONT></TD></TR></TABLE>>];
rl_polish [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>GRPO #7 polish</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">validity+consistency reward on the 0.436 base, revert-on-regress</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.436 unchanged</B></FONT></TD></TR></TABLE>>];
rl_infogain [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>GRPO #8 info-gain</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">add the information-gain reward term + 12 train guesses</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.436 unchanged (RL closed)</B></FONT></TD></TR></TABLE>>];
rl_expert_10row [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>expert-iteration (10-row)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">keep winning rollouts, rebuild with clean think, SFT; teach rows 7–10</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.604→0.646 (the RL that works)</B></FONT></TD></TR></TABLE>>];
rl_expert_tail [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>reachability expert-iter</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">full coverage + tail high-K/high-temp sampling</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>solved 99.5%; SFT reverted</B></FONT></TD></TR></TABLE>>];
rl_grpo_polish [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>GRPO #9 token-level</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">stabilized token-GRPO on the CoT policy (eval-mode forward, k3 KL)</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>6r 0.622 / 10r 0.637 (flat)</B></FONT></TD></TR></TABLE>>];
dpo_commit [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>DPO commit-sharpening</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">DPO on win/loss first-divergence commit pairs; raise P(winning commit)</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.631 (era best, +1.5)</B></FONT></TD></TR></TABLE>>];
dpo_decisive [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>DPO decisive-board</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">clean labels: secret-commit vs wrong consistent word at the same board</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>flat → reverted (think dilutes)</B></FONT></TD></TR></TABLE>>];
dpo_guessonly [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>DPO guess-only</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">score preference on only the 5 committed letters</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>knife-edge: flat or collapses</B></FONT></TD></TR></TABLE>>];
constraint_aux [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>constraint-aux fine-tune</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">extra aux: keep greens / reuse yellows / no repeats</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>no-op (zero gradient, OOD) → reverted</B></FONT></TD></TR></TABLE>>];
dagger_v1 [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>DAgger v1</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">relabel its bad boards with teacher's word, SFT (corrections ~1:6)</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>reverted (under-weighted)</B></FONT></TD></TR></TABLE>>];
rl_grpo_reward [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>GRPO #10 + shaped reward</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">token-GRPO with the full updated reward (repeat/drop-present)</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>greedy DECLINED 0.615→0.583 (proxy-hack)</B></FONT></TD></TR></TABLE>>];
rl_grpo_guessonly [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>GRPO guess-only credit</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">credit only the 5 guess letters</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>KL EXPLODED 0.01→12.7</B></FONT></TD></TR></TABLE>>];
dagger_v2 [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>DAgger v2 (full coverage)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">all secrets, corrections upweighted ×4</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>null (later clean → 0.14–0.17)</B></FONT></TD></TR></TABLE>>];
}
subgraph cluster_audit { label=<<B>AUDIT → CLEAN RE-RUN (the honesty pivot)</B>>; fontsize=13; color="#cccccc"; style="rounded"; bgcolor="#fbfbfb";
audit [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#e8c98a"><B>4-agent adversarial audit</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>method — no model</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">Lead/Code/Telemetry/Devil's-Advocate, every claim a file:line receipt</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>library OK; pipeline contaminated</B></FONT></TD></TR></TABLE>>];
clean_rerun [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>clean re-run (leak-free)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">the 0.616 recipe with train-only pools, disjoint VAL/TEST</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.616 → 0.166 — OVERTURNS</B></FONT></TD></TR></TABLE>>];
overnight_clean_sft [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>overnight clean SFT</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">the clean ephemeral-CoT+aux base (3-seed)</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.166 (strict held-out)</B></FONT></TD></TR></TABLE>>];
overnight_clean_dpo [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>clean DPO</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">DPO on the clean base</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.166 null (gain was contamination)</B></FONT></TD></TR></TABLE>>];
overnight_clean_grpo [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>clean GRPO</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">stabilized GRPO + full reward on clean base</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.166 null (11th GRPO)</B></FONT></TD></TR></TABLE>>];
overnight_clean_dagger [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>clean DAgger</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">failure-state relabel on clean base</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.144 (hurts)</B></FONT></TD></TR></TABLE>>];
overnight_clean_dagger2 [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>clean DAgger ×2</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">full-coverage DAgger ×4 corrections</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.169</B></FONT></TD></TR></TABLE>>];
wordley_audit [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#e8c98a"><B>cross-impl audit: colleague's wordley</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>0.5B-era params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">audit a separate impl (free-gen Plan A): honest? check inference + training pipeline + split</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>INFERENCE honest (beam/26-letters, non-word wastes a turn, no dict; split disjoint) BUT training LEAK — consistency set drawn from FULL dict incl held-out -> 100% of held-out trained as spelling targets (72% tight, 26% near-determined); 46.65->61.7% = max-leak config. Ours stays clean</B></FONT></TD></TR></TABLE>>];
}
subgraph cluster_scale { label=<<B>FAIR RE-BUILD + SCALE SWEEP (clean)</B>>; fontsize=13; color="#cccccc"; style="rounded"; bgcolor="#fbfbfb";
fair_stage1 [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#7ec8d0"><B>FAIR model · STAGE-1</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">candidate/teacher pools = full dictionary (knows spelling, not answer-hood); pretrain 30 + aux λ=1.0</FONT></TD></TR><TR><TD BGCOLOR="#1f5e66"><FONT COLOR="white" POINT-SIZE="10"><B>0.281 · valid 0.662 · avg 4.33</B></FONT></TD></TR></TABLE>>];
distill_constrained [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>constrained self-distillation</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">imitate its own dictionary-constrained rollouts (+aux), eval free-gen</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>valid 0.62→0.80, win flat (traded)</B></FONT></TD></TR></TABLE>>];
infogain_xit_c [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>info-gain XIT (constrained)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">constrained rollouts (wheel), keep high-info-gain turns, SFT free-gen</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.281 null</B></FONT></TD></TR></TABLE>>];
infogain_xit_f [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>info-gain XIT (free / STaR)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">same, free-gen rollouts (no wheel)</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.281 null</B></FONT></TD></TR></TABLE>>];
dpo_fair [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>DPO on fair base</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">the contaminated-winning DPO recipe, clean base</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.281 null (every epoch reverted)</B></FONT></TD></TR></TABLE>>];
grpo_full_fair [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>stage-4 long GRPO</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">stabilized GRPO + full reward from the distilled base, 150 updates</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>flat ~0.33 null</B></FONT></TD></TR></TABLE>>];
scale_tiny [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>scale — tiny</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>1.2M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">fair recipe, smallest net</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.163 (underfits)</B></FONT></TD></TR></TABLE>>];
scale_base [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>scale — base</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>12M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">fair recipe, mid net</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.251 (gap grows)</B></FONT></TD></TR></TABLE>>];
scale_xl [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>scale — xl</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>99M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">fair recipe, largest net</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.270 · valid 0.591 — turns over (< 50M)</B></FONT></TD></TR></TABLE>>];
rft_stage2 [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>stage-2 RFT (on-policy distill)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">distill the model's OWN best-of-N winning games back into greedy (RAFT)</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>win ~noise (0.32 selection peak); valid 0.681</B></FONT></TD></TR></TABLE>>];
validity_max [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>validity-max (aux 3)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">crank aux + distill own constrained-decode games to push in-weights spelling</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>CLEAN 0.281 (valid 0.71) — lever works</B></FONT></TD></TR></TABLE>>];
validity_max_v2 [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>validity-max v2 (aux 6)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">push aux + more always-valid self-distill; honest clean protocol (no dict)</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>CLEAN 0.302 (valid 0.76)</B></FONT></TD></TR></TABLE>>];
validity_max_v3 [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>validity-max v3 (aux 8)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">push aux harder still — validity lever PLATEAUS</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>CLEAN 0.264 (valid 0.77) — no gain, noise band</B></FONT></TD></TR></TABLE>>];
validity_max_v4 [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#7ec8d0"><B>validity-max v4 (aux6, ALL train)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">same recipe, scaled to all 1852 train secrets — DATA lever (better deduction generalization)</FONT></TD></TR><TR><TD BGCOLOR="#1f5e66"><FONT COLOR="white" POINT-SIZE="10"><B>CLEAN 0.332/0.335 (2 seeds) — BEST honest, verified</B></FONT></TD></TR></TABLE>>];
infill [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>green-conditioned infill (user idea)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">green+yellow template as INPUT; clue-aware aux training-only. Valid opener (slate) but template-with-content breaks constrained turns</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>FAILED 0.05 (λ6/λ1 both) — explicit template net-negative, hurts like structured-context</B></FONT></TD></TR></TABLE>>];
dense_encode [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>dense constraint-state (mm V2 port)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">honest port of colleague's '90%' V2: replace raw history w/ clean digest (greens-by-pos, yellows-w-excluded-pos, grays-w-exact-count) as the ONLY input; dense-format warm-up, free char-gen, train-only secrets, disjoint TEST</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>REJECTED 0.065 TEST — MEMORIZES (train 0.317 vs held 0.075); their 90% was contamination, not the encoding</B></FONT></TD></TR></TABLE>>];
grpo_honest [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>honest free-gen GRPO on 50M base</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">FIRST clean-lineage RL: correct free-gen GRPO (CoT-bearing rollout, win-dominance+non-word reward, group-rel adv no/std, clipped surrogate, k3 KL to frozen ref) on validity_max_v4; fixed the dropout-in-ratio bug so RL works mechanically; train-only secrets, eval clean</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>NULL 0.332 TEST — train win 0.41->0.60 (+19pt) but held-out FLAT; RL sharpens/memorizes train, zero transfer (the generalization wall, now via correct RL)</B></FONT></TD></TR></TABLE>>];
reason_cot [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>reasoning-CoT (derive constraints)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">condition on RAW history, DERIVE + write out the constraint state (greens/yellows/grays) as CoT, then guess; teacher supervises the derivation (train-only); free ephemeral decode</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>NULL (stopped ep20, win ~0.04) — reasoning COLLAPSED (derive-acc pinned 0.679 = all-BLANK); model routes around it. Restating a deterministic fn of the input adds NO info</B></FONT></TD></TR></TABLE>>];
iter_refine [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>iterative refine (draft->edit)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">learned edit operator: condition on history + a DRAFT word, output a better word; at play feed each guess back K passes (full-word lookahead); lean 36-vocab; near-miss/identity/random draft training</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>NULL 0.098 TEST — refinement = IDENTITY (passes 0=1=3 byte-identical); model ignores the draft. Same route-around as reason-CoT</B></FONT></TD></TR></TABLE>>];
poe [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>product-of-experts (G x C)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>2x≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">fuse FROZEN generator G (validity_max_v4, common words) x a NEW consistency expert C (trained answer-agnostic on full-dict clue->consistent-word, 50k states) at the decode logits; beta sweep incl beta=0=G-alone; honest (both nets forward-pass, no dict/engine)</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>NULL 0.332 (delta +0.000 vs G) — C adds nothing at low beta, HURTS at high; learns 'a consistent word' but can't pin THE answer on held-out (the wall, relocated to fusion)</B></FONT></TD></TR></TABLE>>];
ambiguity_diag [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#e8c98a"><B>ambiguity diagnostic</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>method — no model</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">measure WHY G loses: |consistent set| at each losing guess (exact full-dict scan)</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>KEY: wall is NOT ambiguity — 61% of losses have <=3 consistent words, 20% UNIQUELY determined yet missed; ceiling +0.41 if tight cases solved</B></FONT></TD></TR></TABLE>>];
poe_sharp [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>sharp consistency expert (endgame)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>2x≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">PoE with C retrained on TIGHT realistic states (the endgame curriculum); decisive test = can C produce the UNIQUE consistent word on held-out unique-answer states?</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>NULL 0.332 (delta +0.000); size-1 deduction acc 0.07 held / 0.14 train — constraint->unique-answer is a discrete SEARCH, not a learnable-generalizable fn. THE root cause of the 0.34 wall</B></FONT></TD></TR></TABLE>>];
format_bakeoff [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>context-format bake-off (interleaved/keyboard)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">fair from-scratch test of board LAYOUTS (letters-then-colors vs each letter beside its clue vs alphabet-state suffix); metric=clue-RESPECT (0.338 model violates clues 14%); full-367-TEST</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>NULL — interleaved respect 0.402 vs baseline 0.368 (+0.03, offset by worse spelling); early leads were learning-SPEED artifacts that converge to equal. Layout doesn't change converged tracking; deduction wall unmoved</B></FONT></TD></TR></TABLE>>];
cot_width [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>more/less CoT (candidate-search width)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">drift-free: force the real 0.338 model to emit N candidate-blocks before commit (N=1 less, N=10 more); also retrained K=6 A/B</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>NULL/NEG — more CoT MONOTONICALLY hurts (win N1 0.332 -> N3 0.264 -> N6 0.054 -> N10 0.000); less (N=1) ≈ native; retrained K=6 +0.014 sub-σ. CoT value = SINGLE draft-then-commit, not multi-candidate search. Formatting thread closed</B></FONT></TD></TR></TABLE>>];
data_expand [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>data lever: +common words (held excl)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">add 3692 common 5-letter words (wordfreq, held-out EXCLUDED 0-leak) -> 5520 secrets, warm-start v4; parallel teacher-gen (76s vs 25min)</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>NULL — TEST 0.272 vs 1852 control 0.267 (+0.005 noise); non-answer commons are a diff distribution, don't transfer. Data lever MAXED at the answer set</B></FONT></TD></TR></TABLE>>];
freq_aux [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>common-word prior (freq-weighted aux)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">reweight the validity-aux toward COMMON valid words (wordfreq) — give the model English-frequency knowledge in-weights; warm-start v4, A/B vs uniform-aux</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>NULL/NEG — uniform 0.338 vs freq λ0.3 0.283 / λ0.7 0.221 / λ3 collapse; freq HURTS clue-respect (0.89->0.55). Knowledge is NOT the bottleneck; context-free commonness conflicts with clue-consistency</B></FONT></TD></TR></TABLE>>];
control_teacher [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#e8c98a"><B>teacher-only control (noise-buster)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">plain gentle re-train, no special ingredients — same VAL-selection procedure</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>VAL 0.365 by chance → TEST 0.259 — proves the win gains are noise</B></FONT></TD></TR></TABLE>>];
overnight_train_null [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>overnight training variants (#2 sweep)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">mixed answers+dict, curriculum (full-dict->answers), full-dict 16-epoch — all warm-start v4, honest (held-out excl secrets+guesses), select by VAL win</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>ALL NULL/NEG on held-out TEST: mixed 0.256/0.347, curriculum 0.327/0.384, fulldict16 0.286/0.350 (greedy/+AR) — raise clue-respect but lower win; distribution shift hurts. Training is exhausted; decode is the only lever</B></FONT></TD></TR></TABLE>>];
rope_ab [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>RoPE vs learned pos-emb A/B</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">swap learned positional embedding for rotary (RoPE) in a from-scratch A/B; only the positional mechanism differs</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>NULL — from-scratch clue-respect ~= control; positional encoding is not the bottleneck</B></FONT></TD></TR></TABLE>>];
denoise_spell [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>denoise/contrastive spell-sharpen (#2)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">continue v4 + contrastive margin: real committed word > its Hamming-1 non-word (+ full-dict unconditional); aimed at the 226/228 one-letter-off commit fumble; honest (no held-out answer-mapping)</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>NULL/NEG — margin already saturated (~0.10), little gradient; greedy 0.281 (vs 0.332), +AR 0.376 (vs 0.385). The RELATIVE spelling lever is tapped; absolute prob on held-out is the plateaued gap. Wall = constrained RETRIEVAL (we already pretrain all-word spelling), not vocab</B></FONT></TD></TR></TABLE>>];
}
subgraph cluster_inf { label=<<B>INFERENCE PROBES (decoding, contaminated bases)</B>>; fontsize=13; color="#cccccc"; style="rounded"; bgcolor="#fbfbfb";
beam_dict [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#aac4dd"><B>beam + dictionary decode</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈25M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">beam search ± constrained to real words via trie (no training)</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>greedy 0.392 → beam+dict 0.580</B></FONT></TD></TR></TABLE>>];
norepeat_decode [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#aac4dd"><B>beam+dict + no-repeat</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈25M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">beam(10) + dict-trie + never re-emit a prior guess</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.596</B></FONT></TD></TR></TABLE>>];
turn_budget [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#aac4dd"><B>turn-budget probe</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈25M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">same weights, allow 6/8/10 guesses</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.392 flat</B></FONT></TD></TR></TABLE>>];
passk [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#aac4dd"><B>pass@N probe</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">sample N full games/secret, count any win</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>greedy 0.453 → pass@10 0.787</B></FONT></TD></TR></TABLE>>];
self_consistency [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#aac4dd"><B>self-consistency vote</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">sample 12 traces/turn, commit the majority vote (no filter)</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>vote 0.627 vs pass@12 0.953</B></FONT></TD></TR></TABLE>>];
}
subgraph cluster_inf2 { label=<<B>INFERENCE on the CLEAN fair weights (aided)</B>>; fontsize=13; color="#cccccc"; style="rounded"; bgcolor="#fbfbfb";
beam_decode [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#aac4dd"><B>honest beam (model's own dist)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">commit highest-joint-prob 5-letter word under the model's OWN logits (no dict), B=1/4/8/16</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>NULL — B=1/4/8 byte-IDENTICAL (0.332); greedy argmax already = the highest-joint word; better decoding recovers nothing</B></FONT></TD></TR></TABLE>>];
constrained_decode [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#aac4dd"><B>constrained-decode diagnostic</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">greedy masked to real-word spellings; model still deduces</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.281 → 0.436 · valid 1.0 — KNOWS the words</B></FONT></TD></TR></TABLE>>];
bestof16 [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#aac4dd"><B>best-of-16 (valid vote)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">sample 16, keep real words, majority vote</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.632 · valid 0.925</B></FONT></TD></TR></TABLE>>];
bestof16_nodict [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#aac4dd"><B>best-of-16 NO dict</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">same, but keep non-words too</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.243 — compute alone HURTS</B></FONT></TD></TR></TABLE>>];
bestof64 [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#aac4dd"><B>best-of-64</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">N=64</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.703</B></FONT></TD></TR></TABLE>>];
bestof128 [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#7fa8cf"><B>best-of-128 — BEST</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">N=128</FONT></TD></TR><TR><TD BGCOLOR="#1f5e66"><FONT COLOR="white" POINT-SIZE="10"><B>0.719 · valid 0.979 (plateau)</B></FONT></TD></TR></TABLE>>];
beam_trie [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#aac4dd"><B>beam over real-word trie</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">sequence-level argmax over valid words (best deterministic)</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.550 · valid 1.0</B></FONT></TD></TR></TABLE>>];
ensemble [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#aac4dd"><B>ensemble committee (no dict)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>5x≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">5 trained models vote per board, majority (no dictionary) — pure multi-model test-time compute</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.283 — NULL (averages toward weaker members; below v4 0.332)</B></FONT></TD></TR></TABLE>>];
eval_db [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#e8c98a"><B>eval DB + failure reframe</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">versioned SQLite eval store (scripts/eval_db.py, faithful to validity_max.play) with per-turn + CoT capture on validity_max_v4, full 367 TEST</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>REFRAME — 93% of losses = non-word SPELLING (deduction is only 7%); non-word rate 0%@51+ left -> 40%@determined; CoT holds the answer only 18% when determined. The wall is constrained lexical retrieval</B></FONT></TD></TR></TABLE>>];
sampling_decode [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>honest sampling decode (T=0.6)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">SAME validity_max_v4 weights, ZERO training change; greedy argmax -> temperature sampling, non-word = recoverable wasted turn (counted, not fed back, play continues) vs greedy's fatal break</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.358 +/-0.018 (5 seeds) vs greedy 0.332 = +0.026 — first clean BEAT of the greedy wall, honest (no dict)</B></FONT></TD></TR></TABLE>>];
antirepeat [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#a9d3d9"><B>anti-repeat sampling (T=0.6)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">sampling + resample for a not-yet-EMITTED word (honest: uses only the game's own guess history, no dict/clue-filter; the lever wordley also uses) — forces fresh spelling attempts instead of re-fumbling the same Hamming-1 non-word</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>0.385 +/-0.022 (5 seeds) — +0.052 over greedy</B></FONT></TD></TR></TABLE>>];
decode_opt [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#7ec8d0"><B>decode optimum (T=0.5, tries=16)</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">overnight sweep of 12 honest approaches: LOWER sampling temp (~0.5, less noise) + bigger anti-repeat resample budget; same v4 weights, inference-only</FONT></TD></TR><TR><TD BGCOLOR="#1f5e66"><FONT COLOR="white" POINT-SIZE="10"><B>0.406 +/-0.014 (multi-seed) — BEST honest, +0.074 over greedy. Lower T beats T=0.6; forcing an opener HURTS; temp-schedules marginal</B></FONT></TD></TR></TABLE>>];
}
subgraph cluster_dep { label=<<B>DEPLOYED FRAMING</B>>; fontsize=13; color="#cccccc"; style="rounded"; bgcolor="#fbfbfb";
deployed [label=<<TABLE BORDER="0" CELLBORDER="1" CELLSPACING="0" CELLPADDING="5"><TR><TD BGCOLOR="#d9c89a"><B>deployed real-Wordle player</B></TD></TR><TR><TD BGCOLOR="#efefef"><FONT POINT-SIZE="9" COLOR="#222222"><B>≈50M params</B></FONT></TD></TR><TR><TD ALIGN="LEFT"><FONT POINT-SIZE="9">trained on all 2,315 known answers (fixed set = the real game)</FONT></TD></TR><TR><TD BGCOLOR="#4a4a4a"><FONT COLOR="white" POINT-SIZE="10"><B>~0.62 (legit deployed, framing)</B></FONT></TD></TR></TABLE>>];
}
validity_max_v4 -> eval_db [label="failure analysis"];
eval_db -> sampling_decode [label="93% losses=spelling -> retries"];
sampling_decode -> antirepeat [label="resample novel word"];
antirepeat -> decode_opt [label="sweep: lower T + bigger budget"];
denoise_spell -> overnight_train_null [label="more training variants"];
validity_max_v4 -> rope_ab [label="RoPE A/B"];
eval_db -> wordley_audit [label="cross-impl compare"];
eval_db -> denoise_spell [label="#2 sharpen commit spelling"];
sft_3m -> sft_5m [label="scale+depth"];
sft_5m -> sft_25m [label="5x params"];
sft_25m -> sft_25m_div [label="diverse secrets"];
sft_25m -> scale_99m [label="pure scale"];
scale_99m -> codesign_50m [label="redesign+curriculum"];
codesign_50m -> sft_deep [label="isolate redesign, converge"];
sft_deep -> sft_aux [label="add spelling loss"];
sft_aux -> sft_aux_xl [label="scale aux"];
sft_3m -> bpe_12m [label="swap to BPE"];
bpe_12m -> bpe_50m [label="capacity test"];
oreo_11m -> oreo_50m [label="scale"];
sft_aux -> structured_ctx [label="add state block"];
sft_aux -> passk [label="pass@N probe"];
sft_25m -> beam_dict [label="beam+dict"];
sft_25m -> norepeat_decode [label="+no-repeat"];
sft_25m -> turn_budget [label="more guesses"];
passk -> cot_14m [label="reason to surface latent"];
cot_14m -> cot_50m [label="scale CoT"];
cot_50m -> cot_50m_aux [label="+aux"];
cot_50m -> cot_show [label="integrity A/B"];
cot_show -> cot_eph [label="throwaway scratchpad (fix leak)"];
sft_deep -> cot_eph [label="matched baseline"];
cot_eph -> cot_eph_aux [label="stack +aux"];
sft_aux -> cot_eph_aux [label="the spelling lever"];
sft_5m -> grpo_5m [label="RL?"];
grpo_5m -> grpo_5m_div [label="diverse"];
sft_25m -> self_distill_25m [label="bank spelling"];
self_distill_25m -> rl_consistency [label="legality reward"];
self_distill_25m -> rl_perguess [label="per-guess credit"];
self_distill_25m -> rl_dict [label="dict-in-loop"];
self_distill_25m -> rl_constrained [label="surface the answer"];
sft_aux -> rl_polish [label="polish best"];
sft_aux -> rl_infogain [label="info-gain term"];
cot_eph_aux -> rl_expert_10row [label="expert-iter"];
rl_expert_10row -> rl_expert_tail [label="reachability"];
rl_expert_10row -> rl_grpo_polish [label="token-GRPO"];
cot_eph_aux -> self_consistency [label="vote vs pass@N"];
cot_eph_aux -> dpo_commit [label="DPO commit"];
cot_eph_aux -> dpo_decisive [label="clean pairs"];
dpo_commit -> dpo_guessonly [label="guess-only"];
cot_eph_aux -> constraint_aux [label="train-in reward"];
dpo_commit -> dagger_v1 [label="failure relabel"];
dpo_commit -> rl_grpo_reward [label="GRPO+reward"];
dpo_commit -> rl_grpo_guessonly [label="guess-only credit"];
dagger_v1 -> dagger_v2 [label="full coverage"];
cot_eph_aux -> audit [label="is it correct?"];
dpo_commit -> audit [label="audit 0.631 too"];
cot_eph_aux -> clean_rerun [label="re-run leak-free"];
clean_rerun -> overnight_clean_sft [label="clean base"];
overnight_clean_sft -> overnight_clean_dpo [label="DPO"];
overnight_clean_sft -> overnight_clean_grpo [label="GRPO"];
overnight_clean_sft -> overnight_clean_dagger [label="DAgger"];
overnight_clean_dagger -> overnight_clean_dagger2 [label="×2"];
overnight_clean_sft -> fair_stage1 [label="know dict, hide answers"];
fair_stage1 -> distill_constrained [label="push validity"];
fair_stage1 -> infogain_xit_c [label="info-gain XIT"];
fair_stage1 -> infogain_xit_f [label="free / STaR"];
fair_stage1 -> dpo_fair [label="retry DPO clean"];
distill_constrained -> grpo_full_fair [label="long GRPO"];
fair_stage1 -> scale_tiny [label="shrink"];
fair_stage1 -> scale_base [label="mid"];
fair_stage1 -> scale_xl [label="grow"];
fair_stage1 -> rft_stage2 [label="on-policy self-distill"];
fair_stage1 -> validity_max [label="push in-weights validity"];
rft_stage2 -> control_teacher [label="attribution: ablate ingredients"];
validity_max -> validity_max_v2 [label="push aux harder (clean protocol)"];
validity_max_v2 -> validity_max_v3 [label="aux 8 — plateaus"];
validity_max_v3 -> validity_max_v4 [label="aux6 + ALL train data"];
fair_stage1 -> infill [label="clue-aware infill (template input, clue logic training-only)"];
fair_stage1 -> constrained_decode [label="mask spelling (diagnostic)"];
fair_stage1 -> bestof16 [label="test-time compute"];
fair_stage1 -> bestof16_nodict [label="no-dict ablation"];
bestof16 -> bestof64 [label="N=64"];
bestof64 -> bestof128 [label="N=128"];
fair_stage1 -> beam_trie [label="beam"];
validity_max_v4 -> ensemble [label="commit-by-committee vote"];
validity_max_v4 -> dense_encode [label="honest port of mm V2 dense encoding"];
validity_max_v4 -> grpo_honest [label="honest free-gen GRPO (CoT-bearing, RL question)"];
validity_max_v4 -> reason_cot [label="execute deduction via reasoning (not amortize)"];
reason_cot -> iter_refine [label="cross-pass computation instead of restating"];
iter_refine -> poe [label="multiplicative fusion (can't be routed around)"];
poe -> ambiguity_diag [label="why does G lose? measure the consistent set"];
ambiguity_diag -> poe_sharp [label="losses are tight -> sharpen C on the endgame"];
ambiguity_diag -> format_bakeoff [label="14% clue-violations -> does layout reduce them?"];
format_bakeoff -> cot_width [label="layout null -> does MORE CoT (search) help?"];
validity_max_v4 -> data_expand [label="more answer-LIKE data (common words)?"];
validity_max_v4 -> beam_decode [label="better decoding of the same weights?"];
validity_max_v4 -> freq_aux [label="give it English word-frequency knowledge?"];
cot_eph_aux -> deployed [label="deployed framing"];
dpo_commit -> deployed [label="best deployed"];
}