Skip to content
nano-gpt-data-curation

gpt-5.6-sol

nano-gpt-data-curation — graded by the task's verifier against a measured baseline.

backend
modal
variant
codex azure_ai/gpt-5.6-sol
metric
held-out multi-domain perplexity
actions
259
duration
1h 6m
recorded
2026-08-03

run iddclm-nano-curation__d9qdiJV

Everything gpt-5.6-sol did on this task, in order.

download JSON ↗
codexgpt-5.6-sol270 turns11 narration259 reasoning259 action259 observation1 final
#226
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolexec
args
{
  "input": "const r = await tools.exec_command({cmd:\"python -u submission/curate.py --chrome-max 8 --output /tmp/candidate_chrome8.json\",\"workdir\":\"/workspace\",\"yield_time_ms\":1000,\"max_output_tokens\":5000,\"tty\":true}); text(JSON.stringify(r))\n"
}
exec result
Script completed
Wall time 1.2 seconds
Output:
{"chunk_id":"ccae2c","wall_time_seconds":1.001696627,"session_id":81975,"original_token_count":0,"output":""}
#227
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolexec
args
{
  "input": "const r = await tools.write_stdin({session_id:81975,chars:\"\",yield_time_ms:30000,max_output_tokens:5000}); text(JSON.stringify(r))\n"
}
exec result
Script running with cell ID 155
Wall time 11.0 seconds
Output:
#228
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolwait
args
{
  "cell_id": "155",
  "yield_time_ms": 30000,
  "max_tokens": 5000
}
wait result
Script completed
Wall time 15.1 seconds
Output:
{"chunk_id":"d353bd","wall_time_seconds":30.002521308,"session_id":81975,"original_token_count":0,"output":""}
#229
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolexec
args
{
  "input": "const r = await tools.write_stdin({session_id:81975,chars:\"\",yield_time_ms:30000,max_output_tokens:5000}); text(JSON.stringify(r))\n"
}
exec result
Script running with cell ID 156
Wall time 11.0 seconds
Output:
#230
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolwait
args
{
  "cell_id": "156",
  "yield_time_ms": 30000,
  "max_tokens": 5000
}
wait result
Script completed
Wall time 14.1 seconds
Output:
{"chunk_id":"4acb46","wall_time_seconds":30.001940219,"session_id":81975,"original_token_count":0,"output":""}
#231
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolexec
args
{
  "input": "const r = await tools.write_stdin({session_id:81975,chars:\"\",yield_time_ms:30000,max_output_tokens:5000}); text(JSON.stringify(r))\n"
}
exec result
Script running with cell ID 157
Wall time 11.0 seconds
Output:
#232
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolwait
args
{
  "cell_id": "157",
  "yield_time_ms": 30000,
  "max_tokens": 5000
}
wait result
Script completed
Wall time 12.9 seconds
Output:
{"chunk_id":"ec3f83","wall_time_seconds":30.000953504,"session_id":81975,"original_token_count":47,"output":"Token indices sequence length is longer than the specified maximum sequence length for this model (1079 > 1024). Running this sequence through the model will result in indexing errors\r\n"}
#233
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolexec
args
{
  "input": "const r = await tools.write_stdin({session_id:81975,chars:\"\",yield_time_ms:30000,max_output_tokens:5000}); text(JSON.stringify(r))\n"
}
exec result
Script running with cell ID 158
Wall time 11.0 seconds
Output:
#234
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolwait
args
{
  "cell_id": "158",
  "yield_time_ms": 30000,
  "max_tokens": 5000
}
wait result
Script completed
Wall time 0.0 seconds
Output:
{"chunk_id":"f3b0ea","wall_time_seconds":11.153853553,"exit_code":0,"original_token_count":35,"output":"eligible [7000, 7000, 7000, 4333] selected [3725, 3770, 4179, 3622] estimated exact tokens [3351711, 3350854, 3350250, 3350596] ids 15296\r\n"}
#235
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolexec
args
{
  "input": "const r = await tools.exec_command({cmd:\"python pack_selection.py /tmp/candidate_chrome8.json /tmp/chrome8.npy && python train_nano.py --train_npy /tmp/chrome8.npy --heldout_npy data/multi_dev.npy --out_json /tmp/chrome8_result.json\",\"workdir\":\"/workspace\",\"yield_time_ms\":1000,\"max_output_tokens\":5000,\"tty\":true}); text(JSON.stringify(r))\n"
}
exec result
Script completed
Wall time 1.2 seconds
Output:
{"chunk_id":"2d6f53","wall_time_seconds":1.001807852,"session_id":15608,"original_token_count":0,"output":""}
#236
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolexec
args
{
  "input": "const r = await tools.exec_command({cmd:\"python - <<'PY'\\nimport json,collections\\nfor p in ['submission/selection.json','/tmp/candidate_chrome8.json']:\\n s=json.load(open(p));print(p,len(s),sum(i>=120000 for i in s),collections.Counter(i//20000 for i in s))\\nPY\\ncat .timer/remaining_secs\",\"workdir\":\"/workspace\",\"yield_time_ms\":10000,\"max_output_tokens\":2000}); text(r.output)\n"
}
exec result
Script completed
Wall time 0.1 seconds
Output:
submission/selection.json 15131 1758 Counter({4: 2462, 2: 2338, 3: 2318, 1: 2294, 0: 2265, 5: 1696, 8: 677, 7: 622, 6: 404, 9: 55})
/tmp/candidate_chrome8.json 15296 1999 Counter({4: 2446, 2: 2322, 3: 2308, 1: 2285, 0: 2247, 5: 1689, 8: 771, 7: 712, 6: 451, 9: 65})
2146
#237
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolexec
args
{
  "input": "const r = await tools.write_stdin({session_id:15608,chars:\"\",yield_time_ms:30000,max_output_tokens:5000}); text(JSON.stringify(r))\n"
}
exec result
Script running with cell ID 161
Wall time 11.0 seconds
Output:
#238
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolwait
args
{
  "cell_id": "161",
  "yield_time_ms": 30000,
  "max_tokens": 5000
}
wait result
Script completed
Wall time 14.3 seconds
Output:
{"chunk_id":"641205","wall_time_seconds":30.001740351,"session_id":15608,"original_token_count":61,"output":"Token indices sequence length is longer than the specified maximum sequence length for this model (1079 > 1024). Running this sequence through the model will result in indexing errors\r\npacked 12000000 tokens from 15296 ids -> /tmp/chrome8.npy\r\n"}
#239
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolexec
args
{
  "input": "const r = await tools.write_stdin({session_id:15608,chars:\"\",yield_time_ms:30000,max_output_tokens:5000}); text(JSON.stringify(r))\n"
}
exec result
Script running with cell ID 162
Wall time 11.0 seconds
Output:
#240
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolwait
args
{
  "cell_id": "162",
  "yield_time_ms": 30000,
  "max_tokens": 5000
}
wait result
Script completed
Wall time 14.2 seconds
Output:
{"chunk_id":"04f88e","wall_time_seconds":30.001351036,"session_id":15608,"original_token_count":0,"output":""}
#241
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolexec
args
{
  "input": "const r = await tools.write_stdin({session_id:15608,chars:\"\",yield_time_ms:30000,max_output_tokens:5000}); text(JSON.stringify(r))\n"
}
exec result
Script completed
Wall time 5.9 seconds
Output:
{"chunk_id":"f2f50a","wall_time_seconds":5.917066886,"exit_code":0,"original_token_count":16,"output":"HELDOUT_PPL 335.047  loss 5.8143  train_s 54  /tmp/chrome8.npy\r\n"}
#242
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolexec
args
{
  "input": "const patch = \"*** Begin Patch\\n*** Update File: /workspace/submission/curate.py\\n@@\\n def doc_score(text, models, totals, broad, broad_total, vocab, blog, dlogs, bun, duns,\\n-              rawlog, rawun, dsir_weight, margin_weight, chrome_max):\\n+              rawlog, rawun, dsir_weight, margin_weight):\\n@@\\n-    chrome = sum(lo[:12000].count(x) for x in\\n-                 (' home ', ' login', ' register', ' search', ' menu', ' navigation',\\n-                  ' download', ' contact', ' password', ' sitemap', ' faq',\\n-                  ' categories', ' all rights reserved'))\\n-    if chrome > chrome_max:\\n-        return None\\n@@\\n-    ap.add_argument('--chrome-max', type=int, default=1000000,\\n-                    help='reject pages above this navigation/chrome marker count')\\n@@\\n                           blog, dlogs, bun, duns, rawlog, rawun, a.dsir_weight,\\n-                          a.margin_weight, a.chrome_max)\\n+                          a.margin_weight)\\n*** End Patch\"; text(await tools.apply_patch(patch))\n"
}
exec result
Script completed
Wall time 0.0 seconds
Output:
{}
#243
narration

The final side test confirmed that a blunt navigation-marker cap is counterproductive (335.05): it removes useful technical forum material along with boilerplate. I’ve discarded that cap. The frozen choice remains the reproducible moderate-DSIR, deduplicated, equal-register selection at 321.51 dev perplexity. I’m doing the final independent pack/validity check now.

#244
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolexec
args
{
  "input": "const r = await tools.exec_command({cmd:\"python -m py_compile submission/curate.py && python pack_selection.py submission/selection.json /tmp/final_verify.npy && sha256sum /tmp/final_verify.npy /tmp/dsir.npy && python - <<'PY'\\nimport json, numpy as np\\ns=json.load(open('submission/selection.json')); a=np.load('/tmp/final_verify.npy')\\nassert len(s)==len(set(s)) and all(type(i) is int and 0<=i<182016 for i in s)\\nassert len(a)==12_000_000 and a.dtype==np.uint16\\nprint({'ids':len(s),'unique_ids':len(set(s)),'packed_tokens':len(a),'dtype':str(a.dtype)})\\nPY\",\"workdir\":\"/workspace\",\"yield_time_ms\":1000,\"max_output_tokens\":5000,\"tty\":true}); text(JSON.stringify(r))\n"
}
exec result
Script completed
Wall time 1.2 seconds
Output:
{"chunk_id":"08f04c","wall_time_seconds":1.00187144,"session_id":47099,"original_token_count":0,"output":""}
#245
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolexec
args
{
  "input": "const r = await tools.write_stdin({session_id:47099,chars:\"\",yield_time_ms:30000,max_output_tokens:5000}); text(JSON.stringify(r))\n"
}
exec result
Script running with cell ID 166
Wall time 11.0 seconds
Output:
#246
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolwait
args
{
  "cell_id": "166",
  "yield_time_ms": 30000,
  "max_tokens": 5000
}
wait result
Script completed
Wall time 13.2 seconds
Output:
{"chunk_id":"6ccbb8","wall_time_seconds":28.503088577,"exit_code":0,"original_token_count":126,"output":"Token indices sequence length is longer than the specified maximum sequence length for this model (1079 > 1024). Running this sequence through the model will result in indexing errors\r\npacked 12000000 tokens from 15131 ids -> /tmp/final_verify.npy\r\n1ed116a17174a83ee0e35fe8ae59c0e6b5c46ae745bf7f0ccc4d99e1b0c7884f  /tmp/final_verify.npy\r\n1ed116a17174a83ee0e35fe8ae59c0e6b5c46ae745bf7f0ccc4d99e1b0c7884f  /tmp/dsir.npy\r\n{'ids': 15131, 'unique_ids': 15131, 'packed_tokens': 12000000, 'dtype': 'uint16'}\r\n"}
#247
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolexec
args
{
  "input": "const r = await tools.exec_command({cmd:\"sed -n '1,260p' submission/curate.py; printf '\\\\nCLAIM\\\\n'; cat claim.md; git status --short 2>/dev/null || true; cat .timer/remaining_secs\",\"workdir\":\"/workspace\",\"yield_time_ms\":10000,\"max_output_tokens\":30000}); text(r.output)\n"
}
exec result
Script completed
Wall time 0.1 seconds
Output:
#!/usr/bin/env python3
"""Rank the raw pool by document quality and broad-target register similarity.

The disclosed development set is four equal contiguous registers.  We use it
only to estimate four smoothed *word unigram* distributions (a deliberately
low-capacity, auditable style signal), plus a target-vs-random-pool importance
ratio. Documents must first pass language, length, boilerplate, diversity, and
repetition checks. Eligible documents are scored by prose hygiene, moderate
importance weight, and their margin for one of the four register distributions.
The final list is a token-aware fair queue giving each register 3M GPT-2 tokens.

No pool id, text fragment, site, or manual allow-list is embedded in this file.
"""
import argparse, collections, json, math, os, re
from pathlib import Path

import numpy as np
from transformers import AutoTokenizer

WORD_RE = re.compile(r"[A-Za-z]+(?:'[A-Za-z]+)?|\d+(?:[.,:]\d+)*")
NON_PROSE_RE = re.compile(r"[^A-Za-z\s]")
UPPER_RE = re.compile(r"[A-Z]")
BAD = (
    "skip to content", "privacy policy", "cookie policy", "all rights reserved",
    "sign in", "log in", "register now", "shopping cart", "javascript is required",
    "forgot your password", "terms of service", "view cart", "menu home",
    "categories home", "wordpress", "enable javascript", "subscribe to our newsletter",
)
REJECT = (
    "parked at loopia", "this domain has been purchased and parked",
    "home copyright complain", "javascript is required for login",
    "username password remember me lost your password", "enable javascript to use this site",
)
TECH = re.compile(r"(?i)(?:<code>|<pre>|stack\s*overflow|\bpython\b|\bjava(?:script)?\b|\bc\+\+\b|\bsql\b|\blinux\b|\bapi\b|\bfunction\b|\bclass\b|\bvariable\b|\bserver\b|\bdatabase\b|\bexception\b|\berror\b|\bcompile\b|\bcommand\b)")
NEWS = re.compile(r"(?i)(?:\breuters\b|\bassociated press\b|\baccording to\b|\bsaid (?:the|he|she|mr|ms)\b|\breported\b|\bspokes(?:man|woman|person)\b|\bnews\b)")
ENC = re.compile(r"(?i)(?:\breferences\b|\bexternal links\b|\bwas born\b|\bis (?:an?|the)\b|\bconsists of\b|\bis located\b|\bpopulation\b|\bspecies\b|\bcentury\b)")


def words(text):
    return WORD_RE.findall(text.lower())


def reference_models(dev_path, tok):
    ids = np.load(dev_path)
    models, totals = [], []
    for k in range(4):
        s = tok.decode(ids[k * len(ids)//4:(k+1) * len(ids)//4])
        c = collections.Counter(words(s))
        models.append(c); totals.append(sum(c.values()))
    broad = sum(models, collections.Counter())
    vocab = len(broad)
    broad_total = sum(totals)
    blog = {w: math.log((v+.25)/(broad_total+.25*vocab)) for w,v in broad.items()}
    dlogs = [{w: math.log((v+.20)/(tot+.20*vocab)) for w,v in c.items()}
             for c,tot in zip(models, totals)]
    bun = math.log(.25/(broad_total+.25*vocab))
    duns = [math.log(.20/(tot+.20*vocab)) for tot in totals]
    return models, totals, broad, broad_total, vocab, blog, dlogs, bun, duns


def doc_score(text, models, totals, broad, broad_total, vocab, blog, dlogs, bun, duns,
              rawlog, rawun, dsir_weight, margin_weight):
    nchar = len(text)
    if nchar < 700 or nchar > 50000:
        return None
    lo = text.lower()
    if any(x in lo for x in REJECT):
        return None
    bad = sum(lo.count(x) for x in BAD)
    if bad >= 4:
        return None
    char_probe = text[:8000]
    alpha = (len(char_probe) - len(NON_PROSE_RE.findall(char_probe))) / len(char_probe)
    if alpha < .72:
        return None
    word_probe = text[:20000]
    ws = words(word_probe)
    n = int(len(ws) * nchar / max(1, len(word_probe)))
    if n < 120 or n > 8500:
        return None
    # Effective lexical diversity on a bounded prefix catches keyword/menu spam
    # without punishing legitimate long articles for reusing function words.
    probe = ws[:1200]
    unique = len(set(probe)) / len(probe)
    if unique < .22:
        return None
    punct = sum(text.count(x) for x in ".?!") / n
    if punct < .010 or punct > .18:
        return None
    upper = len(UPPER_RE.findall(char_probe)) / max(1, alpha*len(char_probe))
    if upper > .19:
        return None
    # Cap word influence so a page cannot win merely by repeating target-common
    # words. Additive smoothing makes this stable for unseen vocabulary.
    sample = ws[:256]
    pooled_ll = sum(blog.get(w, bun) for w in sample) / len(sample)
    dsir = sum(max(-3.0, min(3.0, blog.get(w, bun) - rawlog.get(w, rawun)))
               for w in sample) / len(sample)
    dll = []
    for dl, dun in zip(dlogs, duns):
        dll.append(sum(dl.get(w, dun) for w in sample) / len(sample))
    # Rule evidence only breaks ambiguous unigram assignments. It cannot rescue
    # a low-quality document and is intentionally small.
    # Cheap literal marker counts; the unigram distributions do the actual
    # register classification.
    hp = lo[:12000]
    hints = [sum(hp.count(x) for x in (' references',' external links',' was born',' century',' species')),
             0,
             sum(hp.count(x) for x in ('reuters','according to',' reported',' spokesman',' said ')),
             sum(hp.count(x) for x in ('<code>','<pre>',' python ',' javascript ',' sql ',' linux ',' error ',' function '))]
    adjusted = [dll[k] + .035*math.log1p(hints[k]) for k in range(4)]
    domain = max(range(4), key=lambda k: adjusted[k])
    margin = adjusted[domain] - sorted(adjusted)[-2]
    # Smooth hygiene preferences: full articles, normal sentences, target-like
    # lexical diversity, and little chrome. Raw unigram likelihood is downweighted:
    # otherwise generic SEO filler made of common words outranks real articles.
    min_unique = (.40, .34, .37, .28)[domain]
    if unique < min_unique:
        return None
    length_fit = -abs(math.log(max(n, 1) / 850.0)) * .10
    sent_fit = -abs(math.log(max(punct, 1e-4) / .045)) * .13
    unique_fit = -1.20*abs(unique - (.52, .48, .50, .44)[domain])
    hygiene = length_fit + sent_fit + unique_fit - .25*bad
    score = (.15*pooled_ll + dsir_weight*dsir + hygiene
             + margin_weight*min(margin, 1.0))
    return domain, score, n


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument('--pool', default='/workspace/data/pool.jsonl')
    ap.add_argument('--dev', default='/workspace/data/multi_dev.npy')
    ap.add_argument('--output', default='/workspace/submission/selection.json')
    ap.add_argument('--mix', default='1,1,1,1',
                    help='relative token shares for encyclopedia, web, news, technical')
    ap.add_argument('--dsir-weight', type=float, default=0.30,
                    help='weight on capped target-vs-random-pool unigram log odds')
    ap.add_argument('--margin-weight', type=float, default=0.10,
                    help='weight on assigned-register likelihood margin')
    ap.add_argument('--mode', choices=['balanced','quality'], default='balanced')
    a = ap.parse_args()
    mix = [float(x) for x in a.mix.split(',')]
    if len(mix) != 4 or any(x <= 0 for x in mix):
        raise ValueError('--mix requires four positive comma-separated weights')
    targets = [12_000_000*x/sum(mix) + 350_000 for x in mix]
    tok = AutoTokenizer.from_pretrained('gpt2', local_files_only=True)
    refs = reference_models(a.dev, tok)
    models, totals, broad, broad_total, vocab, blog, dlogs, bun, duns = refs
    raw = collections.Counter()
    with open(a.pool) as f:
        for line in f:
            row = json.loads(line)
            # A deterministic ~1/23 sample estimates the raw-pool distribution.
            if ((row['id']*2654435761) & 0xffffffff) % 23 == 0:
                raw.update(words(row['text'][:20000]))
    raw_total = sum(raw.values())
    rawlog = {w: math.log((n+.25)/(raw_total+.25*vocab)) for w,n in raw.items()}
    rawun = math.log(.25/(raw_total+.25*vocab))
    queues = [[] for _ in range(4)]
    allq = []
    with open(a.pool) as f:
        for line in f:
            row = json.loads(line)
            z = doc_score(row['text'], models, totals, broad, broad_total, vocab,
                          blog, dlogs, bun, duns, rawlog, rawun, a.dsir_weight,
                          a.margin_weight)
            if z is None:
                continue
            d, score, nw = z
            # word count / .75 is a close, conservative token estimate. Exact
            # tokenization happens below only for competitive candidates.
            item = (score, row['id'], nw)
            queues[d].append(item); allq.append(item)
    if a.mode == 'quality':
        queues = [allq, [], [], []]
    for q in queues:
        q.sort(reverse=True)
        # 7k candidates per register is far beyond what a 3.35M-token queue
        # can consume, and bounds peak memory during the exact-token pass.
        del q[20000 if a.mode == 'quality' else 7000:]

    # Tokenize in score order and fill queues to a little beyond the actual 3M
    # quotas. The emitted order cycles domains, so the packer's 12M cutoff is
    # balanced even when documents differ in size.
    chosen = [[] for _ in range(4)]
    texts = {}
    needed = {x[1] for q in queues for x in q}
    with open(a.pool) as f:
        for line in f:
            row = json.loads(line)
            if row['id'] in needed:
                texts[row['id']] = row['text']
    seen_content = set()
    for d, q in enumerate(queues):
        total = 0
        for score, ident, nw in q:
            main_segment = max(texts[ident].split('<|endoftext|>'), key=len)
            fp = ' '.join(WORD_RE.findall(main_segment.lower())[:160])
            if fp in seen_content:
                continue
            seen_content.add(fp)
            nt = len(tok(texts[ident], add_special_tokens=False).input_ids) + 1
            chosen[d].append((ident, nt))
            total += nt
            limit = (targets[d] if a.mode == 'balanced' else 13_000_000)
            if total >= limit:
                break
    out = []
    pos = [0]*4; emitted = [0]*4
    active = [d for d in range(4) if chosen[d]]
    # Weighted fair queuing makes every prefix (including the packer's exact
    # 12M cutoff) approach the requested token mixture.
    while active:
        d = min(active, key=lambda k: emitted[k]/mix[k])
        ident, nt = chosen[d][pos[d]]
        out.append(ident); emitted[d] += nt; pos[d] += 1
        if pos[d] >= len(chosen[d]):
            active.remove(d)
    Path(a.output).parent.mkdir(parents=True, exist_ok=True)
    with open(a.output, 'w') as f:
        json.dump(out, f)
    print('eligible', [len(q) for q in queues], 'selected', [len(q) for q in chosen],
          'estimated exact tokens', [sum(n for _, n in q) for q in chosen], 'ids', len(out))


if __name__ == '__main__':
    main()

CLAIM
# Hypothesis

At a 12M-token budget, a small model benefits more from clean documents that are disproportionately characteristic of the four disclosed target registers than from either random web text or a single global “quality” ranking. The selection should therefore combine moderate target-to-pool importance weighting with an explicit equal token mixture and content deduplication.

# Mechanism

`curate.py` learns only smoothed word-unigram statistics: one model for each equal dev quarter and one background model from a deterministic 1/23 sample of the raw pool. After hard prose, boilerplate, diversity, and template checks, it ranks a document using capped target/background log odds, ordinary target coverage, lexical hygiene, and a small register-margin term. Normalized 160-word prefixes remove repeated pages. Weighted fair queuing keeps every prefix near 25% encyclopedia, 25% edited web, 25% news, and 25% technical Q&A, including the exact 12M-token cutoff.

An observable prediction independent of model perplexity is that the consumed prefix has approximately 3M GPT-2 tokens from each inferred register, no repeated normalized 160-word prefixes, and a higher mean capped target/background log-odds score than a length-matched random pool sample. It should also contain markedly fewer parked-domain, login, navigation, and low-diversity templates.

# Falsification

The hypothesis is falsified if a token audit does not show the predicted equal mixture, uniqueness, and importance-score lift, or if manual blinded samples show that high-ranked documents contain as much boilerplate as matched random documents. More strongly, holding hygiene and token counts fixed while shuffling rank within each register should not match the curated selection if importance ranking is causal; collapsing the four queues should not preserve coverage if balancing is causal.

# Transfer

The method should transfer to a disjoint hidden sample because it uses low-capacity word distributions, capped contributions, generic prose checks, and no document/site allow-list or hand-picked IDs. For a new target, replace the four reference partitions and retain the background estimation, hygiene filters, deduplication, and token-aware fair queue; mixture weights should follow the new evaluation proportions.
2006
#248
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolexec
args
{
  "input": "const patch = \"*** Begin Patch\\n*** Update File: /workspace/submission/curate.py\\n@@\\n def doc_score(text, models, totals, broad, broad_total, vocab, blog, dlogs, bun, duns,\\n-              rawlog, rawun, dsir_weight, margin_weight):\\n+              rawlog, rawun, dsir_weight, margin_weight, nav_max):\\n@@\\n     if n < 120 or n > 8500:\\n         return None\\n+    lead = ws[:160]\\n+    nav_words = {'home','login','register','search','menu','navigation','download',\\n+                 'contact','password','sitemap','faq','categories','copyright','cart'}\\n+    if sum(w in nav_words for w in lead) / len(lead) > nav_max:\\n+        return None\\n@@\\n     ap.add_argument('--mode', choices=['balanced','quality'], default='balanced')\\n+    ap.add_argument('--nav-max', type=float, default=1.0,\\n+                    help='maximum navigation-word fraction in the first 160 words')\\n@@\\n                           blog, dlogs, bun, duns, rawlog, rawun, a.dsir_weight,\\n-                          a.margin_weight)\\n+                          a.margin_weight, a.nav_max)\\n*** End Patch\"; text(await tools.apply_patch(patch))\n"
}
exec result
Script completed
Wall time 0.0 seconds
Output:
{}
#249
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolexec
args
{
  "input": "const r = await tools.exec_command({cmd:\"python -u submission/curate.py --nav-max 0.06 --output /tmp/candidate_nav.json\",\"workdir\":\"/workspace\",\"yield_time_ms\":1000,\"max_output_tokens\":5000,\"tty\":true}); text(JSON.stringify(r))\n"
}
exec result
Script completed
Wall time 1.2 seconds
Output:
{"chunk_id":"ea46f1","wall_time_seconds":1.002405912,"session_id":99480,"original_token_count":0,"output":""}
#250
reasoning (redacted)

encrypted chain-of-thought — recorded, not exposed

toolexec
args
{
  "input": "const r = await tools.write_stdin({session_id:99480,chars:\"\",yield_time_ms:30000,max_output_tokens:5000}); text(JSON.stringify(r))\n"
}
exec result
Script running with cell ID 170
Wall time 11.0 seconds
Output: