From d4a17dc89138c9ff6b6271b70a7d04d176a9da5e Mon Sep 17 00:00:00 2001 From: almondsun Date: Mon, 13 Jul 2026 01:50:48 -0500 Subject: [PATCH 1/2] preregister Hamlet replication --- configs/gptiny_hamlet_bytebpe512_sealed.yaml | 39 +++++++++++ configs/gptiny_hamlet_char_sealed.yaml | 37 ++++++++++ .../027-hamlet-external-distribution.md | 70 +++++++++++++++++++ 3 files changed, 146 insertions(+) create mode 100644 configs/gptiny_hamlet_bytebpe512_sealed.yaml create mode 100644 configs/gptiny_hamlet_char_sealed.yaml create mode 100644 experiments/027-hamlet-external-distribution.md diff --git a/configs/gptiny_hamlet_bytebpe512_sealed.yaml b/configs/gptiny_hamlet_bytebpe512_sealed.yaml new file mode 100644 index 0000000..dedceea --- /dev/null +++ b/configs/gptiny_hamlet_bytebpe512_sealed.yaml @@ -0,0 +1,39 @@ +data: + input_path: data/raw/hamlet_body.txt + prepared_path: data/processed/hamlet_corpus.txt + manifest_path: data/processed/hamlet_corpus_sealed_manifest.json + tokenizer_path: data/processed/hamlet_tokenizer_bytebpe512_sealed.json + tokenizer_type: byte_bpe + bpe_vocab_size: 512 + bpe_min_frequency: 2 + block_size: 37 + train_split: 0.8 + validation_split: 0.1 + +model: + vocab_size: 512 + block_size: 37 + n_layer: 4 + n_head: 4 + n_embd: 128 + dropout: 0.1 + +train: + run_name: gptiny_hamlet_bytebpe512_sealed + runs_dir: runs + batch_size: 27 + max_steps: 5000 + learning_rate: 0.001 + weight_decay: 0.0 + log_interval: 100 + eval_interval: 250 + eval_batches: null + early_stopping_patience: 3 + early_stopping_min_delta: 0.0 + sample_prompt: "To be" + sample_max_new_tokens: 100 + sample_temperature: 1.0 + sample_top_k: null + sample_seed: 1337 + sample_greedy: false + seed: 1337 diff --git a/configs/gptiny_hamlet_char_sealed.yaml b/configs/gptiny_hamlet_char_sealed.yaml new file mode 100644 index 0000000..4bfc19f --- /dev/null +++ b/configs/gptiny_hamlet_char_sealed.yaml @@ -0,0 +1,37 @@ +data: + input_path: data/raw/hamlet_body.txt + prepared_path: data/processed/hamlet_corpus.txt + manifest_path: data/processed/hamlet_corpus_sealed_manifest.json + tokenizer_path: data/processed/hamlet_tokenizer_char_sealed.json + tokenizer_type: char + block_size: 64 + train_split: 0.8 + validation_split: 0.1 + +model: + vocab_size: 256 + block_size: 64 + n_layer: 4 + n_head: 4 + n_embd: 128 + dropout: 0.1 + +train: + run_name: gptiny_hamlet_char_sealed + runs_dir: runs + batch_size: 16 + max_steps: 5000 + learning_rate: 0.001 + weight_decay: 0.0 + log_interval: 100 + eval_interval: 250 + eval_batches: null + early_stopping_patience: 3 + early_stopping_min_delta: 0.0 + sample_prompt: "To be" + sample_max_new_tokens: 100 + sample_temperature: 1.0 + sample_top_k: null + sample_seed: 1337 + sample_greedy: false + seed: 1337 diff --git a/experiments/027-hamlet-external-distribution.md b/experiments/027-hamlet-external-distribution.md new file mode 100644 index 0000000..c0e2e7a --- /dev/null +++ b/experiments/027-hamlet-external-distribution.md @@ -0,0 +1,70 @@ +# 027 — Hamlet External-Distribution Replication + +## Preregistration + +This section was committed before downloading, inspecting, preparing, tokenizing, training on, or +evaluating the Hamlet corpus. Later sections will preserve this declaration and separately report +observations. + +### Question And Hypothesis + +Does the frozen ByteBPE512 decision from milestones 025–026 retain its direction on a dramatic +play rather than narrative prose? + +The directional hypothesis is that ByteBPE512 will have lower sealed-test bits per character (BPC) +than the character tokenizer. The null outcome includes a tie or a character advantage. No minimum +effect size is required, and the observed margin will be reported without rounding before the +comparison is calculated. + +### Corpus Contract + +- Source: *Hamlet* by William Shakespeare, Project Gutenberg ebook #1524. +- URL: `https://www.gutenberg.org/cache/epub/1524/pg1524.txt`. +- Extraction: the complete text between the unique ordered Gutenberg START/END markers, with no + character-budget truncation. +- Normalization: the existing `prepare_corpus.py` rules only. +- Split: chronological 80% train, 10% validation, 10% sealed test after normalization. +- The raw download, extracted body, prepared corpus, tokenizers, checkpoints, and run artifacts stay + ignored; hashes and exact counts will be recorded here. + +Hamlet was selected because dialogue, verse, speaker labels, and stage directions differ +structurally from the Alice and Peter Pan prose distributions. No alternative corpus will be +substituted based on model results. Substitution is permitted only if the declared source cannot be +fetched or fails the existing deterministic Gutenberg marker contract, and any substitution must be +recorded before training. + +### Frozen Comparison + +Both models use seed 1337, 4 layers, 4 heads, width 128, dropout 0.1, AdamW at `1e-3`, zero weight +decay, full validation every 250 steps, patience 3, and a 5,000-step ceiling. Character uses context +64 and batch 16. ByteBPE512 uses the existing boundary-aware lossless tokenizer, context 37, batch +27, vocabulary target 512, and minimum merge frequency 2. These settings match milestone 026 and +will not change after corpus access. + +The committed configurations are: + +- `configs/gptiny_hamlet_char_sealed.yaml` +- `configs/gptiny_hamlet_bytebpe512_sealed.yaml` + +### Decision And Access Rule + +Tokenizer fitting uses training text only. Gradient updates use training tokens only. Validation +selects each run's `best_checkpoint.pt` and drives early stopping. Both training runs must finish +before either test segment is tokenized or scored. Each best checkpoint is then evaluated exactly +once with full target coverage using `scripts/evaluate_test.py`. + +The primary comparison is + +\[ +\Delta_{\text{Hamlet}} = +\operatorname{BPC}_{\text{ByteBPE512,test}}- +\operatorname{BPC}_{\text{character,test}}. +\] + +The hypothesis is supported when \(\Delta_{\text{Hamlet}}<0\). Validation results, stopping steps, +token counts, baselines, test loss, test BPC, checkpoint hashes, and validation-to-test gaps will all +be reported. Test results will not trigger configuration changes or additional Hamlet runs. + +## Observations + +Pending execution of the preregistered protocol. From b4270c46ec83041f84614ae371ad24974eebfb0d Mon Sep 17 00:00:00 2001 From: almondsun Date: Mon, 13 Jul 2026 02:07:08 -0500 Subject: [PATCH 2/2] record Hamlet replication results --- AGENTS.md | 8 +- README.md | 5 +- docs/codex/build-and-test.md | 9 +- docs/codex/experiments.md | 5 + docs/experiments.md | 9 +- docs/training.md | 4 + .../027-hamlet-external-distribution.md | 130 +++++++++++++++++- notes/06-reproducibility.md | 24 ++++ 8 files changed, 182 insertions(+), 12 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index a3b6e76..ab83552 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -90,10 +90,10 @@ If a relevant check cannot be run, state why and what remains unverified. ## Current Technical Status -Milestone 026 is the latest modeling evidence. Frozen 80/10/10 runs evaluated once on sealed -terminal segments confirm ByteBPE512 over character by `0.0614` BPC on Alice and `0.0258` on Peter -Pan. Test BPC is worse than validation for all four models. These test segments are consumed and -must not guide further tuning; future modeling needs a new untouched evaluation distribution. +Milestone 027 is the latest modeling evidence. It was preregistered before Hamlet corpus access and +confirms ByteBPE512 over character by `0.0673` sealed-test BPC on a dramatic play. Hamlet test BPC +is better than validation for both models, reversing milestone 026's harder-tail pattern. Alice, +Peter Pan, and Hamlet tests are consumed and must not guide further tuning. ## Safety And Security diff --git a/README.md b/README.md index 5d25bb7..4efd872 100644 --- a/README.md +++ b/README.md @@ -17,8 +17,8 @@ For a fast technical review, inspect: 1. [`docs/architecture.md`](docs/architecture.md) for boundaries and the end-to-end pipeline. 2. [`docs/experiments.md`](docs/experiments.md) for the milestone index. -3. [`experiments/026-sealed-test-evaluation.md`](experiments/026-sealed-test-evaluation.md) - for the latest confirmatory held-out evidence and its limitations. +3. [`experiments/027-hamlet-external-distribution.md`](experiments/027-hamlet-external-distribution.md) + for the latest preregistered external-distribution evidence and its limitations. 4. [`src/smallm/model/`](src/smallm/model/) for the from-scratch GPTiny model. 5. [`src/smallm/training/`](src/smallm/training/) for training and run artifacts. 6. [`src/smallm/data/`](src/smallm/data/) for corpus and tokenizer contracts. @@ -51,6 +51,7 @@ chronological 90/10 split unless noted otherwise. | [024](experiments/024-cross-corpus-robustness.md) | Peter Pan character vs ByteBPE512 | Full-validation best bits/character | `2.1721` vs `2.1539` | ByteBPE512 replicates the direction on a second book, but only by 0.83%. | | [025](experiments/025-corpus-by-seed-matrix.md) | 2 tokenizers × 2 corpora × 3 seeds | Paired ByteBPE512 minus character BPC | `-0.0619` Alice; `-0.0252` Peter Pan | ByteBPE512 wins all six pairs; effect magnitude is corpus-dependent. | | [026](experiments/026-sealed-test-evaluation.md) | Frozen decision on terminal 10% test segments | ByteBPE512 minus character test BPC | `-0.0614` Alice; `-0.0258` Peter Pan | The tokenizer decision survives one-shot full-coverage tests on both books. | +| [027](experiments/027-hamlet-external-distribution.md) | Preregistered Hamlet play replication | ByteBPE512 minus character test BPC | `-0.0673` | The direction replicates on drama, while validation/test difficulty and effect size remain corpus-dependent. | Token-level loss and perplexity are not directly comparable between character and BPE tokenizers because they predict different units. The tokenizer diff --git a/docs/codex/build-and-test.md b/docs/codex/build-and-test.md index 17091df..027743b 100644 --- a/docs/codex/build-and-test.md +++ b/docs/codex/build-and-test.md @@ -176,8 +176,9 @@ current task or explicitly quoted from a prior report. ## Current Milestone Status -Milestone 026 is the current evaluation contract: legacy two-way splits remain compatible, while -explicit validation fractions reserve a chronological test region that training leaves unencoded -and unscored. +Milestone 027 uses the milestone-026 evaluation contract: legacy two-way splits remain compatible, +while explicit validation fractions reserve a chronological test region that training leaves +unencoded and unscored. Best-checkpoint test evaluation verifies corpus/checkpoint identity, requires full coverage, and -refuses artifact overwrite. +refuses artifact overwrite. The preregistered Hamlet replication confirms the workflow on an +external dramatic-play distribution. diff --git a/docs/codex/experiments.md b/docs/codex/experiments.md index c5b74c5..e7dfc32 100644 --- a/docs/codex/experiments.md +++ b/docs/codex/experiments.md @@ -175,3 +175,8 @@ Milestone 026 introduces an optional `validation_split`; explicit 80/10/10 confi region unavailable to tokenizer fitting, training, early stopping, and checkpoint selection. One-shot best-checkpoint test evaluation confirms ByteBPE512 by `0.0614` BPC on Alice and `0.0258` on Peter Pan. Those test segments are consumed and must not guide further tuning. + +Milestone 027 was preregistered in commit `d4a17dc` before Hamlet corpus access. Under the unchanged +seed-1337 sealed protocol, ByteBPE512 beats character by `0.0673` test BPC on the dramatic-play +distribution. Hamlet's terminal segment is easier than validation for both models, so gap direction +and effect magnitude remain corpus-dependent. The Hamlet test is consumed. diff --git a/docs/experiments.md b/docs/experiments.md index 7153245..6a9101c 100644 --- a/docs/experiments.md +++ b/docs/experiments.md @@ -33,6 +33,7 @@ not a replacement for the original reports. | [024 Cross-Corpus Robustness](../experiments/024-cross-corpus-robustness.md) | External validity | On near-size-matched Peter Pan, ByteBPE512 narrowly beats character at `2.1539` versus `2.1721` BPC. | | [025 Corpus-by-Seed Matrix](../experiments/025-corpus-by-seed-matrix.md) | Factorial robustness | ByteBPE512 wins all six paired comparisons; mean advantage is `0.0619` BPC on Alice and `0.0252` on Peter Pan. | | [026 Sealed Test Evaluation](../experiments/026-sealed-test-evaluation.md) | Confirmatory evaluation | On untouched terminal segments, ByteBPE512 beats character by `0.0614` BPC on Alice and `0.0258` on Peter Pan. | +| [027 Hamlet External-Distribution Replication](../experiments/027-hamlet-external-distribution.md) | Preregistered external validity | ByteBPE512 beats character by `0.0673` sealed-test BPC on a dramatic play; the terminal region is easier than validation for both models. | ## Topic Shortcuts @@ -60,7 +61,8 @@ not a replacement for the original reports. [021](../experiments/021-boundary-aware-byte-bpe.md), [024](../experiments/024-cross-corpus-robustness.md), [025](../experiments/025-corpus-by-seed-matrix.md), - [026](../experiments/026-sealed-test-evaluation.md). + [026](../experiments/026-sealed-test-evaluation.md), + [027](../experiments/027-hamlet-external-distribution.md). ## Current Status @@ -105,3 +107,8 @@ Milestone 026 freezes that decision and evaluates new 80/10/10 runs once on term ByteBPE512 reaches test BPC `2.1178` versus `2.1792` on Alice and `2.2484` versus `2.2742` on Peter Pan. The direction and approximate margins survive, while every model's test BPC is worse than its validation BPC. + +Milestone 027 preregisters Hamlet before corpus access and transfers the same frozen protocol to a +dramatic play. ByteBPE512 reaches sealed-test BPC `2.2546` versus character's `2.3219`, an advantage +of `0.0673`. Both terminal results are better than validation, reversing milestone 026's gap +direction and reinforcing that chronological difficulty is corpus-dependent. diff --git a/docs/training.md b/docs/training.md index 8ea2da6..7b5bb2a 100644 --- a/docs/training.md +++ b/docs/training.md @@ -49,6 +49,7 @@ compileall, and Markdown links. `make smoke` writes an ignored smoke run, and | `configs/gptiny_char_5k_lr1e-3_earlystop*.yaml` | Three-seed Alice character early-stopping controls. | | `configs/gptiny_peterpan_*_earlystop_seed*.yaml` | Additional Peter Pan matrix seeds. | | `configs/gptiny_{alice,peterpan}_{char,bytebpe512}_sealed.yaml` | Frozen 80/10/10 confirmatory runs. | +| `configs/gptiny_hamlet_{char,bytebpe512}_sealed.yaml` | Preregistered 80/10/10 dramatic-play replication. | ## Corpus Preparation @@ -231,6 +232,9 @@ is robust within the matrix while the effect magnitude remains corpus-dependent. Experiment 026 adds a three-way chronological contract and evaluates the frozen decision once. ByteBPE512 beats character on sealed test BPC by `0.0614` on Alice and `0.0258` on Peter Pan. All four test results are worse than validation, and these terminal segments are now consumed evidence. +Experiment 027 preregisters a structurally different Hamlet distribution before corpus access. +ByteBPE512 beats character by `0.0673` sealed-test BPC. Both terminal results are better than +validation, demonstrating that chronological gap direction is not stable across corpora. ## Artifact Policy diff --git a/experiments/027-hamlet-external-distribution.md b/experiments/027-hamlet-external-distribution.md index c0e2e7a..5eb8bf9 100644 --- a/experiments/027-hamlet-external-distribution.md +++ b/experiments/027-hamlet-external-distribution.md @@ -67,4 +67,132 @@ be reported. Test results will not trigger configuration changes or additional H ## Observations -Pending execution of the preregistered protocol. +The preregistration above was committed as `d4a17dc89138c9ff6b6271b70a7d04d176a9da5e` at +`2026-07-13T01:50:48-05:00`, before the first network request. The declared protocol was executed +without substitution or post-access configuration changes. + +### Corpus And Provenance + +The Project Gutenberg download is 202 KiB with SHA-256 +`31584c19795779431b5933499a45b9cecb03e8c246b6ea7e245a2b6d65e8e784`. Deterministic marker +extraction produced 177,962 characters with SHA-256 +`ebce1da4e6c704d3969efe6c56bd09c9392f1291c59f857348a4b9802696691f`. Standard normalization +produced 177,932 characters, 69 distinct source characters, and SHA-256 +`cf18ea4afacfe22a86e74ae5f524a2017cd6f1079bc5b72d884cea5498567c1e`. + +| train characters | validation characters | sealed-test characters | +| ---: | ---: | ---: | +| 142,345 | 17,793 | 17,794 | + +The character tokenizer has 70 tokens including its unknown token. ByteBPE512 reached its declared +512-token vocabulary and encoded training text in 86,403 tokens, or 1.6475 source characters per +token. Vocabulary-dependent embeddings yield 819,526 character-model parameters and 929,664 +ByteBPE512 parameters; this remains an architecture-family comparison rather than a +parameter-matched comparison. + +### Validation Baselines + +| tokenizer | uniform loss | unigram loss | add-one bigram loss | +| --- | ---: | ---: | ---: | +| character | 4.2485 | 3.3269 | 2.5210 | +| ByteBPE512 | 6.2383 | 4.4516 | 3.7830 | + +These are validation token losses and are comparable only within a tokenizer. + +### Training And Frozen Selection + +Both runs completed before either test evaluation. Their summaries contained the 17,794-character +sealed count but no test tokens, loss, or BPC. + +| tokenizer | actual steps | best step | best validation loss | best validation BPC | final validation BPC | duration | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | +| character | 3,500 | 2,750 | 1.756849 | 2.534598 | 2.547963 | 349.4s | +| ByteBPE512 | 3,000 | 2,250 | 2.802604 | **2.525928** | 2.578391 | 263.5s | + +ByteBPE512's validation advantage is 0.008670 BPC. Both tokenizers stopped early after three exact +non-improvements, and both final checkpoints are worse than their selected best checkpoints. + +Run paths are: + +- `runs/gptiny_hamlet_char_sealed/2026-07-13_01-52-34` +- `runs/gptiny_hamlet_bytebpe512_sealed/2026-07-13_01-59-06` + +### One-Shot Sealed Test + +The two best checkpoints were evaluated once, after both training jobs completed. + +| tokenizer | test tokens | target tokens | target characters | test loss | test BPC | +| --- | ---: | ---: | ---: | ---: | ---: | +| character | 17,794 | 17,793 | 17,793 | 1.609405 | 2.321881 | +| ByteBPE512 | 10,813 | 10,812 | 17,792 | 2.571678 | **2.254615** | + +Both artifacts report full target-token coverage. ByteBPE512's first autoregressive input token +spans two source characters, while the character model's spans one; therefore their scored target +character counts differ by one. Each BPC denominator matches the characters completed by its scored +target tokens. + +The preregistered contrast is + +\[ +\Delta_{\text{Hamlet}}=2.2546152062724256-2.321880862208522 +=\mathbf{-0.06726565593609646}\ \text{BPC}. +\] + +The directional hypothesis is supported. ByteBPE512 now beats character on sealed test BPC for +Alice, Peter Pan, and Hamlet. The effect size is not stable: Hamlet's validation margin is only +0.008670 BPC, while its test margin is 0.067266 BPC. + +Unlike milestone 026, Hamlet's terminal segment is easier than its middle validation segment. The +test-minus-best-validation gap is -0.212717 BPC for character and -0.271313 for ByteBPE512. This +does not prove improved generalization; chronological regions can differ in speaker mix, scene +structure, verse density, and intrinsic entropy. It does show that the prior harder-tail pattern is +not distribution-invariant. + +Checkpoint SHA-256 values are: + +- character: `cc75d4fdb311f7b61da2ba1f0b2d674e3f4f7f7c42a24eafbc9246320ffff35d` +- ByteBPE512: `36022b1879f8c664c2a0c89bfc426bf94721b7805822b48f08fd4b31580fe224` + +### Exact Commands + +```bash +curl -fL https://www.gutenberg.org/cache/epub/1524/pg1524.txt \ + -o data/raw/hamlet_gutenberg.txt +sha256sum data/raw/hamlet_gutenberg.txt +.venv/bin/python scripts/extract_gutenberg.py \ + --input data/raw/hamlet_gutenberg.txt --output data/raw/hamlet_body.txt +.venv/bin/python scripts/prepare_corpus.py \ + --input data/raw/hamlet_body.txt --output data/processed/hamlet_corpus.txt \ + --stats data/processed/hamlet_corpus_stats.json \ + --manifest data/processed/hamlet_corpus_sealed_manifest.json \ + --source-name "Hamlet by William Shakespeare" \ + --source-note "Project Gutenberg ebook #1524; complete body between unique START/END markers; fetched from https://www.gutenberg.org/cache/epub/1524/pg1524.txt" \ + --train-split 0.8 --validation-split 0.1 +.venv/bin/python scripts/prepare_data.py --config configs/gptiny_hamlet_char_sealed.yaml +.venv/bin/python scripts/evaluate_baselines.py --config configs/gptiny_hamlet_char_sealed.yaml +.venv/bin/python scripts/prepare_data.py --config configs/gptiny_hamlet_bytebpe512_sealed.yaml +.venv/bin/python scripts/evaluate_baselines.py --config configs/gptiny_hamlet_bytebpe512_sealed.yaml +.venv/bin/python scripts/train.py --config configs/gptiny_hamlet_char_sealed.yaml +.venv/bin/python scripts/train.py --config configs/gptiny_hamlet_bytebpe512_sealed.yaml +.venv/bin/python scripts/evaluate_test.py \ + --run runs/gptiny_hamlet_char_sealed/2026-07-13_01-52-34 --checkpoint-kind best +.venv/bin/python scripts/evaluate_test.py \ + --run runs/gptiny_hamlet_bytebpe512_sealed/2026-07-13_01-59-06 --checkpoint-kind best +make PYTHON=.venv/bin/python check +make PYTHON=.venv/bin/python audit +``` + +### Limitations And Decision + +- This is one seed on one play, not a variance estimate for drama or Shakespeare. +- Alice, Peter Pan, and Hamlet remain canonical English literary texts despite the structural shift. +- Vocabulary-dependent parameter counts remain unequal. +- The one-token autoregressive prefix causes a one-character difference in scored BPC support. +- A chronological split measures forward textual position, not exchangeable sampling. +- The Hamlet test segment is now consumed and cannot guide further tuning or reruns. + +The frozen ByteBPE512 decision survives the preregistered external-distribution test. The honest +claim is directional robustness across three texts, not a universal margin or universal +generalization gap. A stronger next milestone would preregister a small corpus-family panel that +includes nonfiction or speeches and multiple seeds, with new sealed segments defined before any +model access. diff --git a/notes/06-reproducibility.md b/notes/06-reproducibility.md index 2e482a0..739eee8 100644 --- a/notes/06-reproducibility.md +++ b/notes/06-reproducibility.md @@ -188,3 +188,27 @@ higher irreducible entropy or a shifted style. Comparing tokenizer contrasts validation and test asks a different question: whether the frozen decision preserves its direction under forward distribution shift. Milestone 026 finds \(G>0\) for every model while \(\Delta_c<0\) on both sealed tests. + +### Preregistration and distributional replication + +Preregistration separates a hypothesis from its outcome in time. A useful repository-native record +commits the source identity, extraction rule, split, seeds, configurations, stopping policy, primary +metric, directional contrast, and test-access rule before data or model access. Git history then +makes later changes observable. It does not prevent misconduct, but it sharply reduces ambiguity +about which choices preceded the result. + +External-distribution replication asks whether a fixed decision transfers when the data-generating +process changes. Let \(c\) index corpora and +\(\Delta_c=\operatorname{BPC}_{\mathrm{candidate},c}- +\operatorname{BPC}_{\mathrm{control},c}\). Repeatedly observing +\(\operatorname{sign}(\Delta_c)<0\) supports directional robustness. It does not imply a common +effect size: heterogeneity in \(|\Delta_c|\) is evidence that the tokenizer interacts with corpus +structure. + +The chronological generalization gap +\(G_c=\operatorname{BPC}_{\mathrm{test},c}-\operatorname{BPC}_{\mathrm{val},c}\) is likewise not a +model-only property. Milestone 026 finds \(G_c>0\) for Alice and Peter Pan, while preregistered +milestone 027 finds \(G_c<0\) for both Hamlet models. Textual position can alter speaker mix, +chapter or scene structure, punctuation, verse density, and intrinsic entropy. A professional +interpretation therefore reports both the model contrast and the regional difficulty shift rather +than labeling every positive gap "overfitting" or every negative gap "improved generalization."