From ea4129b6947a94f95fdf065f95186eeecc0aee27 Mon Sep 17 00:00:00 2001 From: almondsun Date: Sun, 12 Jul 2026 23:43:48 -0500 Subject: [PATCH] Add cross-corpus robustness milestone --- AGENTS.md | 9 +- README.md | 5 +- ..._bytebpe512_5k_lr1e-3_ctx37_earlystop.yaml | 38 ++++++ ...iny_peterpan_char_5k_lr1e-3_earlystop.yaml | 36 ++++++ docs/codex/build-and-test.md | 2 +- docs/codex/experiments.md | 4 + docs/experiments.md | 9 +- docs/training.md | 6 + experiments/024-cross-corpus-robustness.md | 116 ++++++++++++++++++ notes/06-reproducibility.md | 39 ++++++ scripts/extract_gutenberg.py | 34 +++++ src/smallm/data/__init__.py | 2 + src/smallm/data/corpus.py | 19 +++ tests/test_corpus_preparation.py | 39 ++++++ 14 files changed, 350 insertions(+), 8 deletions(-) create mode 100644 configs/gptiny_peterpan_bytebpe512_5k_lr1e-3_ctx37_earlystop.yaml create mode 100644 configs/gptiny_peterpan_char_5k_lr1e-3_earlystop.yaml create mode 100644 experiments/024-cross-corpus-robustness.md create mode 100644 scripts/extract_gutenberg.py diff --git a/AGENTS.md b/AGENTS.md index 709c3db..ed5e5de 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -90,10 +90,11 @@ If a relevant check cannot be run, state why and what remains unverified. ## Current Technical Status -Milestone 023 is the latest modeling evidence. Across preregistered seeds 1337, 2027, and 4242, -ByteBPE512 early stopping reaches best full-validation BPC `2.0225 ± 0.0124` (population SD), range -`2.0083–2.0384`; all seeds beat the character control. Future work should test split or corpus -robustness rather than select a favorable seed or add another narrow hyperparameter. +Milestone 024 is the latest modeling evidence. On a deterministically extracted, near-size-matched +Peter Pan corpus, ByteBPE512 reaches best BPC `2.1539` versus `2.1721` for character. This +replicates the direction found on Alice, but the 0.83% margin from one seed is not a stable +effect-size estimate. Milestone 023 remains the seed-robustness reference (`2.0225 ± 0.0124` on +Alice); the next strong test is a corpus-by-seed matrix. ## Safety And Security diff --git a/README.md b/README.md index 9ac8cd0..613aabc 100644 --- a/README.md +++ b/README.md @@ -17,8 +17,8 @@ For a fast technical review, inspect: 1. [`docs/architecture.md`](docs/architecture.md) for boundaries and the end-to-end pipeline. 2. [`docs/experiments.md`](docs/experiments.md) for the milestone index. -3. [`experiments/021-boundary-aware-byte-bpe.md`](experiments/021-boundary-aware-byte-bpe.md) - for the latest corrected modeling evidence and its limitations. +3. [`experiments/024-cross-corpus-robustness.md`](experiments/024-cross-corpus-robustness.md) + for the latest cross-corpus modeling evidence and its limitations. 4. [`src/smallm/model/`](src/smallm/model/) for the from-scratch GPTiny model. 5. [`src/smallm/training/`](src/smallm/training/) for training and run artifacts. 6. [`src/smallm/data/`](src/smallm/data/) for corpus and tokenizer contracts. @@ -48,6 +48,7 @@ chronological 90/10 split unless noted otherwise. | [021](experiments/021-boundary-aware-byte-bpe.md) | Boundary-aware ByteBPE320/512 | Full-validation best bits/character | `2.0286` / `2.0083` | Both beat the corrected character and BPE128 controls; ByteBPE512 overfit after step 1750. | | [022](experiments/022-early-stopping-and-regularization.md) | ByteBPE512 early stopping / weight decay | Actual steps and best BPC | `2500`, `2.0083` / `2.0080` | Early stopping halves runtime and limits final degradation; weight decay `0.01` is effectively neutral. | | [023](experiments/023-multi-seed-robustness.md) | ByteBPE512 across three seeds | Best BPC mean ± population SD | `2.0225 ± 0.0124` | All tested seeds beat the character control; stopping varies from 2500–3000 steps. | +| [024](experiments/024-cross-corpus-robustness.md) | Peter Pan character vs ByteBPE512 | Full-validation best bits/character | `2.1721` vs `2.1539` | ByteBPE512 replicates the direction on a second book, but only by 0.83%. | Token-level loss and perplexity are not directly comparable between character and BPE tokenizers because they predict different units. The tokenizer diff --git a/configs/gptiny_peterpan_bytebpe512_5k_lr1e-3_ctx37_earlystop.yaml b/configs/gptiny_peterpan_bytebpe512_5k_lr1e-3_ctx37_earlystop.yaml new file mode 100644 index 0000000..b6a09d7 --- /dev/null +++ b/configs/gptiny_peterpan_bytebpe512_5k_lr1e-3_ctx37_earlystop.yaml @@ -0,0 +1,38 @@ +data: + input_path: data/raw/peter_pan_body.txt + prepared_path: data/processed/peter_pan_corpus.txt + manifest_path: data/processed/peter_pan_corpus_manifest.json + tokenizer_path: data/processed/peter_pan_tokenizer_bytebpe512.json + tokenizer_type: byte_bpe + bpe_vocab_size: 512 + bpe_min_frequency: 2 + block_size: 37 + train_split: 0.9 + +model: + vocab_size: 512 + block_size: 37 + n_layer: 4 + n_head: 4 + n_embd: 128 + dropout: 0.1 + +train: + run_name: gptiny_peterpan_bytebpe512_5k_lr1e-3_ctx37_earlystop + runs_dir: runs + batch_size: 27 + max_steps: 5000 + learning_rate: 0.001 + weight_decay: 0.0 + log_interval: 100 + eval_interval: 250 + eval_batches: null + early_stopping_patience: 3 + early_stopping_min_delta: 0.0 + sample_prompt: Once + sample_max_new_tokens: 100 + sample_temperature: 1.0 + sample_top_k: null + sample_seed: 1337 + sample_greedy: false + seed: 1337 diff --git a/configs/gptiny_peterpan_char_5k_lr1e-3_earlystop.yaml b/configs/gptiny_peterpan_char_5k_lr1e-3_earlystop.yaml new file mode 100644 index 0000000..bf766c1 --- /dev/null +++ b/configs/gptiny_peterpan_char_5k_lr1e-3_earlystop.yaml @@ -0,0 +1,36 @@ +data: + input_path: data/raw/peter_pan_body.txt + prepared_path: data/processed/peter_pan_corpus.txt + manifest_path: data/processed/peter_pan_corpus_manifest.json + tokenizer_path: data/processed/peter_pan_tokenizer_char.json + tokenizer_type: char + block_size: 64 + train_split: 0.9 + +model: + vocab_size: 256 + block_size: 64 + n_layer: 4 + n_head: 4 + n_embd: 128 + dropout: 0.1 + +train: + run_name: gptiny_peterpan_char_5k_lr1e-3_earlystop + runs_dir: runs + batch_size: 16 + max_steps: 5000 + learning_rate: 0.001 + weight_decay: 0.0 + log_interval: 100 + eval_interval: 250 + eval_batches: null + early_stopping_patience: 3 + early_stopping_min_delta: 0.0 + sample_prompt: Once + sample_max_new_tokens: 100 + sample_temperature: 1.0 + sample_top_k: null + sample_seed: 1337 + sample_greedy: false + seed: 1337 diff --git a/docs/codex/build-and-test.md b/docs/codex/build-and-test.md index 7ad31b6..9e11684 100644 --- a/docs/codex/build-and-test.md +++ b/docs/codex/build-and-test.md @@ -50,7 +50,7 @@ The individual commands remain canonical and are listed below. python -m pytest ``` -Current expected result after milestone 023+: at least 141 tests passing with at least 90% coverage. +Current expected result after milestone 024+: at least 146 tests passing with at least 90% coverage. ### Compile Check diff --git a/docs/codex/experiments.md b/docs/codex/experiments.md index 6b55645..ba01d6f 100644 --- a/docs/codex/experiments.md +++ b/docs/codex/experiments.md @@ -161,3 +161,7 @@ neutral at best BPC `2.0080`; future work should test robustness across seeds or Milestone 023 runs the unregularized ByteBPE512 early-stopping setup at seeds 1337, 2027, and 4242. Best BPC is `2.0225 ± 0.0124` (population SD), range `2.0083–2.0384`; every seed beats the character control. The next modeling question should test split/corpus robustness, not select a seed. + +Milestone 024 uses a deterministically extracted, near-size-matched Peter Pan corpus. ByteBPE512 +reaches best BPC `2.1539` versus `2.1721` for character. The direction replicates, but the 0.0181 +BPC margin from one shared seed is weaker evidence than a corpus-by-seed matrix. diff --git a/docs/experiments.md b/docs/experiments.md index b8a315a..ab245a7 100644 --- a/docs/experiments.md +++ b/docs/experiments.md @@ -30,6 +30,7 @@ not a replacement for the original reports. | [021 Boundary-Aware Byte BPE](../experiments/021-boundary-aware-byte-bpe.md) | Tokenizer design | Lossless boundary-aware ByteBPE320/512 beat both corrected controls on best BPC; ByteBPE512 reached `2.0083` but overfit early. | | [022 Early Stopping and Regularization](../experiments/022-early-stopping-and-regularization.md) | Training control | Patience-3 stopping reproduced the step-1750 optimum and halved runtime; weight decay `0.01` was effectively neutral. | | [023 Multi-Seed Robustness](../experiments/023-multi-seed-robustness.md) | Robustness | Three preregistered seeds average best BPC `2.0225 ± 0.0124`; every seed beats the corrected character control. | +| [024 Cross-Corpus Robustness](../experiments/024-cross-corpus-robustness.md) | External validity | On near-size-matched Peter Pan, ByteBPE512 narrowly beats character at `2.1539` versus `2.1721` BPC. | ## Topic Shortcuts @@ -54,7 +55,8 @@ not a replacement for the original reports. [017](../experiments/017-best-checkpoint-evaluation.md). - Tokenization: [016](../experiments/016-tokenization-study.md), [020](../experiments/020-bpe-context-and-learning-rate.md), - [021](../experiments/021-boundary-aware-byte-bpe.md). + [021](../experiments/021-boundary-aware-byte-bpe.md), + [024](../experiments/024-cross-corpus-robustness.md). ## Current Status @@ -85,3 +87,8 @@ Milestone 023 measures seed sensitivity directly. Across seeds 1337, 2027, and 4 `2.0225 ± 0.0124` with range `2.0083–2.0384`; best step ranges 1,750–2,250 and stop step 2,500–3,000. The tokenizer result survives all tested seeds, while the observed seed spread confirms that milestone 022's tiny weight-decay delta was not decision-grade evidence. + +Milestone 024 changes the data distribution to a near-size-matched Peter Pan corpus. ByteBPE512 +reaches best BPC `2.1539` versus `2.1721` for character, reproducing the direction with a much +smaller 0.83% margin. This supports limited cross-corpus robustness, not a universal tokenizer +advantage; a corpus-by-seed matrix is the next stronger test. diff --git a/docs/training.md b/docs/training.md index 951f4d0..be572b2 100644 --- a/docs/training.md +++ b/docs/training.md @@ -43,6 +43,8 @@ compileall, and Markdown links. `make smoke` writes an ignored smoke run, and | `configs/gptiny_bpe128_5k_lr1e-3_ctx42.yaml` | Character-context-matched BPE128 control. | | `configs/gptiny_bpe128_5k_lr5e-4_ctx42.yaml` | Matched-context lower-learning-rate BPE128 run. | | `configs/gptiny_bpe128_5k_lr5e-4_ctx64.yaml` | Longer-context lower-learning-rate BPE128 run. | +| `configs/gptiny_peterpan_char_5k_lr1e-3_earlystop.yaml` | Peter Pan character control. | +| `configs/gptiny_peterpan_bytebpe512_5k_lr1e-3_ctx37_earlystop.yaml` | Context-matched Peter Pan ByteBPE512 run. | ## Corpus Preparation @@ -179,6 +181,10 @@ meaningful evidence of improvement. Experiment 023 repeats the unregularized early-stopping run across seeds 1337, 2027, and 4242. Best BPC is `2.0225 ± 0.0124` (population SD), and all three runs remain better than the corrected character control. Stop steps vary from 2,500 to 3,000. +Experiment 024 tests a second book. On the near-size-matched Peter Pan corpus, ByteBPE512 reaches +best BPC `2.1539` versus `2.1721` for character and stops at step 2,750. The 0.83% advantage is a +cross-corpus replication in direction, but too small and sparsely sampled to establish a stable +effect size. ## Artifact Policy diff --git a/experiments/024-cross-corpus-robustness.md b/experiments/024-cross-corpus-robustness.md new file mode 100644 index 0000000..281e72f --- /dev/null +++ b/experiments/024-cross-corpus-robustness.md @@ -0,0 +1,116 @@ +# 024 — Cross-Corpus Robustness + +## Goal + +Test whether the fixed ByteBPE512 early-stopping protocol still beats a character control on a +different public-domain book. Peter Pan was chosen before training; the corpus, character budget, +seed, architecture, optimizer, validation schedule, and decoding controls were fixed before either +model result was observed. + +## Corpus And Provenance + +The source is *Peter Pan* by J. M. Barrie, Project Gutenberg ebook #16, fetched from +`https://www.gutenberg.org/cache/epub/16/pg16.txt`. The downloaded UTF-8 file has SHA-256 +`6b08714281fe38266a756741e4c62915cda7536c2f78cca501f7fd53f3f445ae`. + +`scripts/extract_gutenberg.py` selects the text between the unique ordered Gutenberg START/END +markers, removes leading blank lines, and retains the first 144,530 body characters. That raw body +has SHA-256 `3b8bb7fc929423926bbefe595cd70a8c58e10e6227a90b0916515750e08aa97d`. +Standard normalization produces 144,489 characters (130,040 train; 14,449 validation), 79 distinct +characters, and SHA-256 +`16e4f26e7e5287dccced8520bd67965666c51c69afe0ed752f3b7d92f4693612`. +The budget is matched before normalization to the Alice study; the prepared corpora are therefore +near-matched, not byte-identical in size. + +## Setup + +Both runs use seed 1337, a chronological 90/10 split, full validation every 250 steps, patience 3, +a 5,000-step ceiling, 4 layers, 4 heads, width 128, dropout 0.1, AdamW at `1e-3`, and no weight +decay. The character model uses 64 tokens and batch 16. ByteBPE512 is fitted on training text only +and uses 37 tokens and batch 27. Its training compression is 1.7271 characters/token, so its +37-token window covers about 63.90 characters versus 64 for the character model. + +Vocabulary-dependent embeddings make the parameter counts unequal: 822,096 for character and +929,664 for ByteBPE512. This is the same architecture family, not a parameter-matched comparison. + +## Baselines + +| tokenizer | uniform loss | unigram loss | add-one bigram loss | +| --- | ---: | ---: | ---: | +| character (80 tokens) | 4.3820 | 3.1119 | 2.4078 | +| ByteBPE512 | 6.2383 | 4.3080 | 3.6856 | + +Token losses are not comparable across tokenizers; these rows are within-tokenizer references. + +## Validation Results + +| tokenizer | actual steps | best step | best token loss | best BPC | final BPC | stop reason | duration | +| --- | ---: | ---: | ---: | ---: | ---: | --- | ---: | +| character | 5,000 | 4,750 | 1.505551 | 2.172051 | 2.178503 | max steps | 496.5s | +| ByteBPE512 | 2,750 | 2,000 | 2.523778 | **2.153930** | 2.179118 | early stopping | 259.7s | + +ByteBPE512 wins by 0.018121 BPC, about 0.83% relative to the character result. The direction matches +Alice, but this margin is comparable to the seed variation measured for ByteBPE512 in milestone +023. One seed on one additional book therefore supports cross-corpus replication of the direction, +not a stable effect-size estimate. + +Early stopping again removes unnecessary training: ByteBPE512 stops after 55% of the ceiling. The +character model continues improving until step 4,750 and never triggers patience, showing that a +single stopping schedule need not imply similar learning dynamics across tokenizers or corpora. + +## Controlled Generation + +Prompt `Once`, 100 new tokens. Seeded decoding uses temperature 0.8, top-k 10, seed 1337. + +| tokenizer/checkpoint | greedy distinct-2 | seeded distinct-2 | +| --- | ---: | ---: | +| character best | 0.2524 | 0.6893 | +| character final | 0.2136 | 0.6117 | +| ByteBPE512 best | 0.2662 | 0.5238 | +| ByteBPE512 final | 0.5138 | 0.6011 | + +Best character seeded output begins `Once. He was the same out them to first her.` Best ByteBPE512 +seeded output begins `Once you to be a little house.` Both have local book-like texture but weak +syntax and no sustained coherence. Greedy outputs repeat phrases. Validation BPC and distinct-2 do +not identify the same checkpoint, so generation remains a separate diagnostic rather than proof of +language quality. + +## Exact Commands + +```bash +curl -fL https://www.gutenberg.org/cache/epub/16/pg16.txt -o data/raw/peter_pan_gutenberg.txt +sha256sum data/raw/peter_pan_gutenberg.txt +uv run --frozen --extra dev python scripts/extract_gutenberg.py \ + --input data/raw/peter_pan_gutenberg.txt --output data/raw/peter_pan_body.txt \ + --max-characters 144530 +uv run --frozen --extra dev python scripts/prepare_corpus.py \ + --input data/raw/peter_pan_body.txt --output data/processed/peter_pan_corpus.txt \ + --stats data/processed/peter_pan_corpus_stats.json \ + --manifest data/processed/peter_pan_corpus_manifest.json \ + --source-name "Peter Pan by J. M. Barrie" \ + --source-note "Project Gutenberg ebook #16; body between START/END markers; first 144530 body characters" +uv run --frozen --extra dev python scripts/prepare_data.py --config configs/gptiny_peterpan_char_5k_lr1e-3_earlystop.yaml +uv run --frozen --extra dev python scripts/evaluate_baselines.py --config configs/gptiny_peterpan_char_5k_lr1e-3_earlystop.yaml +uv run --frozen --extra dev python scripts/train.py --config configs/gptiny_peterpan_char_5k_lr1e-3_earlystop.yaml +uv run --frozen --extra dev python scripts/prepare_data.py --config configs/gptiny_peterpan_bytebpe512_5k_lr1e-3_ctx37_earlystop.yaml +uv run --frozen --extra dev python scripts/evaluate_baselines.py --config configs/gptiny_peterpan_bytebpe512_5k_lr1e-3_ctx37_earlystop.yaml +uv run --frozen --extra dev python scripts/train.py --config configs/gptiny_peterpan_bytebpe512_5k_lr1e-3_ctx37_earlystop.yaml +uv run --frozen --extra dev make check +``` + +Run inspection used `scripts/show_run.py`. Generation used `scripts/generate.py` for `best` and +`final`, once with `--greedy --diagnostics` and once with +`--temperature 0.8 --top-k 10 --seed 1337 --diagnostics`. + +## Limitations And Next Step + +- Peter Pan is a new book but remains English literary prose from Project Gutenberg. +- One shared seed cannot separate corpus effects from seed-by-corpus interactions. +- The chronological tail may differ in difficulty between books. +- Parameter counts differ because vocabularies differ. +- Repeated validation is used for checkpoint selection; there is no sealed test set. +- Distinct-n does not measure factuality, grammar, or long-range coherence. + +The next decision-grade experiment is a preregistered corpus-by-seed matrix: the same three seeds on +Alice and Peter Pan, ideally with a sealed terminal test segment. That would estimate tokenizer, +corpus, seed, and interaction effects instead of extrapolating from one run per new distribution. diff --git a/notes/06-reproducibility.md b/notes/06-reproducibility.md index dcbf78b..2bfad16 100644 --- a/notes/06-reproducibility.md +++ b/notes/06-reproducibility.md @@ -81,3 +81,42 @@ Never discard a completed seed because it weakens the conclusion, and never repo the expected result. A hyperparameter difference much smaller than seed-to-seed spread is not robust evidence. Decoding randomness is held fixed when comparing training seeds so observed generation variation comes from model training rather than a second uncontrolled random stream. + +### External validity and corpus-by-seed interactions + +A random split of one book estimates performance on held-out text from essentially the same source; +it does not establish robustness to a new author, style, vocabulary, or document structure. A new +book changes the empirical distribution \(P_C(X)\), so a tokenizer comparison can be modeled as + +\[ +y_{tcs}=\mu+\alpha_t+\beta_c+(\alpha\beta)_{tc}+u_s+\varepsilon_{tcs}, +\] + +where \(t\) is tokenizer, \(c\) corpus, \(s\) seed, \(\alpha_t\) is the tokenizer effect, +\(\beta_c\) is corpus difficulty, \((\alpha\beta)_{tc}\) is the tokenizer-by-corpus interaction, +and \(u_s\) captures training randomness. One seed on a second corpus observes a cell but cannot +separate interaction from seed noise. A balanced corpus-by-seed matrix can. + +Matching corpus length controls the amount of text, not its entropy, lexical diversity, boundary +statistics, or chronological-tail difficulty. Matching token context also requires converting to +characters: + +\[ +L_{\mathrm{chars}}=L_{\mathrm{tokens}}\frac{N_{\mathrm{train\ chars}}}{N_{\mathrm{train\ tokens}}}. +\] + +This is an average receptive-field proxy; individual ByteBPE sequences still vary in character +length. BPC makes held-out likelihood comparable across lossless tokenizers, + +\[ +\operatorname{BPC}=\frac{-\log_2 p(x_{1:n})}{n}, +\] + +but comparable metrics do not remove design confounds such as vocabulary-dependent parameter +counts or checkpoint selection on the same validation tail. + +External-validity claims should therefore form a ladder: same split across seeds; deterministic +alternative splits; a second source; multiple source families; finally, a sealed test distribution. +Each rung supports a wider claim, and none licenses the next one automatically. Milestone 024 +reaches the second-source rung: it replicates ByteBPE512's direction on Peter Pan, while its narrow +margin and single seed leave the interaction term unresolved. diff --git a/scripts/extract_gutenberg.py b/scripts/extract_gutenberg.py new file mode 100644 index 0000000..1f8dc83 --- /dev/null +++ b/scripts/extract_gutenberg.py @@ -0,0 +1,34 @@ +from __future__ import annotations + +import argparse +from pathlib import Path + +from smallm.data import extract_gutenberg_body +from smallm.utils.io import atomic_write_text + +_MAX_INPUT_BYTES = 20_000_000 + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("--input", required=True) + parser.add_argument("--output", required=True) + parser.add_argument("--max-characters", type=int) + args = parser.parse_args() + + input_path = Path(args.input) + if not input_path.is_file(): + parser.error(f"input file not found: {input_path}") + if input_path.stat().st_size > _MAX_INPUT_BYTES: + parser.error(f"input exceeds {_MAX_INPUT_BYTES} bytes") + try: + text = input_path.read_text(encoding="utf-8") + body = extract_gutenberg_body(text, max_characters=args.max_characters) + except (OSError, UnicodeError, ValueError) as exc: + parser.error(str(exc)) + atomic_write_text(args.output, body) + print(f"extracted {len(body)} characters to {args.output}") + + +if __name__ == "__main__": + main() diff --git a/src/smallm/data/__init__.py b/src/smallm/data/__init__.py index 88760ff..4eff05f 100644 --- a/src/smallm/data/__init__.py +++ b/src/smallm/data/__init__.py @@ -7,6 +7,7 @@ clean_corpus_text, corpus_manifest, corpus_stats, + extract_gutenberg_body, file_sha256, load_prepared_corpus, ) @@ -26,6 +27,7 @@ "clean_corpus_text", "corpus_manifest", "corpus_stats", + "extract_gutenberg_body", "file_sha256", "load_prepared_corpus", "load_tokenizer", diff --git a/src/smallm/data/corpus.py b/src/smallm/data/corpus.py index aaf782b..37eb383 100644 --- a/src/smallm/data/corpus.py +++ b/src/smallm/data/corpus.py @@ -1,6 +1,7 @@ from __future__ import annotations import hashlib +import re from collections import Counter from datetime import datetime, timezone from pathlib import Path @@ -13,6 +14,24 @@ "ensure final newline", ] +_GUTENBERG_START = re.compile(r"^\*\*\* START OF (?:THE|THIS) PROJECT GUTENBERG EBOOK .+ \*\*\*$") +_GUTENBERG_END = re.compile(r"^\*\*\* END OF (?:THE|THIS) PROJECT GUTENBERG EBOOK .+ \*\*\*$") + + +def extract_gutenberg_body(text: str, *, max_characters: int | None = None) -> str: + if max_characters is not None and max_characters <= 0: + raise ValueError("max_characters must be positive or null") + normalized = text.replace("\r\n", "\n").replace("\r", "\n") + lines = normalized.splitlines(keepends=True) + starts = [index for index, line in enumerate(lines) if _GUTENBERG_START.fullmatch(line.strip())] + ends = [index for index, line in enumerate(lines) if _GUTENBERG_END.fullmatch(line.strip())] + if len(starts) != 1 or len(ends) != 1 or starts[0] >= ends[0]: + raise ValueError("expected exactly one ordered Project Gutenberg START/END marker pair") + body = "".join(lines[starts[0] + 1 : ends[0]]).lstrip("\n") + if not body.strip(): + raise ValueError("Project Gutenberg body is empty") + return body if max_characters is None else body[:max_characters] + def clean_corpus_text(text: str) -> str: normalized = text.replace("\r\n", "\n").replace("\r", "\n") diff --git a/tests/test_corpus_preparation.py b/tests/test_corpus_preparation.py index 66dc58b..5843e54 100644 --- a/tests/test_corpus_preparation.py +++ b/tests/test_corpus_preparation.py @@ -8,11 +8,50 @@ clean_corpus_text, corpus_manifest, corpus_stats, + extract_gutenberg_body, file_sha256, load_prepared_corpus, ) +def test_extract_gutenberg_body_removes_markers_and_truncates(): + text = ( + "header\r\n" + "*** START OF THE PROJECT GUTENBERG EBOOK TEST ***\r\n" + "\r\nalpha\r\nbeta\r\n" + "*** END OF THE PROJECT GUTENBERG EBOOK TEST ***\r\nfooter" + ) + + assert extract_gutenberg_body(text) == "alpha\nbeta\n" + assert extract_gutenberg_body(text, max_characters=5) == "alpha" + + +@pytest.mark.parametrize( + "text", + [ + "no markers", + "*** START OF THE PROJECT GUTENBERG EBOOK TEST ***\nbody", + ( + "*** END OF THE PROJECT GUTENBERG EBOOK TEST ***\n" + "*** START OF THE PROJECT GUTENBERG EBOOK TEST ***" + ), + ], +) +def test_extract_gutenberg_body_rejects_invalid_markers(text): + with pytest.raises(ValueError, match="marker pair"): + extract_gutenberg_body(text) + + +def test_extract_gutenberg_body_rejects_invalid_limit_or_empty_body(): + with pytest.raises(ValueError, match="max_characters"): + extract_gutenberg_body("text", max_characters=0) + with pytest.raises(ValueError, match="empty"): + extract_gutenberg_body( + "*** START OF THE PROJECT GUTENBERG EBOOK TEST ***\n\n" + "*** END OF THE PROJECT GUTENBERG EBOOK TEST ***\n" + ) + + def test_clean_corpus_text_normalizes_whitespace_conservatively(): raw = "alpha \r\n\r\n\r\nbeta\t \n\ngamma"