Source-1: weights, loader and model card
Browse files- .gitattributes +2 -0
- AUTHORS +8 -0
- CREDITS_BOOKS.tsv +0 -0
- EVALUATION.md +1042 -0
- LICENSE +202 -0
- NOTICE +801 -0
- README.md +409 -0
- calibration.json +106 -0
- config.json +80 -0
- examples/expected_output.jsonl +6 -0
- examples/sample.jsonl +6 -0
- heads.safetensors +3 -0
- images/source1_benchmark.png +3 -0
- model.fp32.safetensors +3 -0
- model.safetensors +3 -0
- requirements.txt +8 -0
- source1.json +302 -0
- source1.py +1011 -0
- tokenizer.json +3 -0
- tokenizer_config.json +23 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,5 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
images/source1_benchmark.png filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
AUTHORS
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# This is the list of Source-1's significant contributors.
|
| 2 |
+
#
|
| 3 |
+
# "The Source-1 Authors", as used in the copyright notices of this repository
|
| 4 |
+
# (for example "Copyright 2026 The Source-1 Authors" in NOTICE), means the
|
| 5 |
+
# people listed below. Each author is listed by their Hugging Face handle.
|
| 6 |
+
|
| 7 |
+
The Source-1 Authors:
|
| 8 |
+
msmth
|
CREDITS_BOOKS.tsv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
EVALUATION.md
ADDED
|
@@ -0,0 +1,1042 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Source-1 evaluation
|
| 2 |
+
|
| 3 |
+
Source-1 was compared with its teacher and 16 public quality scorers on three test sets. An independent proprietary
|
| 4 |
+
LLM grader scored every chunk with Source-1's 13-field rubric. Each number says how closely a model's ranking agrees
|
| 5 |
+
with the grader's, so it measures agreement with Source-1's own rubric, on home ground.
|
| 6 |
+
|
| 7 |
+
## Summary
|
| 8 |
+
|
| 9 |
+
Rank agreement (Spearman) with the grader's overall score. Each model is read on the chunks it scored.
|
| 10 |
+
|
| 11 |
+
| test set | chunks | Source-1 (307M) | propella-1 4B (4.0B) | FineWeb-Edu classifier (English only) | teacher (open-weight, 27B) |
|
| 12 |
+
|---|---|---|---|---|---|
|
| 13 |
+
| held-out set, 53 languages (main result) | 495 | 0.900 | 0.756 | 0.529 | 0.912 |
|
| 14 |
+
| English exam | 413 | 0.921 | 0.820 | 0.453 | 0.912 |
|
| 15 |
+
| 12-language exam | 352 | 0.895 | 0.637 | - | 0.880 |
|
| 16 |
+
|
| 17 |
+
propella-1 4B scored 493, 412 and 350 of these chunks. The FineWeb-Edu classifier is read on the 159 English held-out
|
| 18 |
+
chunks, where Source-1 scores 0.864. The English exam has 414 chunks; the teacher scored 413, and comparisons use those.
|
| 19 |
+
|
| 20 |
+
- Source-1 agrees with the grader more closely than each of the 16 public scorers, on every set where they were
|
| 21 |
+
compared. All 39 of these comparisons ([how they are counted](#every-public-scorer)) have 95% intervals clear of
|
| 22 |
+
zero.
|
| 23 |
+
- It is level with its teacher within noise, with about 1/88 of the teacher's parameters. On the held-out set it is
|
| 24 |
+
0.012 lower (95% interval of the difference -0.03 to +0.004).
|
| 25 |
+
- The closest public scorer, propella-1 4B, trails by 0.144 on the held-out set. A more generous reading of it,
|
| 26 |
+
chosen after the results were known, narrows the gap to 0.064-0.081 ([details](#how-propella-1-is-read)).
|
| 27 |
+
- At its shipped drop line (the default `keep` flag), Source-1 catches 43 of the 64 chunks the grader drops on the
|
| 28 |
+
held-out set. The teacher, at its own keep flags, catches 46.
|
| 29 |
+
|
| 30 |
+
The held-out set is the main result. The exams count less, because they helped choose the teacher
|
| 31 |
+
([why](#independence-from-development)).
|
| 32 |
+
|
| 33 |
+
## How we measured
|
| 34 |
+
|
| 35 |
+
**The grader.** Its overall score and keep flag are computed from its 13 fields with the rubric's own formula, as for
|
| 36 |
+
Source-1. Its drops are the chunks that hit the rubric's hard filters (spam, boilerplate or toxic text). Its labels
|
| 37 |
+
were never trained on.
|
| 38 |
+
|
| 39 |
+
**Home ground.** The held-out documents come from the same kinds of sources as the training data. The public scorers
|
| 40 |
+
were built for their own definitions of quality, most for educational value, and are not wrong when they disagree with
|
| 41 |
+
this rubric.
|
| 42 |
+
|
| 43 |
+
**The three test sets.**
|
| 44 |
+
|
| 45 |
+
| test set | chunks | languages | grader drops | what it is |
|
| 46 |
+
|---|---|---|---|---|
|
| 47 |
+
| held-out set (main result) | 495, one per document | 53 | 64 | documents from Source-1's held-out test split, never trained or calibrated on |
|
| 48 |
+
| English exam | 414, from 332 documents (413 scored by the teacher) | English | 25 | 200 chunks drawn at random, plus 214 harder cases added on purpose |
|
| 49 |
+
| 12-language exam | 352, one per document | 12 | 46 | web text drawn at random from FineWeb-2, about 30 chunks per language |
|
| 50 |
+
|
| 51 |
+
**95% intervals.** Each difference between two models comes with a 95% interval from a paired bootstrap over
|
| 52 |
+
documents. When the interval excludes zero, chance alone is an unlikely explanation for the difference.
|
| 53 |
+
|
| 54 |
+
**How the public scorers were run.** Each public scorer ran from the pinned revision in
|
| 55 |
+
[Appendix C](#appendix-c-public-scorer-repositories), following its model card and its own code or prompt where it
|
| 56 |
+
publishes one, except as noted here. propella-1 and EAI-Distill were decoded greedily, although the propella-1 4B and
|
| 57 |
+
EAI-Distill repositories default to sampling. propella-1 ran in plain transformers, without the SGLang server and JSON
|
| 58 |
+
grammar its card recommends; the serving engine mainly affects speed, so we claim no speed comparison with it.
|
| 59 |
+
|
| 60 |
+
<details>
|
| 61 |
+
<summary>How the public scorers were run, in full</summary>
|
| 62 |
+
|
| 63 |
+
### How the public scorers were run
|
| 64 |
+
|
| 65 |
+
- Each public scorer ran from the repository and revision listed in
|
| 66 |
+
[Appendix C](#appendix-c-public-scorer-repositories), following its model card and its own code or prompt where it
|
| 67 |
+
publishes one.
|
| 68 |
+
- The encoder classifiers ran in Hugging Face transformers as their model cards show, in the precision each card
|
| 69 |
+
states (bfloat16 where it says so, float32 otherwise).
|
| 70 |
+
- The two fastText models ran in the fasttext library. They read the whole text with newlines turned into spaces, as
|
| 71 |
+
the DCLM code does.
|
| 72 |
+
- EAI-Distill ran in transformers in float32 with greedy decoding (its repository's default settings sample).
|
| 73 |
+
- propella-1 ran in plain transformers in bfloat16, with its repository's own prompt and chat template, greedy
|
| 74 |
+
decoding and no JSON grammar. Its card recommends serving with SGLang and a JSON grammar, and the 4B's default
|
| 75 |
+
settings sample at temperature 0.7. The serving engine mainly affects speed. Greedy decoding without a grammar gives
|
| 76 |
+
the same tokens as grammar-constrained greedy decoding, up to the first token the grammar would forbid, and it
|
| 77 |
+
avoids sampling noise. An answer that did not parse got one retry with more new tokens
|
| 78 |
+
([details](#propella-1s-answers)). We claim no speed comparison with propella-1.
|
| 79 |
+
|
| 80 |
+
</details>
|
| 81 |
+
|
| 82 |
+
### Independence from development
|
| 83 |
+
|
| 84 |
+
- The exam sets helped choose the teacher and its prompt setup, by agreement with the exam grader's labels, and
|
| 85 |
+
Source-1's backbone was kept over two other multilingual encoders in a comparison that looked at these sets.
|
| 86 |
+
- Grades and reviews by proprietary LLMs, among them the grader's model family, informed the rubric, the drop-line
|
| 87 |
+
candidates and floor, and the data filters.
|
| 88 |
+
- AI assistants helped draft the rubric text and write the project's code. The grader's model family includes one of
|
| 89 |
+
the assistants that helped draft the rubric text.
|
| 90 |
+
- The held-out set was built after the teacher and its prompt were fixed. It was not used to choose the model. It was
|
| 91 |
+
not hidden, though: the candidate drop lines were drafted after its grades had been seen.
|
| 92 |
+
- No grade or review by a proprietary LLM was ever used as a label, a training target or a training example.
|
| 93 |
+
|
| 94 |
+
<details>
|
| 95 |
+
<summary>Independence from development, in full</summary>
|
| 96 |
+
|
| 97 |
+
What grades from the grader's model family did and did not influence:
|
| 98 |
+
|
| 99 |
+
- **The exam sets helped choose the teacher.** The English and 12-language exams are the samples on which the teacher
|
| 100 |
+
was vetted and its prompt setup was chosen, by agreement with the exam grader's labels. The prompt setup covers the
|
| 101 |
+
answer format, reasoning on or off, and a rubric rule for pages stitched together from unrelated or scrambled text,
|
| 102 |
+
added after reviewing disagreements on these samples. The backbone was also kept over two other multilingual
|
| 103 |
+
encoders in a comparison that looked at these sets. So the exams are not independent of the grader: Source-1's
|
| 104 |
+
teacher was in part selected to agree with it there. They are reported, but they count less than the held-out set.
|
| 105 |
+
- **The held-out set did not.** It was built after the teacher and its prompt had been fixed, and was not used to
|
| 106 |
+
choose either. It is called held-out because Source-1 never trained or calibrated on it. But it was not hidden
|
| 107 |
+
during development: Source-1's results on it were checked, and the candidate drop lines were drafted after its
|
| 108 |
+
grades had been seen. The released model was not chosen by its score on this set: it is the model trained on the
|
| 109 |
+
final, fully cleaned data.
|
| 110 |
+
- **The rubric and the drop-line rule.** The choice between rubric revisions, the list of candidate drop lines and the
|
| 111 |
+
0.95 keep-agreement floor of the drop-line rule were decided using grades from proprietary LLM graders, among them
|
| 112 |
+
models of the grader's family, on samples of documents, some of them training chunks. The drop line itself was then
|
| 113 |
+
picked by that fixed rule on the teacher's validation labels. The grader's labels were not used to pick it.
|
| 114 |
+
- **Training.** Every training label comes from the open-weight teacher. The grader's labels, and every other grade or
|
| 115 |
+
review by a proprietary LLM, were never used as labels, training targets or training examples. The learning rate
|
| 116 |
+
and the natural language mix came from a short sweep (two learning rates, two language mixes, half an epoch each)
|
| 117 |
+
on a smaller training set labeled by the same teacher, read on validation agreement with the teacher. The epoch
|
| 118 |
+
count came from the same validation curves. The checkpoint is the final step, which had the best validation
|
| 119 |
+
agreement with the teacher (Spearman 0.955). No graded test set was used for these choices.
|
| 120 |
+
- **Data filtering.** Samples reviewed by a proprietary LLM measured how often the license and table-of-contents
|
| 121 |
+
filters missed or over-fired. This informed which collections were filtered out; the reviewed documents were then
|
| 122 |
+
kept out of training. AI assistants also reviewed source terms and document notices across the training data,
|
| 123 |
+
which informed the license rules. They also helped draft the rubric text and write the project's code.
|
| 124 |
+
|
| 125 |
+
</details>
|
| 126 |
+
|
| 127 |
+
<details>
|
| 128 |
+
<summary>The test sets in detail: composition, near-duplicate check, safety filter</summary>
|
| 129 |
+
|
| 130 |
+
### The test sets in detail
|
| 131 |
+
|
| 132 |
+
- **Held-out set** (the main result): 495 chunks in 53 languages, one per document. All come from Source-1's held-out
|
| 133 |
+
test split, drawn from the four data stages (web 129, multilingual web 276, conversations/code/synthetic 48, open
|
| 134 |
+
books 42). The grader drops 64 of them. 159 chunks are English and 29 Chinese; most other languages have 1 to 12
|
| 135 |
+
chunks. Source-1 never trained or calibrated on these documents. They come from the same kinds of sources as the
|
| 136 |
+
training data. 42 of the 495 chunks (8.5%) come from sources that were later removed from training under the
|
| 137 |
+
license rules (19 from DCLM-baseline, 16 raw Common Crawl pages, and 7 from collections with unreliable or gated
|
| 138 |
+
license terms). The test split also keeps documents that the license filtering removed from training.
|
| 139 |
+
- **English exam**: 414 chunks from 332 whole documents split into chunks. 200 chunks from 196 documents were drawn at
|
| 140 |
+
random from web, wiki, Common Pile, math and code sources (the **random-sample** chunks and documents). 214 chunks
|
| 141 |
+
from 136 documents were added on purpose to cover harder cases (long documents 57 chunks, academic 50, math 28,
|
| 142 |
+
code 27, spam 23, toxic 20, fiction 9). The grader drops 25 of the chunks (24 documents). The added chunks make the
|
| 143 |
+
set unlike a random draw. They move each model's numbers, up for some models and down for others (Source-1: rank
|
| 144 |
+
0.919 on the random-sample chunks alone and 0.921 on all 414; AUC 0.987 and 0.979). The teacher has no score for
|
| 145 |
+
one of the 414 chunks (a random-sample chunk). So the comparisons with the teacher and the public scorers' English
|
| 146 |
+
rows use the 413 chunks it scored (411 or 412 where a public scorer also lacks one). The 512-token comparison uses
|
| 147 |
+
all 414.
|
| 148 |
+
- **12-language exam**: 352 chunks, one per document, all drawn at random from FineWeb-2 web text in 12 languages
|
| 149 |
+
(ar, bn, de, es, hi, ja, ko, ru, sw, th, vi, zh; about 30 each). It has no code or math: 351 of the 352 chunks are
|
| 150 |
+
plain text by the grader's label. The grader drops 46.
|
| 151 |
+
|
| 152 |
+
Every exam document was checked against the training data (exact text, URL, title, long-line and shingle matching).
|
| 153 |
+
No copies were found.
|
| 154 |
+
|
| 155 |
+
#### Near-duplicate check
|
| 156 |
+
|
| 157 |
+
The 495 held-out chunks were compared with every training and validation record. The check measures the share of a
|
| 158 |
+
chunk's distinctive 40-character pieces that a single record contains. No chunk is covered 80% or more by any record.
|
| 159 |
+
4 are covered 64% to 71%: three short files from code collections and one US government record, 98 to 345 tokens
|
| 160 |
+
each, all short templated texts. 9 in all are covered 20% or more. These chunks stay in the reported set. Without the
|
| 161 |
+
4, Source-1's rank agreement is 0.901 (491 chunks); without all 9, 0.900.
|
| 162 |
+
|
| 163 |
+
#### Safety filter
|
| 164 |
+
|
| 165 |
+
A safety filter fixed before any result was seen removed a small number of documents from every evaluation set and
|
| 166 |
+
from Source-1's training, validation and test data. Documents that substantially copy removed text were removed from
|
| 167 |
+
its data as well. Every model is compared on the same filtered chunks.
|
| 168 |
+
|
| 169 |
+
</details>
|
| 170 |
+
|
| 171 |
+
<details>
|
| 172 |
+
<summary>The grader in detail, and how much its grades move when repeated</summary>
|
| 173 |
+
|
| 174 |
+
### The grader in detail
|
| 175 |
+
|
| 176 |
+
Strictly, these are two graders from one proprietary model family: one graded the exams, another the held-out set.
|
| 177 |
+
This file calls either "the grader". That model family also includes one of the assistants that helped draft the
|
| 178 |
+
rubric text. The exams were graded with an older revision of the rubric, from before the rubric settled how to score
|
| 179 |
+
ads.
|
| 180 |
+
|
| 181 |
+
The grader returns the 13 fields only. Its reference overall score and keep flag are computed from its scores in the
|
| 182 |
+
same way as Source-1's: the overall formula of the rubric (see [README.md](README.md#what-it-returns)), and keep =
|
| 183 |
+
false when the rubric's hard filters match (`toxicity >= 4`, `spam_seo >= 4` or `boilerplate >= 4.5`). "The grader's
|
| 184 |
+
drops" in this file are those chunks: spam, boilerplate or toxic text under the hard filters.
|
| 185 |
+
|
| 186 |
+
The same sets were scored by the teacher and by the 16 public quality scorers, open-weight models run as described
|
| 187 |
+
under [How the public scorers were run](#how-the-public-scorers-were-run).
|
| 188 |
+
|
| 189 |
+
#### Grader self-agreement
|
| 190 |
+
|
| 191 |
+
The exam grader also graded part of each exam a second time. The two gradings agree at rank 0.904 on 97 English
|
| 192 |
+
random-sample documents and 0.945 on 58 documents of the 12-language exam. This is a rough noise ceiling: it shows how
|
| 193 |
+
much one grading moves when repeated. It is not a hard upper bound, because a model can agree with one grading more
|
| 194 |
+
closely than two noisy gradings agree with each other.
|
| 195 |
+
|
| 196 |
+
</details>
|
| 197 |
+
|
| 198 |
+
### The one-line header
|
| 199 |
+
|
| 200 |
+
<details>
|
| 201 |
+
<summary>The one-line header that Source-1, the teacher and the grader saw</summary>
|
| 202 |
+
|
| 203 |
+
|
| 204 |
+
Source-1, the teacher and the grader saw each chunk after a one-line header. It gives the chunk's source type
|
| 205 |
+
("dataset record" for 478 of the 495 held-out chunks, and a code-file type such as "Python source file" for the other
|
| 206 |
+
17), its part number when it is one part of a longer document and, for 121 of the 495 held-out chunks, a title. The
|
| 207 |
+
public scorers read the text alone. On the 1,261 chunks of the held-out set and both exams, dropping the whole header
|
| 208 |
+
moved Source-1's overall score by 0.03 on average (at most 0.655). Dropping a title moved it by 0.05 on average on the
|
| 209 |
+
chunks that had one. Adding a URL, which no training input had, moved it by up to 0.7 and did not improve its rank
|
| 210 |
+
agreement with the grader. So `source1.py` does not show a URL to the model by default.
|
| 211 |
+
|
| 212 |
+
</details>
|
| 213 |
+
|
| 214 |
+
<details>
|
| 215 |
+
<summary>Metric definitions: rank, AUC, matched keep rate, intervals</summary>
|
| 216 |
+
|
| 217 |
+
### Metric definitions
|
| 218 |
+
|
| 219 |
+
Computed per chunk:
|
| 220 |
+
|
| 221 |
+
- **Rank**: Spearman correlation between a model's overall score and the grader's overall score.
|
| 222 |
+
- **AUC**: how well a model's score separates the grader's drops from its keeps (area under the ROC curve; a low score
|
| 223 |
+
means drop; ties count one half).
|
| 224 |
+
- **Matched keep rate**: every model keeps the same share of chunks the grader keeps (87% on the held-out set), taking
|
| 225 |
+
its highest-scored chunks. **Keep agreement** is the share of chunks where the model's keep/drop matches the
|
| 226 |
+
grader's. **Drop recall** is the share of the grader's drops the model also drops. This puts scorers with different
|
| 227 |
+
scales on the same operating point. Chunks tied at the cut are kept fractionally (the expected value over every
|
| 228 |
+
order of the tied chunks), so a count of drops caught can be fractional; such counts are given as "about".
|
| 229 |
+
- For Source-1 and the teacher, AUC and the matched keep rate use the overall score before it is clipped to 0, so
|
| 230 |
+
heavily penalized chunks are not tied at zero.
|
| 231 |
+
- Intervals at a drop line are Wilson 95% intervals.
|
| 232 |
+
- **Differences between two models** (Source-1 minus the other, on the chunks both scored) come with paired bootstrap
|
| 233 |
+
95% intervals: documents resampled with replacement, 2,000 resamples, seed 0, the same resamples for both models,
|
| 234 |
+
percentile intervals. Interval ends are given to two decimals (three when they are close to zero), because the
|
| 235 |
+
third decimal moves with the random seed.
|
| 236 |
+
|
| 237 |
+
</details>
|
| 238 |
+
|
| 239 |
+
## Results
|
| 240 |
+
|
| 241 |
+
### Held-out set
|
| 242 |
+
|
| 243 |
+
The table shows Source-1, the teacher and the three multilingual public scorers that agree best with the grader. Keep
|
| 244 |
+
agreement and drop recall are read at a matched keep rate: each model keeps its top-scored 87% of chunks, the share
|
| 245 |
+
the grader keeps ([Metric definitions](#metric-definitions)).
|
| 246 |
+
|
| 247 |
+
| model | chunks | rank | AUC | keep agreement (87% kept) | drop recall (87% kept) |
|
| 248 |
+
|---|---|---|---|---|---|
|
| 249 |
+
| **Source-1** | 495 | 0.900 | 0.946 | 0.927 | 0.719 (46/64) |
|
| 250 |
+
| Teacher (open-weight 27B LLM) | 495 | 0.912 | 0.948 | 0.919 | 0.688 (44/64) |
|
| 251 |
+
| propella-1 4B | 493 | 0.756 | 0.862 | 0.886 | 0.562 (36/64) |
|
| 252 |
+
| propella-1 1.7B | 495 | 0.736 | 0.855 | 0.896 | 0.599 (about 38/64) |
|
| 253 |
+
| JQL-Edu (mean of 3 balanced heads) | 495 | 0.600 | 0.737 | 0.826 | 0.328 (21/64) |
|
| 254 |
+
|
| 255 |
+
- At this matched rate Source-1 catches 46 of the grader's 64 drops and the teacher 44, a difference within noise. At
|
| 256 |
+
each model's own drop line (for the teacher, its own keep flags) they catch 43 and 46
|
| 257 |
+
([The drop line](#the-drop-line)). Source-1 and the teacher are also level within noise on rank and AUC.
|
| 258 |
+
- The lead over propella-1 4B, the closest public scorer, is at least as large outside English: 0.916 against 0.765
|
| 259 |
+
on the 335 non-English chunks.
|
| 260 |
+
|
| 261 |
+
<details>
|
| 262 |
+
<summary>Held-out set in detail: every interval, and results by data stage and for Chinese</summary>
|
| 263 |
+
|
| 264 |
+
### Held-out set in detail
|
| 265 |
+
|
| 266 |
+
- Every public scorer ranks the held-out set well below Source-1 on this rubric. The closest, propella-1 4B, reaches
|
| 267 |
+
0.756 on the 493 chunks it scored (Source-1 minus propella-1 4B: +0.144, 95% interval +0.11 to +0.18). It is also
|
| 268 |
+
further from the grader on the drops: AUC 0.862 against Source-1's 0.946 (+0.084, +0.05 to +0.12). At the matched
|
| 269 |
+
rate it catches 36 of the 64 drops to Source-1's 46 (drop recall +0.156, +0.06 to +0.25). Most public scorers do
|
| 270 |
+
not target spam, boilerplate or toxicity, which is what the grader drops. With propella-1's own ratings for them
|
| 271 |
+
added, its AUC gap is +0.049 (+0.02 to +0.07; see [How propella-1 is read](#how-propella-1-is-read)).
|
| 272 |
+
- The lead over propella-1 4B is at least as large outside English: 0.916 against 0.765 on the 335 non-English chunks
|
| 273 |
+
in 52 languages (+0.151, +0.11 to +0.20; 34 of them are in languages propella-1's card does not list). It is +0.117
|
| 274 |
+
(+0.06 to +0.18) on the 158 English chunks it scored (of 159).
|
| 275 |
+
- Source-1 ranks 0.012 below its teacher (0.900 vs 0.912; 95% interval of the difference -0.03 to +0.004), also
|
| 276 |
+
outside English (-0.012, -0.03 to +0.006). It is level with it on AUC (0.946 vs 0.948; -0.02 to +0.01). At the
|
| 277 |
+
matched rate it catches 46 of the grader's 64 drops and the teacher 44, a difference within noise (drop recall
|
| 278 |
+
+0.031, -0.04 to +0.10).
|
| 279 |
+
|
| 280 |
+
By data stage, and for Chinese (same chunks; drops caught at each model's own drop line):
|
| 281 |
+
|
| 282 |
+
| model | web (129) | multilingual web (276) | conversations, code, synthetic (48) | open books (42) | Chinese (29) |
|
| 283 |
+
|---|---|---|---|---|---|
|
| 284 |
+
| **Source-1**, rank | 0.891 | 0.905 | 0.728 | 0.677 | 0.934 |
|
| 285 |
+
| Teacher, rank | 0.902 | 0.917 | 0.821 | 0.741 | 0.938 |
|
| 286 |
+
| **Source-1**, drops caught | 14 of 17 | 28 of 44 | 1 of 2 | 0 of 1 | 11 of 14 |
|
| 287 |
+
| Teacher, drops caught | 14 of 17 | 29 of 44 | 2 of 2 | 1 of 1 | 10 of 14 |
|
| 288 |
+
|
| 289 |
+
The held-out set has 159 English chunks and 1 to 12 chunks for most other languages (29 for Chinese), so
|
| 290 |
+
per-language results are noisy. Groups with fewer than 50 chunks are indicative only.
|
| 291 |
+
|
| 292 |
+
</details>
|
| 293 |
+
|
| 294 |
+
### Every public scorer
|
| 295 |
+
|
| 296 |
+
Source-1 ranks the chunks closer to the grader than each of the 16 public scorers, on every set each was run on. All
|
| 297 |
+
39 of these rank-agreement leads have 95% intervals clear of zero, also after a Bonferroni adjustment, while the
|
| 298 |
+
teacher's intervals all include zero. The smallest lead in size is +0.101 (+0.07 to +0.14), over propella-1 4B on the
|
| 299 |
+
English exam.
|
| 300 |
+
|
| 301 |
+
<details>
|
| 302 |
+
<summary>All 17 public-scorer rows with intervals, and how the 39 comparisons are counted</summary>
|
| 303 |
+
|
| 304 |
+
We compared 16 public scorers. The tables have 17 public rows, because the Nemotron-CC 3-way ensemble is computed
|
| 305 |
+
from three of the 16 (the two NeMo Curator classifiers and DCLM fastText). That gives 39 differences with intervals:
|
| 306 |
+
17 on the held-out set, 17 on the English exam and 5 on the 12-language exam, where only the five multilingual
|
| 307 |
+
scorers were run. Each cell is a rank agreement with the grader, and the lead is Source-1 minus that scorer on exactly
|
| 308 |
+
the chunks it scored. The held-out set has 495 chunks unless noted; the English exam 411 to 413; the 12-language exam
|
| 309 |
+
352 unless noted.
|
| 310 |
+
|
| 311 |
+
Multilingual scorers, and the teacher:
|
| 312 |
+
|
| 313 |
+
| model | held-out: rank | held-out: Source-1 lead (95% interval) | English exam: rank | English exam: lead | 12 languages: rank | 12 languages: lead |
|
| 314 |
+
|---|---|---|---|---|---|---|
|
| 315 |
+
| **Source-1** | 0.900 | - | 0.921 | - | 0.895 | - |
|
| 316 |
+
| Teacher (open-weight 27B LLM) | 0.912 | -0.012 (-0.03 to +0.004) | 0.912 | +0.009 (-0.01 to +0.02) | 0.880 | +0.015 (-0.007 to +0.04) |
|
| 317 |
+
| propella-1 4B | 0.756 (493 chunks) | +0.144 (+0.11 to +0.18) | 0.820 | +0.101 (+0.07 to +0.14) | 0.637 (350 chunks) | +0.258 (+0.19 to +0.33) |
|
| 318 |
+
| propella-1 1.7B | 0.736 | +0.163 (+0.13 to +0.20) | 0.812 | +0.108 (+0.07 to +0.14) | 0.631 | +0.264 (+0.20 to +0.33) |
|
| 319 |
+
| JQL-Edu | 0.600 | +0.299 (+0.25 to +0.36) | 0.541 | +0.380 (+0.30 to +0.46) | 0.379 | +0.516 (+0.42 to +0.61) |
|
| 320 |
+
| FinePDFs-Edu | 0.476 | +0.424 (+0.36 to +0.49) | 0.787 | +0.134 (+0.10 to +0.18) | 0.332 | +0.563 (+0.47 to +0.66) |
|
| 321 |
+
| FineWeb2-HQ (21 of the 53 languages) | 0.449 (342 chunks) | +0.456 (+0.38 to +0.54) | 0.448 | +0.473 (+0.39 to +0.56) | 0.496 (205 chunks) | +0.415 (+0.31 to +0.53) |
|
| 322 |
+
|
| 323 |
+
English-only scorers (held-out set: its 159 English chunks):
|
| 324 |
+
|
| 325 |
+
| model | held-out: rank | held-out: Source-1 lead (95% interval) | English exam: rank | English exam: lead |
|
| 326 |
+
|---|---|---|---|---|
|
| 327 |
+
| **Source-1** | 0.864 | - | 0.921 | - |
|
| 328 |
+
| FineWeb-Edu classifier | 0.529 | +0.336 (+0.22 to +0.45) | 0.453 | +0.468 (+0.38 to +0.56) |
|
| 329 |
+
| DCLM fastText (OH+ELI5) | 0.299 | +0.565 (+0.42 to +0.71) | 0.193 | +0.728 (+0.63 to +0.83) |
|
| 330 |
+
| NeMo Curator edu (Nemotron-4 labels) | 0.412 | +0.452 (+0.32 to +0.59) | 0.260 | +0.661 (+0.55 to +0.78) |
|
| 331 |
+
| NeMo Curator edu (Mixtral labels) | 0.498 | +0.366 (+0.24 to +0.50) | 0.422 | +0.499 (+0.41 to +0.59) |
|
| 332 |
+
| Meta-rater reasoning | 0.580 | +0.284 (+0.19 to +0.39) | 0.710 | +0.211 (+0.15 to +0.27) |
|
| 333 |
+
| Meta-rater readability | 0.574 | +0.291 (+0.19 to +0.40) | 0.619 | +0.302 (+0.24 to +0.37) |
|
| 334 |
+
| Meta-rater cleanliness | 0.597 | +0.267 (+0.18 to +0.37) | 0.698 | +0.223 (+0.17 to +0.28) |
|
| 335 |
+
| Meta-rater professionalism | 0.538 | +0.327 (+0.23 to +0.43) | 0.688 | +0.233 (+0.18 to +0.30) |
|
| 336 |
+
| EAI-Distill 0.5B | 0.592 | +0.273 (+0.18 to +0.37) | 0.685 | +0.236 (+0.19 to +0.29) |
|
| 337 |
+
| NVIDIA quality classifier (DeBERTa) | 0.326 | +0.539 (+0.39 to +0.70) | 0.181 | +0.740 (+0.59 to +0.90) |
|
| 338 |
+
| Dolma 3 fastText quality | 0.341 | +0.524 (+0.39 to +0.66) | 0.386 | +0.535 (+0.46 to +0.62) |
|
| 339 |
+
| Nemotron-CC 3-way ensemble (derived) | 0.353 | +0.511 (+0.37 to +0.67) | 0.295 | +0.626 (+0.53 to +0.72) |
|
| 340 |
+
|
| 341 |
+
- Every one of the 39 public-scorer intervals excludes zero, and the leads stay clear of zero after a Bonferroni
|
| 342 |
+
adjustment for the 39 comparisons. The teacher's intervals all include zero.
|
| 343 |
+
- This holds for rank agreement. The other measures are much noisier on the 159 English held-out chunks (see
|
| 344 |
+
[AUC and noise](#auc-and-noise)).
|
| 345 |
+
|
| 346 |
+
</details>
|
| 347 |
+
|
| 348 |
+
<details>
|
| 349 |
+
<summary>AUC of every scorer, Source-1 on each scorer's chunks, and the noise behind the tables</summary>
|
| 350 |
+
|
| 351 |
+
### AUC and noise
|
| 352 |
+
|
| 353 |
+
AUC against the grader's drops, each scorer on the chunks it scored:
|
| 354 |
+
|
| 355 |
+
| model | held-out: AUC | English exam: AUC | 12 languages: AUC |
|
| 356 |
+
|---|---|---|---|
|
| 357 |
+
| **Source-1** | 0.946 | 0.979 | 0.961 |
|
| 358 |
+
| Teacher (open-weight 27B LLM) | 0.948 | 0.965 | 0.945 |
|
| 359 |
+
| propella-1 4B | 0.862 | 0.924 | 0.836 |
|
| 360 |
+
| propella-1 1.7B | 0.855 | 0.941 | 0.846 |
|
| 361 |
+
| JQL-Edu | 0.737 | 0.789 | 0.645 |
|
| 362 |
+
| FinePDFs-Edu | 0.731 | 0.855 | 0.659 |
|
| 363 |
+
| FineWeb2-HQ | 0.780 | 0.795 | 0.768 |
|
| 364 |
+
| FineWeb-Edu classifier | 0.812 | 0.754 | - |
|
| 365 |
+
| DCLM fastText (OH+ELI5) | 0.741 | 0.576 | - |
|
| 366 |
+
| NeMo Curator edu (Nemotron-4 labels) | 0.706 | 0.645 | - |
|
| 367 |
+
| NeMo Curator edu (Mixtral labels) | 0.788 | 0.775 | - |
|
| 368 |
+
| Meta-rater reasoning | 0.763 | 0.793 | - |
|
| 369 |
+
| Meta-rater readability | 0.817 | 0.832 | - |
|
| 370 |
+
| Meta-rater cleanliness | 0.873 | 0.880 | - |
|
| 371 |
+
| Meta-rater professionalism | 0.735 | 0.777 | - |
|
| 372 |
+
| EAI-Distill 0.5B | 0.786 | 0.835 | - |
|
| 373 |
+
| NVIDIA quality classifier (DeBERTa) | 0.794 | 0.706 | - |
|
| 374 |
+
| Dolma 3 fastText quality | 0.752 | 0.710 | - |
|
| 375 |
+
| Nemotron-CC 3-way ensemble | 0.750 | 0.700 | - |
|
| 376 |
+
|
| 377 |
+
Source-1's own rank / AUC on each scorer's chunks:
|
| 378 |
+
|
| 379 |
+
- held-out set: 0.900 / 0.946 (on all 495 chunks and on propella-1 4B's 493), 0.864 / 0.941 on the 159 English chunks,
|
| 380 |
+
and 0.905 / 0.957 on FineWeb2-HQ's 342;
|
| 381 |
+
- English exam: 0.921 / 0.979 (on each set of 411 to 413 chunks);
|
| 382 |
+
- 12-language exam: 0.895 / 0.961 (on 352 chunks and on propella-1 4B's 350), and 0.910 / 0.962 on FineWeb2-HQ's 205.
|
| 383 |
+
|
| 384 |
+
Noise:
|
| 385 |
+
|
| 386 |
+
- None of the 2,000 resamples put any of the 39 differences at or below zero. Measured against its own noise, the
|
| 387 |
+
smallest lead is 5.3 bootstrap standard deviations above zero (Meta-rater cleanliness on the 159 English held-out
|
| 388 |
+
chunks). So the leads stay clear of zero after a Bonferroni adjustment for the 39 comparisons (normal
|
| 389 |
+
approximation).
|
| 390 |
+
- On the 159 English held-out chunks, with 15 grader drops, the other measures are much noisier. The keep-agreement
|
| 391 |
+
or drop-recall interval at the matched rate reaches zero for 7 of the 12 English-only rows. Two AUC leads are only
|
| 392 |
+
just clear of zero: over the FineWeb-Edu classifier (+0.129, +0.004 to +0.275) and over Meta-rater cleanliness
|
| 393 |
+
(+0.069, +0.008 to +0.133).
|
| 394 |
+
- The English-only scorers are compared on far fewer held-out chunks (159) than the multilingual ones. The exam sets
|
| 395 |
+
are not independent of the grader ([Independence from development](#independence-from-development)).
|
| 396 |
+
- All of this is agreement with Source-1's own rubric. The public scorers were built for their own definitions of
|
| 397 |
+
quality (most for educational value) and are not wrong when they disagree with it.
|
| 398 |
+
|
| 399 |
+
</details>
|
| 400 |
+
|
| 401 |
+
<details>
|
| 402 |
+
<summary>How each public scorer is read: inputs, main scores, EAI-Distill, the Nemotron-CC ensemble</summary>
|
| 403 |
+
|
| 404 |
+
### How each public scorer is read
|
| 405 |
+
|
| 406 |
+
Each public scorer is read through one main score, on the chunks it scored. The English-only scorers are read on
|
| 407 |
+
English chunks and FineWeb2-HQ on its 21 languages. propella-1, JQL-Edu and FinePDFs-Edu (with its fallback model for
|
| 408 |
+
five languages) are read on all 53, although propella-1's card does not list ten of them (az, fil, gu, kk, kn, ml, mr,
|
| 409 |
+
ms, ta, te) and JQL-Edu's backbone covers 52. Of the 16, 11 are English-only and one covers 21 of the 53 languages;
|
| 410 |
+
the other four were run on all 53. The public scorers read the text without Source-1's header.
|
| 411 |
+
|
| 412 |
+
| public scorer | languages it was run on | input it reads | main score used here |
|
| 413 |
+
|---|---|---|---|
|
| 414 |
+
| propella-1 4B | all 53 (its card lists 43 of them) | the whole chunk, up to 50,000 characters | a weighted mean of four of its quality ratings, defined by us (see [How propella-1 is read](#how-propella-1-is-read)) |
|
| 415 |
+
| propella-1 1.7B | all 53 (its card lists 43 of them) | the whole chunk, up to 50,000 characters | as propella-1 4B |
|
| 416 |
+
| JQL-Edu | all 53 (its backbone covers 52) | the first 8,192 tokens | the mean of its three balanced educational-value heads |
|
| 417 |
+
| FinePDFs-Edu | all 53 (a model per language; a fallback model for five) | about 2,000 tokens from the start, and from the end of long texts (the higher score counts) | its educational-value score |
|
| 418 |
+
| FineWeb2-HQ (the per-language classifiers in epfml/FineWeb-HQ-Classifiers; on English text, its FineWeb-HQ classifier) | 21 of the 53 | the first 512 tokens | its probability of high quality |
|
| 419 |
+
| FineWeb-Edu classifier | English | the first 512 tokens | its educational-value score |
|
| 420 |
+
| DCLM fastText (OH+ELI5) | English | the whole text | its probability of the high-quality label |
|
| 421 |
+
| NeMo Curator edu (Nemotron-4 labels) | English | the first 512 tokens | its educational-value score |
|
| 422 |
+
| NeMo Curator edu (Mixtral labels) | English | the first 512 tokens | its educational-value score |
|
| 423 |
+
| Meta-rater reasoning, readability, cleanliness, professionalism (four models) | English | the first 4,096 tokens | each model's expected rating |
|
| 424 |
+
| EAI-Distill 0.5B | English | the whole text up to 30,000 characters (beyond that, the start, a middle part and the end, as its card says) | a 0-5 score defined by us from four of its labels (see [How EAI-Distill and the Nemotron-CC ensemble are read](#how-eai-distill-and-the-nemotron-cc-ensemble-are-read)) |
|
| 425 |
+
| NVIDIA quality classifier (DeBERTa) | English | the first 1,024 tokens | its expected class (low 0, medium 1, high 2) |
|
| 426 |
+
| Dolma 3 fastText quality | English | the whole text | its probability of the high-quality label |
|
| 427 |
+
| Nemotron-CC 3-way ensemble (derived) | English | as its three parts (the two NeMo Curator classifiers and DCLM fastText) | the highest of the three parts' percentile buckets, with percentiles taken within each evaluation set |
|
| 428 |
+
|
| 429 |
+
#### How EAI-Distill and the Nemotron-CC ensemble are read
|
| 430 |
+
|
| 431 |
+
- **EAI-Distill**: reasoning depth and technical correctness are read as their position among the five ordered levels
|
| 432 |
+
(0 to 4) divided by 4. Extraction artifacts and missing content are 1 when the model reports any and 0 when it
|
| 433 |
+
reports none. Main score = 5 x mean(reasoning depth, technical correctness) - extraction artifacts - missing content,
|
| 434 |
+
clipped to 0-5. An indeterminate or abstaining answer is left out of the mean (no score when both are), and a missing
|
| 435 |
+
penalty counts 0. Its answer codes were read with the code tables of the Essential-AI/eai-taxonomy README at commit
|
| 436 |
+
`e8a934d5ca77a05f8daddc73466aedf8a9eb7a6c`.
|
| 437 |
+
- **Nemotron-CC 3-way ensemble**: each part's score becomes a bucket, floor(20 x its percentile rank within the
|
| 438 |
+
evaluation set) (0 to 19), and the ensemble's score is the highest of the three buckets.
|
| 439 |
+
|
| 440 |
+
</details>
|
| 441 |
+
|
| 442 |
+
<details>
|
| 443 |
+
<summary>Educational value alone: Source-1's educational_value field against every public scorer</summary>
|
| 444 |
+
|
| 445 |
+
### Educational value alone
|
| 446 |
+
|
| 447 |
+
Most public scorers were built to rate educational value. Read that way, Source-1's `educational_value` field ranks
|
| 448 |
+
the held-out set closer to the grader's `educational_value` (0.879) than every public scorer's main score does. All
|
| 449 |
+
17 intervals exclude zero, as they do on both exams. The closest is propella-1 4B's composite (0.823; Source-1 +0.056,
|
| 450 |
+
95% interval +0.03 to +0.08). Among the dedicated educational-value classifiers, JQL-Edu trails by +0.174 (+0.14 to
|
| 451 |
+
+0.22) and, on the 159 English chunks, the FineWeb-Edu classifier by +0.238 (+0.16 to +0.33). On the English exam,
|
| 452 |
+
Source-1's `educational_value` reaches 0.904 against 0.614 for the FineWeb-Edu classifier and 0.884 for propella-1 4B.
|
| 453 |
+
The margin over propella-1 4B is small there: +0.020 (+0.002 to +0.040). The grader's `educational_value` follows
|
| 454 |
+
Source-1's rubric anchors, not the annotation prompts these classifiers were trained on. propella-1's own
|
| 455 |
+
educational-value rating agrees less with it (0.776 on the held-out set) than its composite does.
|
| 456 |
+
|
| 457 |
+
</details>
|
| 458 |
+
|
| 459 |
+
### How propella-1 is read
|
| 460 |
+
|
| 461 |
+
<details>
|
| 462 |
+
<summary>The composite we read it through, a more generous reading, and propella-1's answers</summary>
|
| 463 |
+
|
| 464 |
+
propella-1 answers in words, not numbers. We map each of its ordered ratings to integers. We read it through a
|
| 465 |
+
weighted mean of four quality ratings (educational value, reasoning, content quality, information density), weighted
|
| 466 |
+
as in Source-1's quality formula. That leaves out its ratings for commercial bias, content ratio and integrity, and
|
| 467 |
+
content safety, which are close to the grader's drop rules. It was also run on all 53 languages, ten of which its card
|
| 468 |
+
does not list. As a check of a more generous reading, chosen after the results were known, we also restricted it to
|
| 469 |
+
its 43 listed languages and subtracted rubric-style penalties for those ratings. Each was rescaled to 0-5, then 0.5,
|
| 470 |
+
0.4 and 0.8 times the excess over 1 was subtracted for commercial bias, the larger of content ratio and integrity, and
|
| 471 |
+
content safety, as in Source-1's overall score. Held-out set:
|
| 472 |
+
|
| 473 |
+
| reading of propella-1 4B | chunks | propella-1 4B rank | Source-1 rank | Source-1 minus propella-1 4B: rank (95% interval) | AUC (95% interval) |
|
| 474 |
+
|---|---|---|---|---|---|
|
| 475 |
+
| main score, all 53 languages (the tables above) | 493 | 0.756 | 0.900 | +0.144 (+0.11 to +0.18) | +0.084 (+0.05 to +0.12) |
|
| 476 |
+
| main score, its 43 listed languages | 459 | 0.766 | 0.899 | +0.133 (+0.10 to +0.17) | +0.080 (+0.05 to +0.12) |
|
| 477 |
+
| with its own red-flag ratings as penalties, all 53 languages | 493 | 0.828 | 0.900 | +0.072 (+0.05 to +0.10) | +0.049 (+0.02 to +0.07) |
|
| 478 |
+
| with its own red-flag ratings as penalties, its 43 listed languages | 459 | 0.835 | 0.899 | +0.064 (+0.04 to +0.09) | +0.048 (+0.03 to +0.07) |
|
| 479 |
+
|
| 480 |
+
We tried four ways of weighting the penalties: the one above, boilerplate from content ratio alone, from the sum of
|
| 481 |
+
content ratio and integrity, and no rescaling. Across them, on all 53 languages and on its 43 listed ones, the
|
| 482 |
+
held-out rank gap ranges from 0.064 to 0.081. With the penalties as above, the exam gaps are +0.075 (+0.05 to +0.10)
|
| 483 |
+
in English and +0.155 (+0.10 to +0.21) in the 12 languages. The lead holds under every reading we tried, but it
|
| 484 |
+
roughly halves: the 0.144 in the main tables depends on reading propella-1 through its quality ratings alone.
|
| 485 |
+
|
| 486 |
+
The exact mapping: each rating word is mapped to its position in the rating's scale, from 0: educational value (none,
|
| 487 |
+
minimal, basic, moderate, high), reasoning (none, minimal, basic_reasoning, explanatory, analytical), content quality
|
| 488 |
+
(unacceptable, poor, adequate, good, excellent), information density (empty, thin, moderate, adequate, dense). Main
|
| 489 |
+
score = (0.30 x educational value + 0.20 x reasoning + 0.15 x content quality + 0.20 x information density) / 0.85, on
|
| 490 |
+
that 0-4 scale. The penalized reading also maps commercial bias (none, minimal, moderate, heavy, pure_marketing),
|
| 491 |
+
content ratio (complete_content, mostly_content, mixed_content, mostly_navigation, minimal_content) and content safety
|
| 492 |
+
(safe, mild_concerns, nsfw, harmful, illegal) to 0-4 and content integrity (complete, mostly_complete, fragment,
|
| 493 |
+
severely_degraded) to 0-3. It rescales each of them and the main score to 0-5, and subtracts 0.5 x max(0, commercial
|
| 494 |
+
bias - 1) + 0.4 x max(0, max(content ratio, content integrity) - 1) + 0.8 x max(0, content safety - 1), clipped to
|
| 495 |
+
0-5.
|
| 496 |
+
|
| 497 |
+
#### propella-1's answers
|
| 498 |
+
|
| 499 |
+
An answer that did not parse as JSON was run once more, with a limit of 1,536 new tokens instead of 512. On the
|
| 500 |
+
evaluation chunks, the 4B was retried on 1 held-out chunk, and the 1.7B on 4 held-out chunks and 1 English exam chunk.
|
| 501 |
+
After the retry every answer parsed, except that one 1.7B English exam answer: it ran to the token limit, and its
|
| 502 |
+
ratings were read from the raw text. Seven answers used an educational-value word outside its scale (the 4B: 2
|
| 503 |
+
held-out, 1 English exam and 2 12-language exam chunks; the 1.7B: 2 English exam chunks). Those chunks have no main
|
| 504 |
+
score. This is why propella-1 4B is compared on 493 of the 495 held-out chunks and 350 of the 352 12-language chunks.
|
| 505 |
+
|
| 506 |
+
</details>
|
| 507 |
+
|
| 508 |
+
### Exam sets
|
| 509 |
+
|
| 510 |
+
On both exams Source-1 and the teacher are level within noise (English +0.009, 95% interval -0.01 to +0.02;
|
| 511 |
+
12 languages +0.015, -0.007 to +0.04).
|
| 512 |
+
|
| 513 |
+
<details>
|
| 514 |
+
<summary>Exam sets in detail: rank, AUC, whole documents and the drop line per document</summary>
|
| 515 |
+
|
| 516 |
+
### Exam sets in detail
|
| 517 |
+
|
| 518 |
+
| model | English: rank (413 chunks) | English: AUC | English: whole documents (rank) | 12 languages: rank (352 chunks) | 12 languages: AUC |
|
| 519 |
+
|---|---|---|---|---|---|
|
| 520 |
+
| **Source-1** | 0.921 | 0.979 | 0.920 | 0.895 | 0.961 |
|
| 521 |
+
| Teacher | 0.912 | 0.965 | 0.901 | 0.880 | 0.945 |
|
| 522 |
+
|
| 523 |
+
Whole documents: the token-weighted aggregation over each document's chunks, on the English exam's random-sample
|
| 524 |
+
documents (196 for Source-1; 195 for the teacher, which has no score for one of them). Source-1 on the same 195 is also
|
| 525 |
+
0.920.
|
| 526 |
+
|
| 527 |
+
At the shipped drop line, counted per document (the teacher at its own keep flags):
|
| 528 |
+
|
| 529 |
+
| exam documents | grader drops | Source-1: drops caught | Source-1: keep agreement | teacher: drops caught | teacher: keep agreement |
|
| 530 |
+
|---|---|---|---|---|---|
|
| 531 |
+
| English, all 332 | 24 | 14 (0.583) | 96.7% | 14 (0.583) | 96.1% |
|
| 532 |
+
| English, the 196 random-sample documents | 13 | 9 (0.692) | 97.4% | 6 (0.462) | 94.9% |
|
| 533 |
+
| 12 languages, all 352 (all random) | 46 | 31 (0.674) | 94.3% | 28 (0.609) | 93.5% |
|
| 534 |
+
|
| 535 |
+
The exams were graded with an older revision of the rubric, before it settled how to score ads (`spam_seo` 3, kept).
|
| 536 |
+
So part of the gap is rubric drift that affects the teacher and Source-1 alike.
|
| 537 |
+
|
| 538 |
+
</details>
|
| 539 |
+
|
| 540 |
+
### More results
|
| 541 |
+
|
| 542 |
+
### The drop line
|
| 543 |
+
|
| 544 |
+
<details>
|
| 545 |
+
<summary>The drop line: how it was chosen, stricter lines, results on the held-out set</summary>
|
| 546 |
+
|
| 547 |
+
|
| 548 |
+
`keep` is false when the scores cross the drop line stored in `calibration.json`:
|
| 549 |
+
|
| 550 |
+
```
|
| 551 |
+
drop if toxicity >= 4 or spam_seo >= 3.5 or boilerplate >= 4.5
|
| 552 |
+
```
|
| 553 |
+
|
| 554 |
+
The rubric's own hard filters use `spam_seo >= 4`. Pages scored `spam_seo` 3, the level for ads and promotional pages,
|
| 555 |
+
are kept by both lines. The line was chosen on the teacher's labels for the validation split (9,744 chunks) by a fixed
|
| 556 |
+
rule: among five candidate lines, take the highest drop recall whose keep agreement with the teacher stays at or above
|
| 557 |
+
0.95. On that split it agrees with the teacher's keep flags on 96.4% of chunks, catches 77.1% of the teacher's drops
|
| 558 |
+
(803 of 1,042) and drops 9.4% of chunks. On the held-out test split (9,553 chunks, not used for the choice) it agrees
|
| 559 |
+
on 96.7% and catches 80.2% (840 of 1,047). The candidate lines and the 0.95 floor were set with help from grades by
|
| 560 |
+
the grader's model family ([Independence from development](#independence-from-development)).
|
| 561 |
+
|
| 562 |
+
If you need to catch more low-quality text and can afford to lose more good text, pass a stricter line as
|
| 563 |
+
`drop_line`. The candidates trade keep agreement for recall (validation split, against the teacher's keep flags):
|
| 564 |
+
|
| 565 |
+
| drop line | keep agreement | teacher drops caught | wrong drops | share dropped |
|
| 566 |
+
|---|---|---|---|---|
|
| 567 |
+
| `toxicity >= 4 or spam_seo >= 4 or boilerplate >= 4.5` (rubric default) | 0.960 | 0.711 (741/1,042) | 89 | 8.5% |
|
| 568 |
+
| **`toxicity >= 4 or spam_seo >= 3.5 or boilerplate >= 4.5` (shipped)** | **0.964** | **0.771 (803/1,042)** | **114** | **9.4%** |
|
| 569 |
+
| `toxicity >= 4 or spam_seo >= 3 or boilerplate >= 4.5` | 0.929 | 0.830 (865/1,042) | 515 | 14.2% |
|
| 570 |
+
| `toxicity >= 4 or spam_seo >= 3.5 or boilerplate >= 4` | 0.945 | 0.880 (917/1,042) | 408 | 13.6% |
|
| 571 |
+
| `toxicity >= 4 or spam_seo >= 2.5 or boilerplate >= 4.5` | 0.856 | 0.872 (909/1,042) | 1,271 | 22.4% |
|
| 572 |
+
|
| 573 |
+
`calibration.json` also stores one offset per quality score (mean teacher label minus mean model score on the
|
| 574 |
+
validation split). They are small, between -0.016 and -0.003 points, and `source1.py` leaves them off unless you pass
|
| 575 |
+
`apply_offsets=True`. The results in this file use the scores without offsets, as `source1.py` returns them by
|
| 576 |
+
default, except the per-field bias table, which says where it applies them.
|
| 577 |
+
|
| 578 |
+
On the held-out set, against the grader:
|
| 579 |
+
|
| 580 |
+
| model | line | keep agreement | drop recall (caught / grader drops) | wrong drops | share dropped |
|
| 581 |
+
|---|---|---|---|---|---|
|
| 582 |
+
| **Source-1** | shipped line | 0.941 (0.917 to 0.959) | 0.672 (43/64; 0.550 to 0.774) | 8 | 10.3% |
|
| 583 |
+
| Source-1 | rubric default line (`spam_seo >= 4`) | 0.931 | 0.594 (38/64) | 8 | 9.3% |
|
| 584 |
+
| Teacher | its own keep flags | 0.943 | 0.719 (46/64) | 10 | 11.3% |
|
| 585 |
+
|
| 586 |
+
17 of Source-1's 21 misses at the shipped line are also missed by the teacher at its own keep flags, and 16 of the 21
|
| 587 |
+
are in multilingual web text. On the 29 Chinese chunks Source-1's line agrees with the grader on 89.7% and catches 11
|
| 588 |
+
of 14 drops. The candidate lines were drafted after the held-out set's grades had been seen once, so treat these
|
| 589 |
+
held-out drop-line numbers as slightly optimistic. The test-split numbers above do not have this problem.
|
| 590 |
+
|
| 591 |
+
</details>
|
| 592 |
+
|
| 593 |
+
<details>
|
| 594 |
+
<summary>Per-field agreement and bias on the exam sets, and the red flags on the held-out set</summary>
|
| 595 |
+
|
| 596 |
+
### Per-field agreement and bias
|
| 597 |
+
|
| 598 |
+
Agreement per field on the exam sets. For the five quality scores: quadratic-weighted kappa on levels rounded to the
|
| 599 |
+
nearest integer (English: the 200 random-sample chunks, 199 for the teacher; 12 languages: all 352 chunks). For the
|
| 600 |
+
labels: unweighted Cohen's kappa (English: all 414 chunks, 413 for the teacher; 12 languages: all 352 chunks). A dash
|
| 601 |
+
means the kappa is not meaningful. 351 of the 352 chunks in the 12-language exam are plain text by the grader's label,
|
| 602 |
+
so content type has almost no variation there (Source-1 matches the grader on 350 of them, and its kappa is about 0).
|
| 603 |
+
|
| 604 |
+
| field | English: Source-1 kappa | English: teacher kappa | 12 languages: Source-1 kappa | 12 languages: teacher kappa |
|
| 605 |
+
|---|---|---|---|---|
|
| 606 |
+
| educational_value | 0.831 | 0.816 | 0.788 | 0.803 |
|
| 607 |
+
| reasoning_depth | 0.798 | 0.781 | 0.719 | 0.730 |
|
| 608 |
+
| writing_quality | 0.769 | 0.811 | 0.785 | 0.786 |
|
| 609 |
+
| information_density | 0.861 | 0.828 | 0.812 | 0.800 |
|
| 610 |
+
| reliability | 0.739 | 0.790 | 0.735 | 0.743 |
|
| 611 |
+
| format | 0.748 | 0.742 | 0.745 | 0.745 |
|
| 612 |
+
| topic | 0.776 | 0.794 | 0.786 | 0.808 |
|
| 613 |
+
| content_type | 0.833 | 0.864 | - | - |
|
| 614 |
+
|
| 615 |
+
Bias is the mean of model minus grader on rounded levels, on the same chunks as the quality kappas. Source-1's values
|
| 616 |
+
here include its optional calibration offsets (`apply_offsets=True`). Without them, as `source1.py` returns scores by
|
| 617 |
+
default, they differ by at most 0.014 (English: `reasoning_depth` +0.390, `writing_quality` +0.325, `reliability`
|
| 618 |
+
+0.270). The teacher has no offsets.
|
| 619 |
+
|
| 620 |
+
| field | English: Source-1 bias | English: teacher bias | 12 languages: Source-1 bias | 12 languages: teacher bias |
|
| 621 |
+
|---|---|---|---|---|
|
| 622 |
+
| educational_value | +0.280 | +0.307 | +0.111 | +0.142 |
|
| 623 |
+
| reasoning_depth | +0.385 | +0.422 | +0.301 | +0.321 |
|
| 624 |
+
| writing_quality | +0.320 | +0.281 | +0.131 | +0.153 |
|
| 625 |
+
| information_density | -0.105 | -0.085 | -0.196 | -0.159 |
|
| 626 |
+
| reliability | +0.260 | +0.261 | +0.196 | +0.196 |
|
| 627 |
+
|
| 628 |
+
On the held-out set (quadratic-weighted kappa on rounded levels, all 495 chunks), the red flags reach 0.87 for
|
| 629 |
+
`spam_seo`, 0.81 for `boilerplate` and 0.70 for `toxicity` (the teacher: 0.88, 0.81 and 0.73). The gated scores can
|
| 630 |
+
only be compared on the few chunks where both the grader and the model give them: `math_quality` 0.64 on 8 chunks (the
|
| 631 |
+
teacher 0.90) and `code_quality` 0.67 on 23 (the teacher 0.82 on 22). These numbers are very noisy.
|
| 632 |
+
|
| 633 |
+
</details>
|
| 634 |
+
|
| 635 |
+
### Reading the whole chunk
|
| 636 |
+
|
| 637 |
+
<details>
|
| 638 |
+
<summary>Reading the whole chunk: what the text past 512 tokens adds</summary>
|
| 639 |
+
|
| 640 |
+
|
| 641 |
+
The same model was run with every input cut to its first 512 tokens at scoring time. It was trained on full chunks,
|
| 642 |
+
so this measures what the text past 512 tokens adds, not how a model trained for 512 tokens would do. On the held-out
|
| 643 |
+
set, 197 of the 495 chunks fit in 512 tokens and score identically both ways.
|
| 644 |
+
|
| 645 |
+
| chunks | full chunk: rank | first 512 tokens: rank | difference (95% interval) |
|
| 646 |
+
|---|---|---|---|
|
| 647 |
+
| held-out set, all (495; 1,650 tokens on average) | 0.900 | 0.848 | +0.052 (+0.02 to +0.08) |
|
| 648 |
+
| held-out set, 513 to 2,048 tokens (192) | 0.902 | 0.881 | +0.021 (-0.01 to +0.05) |
|
| 649 |
+
| held-out set, over 2,048 tokens (106) | 0.875 | 0.753 | +0.122 (+0.05 to +0.21) |
|
| 650 |
+
| English exam (414) | 0.921 | 0.845 | +0.075 (+0.05 to +0.11) |
|
| 651 |
+
| 12-language exam (352) | 0.895 | 0.853 | +0.042 (+0.02 to +0.08) |
|
| 652 |
+
|
| 653 |
+
The gain comes from the longer chunks. Finding the grader's drops barely changes (held-out AUC 0.946 against 0.943;
|
| 654 |
+
+0.003, -0.01 to +0.01). Only 74 held-out chunks are longer than 4,096 tokens, so this does not test the far end of
|
| 655 |
+
the 8,192-token window.
|
| 656 |
+
|
| 657 |
+
The 512-token limit of some public scorers does not explain their lower agreement on the held-out set. Cut to 512
|
| 658 |
+
tokens, Source-1 still ranks it closer to the grader than propella-1 4B reading the whole chunk (0.848 against 0.756
|
| 659 |
+
on its 493 chunks; +0.092, +0.05 to +0.14). It also ranks closer than each of the four scorers that read 512 tokens
|
| 660 |
+
(leads +0.30 to +0.42, every interval clear of zero). On the English exam the cut model is only level with
|
| 661 |
+
propella-1 4B (+0.026, -0.01 to +0.07).
|
| 662 |
+
|
| 663 |
+
</details>
|
| 664 |
+
|
| 665 |
+
<details>
|
| 666 |
+
<summary>Agreement with the teacher on the test split: the job Source-1 was trained for</summary>
|
| 667 |
+
|
| 668 |
+
### Agreement with the teacher on the test split
|
| 669 |
+
|
| 670 |
+
How closely Source-1 reproduces the teacher's labels on 9,553 test chunks (8,473 documents, 53 languages) it never
|
| 671 |
+
trained on: the job it was trained for. The grader results above instead measure agreement with an independent LLM
|
| 672 |
+
grader applying the same rubric.
|
| 673 |
+
|
| 674 |
+
| chunks | n | overall score rank agreement | keep/drop agreement (rubric default line) | label accuracy | quality score error (MAE, 0-5 scale) |
|
| 675 |
+
|---|---|---|---|---|---|
|
| 676 |
+
| all | 9,553 | 0.953 (0.950 to 0.955) | 0.963 | 0.907 | 0.254 |
|
| 677 |
+
| web | 2,677 | 0.952 | 0.964 | 0.904 | 0.250 |
|
| 678 |
+
| multilingual web | 5,344 | 0.955 | 0.960 | 0.912 | 0.245 |
|
| 679 |
+
| conversations, code, synthetic | 950 | 0.895 | 0.964 | 0.890 | 0.325 |
|
| 680 |
+
| open books | 582 | 0.844 | 0.985 | 0.901 | 0.248 |
|
| 681 |
+
|
| 682 |
+
The keep/drop column applies the rubric's default hard filters to Source-1's scores. With the shipped drop line,
|
| 683 |
+
agreement on this split is 0.967 (see [The drop line](#the-drop-line)). The interval on the overall rank agreement is
|
| 684 |
+
a 95% bootstrap interval over the 8,473 documents (2,000 resamples, seed 0).
|
| 685 |
+
|
| 686 |
+
Label accuracy is 0.877 for format, 0.853 for topic and 0.991 for content type. The gated scores agree least with the
|
| 687 |
+
teacher: kappa 0.58 for `code_quality` (555 chunks where it applies) and 0.50 for `math_quality` (257 chunks).
|
| 688 |
+
Agreement with the teacher is lowest for Gujarati (0.788), Georgian (0.867), Croatian (0.887), Malayalam (0.893),
|
| 689 |
+
Bengali and Marathi (0.897) and Serbian (0.899).
|
| 690 |
+
|
| 691 |
+
</details>
|
| 692 |
+
|
| 693 |
+
<details>
|
| 694 |
+
<summary>Speed, and how much scores move with precision and batching</summary>
|
| 695 |
+
|
| 696 |
+
### Speed
|
| 697 |
+
|
| 698 |
+
Measured with `source1.py` on one RTX 3090 in bf16 with the default batches, excluding load time, on the 495 held-out
|
| 699 |
+
chunks and the 766 exam chunks. Repeated runs on the same GPU differed by up to about 10%. No speed comparison with
|
| 700 |
+
the public scorers or the teacher was made under the same conditions, so none is claimed.
|
| 701 |
+
|
| 702 |
+
| set | chunks | mean input tokens per chunk | chunks/s | input tokens/s | peak VRAM |
|
| 703 |
+
|---|---|---|---|---|---|
|
| 704 |
+
| held-out | 495 | 1,650 | 32.2 | 53,198 | 4.10 GB |
|
| 705 |
+
| exam | 766 | 1,766 | 31.0 | 54,644 | 4.21 GB |
|
| 706 |
+
|
| 707 |
+
Precision and batching: computing in bfloat16, both weight files give the same scores. On these 1,261 chunks, scoring
|
| 708 |
+
each chunk alone instead of in the default batches moved `overall` by up to 0.04 and a single field by up to 0.10 (3
|
| 709 |
+
labels changed, no keep decision). float32 differs from bfloat16 by a similar amount (up to 0.03 on `overall` and 0.09
|
| 710 |
+
on a single field; 4 labels and 1 keep decision changed).
|
| 711 |
+
|
| 712 |
+
</details>
|
| 713 |
+
|
| 714 |
+
## Limitations
|
| 715 |
+
|
| 716 |
+
- **Home ground.** It measures agreement with Source-1's rubric, as applied by graders from one proprietary model
|
| 717 |
+
family whose grades also steered development. It does not show how Source-1 does on text from other sources, or
|
| 718 |
+
that filtering with it trains better language models.
|
| 719 |
+
- **The exams helped choose the teacher**, so they are not independent of the grader.
|
| 720 |
+
- **It misses about a third of the grader's drops** at the shipped line: it catches 43 of 64 on the held-out set, the
|
| 721 |
+
teacher, at its own keep flags, 46.
|
| 722 |
+
- **Like its teacher, it rates some qualities higher than the grader does.** On the English exam, `educational_value`
|
| 723 |
+
is 0.28 levels above the grader on average.
|
| 724 |
+
- **Weaker on books and on conversations, code and synthetic text** (held-out rank 0.677 and 0.728, against about 0.90
|
| 725 |
+
for web text).
|
| 726 |
+
- **One chunk of up to 8,192 tokens at a time.** Nothing outside a chunk is visible to it.
|
| 727 |
+
- **It copies the teacher, biases included.** For example, it can score a thin affiliate page 3 (kept) where the
|
| 728 |
+
rubric says 4 (dropped).
|
| 729 |
+
- **The gated scores are the least reliable fields, and `toxicity` is the weakest red flag.**
|
| 730 |
+
- **Not a fact checker or a safety tool.**
|
| 731 |
+
- **Less data for some languages.** The 16 smallest have 1,249 to 1,470 training chunks each.
|
| 732 |
+
- **License screening has limits.** Notices the patterns miss, and opt-outs outside the text, were not caught.
|
| 733 |
+
- **No reproduction kit.** The numbers cannot be recomputed from this repository alone
|
| 734 |
+
([what is included](#reproducing-the-evaluation)).
|
| 735 |
+
|
| 736 |
+
### Limitations in detail
|
| 737 |
+
|
| 738 |
+
<details>
|
| 739 |
+
<summary>The full text of each limitation</summary>
|
| 740 |
+
|
| 741 |
+
- **Home-ground evaluation.** The benchmark measures agreement with Source-1's rubric as applied by independent LLM
|
| 742 |
+
graders from one proprietary model family (one graded the exams, another the held-out set). Grades from that family,
|
| 743 |
+
among other proprietary LLM graders, also steered the rubric revisions, the choice of the teacher and its prompt
|
| 744 |
+
setup, and the drop-line candidates and floor ([Independence from development](#independence-from-development)).
|
| 745 |
+
The held-out documents come from the same kinds of sources as the training data. The public scorers were built for
|
| 746 |
+
other definitions of quality, read the text without Source-1's header and are each read through one main score. A
|
| 747 |
+
more generous reading of propella-1 halves its gap to Source-1 ([How propella-1 is read](#how-propella-1-is-read)).
|
| 748 |
+
The comparison does not show how Source-1 does on text from other sources, or that filtering with Source-1 trains
|
| 749 |
+
better language models; neither has been tested.
|
| 750 |
+
- **The exams are not independent of the grader.** They are the samples on which the teacher and its prompt setup were
|
| 751 |
+
chosen against the exam grader's labels. The held-out set, which played no part in that choice, is the main result.
|
| 752 |
+
- **It misses about a third of the chunks the grader drops at the shipped line.** On the held-out set the shipped line
|
| 753 |
+
catches 67.2% of the grader's drops (43/64; interval 55.0% to 77.4%); the teacher catches 71.9% (46/64). 17 of
|
| 754 |
+
Source-1's 21 misses are also missed by the teacher, so most of what Source-1 misses its teacher misses too. Most
|
| 755 |
+
misses are in multilingual web text (28 of 44 caught there). If recall matters more than keeping good text, use a
|
| 756 |
+
stricter line ([The drop line](#the-drop-line)) or rank on `overall` and cut lower.
|
| 757 |
+
- **Like its teacher, it rates some qualities higher than the grader does.** On the English exam's 200 random-sample
|
| 758 |
+
chunks, Source-1's `educational_value` is on average 0.28 levels above the grader's (the teacher 0.31). Its
|
| 759 |
+
`reasoning_depth`, `writing_quality` and `reliability` are 0.26 to 0.39 levels above (rounded levels, with or without
|
| 760 |
+
the optional calibration offsets; the teacher: 0.26 to 0.42). For `writing_quality` its bias is larger than the
|
| 761 |
+
teacher's (+0.32 to +0.325 against +0.28). In the 12 languages they are smaller (`reasoning_depth` +0.30,
|
| 762 |
+
`reliability` +0.20, `writing_quality` +0.13 to +0.14, `educational_value` +0.11; the teacher +0.32, +0.20, +0.15
|
| 763 |
+
and +0.14). See [Per-field agreement and bias](#per-field-agreement-and-bias).
|
| 764 |
+
- **Weaker on books and on conversations, code and synthetic text.** Held-out rank is 0.677 for open books and 0.728
|
| 765 |
+
for conversations/code/synthetic (the teacher: 0.741 and 0.821; 42 and 48 chunks), against about 0.90 for web text.
|
| 766 |
+
Against the teacher on the test split it is 0.844 and 0.895. Book labels are nearly constant (long, formal, almost
|
| 767 |
+
always kept), which leaves little signal to learn from.
|
| 768 |
+
- **8,192 tokens per chunk.** Longer documents are split and each chunk is judged on its own. Nothing outside a chunk
|
| 769 |
+
is visible to it, except the "Part i of n" header. Text in scripts that need many tokens per character fills the
|
| 770 |
+
window sooner: in training, 10.5% of Bengali chunks, 7.2% of Georgian, 6.3% of Arabic and 5.0% of Korean chunks
|
| 771 |
+
were longer than 8,192 tokens and were truncated. When you score with `source1.py`, a document longer than the
|
| 772 |
+
window is split into chunks rather than cut, so all of its text is read (the `truncated` field reports the rare
|
| 773 |
+
chunk that still had to be cut). `max_chunks` trades that for speed by scoring only some evenly spaced chunks.
|
| 774 |
+
- **It copies the teacher, biases included.** The rubric puts ads, company pages and product pages at `spam_seo` 3,
|
| 775 |
+
which the shipped line keeps, and thin affiliate and doorway pages at 4, which it drops. The teacher does not always
|
| 776 |
+
follow the second rule and sometimes scores such pages 3, and Source-1 learned from those labels.
|
| 777 |
+
- **The gated scores are the least reliable fields, and toxicity is the weakest red flag.** Against the teacher on
|
| 778 |
+
the test split, kappa is 0.58 for `code_quality` and 0.50 for `math_quality`. Against the grader on the held-out set
|
| 779 |
+
they can be compared only on 8 and 22 to 23 chunks ([Per-field agreement and bias](#per-field-agreement-and-bias)).
|
| 780 |
+
`toxicity` reaches 0.70 against the grader (the teacher 0.73).
|
| 781 |
+
- **Not a fact checker or a safety tool.** `reliability` is a surface judgment of care and plausibility; the model does
|
| 782 |
+
not verify claims. Toxic text is rare in the training data. `toxicity` is meant as a data-filtering red flag, not a
|
| 783 |
+
moderation classifier.
|
| 784 |
+
- **Languages with little data.** The 16 languages with the fewest training chunks (az, et, fil, gu, ka, kk, kn, lv,
|
| 785 |
+
ml, mr, ms, sq, sw, ta, te, ur) have 1,249 to 1,470 each, and Kannada had no books. Agreement with the teacher on the
|
| 786 |
+
test split is lowest for Gujarati (0.788), Georgian (0.867), Croatian (0.887), Malayalam (0.893), Bengali and Marathi
|
| 787 |
+
(0.897) and Serbian (0.899). The held-out set has 1 to 12 chunks for most non-English languages, so per-language
|
| 788 |
+
results there are noisy.
|
| 789 |
+
- **License screening has limits.** Licenses come from each source's metadata. The training documents were also
|
| 790 |
+
screened by pattern matching on their own text: books for NonCommercial, NoDerivatives and all-rights-reserved
|
| 791 |
+
notices in their front and back matter, web pages for such terms and for the sites they come from, and every
|
| 792 |
+
document for text-and-data-mining and AI-training reservations. This screening did not catch notices worded in ways
|
| 793 |
+
the patterns miss, or reservations made outside the text itself (on the terms pages of sites the rules do not list,
|
| 794 |
+
or in machine-readable opt-out signals such as robots.txt). If you find such a document, tell us (see the contact
|
| 795 |
+
section of [README.md](README.md#contact-and-takedown)).
|
| 796 |
+
- **No reproduction kit.** See [Reproducing the evaluation](#reproducing-the-evaluation).
|
| 797 |
+
|
| 798 |
+
</details>
|
| 799 |
+
|
| 800 |
+
## Reproducing and appendices
|
| 801 |
+
|
| 802 |
+
<details>
|
| 803 |
+
<summary>Reproducing the evaluation: what this repository does and does not include</summary>
|
| 804 |
+
|
| 805 |
+
### Reproducing the evaluation
|
| 806 |
+
|
| 807 |
+
What this repository gives you:
|
| 808 |
+
|
| 809 |
+
- Source-1 itself: the weights, `source1.py` and `calibration.json`, which produce the Source-1 scores used here
|
| 810 |
+
(computed in bfloat16 on one RTX 3090 with the default batches; see [Speed](#speed) for how much scores move with
|
| 811 |
+
precision and batching).
|
| 812 |
+
- The metric definitions and bootstrap settings ([Metric definitions](#metric-definitions)).
|
| 813 |
+
- How each public scorer was run ([How the public scorers were run](#how-the-public-scorers-were-run)) and read
|
| 814 |
+
([How each public scorer is read](#how-each-public-scorer-is-read)), with the exact definitions of the scores we
|
| 815 |
+
defined ourselves ([How propella-1 is read](#how-propella-1-is-read),
|
| 816 |
+
[How EAI-Distill and the Nemotron-CC ensemble are read](#how-eai-distill-and-the-nemotron-cc-ensemble-are-read)).
|
| 817 |
+
- The Hugging Face repositories and revisions of the 16 public scorers
|
| 818 |
+
([Appendix C](#appendix-c-public-scorer-repositories)).
|
| 819 |
+
|
| 820 |
+
What it does not include: the ids and texts of the evaluation chunks, the grader's labels, any model's per-chunk
|
| 821 |
+
scores, the script that computes the metrics, and the teacher's prompt and decoding settings. The numbers in this file
|
| 822 |
+
cannot be recomputed from this repository alone.
|
| 823 |
+
|
| 824 |
+
</details>
|
| 825 |
+
|
| 826 |
+
### Appendix A: rubric anchors
|
| 827 |
+
|
| 828 |
+
<details>
|
| 829 |
+
<summary>Appendix A: rubric anchors, and the teacher's extra rules</summary>
|
| 830 |
+
|
| 831 |
+
|
| 832 |
+
| field | 0 | 1 | 2 | 3 | 4 | 5 |
|
| 833 |
+
|---|---|---|---|---|---|---|
|
| 834 |
+
| educational_value | Teaches nothing: spam, ads, navigation, gibberish | Almost nothing to learn: a few incidental facts in promotional, personal or trivial text | Some useful information, but superficial, fragmentary, or mixed with irrelevant material | Useful and coherent; real knowledge or skills, without much depth or completeness | Clearly educational; explains concepts or methods well enough to learn from, minor gaps | Outstanding teaching material, comparable to an excellent textbook or expert tutorial |
|
| 835 |
+
| reasoning_depth | No reasoning: fragments, lists, boilerplate | Bare assertions or opinions | Occasional explanation, mostly unsupported; steps skipped | Explains the why behind key points, some step-by-step structure | Consistent explicit reasoning: derivations, cause and effect, worked examples | Rigorous multi-step reasoning throughout: proofs, careful derivations, thorough analysis |
|
| 836 |
+
| writing_quality | Unreadable: garbled, broken encoding, keyword soup | Very poor: frequent errors, incoherent | Below average: understandable but disorganized or repetitive | Adequate: clear and coherent, minor issues | Good: well organized, fluent, precise | Excellent: publication quality |
|
| 837 |
+
| information_density | No real content | Mostly padding around a little content | Noticeable padding or digressions | Reasonable: mostly on point, some filler | Dense: most sentences carry information | Very dense yet readable |
|
| 838 |
+
| reliability | Fabricated, nonsensical or deceptive | Largely unreliable: many errors, sensational claims | Questionable: some errors or unsupported claims | Generally plausible and consistent, informal or unverifiable | Careful and accurate; shows its work or cites sources | Authoritative: expert-level accuracy, well sourced |
|
| 839 |
+
| spam_seo | None | Minor promotion: a call to action or a brief ad in otherwise genuine content | Noticeable promotion: repeated calls to action, affiliate links, marketing tone | Substantial: the text mainly exists to promote, sell or rank; visible keyword repetition | Mostly spam: keyword stuffing, clickbait, thin affiliate or doorway content | Pure spam: auto-generated SEO text, scams, keyword lists |
|
| 840 |
+
| boilerplate | None: all real content | A little: a stray header, footer or copyright line | Noticeable: roughly 10-25% navigation, cookie or legal text | Substantial: roughly 25-50% templates, menus, link lists | Mostly boilerplate: over half templated or navigational | Entirely boilerplate: auto-generated pages, link lists, cookie banners, error or index pages |
|
| 841 |
+
| toxicity | None | Mild: occasional profanity or rudeness | Moderate: insults, crude humor, mild sexual references | Significant: harassment, demeaning stereotypes, graphic violence, partly explicit | Severe: hate speech, targeted harassment, sexually explicit as the main content | Extreme: violent extremism, dehumanizing hate, incitement, sexual content involving minors |
|
| 842 |
+
| code_quality | Not usable code: garbled, minified, obfuscated | Very poor: likely non-functional fragments, no structure, or auto-generated boilerplate | Poor: may work but messy | Acceptable: readable, plausibly correct, minimal docs | Good: clean, idiomatic, documented | Excellent: exemplary, production quality, instructive |
|
| 843 |
+
| math_quality | Garbled math | Mostly wrong or incoherent | Some correct math, but errors or skipped steps | Generally correct, key steps shown | Correct, clean notation, complete steps | Rigorous and elegant, every step justified |
|
| 844 |
+
|
| 845 |
+
The rubric anchors and the head layout are in `source1.json`. The teacher's prompt also had a few special rules that
|
| 846 |
+
are not in `source1.json`, and Source-1 was trained on labels that follow them, as far as the teacher did:
|
| 847 |
+
|
| 848 |
+
- Pages whose main purpose is to promote or sell a business, product or service (company "about us" pages, product
|
| 849 |
+
and landing pages, shop listings, brochures) are ads: format `product_page` and `spam_seo` 3, even when cleanly
|
| 850 |
+
written. Self-promotional press releases stay `news` with `spam_seo` 3. Selling alone is never a reason for
|
| 851 |
+
`spam_seo` 4 or 5; those levels are for keyword-stuffed text, doorway or thin affiliate pages made to rank, and
|
| 852 |
+
scams. Independent reviews, comparisons and news about products are not ads.
|
| 853 |
+
- Pages stitched together from unrelated or scrambled text (often a keyword title over copied or shuffled
|
| 854 |
+
paragraphs) are `spam_seo` 4-5, with `reliability` and `writing_quality` 0-1.
|
| 855 |
+
- Tag, category, archive and search-result pages, feeds, link directories and other index pages that mostly list
|
| 856 |
+
other pages are `boilerplate` 4, and 5 when the list is all they contain; empty auto-generated stub pages are 4-5.
|
| 857 |
+
- Sexually explicit material as the main content is `toxicity` 4 in any language (5 if it involves minors).
|
| 858 |
+
- General rules: judge only the text shown (a part of a longer document is not penalized for starting or ending
|
| 859 |
+
mid-thought); ignore personal-data placeholders such as `<EMAIL>`; judge every language by its own standards;
|
| 860 |
+
poor machine translation lowers `writing_quality`, and machine-translated filler written to rank is spam.
|
| 861 |
+
|
| 862 |
+
</details>
|
| 863 |
+
|
| 864 |
+
### Appendix B: training in detail
|
| 865 |
+
|
| 866 |
+
<details>
|
| 867 |
+
<summary>Appendix B: training in detail (model, data, filtering, labels, recipe)</summary>
|
| 868 |
+
|
| 869 |
+
|
| 870 |
+
#### Model
|
| 871 |
+
|
| 872 |
+
- Backbone: [mmBERT-base](https://huggingface.co/jhu-clsp/mmBERT-base) (ModernBERT architecture, 22 layers, hidden
|
| 873 |
+
size 768, 8,192-token context; trained by its authors on 3T+ tokens across 1800+ languages). All backbone weights
|
| 874 |
+
were fine-tuned.
|
| 875 |
+
- Heads: 13 linear heads on the mean-pooled final hidden states (about 68k parameters): one softmax head per label
|
| 876 |
+
(10, 15 and 4 classes) and one six-level softmax head per 0-5 field.
|
| 877 |
+
- Total: 307M parameters. Trained with float32 weights in bfloat16 mixed precision; released in bfloat16 (default)
|
| 878 |
+
and float32.
|
| 879 |
+
|
| 880 |
+
#### Data
|
| 881 |
+
|
| 882 |
+
220,346 chunks were labeled. After license filtering, the safety filter and the held-out splits, and after setting
|
| 883 |
+
aside a reserve that was never trained on (about 1% of training documents, picked by a hash of the document id and
|
| 884 |
+
held back for label-quality checks, plus the training and validation documents reviewed during development; see
|
| 885 |
+
[Independence from development](#independence-from-development)), 172,895 chunks (350M tokens, from 151,281 documents)
|
| 886 |
+
were used for training, 9,744 for validation and 9,553 for testing. No label-quality result from the reserve is
|
| 887 |
+
reported here. Documents were split 90/5/5 by a hash of the document id, so no document spans two splits.
|
| 888 |
+
|
| 889 |
+
| stage | what it is | training chunks |
|
| 890 |
+
|---|---|---|
|
| 891 |
+
| Web | A stratified sample of filtered and unfiltered web text, PDFs, wikis, math pages, permissively licensed code, Common Pile sources and toxicity datasets, in 53 languages; low-quality pages included on purpose | 46,389 |
|
| 892 |
+
| Multilingual web | A larger sample of the same kinds of sources, weighted toward languages other than English | 96,778 |
|
| 893 |
+
| Conversations, code, synthetic | Chat and instruction data, permissively licensed code and commits, synthetic and machine-generated text, comments and other short or noisy text, and domain text (law, parliament proceedings, science articles, historical and OCR text) | 17,614 |
|
| 894 |
+
| Open books | Books recorded as openly licensed or public domain, from Project Gutenberg, HAL, Wikibooks, Wikisource, OpenStax, the World Bank, EU and FAO publications, DOAB, OAPEN and others, after removing books whose own text states stricter terms (see below) | 12,114 |
|
| 895 |
+
| **Total** | 53 languages; English is 38.9% of training chunks | **172,895** |
|
| 896 |
+
|
| 897 |
+
Languages: ar, az, bg, bn, ca, cs, da, de, el, en, es, et, fa, fi, fil, fr, gu, he, hi, hr, hu, id, it, ja, ka, kk,
|
| 898 |
+
kn, ko, lt, lv, ml, mr, ms, nl, no, pl, pt, ro, ru, sk, sl, sq, sr, sv, sw, ta, te, th, tr, uk, ur, vi, zh. Every
|
| 899 |
+
non-English language has 1,249 to 3,437 training chunks; the 16 languages with the fewest are listed under
|
| 900 |
+
[Limitations in detail](#limitations-in-detail).
|
| 901 |
+
|
| 902 |
+
To replace documents that the license filtering below removed, 18,113 of the training and validation chunks were
|
| 903 |
+
drawn by fixed sampling rules (no model chose documents) from the same kinds of open sources: FineWeb-2 (including its
|
| 904 |
+
removed-documents part), the FineWeb-Edu annotations, C4, HPLT, FinePDFs, FineMath, permissively licensed code, Common
|
| 905 |
+
Pile and Common Corpus documents, and open books. Every rule below was applied to them too.
|
| 906 |
+
|
| 907 |
+
Filtering before training (training and validation splits; the held-out test split keeps every document the license
|
| 908 |
+
filtering removed, for evaluation only):
|
| 909 |
+
|
| 910 |
+
- License filtering removed 21,576 chunks from 19,720 documents:
|
| 911 |
+
- sources whose terms do not clearly cover this use: DCLM-baseline (its dataset card states that it is intended
|
| 912 |
+
for research use), raw Common Crawl WET pages (no dataset license), Stack v2 Edu (its upstream terms are gated),
|
| 913 |
+
Common Pile's YouTube transcripts (licenses asserted by uploaders over broadcasts), two collections with
|
| 914 |
+
unreliable license metadata, a corpus of third-party social media posts and a toxicity corpus whose texts are
|
| 915 |
+
not covered by its stated license, and French public data under the Licence Ouverte;
|
| 916 |
+
- code: copyleft licenses (GPL, AGPL, LGPL, MPL, EPL), code outside a permissive allow-list (MIT, Apache-2.0, BSD,
|
| 917 |
+
ISC, CC0, Unlicense), and code files whose own header states copyleft, proprietary or NonCommercial terms;
|
| 918 |
+
- books whose own front or back matter states stricter terms than the open license their platform recorded: a
|
| 919 |
+
scan of all 4,170 book documents found 137 that state NonCommercial or NoDerivatives terms or reserve all rights
|
| 920 |
+
with no open grant, and a wider pass over the same pages found 28 that state such terms in other wordings, forbid
|
| 921 |
+
sale or contradict their license record; also books deposited in HAL whose own text states no open license (84)
|
| 922 |
+
and library books from the Norwegian Colossal Corpus published after 1955 (14);
|
| 923 |
+
- other documents whose own text carries such notices (NonCommercial, NoDerivatives or all-rights-reserved
|
| 924 |
+
statements, publishers' copyright notices, text reprinted with permission): 141; web pages under NonCommercial or
|
| 925 |
+
NoDerivatives terms, or from sites whose terms put all their content under such terms (302); pages from sites
|
| 926 |
+
that re-host other people's documents, homework, shadow-library, pirated-novel, lyrics and subtitle sites (529);
|
| 927 |
+
and a few smaller groups (pages offering software cracks, open-education pages with no stated license,
|
| 928 |
+
GFDL-only pages, papers marked closed-access);
|
| 929 |
+
- documents whose own text reserves text-and-data-mining or AI-training rights: a scan of all 192,905 input
|
| 930 |
+
documents found 16.
|
| 931 |
+
- Separately, a pre-specified safety filter removed documents from every split, and a rule fixed in advance also
|
| 932 |
+
removed documents that substantially copy text the safety filter removed (every split).
|
| 933 |
+
- Book pages that are mostly a table of contents were left out of training (99 chunks).
|
| 934 |
+
- Every exam document was checked against the training data (exact text, URL, title, long-line and shingle matching);
|
| 935 |
+
no copies were found.
|
| 936 |
+
|
| 937 |
+
#### Labels
|
| 938 |
+
|
| 939 |
+
- Every training label comes from one teacher: an open-weight 27B LLM scoring each chunk against the 13-field rubric,
|
| 940 |
+
self-hosted on our own and rented GPUs. No human labels were used.
|
| 941 |
+
- No output of a proprietary model was used as a label or a training target. The evaluation grader's outputs were
|
| 942 |
+
never trained on. What grades from proprietary LLMs did inform is listed under
|
| 943 |
+
[Independence from development](#independence-from-development).
|
| 944 |
+
- About 2% of training documents (3,387) come from public datasets of model-written text (synthetic textbooks, chat
|
| 945 |
+
logs, machine-generated-text detection sets, machine translations). They are there so the scorer learns to judge
|
| 946 |
+
such text; the teacher scored them like any other input.
|
| 947 |
+
|
| 948 |
+
#### Recipe
|
| 949 |
+
|
| 950 |
+
| setting | value |
|
| 951 |
+
|---|---|
|
| 952 |
+
| epochs | 2 (5,320 optimizer steps) |
|
| 953 |
+
| tokens per step | 131,072 (about 65 chunks) |
|
| 954 |
+
| max length | 8,192 tokens; 964 training chunks (0.6%) were longer and were truncated |
|
| 955 |
+
| optimizer | AdamW, betas 0.9 / 0.98, eps 1e-6, weight decay 0.01, gradient clipping 1.0 |
|
| 956 |
+
| learning rate | 5e-5 for the backbone, 10x for the heads; 5% warmup, cosine decay to 10% |
|
| 957 |
+
| other | mean pooling, dropout 0.1, bf16 mixed precision, runs of spaces and tabs collapsed to one space before tokenizing (newlines kept), natural language mix (no reweighting), seed 0 |
|
| 958 |
+
| checkpoint | the final step, which had the best validation overall Spearman against the teacher (0.955) |
|
| 959 |
+
| how the settings were chosen | learning rate and language mix: a half-epoch sweep (two learning rates, two language mixes) on a smaller training set labeled by the same teacher, read on validation agreement with the teacher; epochs: the same validation curves (a third epoch added little while validation loss rose) |
|
| 960 |
+
| hardware | one rented NVIDIA H100 NVL, about 1.9 hours |
|
| 961 |
+
|
| 962 |
+
</details>
|
| 963 |
+
|
| 964 |
+
<details>
|
| 965 |
+
<summary>Appendix C: Hugging Face repositories and revisions of the public scorers</summary>
|
| 966 |
+
|
| 967 |
+
### Appendix C: public scorer repositories
|
| 968 |
+
|
| 969 |
+
| public scorer | Hugging Face repository | revision |
|
| 970 |
+
|---|---|---|
|
| 971 |
+
| propella-1 4B | `ellamind/propella-1-4b` | `bf607e62b6afa3e0e8d71c4d08d1429d9a09c82f` |
|
| 972 |
+
| propella-1 1.7B | `ellamind/propella-1-1.7b` | `2cb58fd324fce70e1cb106df20bf4e1d79696021` |
|
| 973 |
+
| JQL-Edu | `JQL-AI/JQL-Edu-Heads` (heads) and the embedding model its card names as the backbone (Snowflake's arctic-embed-m, version 2.0) | `5cb4a2d26c7961950b0facd1d8a390374027b7e4` (heads) and `95c2741480856aa9666782eb4afe11959938017f` (backbone) |
|
| 974 |
+
| FinePDFs-Edu | `HuggingFaceFW/finepdfs_edu_classifier_<code>`, one model per language (list below) | per model |
|
| 975 |
+
| FineWeb2-HQ | `epfml/FineWeb-HQ-Classifiers` (heads) and `FacebookAI/xlm-roberta-base` (backbone) | `1940ba2308cf2b12e530690c1eef183985dfcf29` and `e73636d4f797dec63c3081bb6ed5c7b0bb3f2089` |
|
| 976 |
+
| FineWeb-Edu classifier | `HuggingFaceFW/fineweb-edu-classifier` | `284663cbb2dabf9bda30d8f8cc49601251ee1631` |
|
| 977 |
+
| DCLM fastText (OH+ELI5) | `mlfoundations/fasttext-oh-eli5` | `cd8b714a90f2dbcd3b02cf5fc972e5d7c7f4f107` |
|
| 978 |
+
| NeMo Curator edu (Nemotron-4 labels) | `nvidia/nemocurator-fineweb-nemotron-4-edu-classifier` | `842316292abe5bc78521758f5498d6a05adc0f8b` |
|
| 979 |
+
| NeMo Curator edu (Mixtral labels) | `nvidia/nemocurator-fineweb-mixtral-edu-classifier` | `768fe255b7e7fbe222014e84cc6576a565516523` |
|
| 980 |
+
| Meta-rater reasoning | `opendatalab/meta-rater-reasoning-rating` | `0072a9a83971eb4af6d689dfc64f8f203c45b398` |
|
| 981 |
+
| Meta-rater readability | `opendatalab/meta-rater-readability-rating` | `5bfbee1110869ddcbf23447354a7311374784952` |
|
| 982 |
+
| Meta-rater cleanliness | `opendatalab/meta-rater-cleanliness-rating` | `4403a9535d47cbc7cc99de26b25099335fe2d9b6` |
|
| 983 |
+
| Meta-rater professionalism | `opendatalab/meta-rater-professionalism-rating` | `fc91d4be35fc91de3c65654bb59655ec533a1f61` |
|
| 984 |
+
| EAI-Distill 0.5B | `EssentialAI/eai-distill-0.5b` | `39f51ea6e8f1e959961feea0403c69ecfcc8b342` |
|
| 985 |
+
| NVIDIA quality classifier (DeBERTa) | `nvidia/quality-classifier-deberta` | `401824e175e89d3243bc376dc4ba262516615d81` |
|
| 986 |
+
| Dolma 3 fastText quality | `allenai/dolma3-fasttext-quality-classifier` | `bb89085994fef638ca8dc2ca25169db328e314bb` |
|
| 987 |
+
|
| 988 |
+
FinePDFs-Edu models used on the evaluation sets (`unknown` is its fallback model, used for fil, kn, ml, sw and te):
|
| 989 |
+
|
| 990 |
+
| repository | revision |
|
| 991 |
+
|---|---|
|
| 992 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_als_Latn` | `8f538c2701074964af4941048ec66077b1b6ca1f` |
|
| 993 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_arb_Arab` | `78462a34a522fbda15ad583ffa1cd98781571749` |
|
| 994 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_azj_Latn` | `6860d1ada3da270bce499e82a8177bfc895a87b0` |
|
| 995 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_ben_Beng` | `ca2a231ad78dc1948926cc5aa497240d95eeab40` |
|
| 996 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_bul_Cyrl` | `a23563de023ccabecf4c1e0d2210fe3588e1c381` |
|
| 997 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_cat_Latn` | `95c70a102e3862dc8708fe7b9e6bde361ed0643a` |
|
| 998 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_ces_Latn` | `43c57ff228771a55c4f496a1a680a1a7942463e7` |
|
| 999 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_cmn_Hani` | `b1157788380a284bac35fa96fb19654219f4f9b8` |
|
| 1000 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_dan_Latn` | `c3746210c23a292dc10c74a338addadb80c11d3c` |
|
| 1001 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_deu_Latn` | `eb2176fc3386be57b525a99fdee295b0580d1307` |
|
| 1002 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_ekk_Latn` | `5b66ac177115e31f9b304c56408a18dbadde838c` |
|
| 1003 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_ell_Grek` | `248951027a7d5e7853969863ef9f7628d379271b` |
|
| 1004 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_fas_Arab` | `e3d91254e276f6fd6415c2aa19705441b064bf92` |
|
| 1005 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_fin_Latn` | `b6925f941773d6a0716d8b3da09fed3130af16d6` |
|
| 1006 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_fra_Latn` | `f5050a44f837329386ec89e9c8bb4380aa8765b4` |
|
| 1007 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_guj_Gujr` | `72c1af11acd9085893fb1a6ae83c33ca49eaedaf` |
|
| 1008 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_heb_Hebr` | `95d6c2d065d0f6943d607d5b7eb192ed7efb5bf3` |
|
| 1009 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_hin_Deva` | `dfae45d02aa92adf72842e156d78a107e4f8a82d` |
|
| 1010 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_hrv_Latn` | `ecd3cb19a72491f32f5ce628f1a3b8cf8b9c0be0` |
|
| 1011 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_hun_Latn` | `078f8963005ccb83d24a444c8a87b0cf72443e74` |
|
| 1012 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_ind_Latn` | `ee3e75ff7ddc4c20eea2bf524b6a77b1d224f786` |
|
| 1013 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_ita_Latn` | `b67b3258ab616e68f2c1b61167e2d9b0673e95b1` |
|
| 1014 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_jpn_Jpan` | `3478261183e28a6214b82b85bfa47adc2b4a503f` |
|
| 1015 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_kat_Geor` | `5bf4a56cfa9249479098ce4a250fabb008c1278e` |
|
| 1016 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_kaz_Cyrl` | `f88ddf006ec15795340263cdc1de07c4d8e1a7d6` |
|
| 1017 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_kor_Hang` | `2baea20aa8f6640bd61ed879ba528292335f34ba` |
|
| 1018 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_lit_Latn` | `d4b7281f2b258c2a8057e42c949ab7fbe196c242` |
|
| 1019 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_lvs_Latn` | `0e1693b8ea51c6dfe76f8116604fc29ccf2119ce` |
|
| 1020 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_mar_Deva` | `aae791cf53a4959b66ee67f26acc5479aa38d8e5` |
|
| 1021 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_nld_Latn` | `b307a64f31a3c409d56c450f4f928aad596818e9` |
|
| 1022 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_nob_Latn` | `34166e84a7fb08a905774c637b6ef2eb7b19cc1d` |
|
| 1023 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_pol_Latn` | `58ff15760fa995fb7bea2c33a0761af1f9ce66a9` |
|
| 1024 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_por_Latn` | `d11bd310217f4cdb27522eaadc36202d2df705d4` |
|
| 1025 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_ron_Latn` | `92abfcae91841b726cc3ab71c3122cbfc77eb7b9` |
|
| 1026 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_rus_Cyrl` | `23ba4c39b4565af85282c1f1d1bc8479fdaa482d` |
|
| 1027 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_slk_Latn` | `7c1f7ec820a2d3f6eed0ede492d0417973dedb03` |
|
| 1028 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_slv_Latn` | `8825bc094303a48239f3b0c40a023689ae5a6c14` |
|
| 1029 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_spa_Latn` | `60eb17b37f8ea80fff614b50424359e974a43736` |
|
| 1030 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_srp_Cyrl` | `393b63976a35e266b21b14d91aea89990b4cf8cc` |
|
| 1031 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_swe_Latn` | `3da33c8970f10076e9da02649f51b10d359b2200` |
|
| 1032 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_tam_Taml` | `68239bbdb85ab737aaed970d45d313af9f18f051` |
|
| 1033 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_tha_Thai` | `db02cb1431acfb6baa956e8e09379f20ffd95980` |
|
| 1034 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_tur_Latn` | `dcdccca95c802edf5f1454ad33620e7342261da2` |
|
| 1035 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_ukr_Cyrl` | `1353ea90e4f65f8b33dce0570402f8692c982768` |
|
| 1036 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_unknown` | `d61616d51ece5ce2159936d8ece8aa39a6ef68bc` |
|
| 1037 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_urd_Arab` | `c2019778c7413f5a299d2ab4978acfadcdbe2030` |
|
| 1038 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_v2_eng_Latn` | `90ddef285f67230389057c14b2f6bbfeb70d40ea` |
|
| 1039 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_vie_Latn` | `870370fb168cc1c76549938b13f9cff953def4b7` |
|
| 1040 |
+
| `HuggingFaceFW/finepdfs_edu_classifier_zsm_Latn` | `5ad2ae90901c74585f0f921ab84fac0a52e3cbbe` |
|
| 1041 |
+
|
| 1042 |
+
</details>
|
LICENSE
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
Apache License
|
| 3 |
+
Version 2.0, January 2004
|
| 4 |
+
http://www.apache.org/licenses/
|
| 5 |
+
|
| 6 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 7 |
+
|
| 8 |
+
1. Definitions.
|
| 9 |
+
|
| 10 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 11 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 12 |
+
|
| 13 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 14 |
+
the copyright owner that is granting the License.
|
| 15 |
+
|
| 16 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 17 |
+
other entities that control, are controlled by, or are under common
|
| 18 |
+
control with that entity. For the purposes of this definition,
|
| 19 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 20 |
+
direction or management of such entity, whether by contract or
|
| 21 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 22 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 23 |
+
|
| 24 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 25 |
+
exercising permissions granted by this License.
|
| 26 |
+
|
| 27 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 28 |
+
including but not limited to software source code, documentation
|
| 29 |
+
source, and configuration files.
|
| 30 |
+
|
| 31 |
+
"Object" form shall mean any form resulting from mechanical
|
| 32 |
+
transformation or translation of a Source form, including but
|
| 33 |
+
not limited to compiled object code, generated documentation,
|
| 34 |
+
and conversions to other media types.
|
| 35 |
+
|
| 36 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 37 |
+
Object form, made available under the License, as indicated by a
|
| 38 |
+
copyright notice that is included in or attached to the work
|
| 39 |
+
(an example is provided in the Appendix below).
|
| 40 |
+
|
| 41 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 42 |
+
form, that is based on (or derived from) the Work and for which the
|
| 43 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 44 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 45 |
+
of this License, Derivative Works shall not include works that remain
|
| 46 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 47 |
+
the Work and Derivative Works thereof.
|
| 48 |
+
|
| 49 |
+
"Contribution" shall mean any work of authorship, including
|
| 50 |
+
the original version of the Work and any modifications or additions
|
| 51 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 52 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 53 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 54 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 55 |
+
means any form of electronic, verbal, or written communication sent
|
| 56 |
+
to the Licensor or its representatives, including but not limited to
|
| 57 |
+
communication on electronic mailing lists, source code control systems,
|
| 58 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 59 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 60 |
+
excluding communication that is conspicuously marked or otherwise
|
| 61 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 62 |
+
|
| 63 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 64 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 65 |
+
subsequently incorporated within the Work.
|
| 66 |
+
|
| 67 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 68 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 69 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 70 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 71 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 72 |
+
Work and such Derivative Works in Source or Object form.
|
| 73 |
+
|
| 74 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 75 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 76 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 77 |
+
(except as stated in this section) patent license to make, have made,
|
| 78 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 79 |
+
where such license applies only to those patent claims licensable
|
| 80 |
+
by such Contributor that are necessarily infringed by their
|
| 81 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 82 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 83 |
+
institute patent litigation against any entity (including a
|
| 84 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 85 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 86 |
+
or contributory patent infringement, then any patent licenses
|
| 87 |
+
granted to You under this License for that Work shall terminate
|
| 88 |
+
as of the date such litigation is filed.
|
| 89 |
+
|
| 90 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 91 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 92 |
+
modifications, and in Source or Object form, provided that You
|
| 93 |
+
meet the following conditions:
|
| 94 |
+
|
| 95 |
+
(a) You must give any other recipients of the Work or
|
| 96 |
+
Derivative Works a copy of this License; and
|
| 97 |
+
|
| 98 |
+
(b) You must cause any modified files to carry prominent notices
|
| 99 |
+
stating that You changed the files; and
|
| 100 |
+
|
| 101 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 102 |
+
that You distribute, all copyright, patent, trademark, and
|
| 103 |
+
attribution notices from the Source form of the Work,
|
| 104 |
+
excluding those notices that do not pertain to any part of
|
| 105 |
+
the Derivative Works; and
|
| 106 |
+
|
| 107 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 108 |
+
distribution, then any Derivative Works that You distribute must
|
| 109 |
+
include a readable copy of the attribution notices contained
|
| 110 |
+
within such NOTICE file, excluding those notices that do not
|
| 111 |
+
pertain to any part of the Derivative Works, in at least one
|
| 112 |
+
of the following places: within a NOTICE text file distributed
|
| 113 |
+
as part of the Derivative Works; within the Source form or
|
| 114 |
+
documentation, if provided along with the Derivative Works; or,
|
| 115 |
+
within a display generated by the Derivative Works, if and
|
| 116 |
+
wherever such third-party notices normally appear. The contents
|
| 117 |
+
of the NOTICE file are for informational purposes only and
|
| 118 |
+
do not modify the License. You may add Your own attribution
|
| 119 |
+
notices within Derivative Works that You distribute, alongside
|
| 120 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 121 |
+
that such additional attribution notices cannot be construed
|
| 122 |
+
as modifying the License.
|
| 123 |
+
|
| 124 |
+
You may add Your own copyright statement to Your modifications and
|
| 125 |
+
may provide additional or different license terms and conditions
|
| 126 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 127 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 128 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 129 |
+
the conditions stated in this License.
|
| 130 |
+
|
| 131 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 132 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 133 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 134 |
+
this License, without any additional terms or conditions.
|
| 135 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 136 |
+
the terms of any separate license agreement you may have executed
|
| 137 |
+
with Licensor regarding such Contributions.
|
| 138 |
+
|
| 139 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 140 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 141 |
+
except as required for reasonable and customary use in describing the
|
| 142 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 143 |
+
|
| 144 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 145 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 146 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 147 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 148 |
+
implied, including, without limitation, any warranties or conditions
|
| 149 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 150 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 151 |
+
appropriateness of using or redistributing the Work and assume any
|
| 152 |
+
risks associated with Your exercise of permissions under this License.
|
| 153 |
+
|
| 154 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 155 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 156 |
+
unless required by applicable law (such as deliberate and grossly
|
| 157 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 158 |
+
liable to You for damages, including any direct, indirect, special,
|
| 159 |
+
incidental, or consequential damages of any character arising as a
|
| 160 |
+
result of this License or out of the use or inability to use the
|
| 161 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 162 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 163 |
+
other commercial damages or losses), even if such Contributor
|
| 164 |
+
has been advised of the possibility of such damages.
|
| 165 |
+
|
| 166 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 167 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 168 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 169 |
+
or other liability obligations and/or rights consistent with this
|
| 170 |
+
License. However, in accepting such obligations, You may act only
|
| 171 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 172 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 173 |
+
defend, and hold each Contributor harmless for any liability
|
| 174 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 175 |
+
of your accepting any such warranty or additional liability.
|
| 176 |
+
|
| 177 |
+
END OF TERMS AND CONDITIONS
|
| 178 |
+
|
| 179 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 180 |
+
|
| 181 |
+
To apply the Apache License to your work, attach the following
|
| 182 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 183 |
+
replaced with your own identifying information. (Don't include
|
| 184 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 185 |
+
comment syntax for the file format. We also recommend that a
|
| 186 |
+
file or class name and description of purpose be included on the
|
| 187 |
+
same "printed page" as the copyright notice for easier
|
| 188 |
+
identification within third-party archives.
|
| 189 |
+
|
| 190 |
+
Copyright [yyyy] [name of copyright owner]
|
| 191 |
+
|
| 192 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 193 |
+
you may not use this file except in compliance with the License.
|
| 194 |
+
You may obtain a copy of the License at
|
| 195 |
+
|
| 196 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 197 |
+
|
| 198 |
+
Unless required by applicable law or agreed to in writing, software
|
| 199 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 200 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 201 |
+
See the License for the specific language governing permissions and
|
| 202 |
+
limitations under the License.
|
NOTICE
ADDED
|
@@ -0,0 +1,801 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Source-1
|
| 2 |
+
Copyright 2026 The Source-1 Authors
|
| 3 |
+
|
| 4 |
+
Licensed under the Apache License, Version 2.0 (see the LICENSE file, or
|
| 5 |
+
https://www.apache.org/licenses/LICENSE-2.0). The Source-1 Authors are listed in
|
| 6 |
+
the AUTHORS file.
|
| 7 |
+
|
| 8 |
+
This NOTICE has four parts and one appendix:
|
| 9 |
+
1. Base model: mmBERT-base (MIT License)
|
| 10 |
+
2. How the training scores were made
|
| 11 |
+
3. Data credits
|
| 12 |
+
4. Contact and removal requests
|
| 13 |
+
Appendix A. World Bank works (CC BY 3.0 IGO)
|
| 14 |
+
|
| 15 |
+
|
| 16 |
+
================================================================================
|
| 17 |
+
1. BASE MODEL: mmBERT-base (MIT License)
|
| 18 |
+
================================================================================
|
| 19 |
+
|
| 20 |
+
Source-1 is fine-tuned from mmBERT-base (https://huggingface.co/jhu-clsp/mmBERT-base),
|
| 21 |
+
published by jhu-clsp (the mmBERT authors at Johns Hopkins University CLSP) under the
|
| 22 |
+
MIT License. mmBERT is described in Marone et al., "mmBERT: A Modern Multilingual
|
| 23 |
+
Encoder with Annealed Language Learning", arXiv:2509.06888. The upstream repository
|
| 24 |
+
publishes no copyright line, so none is reproduced here.
|
| 25 |
+
|
| 26 |
+
Changes made by the Source-1 Authors (2026-10-03): all encoder weights were further
|
| 27 |
+
trained, and scoring heads (heads.safetensors) were added. The encoder weights are
|
| 28 |
+
published in bfloat16 (model.safetensors, the default) and in float32
|
| 29 |
+
(model.fp32.safetensors). The tokenizer is mmBERT's.
|
| 30 |
+
|
| 31 |
+
mmBERT's tokenizer is based on the Gemma 2 tokenizer by Google. See the Gemma Terms of
|
| 32 |
+
Use: https://ai.google.dev/gemma/terms
|
| 33 |
+
|
| 34 |
+
MIT License (mmBERT-base):
|
| 35 |
+
|
| 36 |
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
| 37 |
+
of this software and associated documentation files (the "Software"), to deal
|
| 38 |
+
in the Software without restriction, including without limitation the rights
|
| 39 |
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
| 40 |
+
copies of the Software, and to permit persons to whom the Software is
|
| 41 |
+
furnished to do so, subject to the following conditions:
|
| 42 |
+
|
| 43 |
+
The above copyright notice and this permission notice shall be included in all
|
| 44 |
+
copies or substantial portions of the Software.
|
| 45 |
+
|
| 46 |
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
| 47 |
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
| 48 |
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
| 49 |
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
| 50 |
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
| 51 |
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
| 52 |
+
SOFTWARE.
|
| 53 |
+
|
| 54 |
+
|
| 55 |
+
================================================================================
|
| 56 |
+
2. HOW THE TRAINING SCORES WERE MADE
|
| 57 |
+
================================================================================
|
| 58 |
+
|
| 59 |
+
Source-1 was fine-tuned on scores produced by an open-weight 27B LLM teacher scoring a
|
| 60 |
+
13-field rubric (3 labels, 5 quality scores, 3 red flags and 2 gated scores). Every
|
| 61 |
+
training target is a teacher score; no human labels were used. The teacher's weights
|
| 62 |
+
are not part of this release.
|
| 63 |
+
|
| 64 |
+
Source-1 was evaluated with an independent proprietary LLM grader, used for evaluation
|
| 65 |
+
only: the grader's outputs were never used as training targets or training data, and
|
| 66 |
+
the shipped drop line was chosen by a fixed rule on the teacher's labels. Grades and
|
| 67 |
+
reviews by proprietary LLMs from the grader's model family did inform some design
|
| 68 |
+
choices (the rubric revision, the choice of teacher and its prompt on the exam sets,
|
| 69 |
+
the candidate drop lines and which collections were filtered out); see the
|
| 70 |
+
development note in README.md and "Independence from development" in EVALUATION.md.
|
| 71 |
+
|
| 72 |
+
|
| 73 |
+
================================================================================
|
| 74 |
+
3. DATA CREDITS
|
| 75 |
+
================================================================================
|
| 76 |
+
|
| 77 |
+
This part covers the documents in Source-1's training and validation splits: 159,829
|
| 78 |
+
documents (182,639 text chunks). The documents were inputs to be scored; their scores
|
| 79 |
+
came from the teacher described in part 2. No training text is distributed with
|
| 80 |
+
Source-1: the release contains model weights and code only.
|
| 81 |
+
|
| 82 |
+
The credits below follow what each license or publisher asks for. Every work listed
|
| 83 |
+
under a Creative Commons license was modified: it was used to train a classifier.
|
| 84 |
+
Credit for each individual work belongs to the authors and publishers named in it.
|
| 85 |
+
Naming a source does not suggest that its authors, publishers or licensors endorse
|
| 86 |
+
Source-1 or any use of it.
|
| 87 |
+
|
| 88 |
+
Each book is credited under the licence recorded for it by the platform it came from,
|
| 89 |
+
except three works whose own text states an IGO licence (see 3.8). Before training,
|
| 90 |
+
rules applied to each source's terms and to each document's own text kept 19,720
|
| 91 |
+
documents out of the training and validation splits. Source-1 was not trained on them,
|
| 92 |
+
and they are not credited here. They include: sources whose terms limit use to research
|
| 93 |
+
or that carry no dataset licence (DCLM-baseline, raw Common Crawl WET pages); Stack v2
|
| 94 |
+
Edu, whose upstream terms are gated; YouTube transcripts and a corpus of third-party
|
| 95 |
+
social media posts; French public data under the Licence Ouverte; books and other
|
| 96 |
+
documents whose own text states NonCommercial (NC) or NoDerivatives (ND) terms,
|
| 97 |
+
reserves all rights or forbids reproduction without permission, or reserves
|
| 98 |
+
text-and-data-mining or AI-training rights; books deposited in HAL whose own text
|
| 99 |
+
states no open licence; web pages under NC or ND terms, and pages on sites that re-host
|
| 100 |
+
other people's documents; and code whose licence or file header is copyleft,
|
| 101 |
+
proprietary or not on a permissive allow-list.
|
| 102 |
+
|
| 103 |
+
Every book that is not public domain or CC0 is also credited individually in the file
|
| 104 |
+
CREDITS_BOOKS.tsv, which is part of this notice: title, authors, language, licence and
|
| 105 |
+
source URL. Where a work's own text asks to be cited or attributed in a particular way,
|
| 106 |
+
the file gives that citation or attribution: FAO's "Required citation", OpenStax's
|
| 107 |
+
attribution request, the citations requested by Eurydice, JRC and other EU reports, and
|
| 108 |
+
those of some other books and reports (university presses, Frontiers ebooks, research and
|
| 109 |
+
project reports). The World Bank works are credited in Appendix A.
|
| 110 |
+
|
| 111 |
+
License links used below:
|
| 112 |
+
ODC-By 1.0 https://opendatacommons.org/licenses/by/1-0/
|
| 113 |
+
CC BY 4.0 https://creativecommons.org/licenses/by/4.0/
|
| 114 |
+
CC BY 3.0 https://creativecommons.org/licenses/by/3.0/
|
| 115 |
+
CC BY 2.5 https://creativecommons.org/licenses/by/2.5/
|
| 116 |
+
CC BY 2.0 https://creativecommons.org/licenses/by/2.0/
|
| 117 |
+
CC BY 3.0 IGO https://creativecommons.org/licenses/by/3.0/igo/
|
| 118 |
+
CC BY-SA 3.0 IGO https://creativecommons.org/licenses/by-sa/3.0/igo/
|
| 119 |
+
CC BY-SA 4.0 https://creativecommons.org/licenses/by-sa/4.0/
|
| 120 |
+
CC BY-SA 3.0 https://creativecommons.org/licenses/by-sa/3.0/
|
| 121 |
+
CC BY-SA 2.5 https://creativecommons.org/licenses/by-sa/2.5/
|
| 122 |
+
CC0 1.0 https://creativecommons.org/publicdomain/zero/1.0/
|
| 123 |
+
Apache-2.0 https://www.apache.org/licenses/LICENSE-2.0
|
| 124 |
+
MIT https://opensource.org/license/mit
|
| 125 |
+
OPL v3.0 https://www.parliament.uk/site-information/copyright-parliament/open-parliament-licence/
|
| 126 |
+
Common Crawl Terms of Use
|
| 127 |
+
https://commoncrawl.org/terms-of-use
|
| 128 |
+
|
| 129 |
+
|
| 130 |
+
3.1 Open Data Commons Attribution License (ODC-By 1.0)
|
| 131 |
+
--------------------------------------------------------------------------------
|
| 132 |
+
|
| 133 |
+
Contains information from the following databases, which are made available under the
|
| 134 |
+
ODC Attribution License (https://opendatacommons.org/licenses/by/1-0/):
|
| 135 |
+
- FineWeb https://huggingface.co/datasets/HuggingFaceFW/fineweb
|
| 136 |
+
- FineWeb-2 (including its removed-documents part)
|
| 137 |
+
https://huggingface.co/datasets/HuggingFaceFW/fineweb-2
|
| 138 |
+
- FinePDFs https://huggingface.co/datasets/HuggingFaceFW/finepdfs
|
| 139 |
+
- FineWeb-Edu Llama 3 annotations
|
| 140 |
+
https://huggingface.co/datasets/HuggingFaceFW/fineweb-edu-llama3-annotations
|
| 141 |
+
- FineTranslations https://huggingface.co/datasets/HuggingFaceFW/finetranslations
|
| 142 |
+
- FineMath https://huggingface.co/datasets/HuggingFaceTB/finemath
|
| 143 |
+
- Cosmopedia v2 (SmolLM-Corpus)
|
| 144 |
+
https://huggingface.co/datasets/HuggingFaceTB/smollm-corpus
|
| 145 |
+
- C4 (en.noclean) https://huggingface.co/datasets/allenai/c4
|
| 146 |
+
- OpenWebMath https://huggingface.co/datasets/open-web-math/open-web-math
|
| 147 |
+
- WildChat-1M https://huggingface.co/datasets/allenai/WildChat-1M
|
| 148 |
+
|
| 149 |
+
|
| 150 |
+
3.2 Common Crawl
|
| 151 |
+
--------------------------------------------------------------------------------
|
| 152 |
+
|
| 153 |
+
Much of the web text comes from Common Crawl (https://commoncrawl.org): through
|
| 154 |
+
FineWeb, FineWeb-2, FinePDFs, FineMath, OpenWebMath, C4 and FineTranslations. Use was
|
| 155 |
+
subject to the Common Crawl Terms of Use (https://commoncrawl.org/terms-of-use).
|
| 156 |
+
The text of each page belongs to its owners.
|
| 157 |
+
|
| 158 |
+
|
| 159 |
+
3.3 Creative Commons Attribution (CC BY)
|
| 160 |
+
--------------------------------------------------------------------------------
|
| 161 |
+
|
| 162 |
+
Modified: used to train a classifier.
|
| 163 |
+
|
| 164 |
+
- EUR-Lex resources, CC BY 4.0 (see also 3.7).
|
| 165 |
+
https://huggingface.co/datasets/joelniklaus/eurlex_resources
|
| 166 |
+
- Sangraha, synthetic part (AI4Bharat), CC BY 4.0. Machine translations of English
|
| 167 |
+
Wikipedia text (Wikipedia contributors, CC BY-SA).
|
| 168 |
+
https://huggingface.co/datasets/ai4bharat/sangraha
|
| 169 |
+
- PubMed Central Open Access Subset (U.S. National Library of Medicine), articles under
|
| 170 |
+
CC BY, credited to the authors named in each article. https://pmc.ncbi.nlm.nih.gov
|
| 171 |
+
- Wikinews (Wikinews contributors), CC BY 4.0; the German edition CC BY 2.5. The Arabic,
|
| 172 |
+
Persian, French and Swedish editions are listed under 3.4. https://www.wikinews.org
|
| 173 |
+
- Common Corpus (PleIAs), documents marked CC-By: open-access articles from OpenAlex and
|
| 174 |
+
arXiv, EUR-Lex and Eurovoc documents of the European Union (see 3.7), and German
|
| 175 |
+
political speeches, each under the license stated by its source.
|
| 176 |
+
https://huggingface.co/datasets/PleIAs/common_corpus
|
| 177 |
+
- Common Pile v0.1, documents under CC BY 2.5 / 3.0 / 4.0 (see 3.5).
|
| 178 |
+
- Open-access books, each credited to the authors and publishers named in it, under
|
| 179 |
+
the licence the platform records for it:
|
| 180 |
+
Language Science Press (CC BY 4.0) https://langsci-press.org
|
| 181 |
+
OpenStax (Rice University), editions released under CC BY 4.0, credited as each
|
| 182 |
+
book asks, in most of them to "OpenStax and its content contributors" (the
|
| 183 |
+
wording for each book: CREDITS_BOOKS.tsv) https://openstax.org
|
| 184 |
+
OAPEN Library (CC BY 2.0 / 3.0 / 4.0, as stated for each book)
|
| 185 |
+
https://library.oapen.org
|
| 186 |
+
Directory of Open Access Books (CC BY 4.0) https://directory.doabooks.org
|
| 187 |
+
HAL open archive (CC BY 4.0) https://hal.science
|
| 188 |
+
Arabic E-Book Corpus, Hindawi Foundation, books under CC BY 4.0
|
| 189 |
+
https://www.hindawi.org
|
| 190 |
+
EU Publications Office books: see 3.7. FAO books and the works under IGO licences:
|
| 191 |
+
see 3.8. Per-work credits for all of these books: CREDITS_BOOKS.tsv.
|
| 192 |
+
|
| 193 |
+
|
| 194 |
+
3.4 Creative Commons Attribution-ShareAlike (CC BY-SA)
|
| 195 |
+
--------------------------------------------------------------------------------
|
| 196 |
+
|
| 197 |
+
Modified: used to train a classifier.
|
| 198 |
+
|
| 199 |
+
- Wikipedia (Wikipedia contributors), CC BY-SA 4.0 and CC BY-SA 3.0, also available
|
| 200 |
+
under the GNU Free Documentation License. With thanks to Wikipedia's volunteer editors.
|
| 201 |
+
Through https://huggingface.co/datasets/HuggingFaceFW/finewiki (August 2025 dumps) and
|
| 202 |
+
https://huggingface.co/datasets/wikimedia/wikipedia (English).
|
| 203 |
+
- Wikipedia talk-page comments (Wikipedia contributors), CC BY-SA 3.0, through
|
| 204 |
+
https://huggingface.co/datasets/OxAISH-AL-LLM/wiki_toxic
|
| 205 |
+
- Wikisource (Wikisource contributors), CC BY-SA 4.0; the dump used is published under
|
| 206 |
+
CC BY-SA 3.0 and the GFDL. The works themselves are public domain or freely licensed.
|
| 207 |
+
https://wikisource.org and https://huggingface.co/datasets/wikimedia/wikisource
|
| 208 |
+
- Wikibooks (Wikibooks contributors), CC BY-SA 4.0 (older revisions CC BY-SA 3.0).
|
| 209 |
+
https://www.wikibooks.org
|
| 210 |
+
- Wikinews, Arabic, Persian, French and Swedish editions (Wikinews contributors),
|
| 211 |
+
CC BY-SA 4.0. https://www.wikinews.org
|
| 212 |
+
- databricks-dolly-15k (Databricks), CC BY-SA 3.0.
|
| 213 |
+
https://huggingface.co/datasets/databricks/databricks-dolly-15k
|
| 214 |
+
- HC3, wiki_csai subset (Hello-SimpleAI), CC BY-SA 4.0.
|
| 215 |
+
https://huggingface.co/datasets/Hello-SimpleAI/HC3
|
| 216 |
+
- Stack Exchange (Stack Exchange contributors), CC BY-SA 2.5 / 3.0 / 4.0, through
|
| 217 |
+
Common Pile v0.1 (see 3.5).
|
| 218 |
+
- Kanripo (Kanseki Repository) transcriptions of premodern Chinese texts, CC BY-SA 4.0.
|
| 219 |
+
https://github.com/kanripo
|
| 220 |
+
- NDLA (Norwegian Digital Learning Arena) learning resources, CC BY-SA 4.0.
|
| 221 |
+
https://ndla.no
|
| 222 |
+
- Deutsches Textarchiv (DTA core corpus), CC BY-SA 4.0 as stated in each item's TEI
|
| 223 |
+
header. https://www.deutschestextarchiv.de
|
| 224 |
+
- Common Corpus (PleIAs), documents marked CC-By-SA.
|
| 225 |
+
https://huggingface.co/datasets/PleIAs/common_corpus
|
| 226 |
+
- Common Pile v0.1, documents under CC BY-SA 2.5 / 3.0 / 4.0 (see 3.5).
|
| 227 |
+
- Open-access books under CC BY-SA 3.0 / 4.0 from the OAPEN Library, the Directory of
|
| 228 |
+
Open Access Books and HAL, each credited to the authors and publishers named in it.
|
| 229 |
+
Per-work credits for these books and Wikibooks pages: CREDITS_BOOKS.tsv.
|
| 230 |
+
|
| 231 |
+
|
| 232 |
+
3.5 Common Pile v0.1
|
| 233 |
+
--------------------------------------------------------------------------------
|
| 234 |
+
|
| 235 |
+
Includes documents from the Common Pile v0.1 (Kandpal et al., 2025, "The Common Pile
|
| 236 |
+
v0.1: An 8TB Dataset of Public Domain and Openly Licensed Text";
|
| 237 |
+
https://huggingface.co/common-pile), each under the open license recorded in its
|
| 238 |
+
metadata (public domain, CC0, CC BY, CC BY-SA, the Open Parliament Licence, or a
|
| 239 |
+
permissive software license). Collections used: arXiv abstracts and papers,
|
| 240 |
+
Biodiversity Heritage Library, Caselaw Access Project, Data Provenance Initiative,
|
| 241 |
+
Directory of Open Access Books, Foodista, GitHub Archive, LibreTexts, Library of
|
| 242 |
+
Congress, News, OERCommons, peS2o, pre-1929 books, Pressbooks, Project Gutenberg,
|
| 243 |
+
Public Domain Review, Python Enhancement Proposals, Regulations.gov, Stack Exchange,
|
| 244 |
+
Ubuntu IRC, UK Hansard, USGPO, USPTO and WikiTeam wikis.
|
| 245 |
+
Modified (for the documents under a Creative Commons license): used to train a
|
| 246 |
+
classifier.
|
| 247 |
+
|
| 248 |
+
|
| 249 |
+
3.6 UK Parliament (Open Parliament Licence)
|
| 250 |
+
--------------------------------------------------------------------------------
|
| 251 |
+
|
| 252 |
+
Contains Parliamentary information licensed under the Open Parliament Licence v3.0.
|
| 253 |
+
(UK Hansard, through Common Pile v0.1.)
|
| 254 |
+
|
| 255 |
+
|
| 256 |
+
3.7 European Union
|
| 257 |
+
--------------------------------------------------------------------------------
|
| 258 |
+
|
| 259 |
+
- EUR-Lex. Legal documents from EUR-Lex (https://eur-lex.europa.eu),
|
| 260 |
+
© European Union, 1998-2026, re-used under the Commission's document reuse policy,
|
| 261 |
+
Decision 2011/833/EU. Obtained through EUR-Lex resources
|
| 262 |
+
(https://huggingface.co/datasets/joelniklaus/eurlex_resources), and through Common
|
| 263 |
+
Corpus (collections "Eurlex" and "Eurovoc", recorded there as CC-By;
|
| 264 |
+
https://huggingface.co/datasets/PleIAs/common_corpus).
|
| 265 |
+
- European Parliament. Plenary debates from the Europarl corpus v10
|
| 266 |
+
(https://www.statmt.org/europarl/). © European Union - Source: European Parliament
|
| 267 |
+
(https://www.europarl.europa.eu).
|
| 268 |
+
- Publications Office of the European Union. Books from https://op.europa.eu,
|
| 269 |
+
© European Union. The reuse policy of European Commission documents is implemented by
|
| 270 |
+
Commission Decision 2011/833/EU; these books are re-used under CC BY 4.0. Modified:
|
| 271 |
+
used to train a classifier. Where a book asks to be cited in a particular way
|
| 272 |
+
(Eurydice, JRC and some other reports), CREDITS_BOOKS.tsv gives that citation.
|
| 273 |
+
- VoxPopuli European Parliament speech transcripts (CC0 1.0), through Common Corpus.
|
| 274 |
+
|
| 275 |
+
|
| 276 |
+
3.8 International organizations
|
| 277 |
+
--------------------------------------------------------------------------------
|
| 278 |
+
|
| 279 |
+
World Bank (CC BY 3.0 IGO). Includes works of The World Bank, licensed under the Creative
|
| 280 |
+
Commons Attribution 3.0 IGO license (https://creativecommons.org/licenses/by/3.0/igo/).
|
| 281 |
+
Each work is credited in Appendix A as its copyright page asks. Modified: used to train a
|
| 282 |
+
classifier. This credit does not suggest that The World Bank endorses Source-1 or its
|
| 283 |
+
use. To the extent that Source-1 is regarded as an adaptation of these works: This is an
|
| 284 |
+
adaptation of an original work by The World Bank. Views and opinions expressed in the
|
| 285 |
+
adaptation are the sole responsibility of the author or authors of the adaptation and are
|
| 286 |
+
not endorsed by The World Bank. (One work in Appendix A is a joint OECD and World Bank
|
| 287 |
+
publication; its own adaptation notice is given there.)
|
| 288 |
+
|
| 289 |
+
Food and Agriculture Organization of the United Nations (FAO), CC BY 4.0. Includes books
|
| 290 |
+
from https://openknowledge.fao.org, copyright as stated in each book (mostly © FAO), under
|
| 291 |
+
the Creative Commons Attribution 4.0 International licence. Modified: used to train a
|
| 292 |
+
classifier. The citation FAO asks for ("Required citation") is given in CREDITS_BOOKS.tsv
|
| 293 |
+
for every FAO book whose text carries one (114 of the 115). There is no suggestion that FAO
|
| 294 |
+
endorses any specific organization, products or services, including Source-1. To the
|
| 295 |
+
extent that Source-1 is regarded as an adaptation of these works: "This adaptation was not
|
| 296 |
+
created by the Food and Agriculture Organization of the United Nations (FAO). FAO is not
|
| 297 |
+
responsible for the content or accuracy of this adaptation. The original editions shall be
|
| 298 |
+
the authoritative editions."
|
| 299 |
+
|
| 300 |
+
Inter-American Development Bank (CC BY 3.0 IGO). Includes one book published by the
|
| 301 |
+
Inter-American Development Bank (IDB) and deposited in HAL: Allen Blackman, Eduardo
|
| 302 |
+
Cavallo, Bridget Hoffmann and Adrien Vogt-Schilb (eds.), "Peril and Promise: Tackling
|
| 303 |
+
Climate Change in Latin America and the Caribbean", Inter-American Development Bank, 2025
|
| 304 |
+
(ISBN 978-1-59782-574-0, digital edition; https://shs.hal.science/halshs-04977766v1),
|
| 305 |
+
licensed under the Creative Commons Attribution 3.0 IGO license
|
| 306 |
+
(https://creativecommons.org/licenses/by/3.0/igo/legalcode), as its own text states (HAL
|
| 307 |
+
records CC BY 4.0). Also credited in CREDITS_BOOKS.tsv. Modified: used to train a
|
| 308 |
+
classifier. The IDB's name is used here only to credit the IDB, and no IDB logo is used.
|
| 309 |
+
This credit does not suggest that the IDB endorses Source-1 or its use.
|
| 310 |
+
|
| 311 |
+
UNESCO (CC BY-SA 3.0 IGO). Includes two books published by or with UNESCO and deposited
|
| 312 |
+
in HAL, credited in the form UNESCO's terms of use for its open access publications ask
|
| 313 |
+
for:
|
| 314 |
+
- Cléo Lossouarn et al. / Water, Megacities and Global Change / ISBN
|
| 315 |
+
978-92-3-100161-1 – licensed under CC BY-SA 3.0 IGO. (© UNESCO / ARCEAU IdF 2016;
|
| 316 |
+
https://enpc.hal.science/hal-01449109v1)
|
| 317 |
+
- David C. Andolfatto and Thomas Schrom (eds.) / The Restoration of Mangal Bahudvara
|
| 318 |
+
Caitya. A Tashi Gomang Stupa / ISBN 978-9937-9301-6-1 – licensed under CC BY-SA 3.0
|
| 319 |
+
IGO. (© UNESCO 2021; https://hal.sorbonne-universite.fr/hal-03900987v1)
|
| 320 |
+
Each book's own text states the Creative Commons Attribution-ShareAlike 3.0 IGO license
|
| 321 |
+
(https://creativecommons.org/licenses/by-sa/3.0/igo/); HAL records CC BY-SA 4.0. Also
|
| 322 |
+
credited in CREDITS_BOOKS.tsv, which lists the authors HAL records. Only the books' text
|
| 323 |
+
was used. Modified: used to train a classifier. No UNESCO logo is used, and this credit
|
| 324 |
+
does not suggest that UNESCO endorses Source-1 or its use. To the extent that Source-1 is
|
| 325 |
+
regarded as an adaptation of these works: Source-1 is not an official UNESCO publication
|
| 326 |
+
and shall not be considered as such.
|
| 327 |
+
|
| 328 |
+
|
| 329 |
+
3.9 Licence Ouverte / Open Licence 2.0 (Etalab)
|
| 330 |
+
--------------------------------------------------------------------------------
|
| 331 |
+
|
| 332 |
+
No document under the Licence Ouverte is in Source-1's training or validation splits:
|
| 333 |
+
the French public data in Common Corpus and the books deposited in HAL under this
|
| 334 |
+
licence were removed before training.
|
| 335 |
+
|
| 336 |
+
|
| 337 |
+
3.10 Other sources
|
| 338 |
+
--------------------------------------------------------------------------------
|
| 339 |
+
|
| 340 |
+
- HPLT v2.0, cleaned (https://huggingface.co/datasets/HPLT/HPLT2.0_cleaned). HPLT does
|
| 341 |
+
not own the text from which these data were extracted; it licenses the packaging
|
| 342 |
+
under CC0 1.0 and runs a notice-and-takedown policy.
|
| 343 |
+
- Apache-2.0: OpenAssistant oasst2 (https://huggingface.co/datasets/OpenAssistant/oasst2);
|
| 344 |
+
Aya Dataset (https://huggingface.co/datasets/CohereLabs/aya_dataset); Cosmopedia
|
| 345 |
+
(https://huggingface.co/datasets/HuggingFaceTB/cosmopedia); Meta Kaggle Code notebooks
|
| 346 |
+
by Kaggle, through https://huggingface.co/datasets/HuggingFaceTB/issues-kaggle-notebooks
|
| 347 |
+
- MIT: RAID (https://huggingface.co/datasets/liamdugan/raid); hh-rlhf (Bai et al., 2022,
|
| 348 |
+
"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human
|
| 349 |
+
Feedback", arXiv:2204.05862).
|
| 350 |
+
- Source code under permissive licenses (MIT, Apache-2.0, BSD-2-Clause, BSD-3-Clause,
|
| 351 |
+
ISC, CC0-1.0, Unlicense), as recorded for each file's repository, from
|
| 352 |
+
https://huggingface.co/datasets/codeparrot/github-code-clean,
|
| 353 |
+
https://huggingface.co/datasets/codeparrot/github-code,
|
| 354 |
+
https://huggingface.co/datasets/bigcode/commitpackft,
|
| 355 |
+
https://huggingface.co/datasets/bigcode/commitpack, and the Common Pile v0.1
|
| 356 |
+
GitHub Archive collection. Copyright in each file stays with its authors.
|
| 357 |
+
- CC0 1.0 and public domain, with thanks: Civil Comments
|
| 358 |
+
(https://huggingface.co/datasets/google/civil_comments); PleIAs Post-OCR-Correction
|
| 359 |
+
(https://huggingface.co/datasets/PleIAs/Post-OCR-Correction); Project Gutenberg
|
| 360 |
+
(https://www.gutenberg.org); the public-domain and CC0 collections of Common Pile v0.1
|
| 361 |
+
and Common Corpus (court decisions, laws and parliamentary papers from Brazil, China,
|
| 362 |
+
Denmark, Germany, Korea, Russia and the United States; Brazilian Chamber of Deputies
|
| 363 |
+
speeches; VoxPopuli; open-access articles); CC0 articles from PubMed Central; and
|
| 364 |
+
public-domain or CC0 books from Wikisource, the National Diet Library of Japan
|
| 365 |
+
(https://lab.ndl.go.jp), DBNL (https://dbnl.org), Project Ben-Yehuda
|
| 366 |
+
(https://benyehuda.org), the Hindawi Foundation, the Norwegian Colossal Corpus
|
| 367 |
+
(National Library of Norway, through NbAiLab), PleIAs Ukrainian cultural-heritage
|
| 368 |
+
books (Internet Archive) and HAL.
|
| 369 |
+
|
| 370 |
+
|
| 371 |
+
================================================================================
|
| 372 |
+
4. CONTACT AND REMOVAL REQUESTS
|
| 373 |
+
================================================================================
|
| 374 |
+
|
| 375 |
+
Questions about these credits, corrections, and removal or opt-out requests:
|
| 376 |
+
the Community tab of the model repository,
|
| 377 |
+
https://huggingface.co/msmth/Source-1/discussions.
|
| 378 |
+
|
| 379 |
+
|
| 380 |
+
================================================================================
|
| 381 |
+
APPENDIX A. WORLD BANK WORKS (CC BY 3.0 IGO)
|
| 382 |
+
================================================================================
|
| 383 |
+
|
| 384 |
+
Each entry is the attribution the work's copyright page asks for, followed by its record in
|
| 385 |
+
the World Bank Open Knowledge Repository. Modified: used to train a classifier. See 3.8 for
|
| 386 |
+
the adaptation notice and https://creativecommons.org/licenses/by/3.0/igo/ for the license.
|
| 387 |
+
|
| 388 |
+
A01. Alam, Muneeza Mehmood, and Lisa Bagnoli. 2024. Ten Thousand Steps in Her Shoes: The
|
| 389 |
+
Role of Public Transport in Women’s Economic Empowerment. Middle East and North
|
| 390 |
+
Africa Development Report Series. Washington, DC: World Bank.
|
| 391 |
+
doi:10.1596/978-1-46482091-5. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 392 |
+
https://openknowledge.worldbank.org/handle/10986/41786
|
| 393 |
+
|
| 394 |
+
A02. Alsharkas, Zeina, Jean Michel N. Marchat, Enrique Aldaz-Carroll, and Ana
|
| 395 |
+
Goicoechea. 2026. Competing in the Face of Climate Risks: Evidence from Firms and
|
| 396 |
+
Policy Priorities in MENAAP. Middle East, North Africa, Afghanistan, and Pakistan
|
| 397 |
+
Development Report. Washington, DC: World Bank. doi: 10.1596/978-1-4648-2335-0.
|
| 398 |
+
License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 399 |
+
https://openknowledge.worldbank.org/handle/10986/45042
|
| 400 |
+
|
| 401 |
+
A03. Arévalo-Sánchez, Inés, Janet Heisey, Sarang Chaudhary, Timothy Clay, Victoria
|
| 402 |
+
Strokova, Puja Vasudeva Dutta, and Colin Andrews. 2024. The State of Economic
|
| 403 |
+
Inclusion Report 2024: Pathways to Scale. Washington, DC: World Bank.
|
| 404 |
+
doi:10.1596/978-1-4648-2076-2. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 405 |
+
https://openknowledge.worldbank.org/handle/10986/42408
|
| 406 |
+
|
| 407 |
+
A04. Bachas, Pierre, Oyebola Okunogbe, Mahvish Shaukat, and Dario Tortarolo. 2026.
|
| 408 |
+
Raising Revenue Right: A Roadmap for Domestic Resource Mobilization. Policy
|
| 409 |
+
Research Report. World Bank. doi: 10.1596/978-1-4648-2034-2. License: Creative
|
| 410 |
+
Commons Attribution CC BY 3.0 IGO.
|
| 411 |
+
https://openknowledge.worldbank.org/handle/10986/45448
|
| 412 |
+
|
| 413 |
+
A05. Bandlien, Einar H., Sander De Kruijf, Eilen Arctander Vik, Ole Fredrik Ekern, Johan
|
| 414 |
+
Bernhard Siqueland Knudsen, Geoffrey Dyce, Silvana Tordo, and François Bertone.
|
| 415 |
+
2024. Water Management in Oil and Gas Operations: Industry Practice and Policy
|
| 416 |
+
Guidelines for Developing Countries. International Development in Focus.
|
| 417 |
+
Washington, DC: World Bank. doi:10.1596/978-1-4648-2047-2. License: Creative
|
| 418 |
+
Commons Attribution CC BY 3.0 IGO.
|
| 419 |
+
https://openknowledge.worldbank.org/handle/10986/41055
|
| 420 |
+
|
| 421 |
+
A06. Begazo, Tania, Moussa P. Blimpo, and Mark A. Dutz. 2023. Digital Africa:
|
| 422 |
+
Technological Transformation for Jobs. Washington, DC: World Bank. doi:
|
| 423 |
+
978-1-4648-1737-3. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 424 |
+
https://openknowledge.worldbank.org/handle/10986/39491
|
| 425 |
+
|
| 426 |
+
A07. Beylis, Guillermo, Roberto Fattal Jaef, Michael Morris, Ashwini Rekha Sebastian,
|
| 427 |
+
and Rishabh Sinha. 2020. Going Viral: COVID-19 and the Accelerated Transformation
|
| 428 |
+
of Jobs in Latin America and the Caribbean. World Bank Latin American and Caribbean
|
| 429 |
+
Studies. Washington, DC: World Bank. doi:10.1596/978-1-4648-1448-8. License:
|
| 430 |
+
Creative Commons Attribution CC BY 3.0 IGO.
|
| 431 |
+
https://openknowledge.worldbank.org/handle/10986/34413
|
| 432 |
+
|
| 433 |
+
A08. Borgomeo, Edoardo, Claire Chase, Nicolas Salazar Godoy, and Victor Osei Kwadwo.
|
| 434 |
+
2023. Rising from the Depths: Water Security and Fragility in South Sudan.
|
| 435 |
+
International Development in Focus. Washington, DC: World Bank.
|
| 436 |
+
doi:10.1596/978-1-4648-1943-8. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 437 |
+
https://openknowledge.worldbank.org/handle/10986/38379
|
| 438 |
+
|
| 439 |
+
A09. Bossavie, Laurent, and Daniel Garrote-Sánchez. 2022. Safe and Productive Migration
|
| 440 |
+
from the Kyrgyz Republic: Lessons from the COVID-19 Pandemic. International
|
| 441 |
+
Development in Focus. Washington, DC: World Bank. doi:10.1596/978-1-4648-1905-6.
|
| 442 |
+
License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 443 |
+
https://openknowledge.worldbank.org/handle/10986/38290
|
| 444 |
+
|
| 445 |
+
A10. Chapman, Emily Weedon, and Margaux Vinez, eds. 2023. Working Today for a Better
|
| 446 |
+
Tomorrow in Ethiopia: Jobs for Poor and Vulnerable Households. International
|
| 447 |
+
Development in Focus. Washington, DC: World Bank. doi:10.1596/978-1-4648-2020-5.
|
| 448 |
+
License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 449 |
+
https://openknowledge.worldbank.org/handle/10986/40801
|
| 450 |
+
|
| 451 |
+
A11. Chatain, Pierre-Laurent, Emile van der Does de Willebois, and Maud Bökkerink. 2022.
|
| 452 |
+
Preventing Money Laundering and Terrorist Financing: A Practical Guide for Bank
|
| 453 |
+
Supervisors. Second edition. Washington, DC: World Bank.
|
| 454 |
+
doi:10.1596/978-1-4648-1851-6. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 455 |
+
https://openknowledge.worldbank.org/handle/10986/37726
|
| 456 |
+
|
| 457 |
+
A12. Chrimes, Tommy, Bram Gootjes, M. Ayhan Kose, and Collette Wheeler. 2024. The Great
|
| 458 |
+
Reversal: Prospects, Risks, and Policies in International Development Association
|
| 459 |
+
(IDA) Countries. Washington, DC: World Bank. doi: 10.1596/978-1-4648-2145-5.
|
| 460 |
+
License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 461 |
+
https://openknowledge.worldbank.org/handle/10986/41403
|
| 462 |
+
|
| 463 |
+
A13. Corsi, Anna, and Harris Selod. 2023. Land Matters: Can Better Governance and
|
| 464 |
+
Management of Scarcity Prevent a Looming Crisis in the Middle East and North
|
| 465 |
+
Africa? Washington, DC: World Bank. doi:10.1596/978-1-4648-1661-1. License:
|
| 466 |
+
Creative Commons Attribution CC BY 3.0 IGO.
|
| 467 |
+
https://openknowledge.worldbank.org/handle/10986/38384
|
| 468 |
+
|
| 469 |
+
A14. Cruz, Marcio, ed. 2024. Digital Opportunities in African Businesses. Washington,
|
| 470 |
+
DC: World Bank. doi:10.1596/978-1-4648-2088-5. License: Creative Commons
|
| 471 |
+
Attribution CC BY 3.0 IGO.
|
| 472 |
+
https://openknowledge.worldbank.org/handle/10986/41447
|
| 473 |
+
|
| 474 |
+
A15. Dalhuijsen, Emma, Eva Gutierrez, Tatsiana Kliatskova, Rachel Mok, and Martijn Gert
|
| 475 |
+
Jan Regelink. 2023. Greening National Development Financial Institutions: Trends,
|
| 476 |
+
Lessons Learned, and Ways Forward. International Development in Focus. Washington,
|
| 477 |
+
DC: World Bank. doi:10.1596/978-1-4648-2031-1. License: Creative Commons
|
| 478 |
+
Attribution CC BY 3.0 IGO.
|
| 479 |
+
https://openknowledge.worldbank.org/handle/10986/40432
|
| 480 |
+
|
| 481 |
+
A16. Damania, Richard, Ebad Ebadi, Kentaro Mayr, Jason Russ, and Esha Zaveri. 2025.
|
| 482 |
+
Reboot Development: The Economics of a Livable Planet. Washington, DC: World Bank.
|
| 483 |
+
doi:10.1596/978-1-4648-2271-1. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 484 |
+
https://openknowledge.worldbank.org/handle/10986/43522
|
| 485 |
+
|
| 486 |
+
A17. de Nicola, Francesca, Aaditya Mattoo, and Jonathan Timmis. 2025. Firm Foundations
|
| 487 |
+
of Growth: Productivity and Technology in East Asia and Pacific. East Asia and
|
| 488 |
+
Pacific Development Studies. Washington, DC: World Bank. doi:
|
| 489 |
+
10.1596/978-1-4648-2200-1. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 490 |
+
https://openknowledge.worldbank.org/handle/10986/43128
|
| 491 |
+
|
| 492 |
+
A18. Didier, Tatiana, and Ana Paula Cusolito. 2024. Unleashing Productivity through Firm
|
| 493 |
+
Financing. Washington, DC: World Bank. doi:10.1596/978-1-4648-1939-1. License:
|
| 494 |
+
Creative Commons Attribution CC BY 3.0 IGO.
|
| 495 |
+
https://openknowledge.worldbank.org/handle/10986/42194
|
| 496 |
+
|
| 497 |
+
A19. Fernandes, Ana Margarida, and Tristan Reed. 2026. Industrial Policy for
|
| 498 |
+
Development: Approaches in the 21st Century. Policy Research Reports. Washington,
|
| 499 |
+
DC: World Bank. doi:10.1596/978-1-4648-2276-6. License: Creative Commons
|
| 500 |
+
Attribution CC BY 3.0 IGO.
|
| 501 |
+
https://openknowledge.worldbank.org/handle/10986/44244
|
| 502 |
+
|
| 503 |
+
A20. Hallegatte, Stéphane, Catrina Godinho, Jun Rentschler, Paolo Avner, Ira Irina
|
| 504 |
+
Dorband, Camilla Knudsen, Jana Lemke, and Penny Mealy. 2024. Within Reach:
|
| 505 |
+
Navigating the Political Economy of Decarbonization. Climate Change and Development
|
| 506 |
+
Series. Washington, DC: World Bank. doi:10.1596/978-1-4648-1953-7. License:
|
| 507 |
+
Creative Commons Attribution CC BY 3.0 IGO.
|
| 508 |
+
https://openknowledge.worldbank.org/handle/10986/40601
|
| 509 |
+
|
| 510 |
+
A21. Hanusch, Marek, ed. 2023. A Balancing Act for Brazil’s Amazonian States: An Economic
|
| 511 |
+
Memorandum. International Development in Focus. Washington, DC: World Bank.
|
| 512 |
+
doi:10.1596/978-1-4648-1909-4. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 513 |
+
https://openknowledge.worldbank.org/handle/10986/39778
|
| 514 |
+
|
| 515 |
+
A22. Herrera Dappe, Matías, Mathilde Lebrand, and Aiga Stokenberga. 2024. Shrinking
|
| 516 |
+
Economic Distance: Understanding How Markets and Places Can Lower Transport Costs
|
| 517 |
+
in Developing Countries. Sustainable Infrastructure Series. Washington, DC: World
|
| 518 |
+
Bank. doi:10.1596/978-1-4648-2124-0. License: Creative Commons Attribution CC BY
|
| 519 |
+
3.0 IGO.
|
| 520 |
+
https://openknowledge.worldbank.org/handle/10986/42061
|
| 521 |
+
|
| 522 |
+
A23. Holla, Alaka, Norbert Schady, and Joana Silva, eds. 2026. Building Human Capital
|
| 523 |
+
Where It Matters: Homes, Neighborhoods, and Workplaces. World Bank. doi:
|
| 524 |
+
10.1596/978-1-4648-2277-3. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 525 |
+
https://openknowledge.worldbank.org/handle/10986/44282
|
| 526 |
+
|
| 527 |
+
A24. Hou, Xiaohui, Jigyasa Sharma, and Feng Zhao, eds. 2023. Silver Opportunity:
|
| 528 |
+
Building Integrated Services for Older Adults around Primary Health Care.
|
| 529 |
+
Washington, DC: World Bank. doi:10.1596/978-1-4648-1958-2. License: Creative
|
| 530 |
+
Commons Attribution CC BY 3.0 IGO.
|
| 531 |
+
https://openknowledge.worldbank.org/handle/10986/39422
|
| 532 |
+
|
| 533 |
+
A25. Iacovone, Leonardo, Henry Aviomoh, Matias Belacin, Laurent Bossavie, Ana Cusolito,
|
| 534 |
+
Rafael de Hoyos, Gianmarco Ottaviano, Fabian Scheifele, Iván Torre, and Yutaka
|
| 535 |
+
Yoshino. 2025. TIDES of Change: Igniting Productivity Growth in Europe and Central
|
| 536 |
+
Asia. Europe and Central Asia Studies. Washington, DC: World Bank. License:
|
| 537 |
+
Creative Commons Attribution CC BY 3.0 IGO.
|
| 538 |
+
https://openknowledge.worldbank.org/handle/10986/43788
|
| 539 |
+
|
| 540 |
+
A26. Ianchovichina, Elena. 2024. The Evolving Geography of Productivity and Employment:
|
| 541 |
+
Ideas for Inclusive Growth through a Territorial Lens in Latin America and the
|
| 542 |
+
Caribbean. World Bank Latin American and Caribbean Studies. Washington, DC: World
|
| 543 |
+
Bank. doi:10.1596/978-1-4648-1959-9. License: Creative Commons Attribution CC BY
|
| 544 |
+
3.0 IGO.
|
| 545 |
+
https://openknowledge.worldbank.org/handle/10986/40969
|
| 546 |
+
|
| 547 |
+
A27. Iootty, Mariana, Asset Bizhan, and Paulo G. Correa. 2022. Boosting Productivity in
|
| 548 |
+
Kazakhstan with Micro-Level Tools: Analysis and Policy Lessons. International
|
| 549 |
+
Development in Focus. Washington, DC: World Bank. doi:10.1596/978-1-4648-1910-0.
|
| 550 |
+
License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 551 |
+
https://openknowledge.worldbank.org/handle/10986/38460
|
| 552 |
+
|
| 553 |
+
A28. Islamaj, Ergys, Aaditya Mattoo, Agustin Samano, and Matthew Wai-Poi. 2026. Small
|
| 554 |
+
Governments, Big Ambitions: Fiscal Policy in East Asia and Pacific. East Asia and
|
| 555 |
+
Pacific Development Studies. World Bank. doi:10.1596/978-1-4648-2318-3. License:
|
| 556 |
+
Creative Commons Attribution CC BY 3.0 IGO.
|
| 557 |
+
https://openknowledge.worldbank.org/handle/10986/45031
|
| 558 |
+
|
| 559 |
+
A29. Junquera-Varela, Raúl Félix, and Cristian Óliver Lucas-Mas. 2024. Revenue
|
| 560 |
+
Administration Handbook. Washington, DC: World Bank. doi:10.1596/978-1-4648-2053-3.
|
| 561 |
+
License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 562 |
+
https://openknowledge.worldbank.org/handle/10986/41090
|
| 563 |
+
|
| 564 |
+
A30. Kassa, Woubet, Hiau Looi Kee, and Jean-Christophe Maur. 2026. Integrating Africa:
|
| 565 |
+
From Threads to Hubs. Africa Development Forum series. Washington, DC: World Bank.
|
| 566 |
+
doi: 10.1596/978-1-4648-2320-6. License: Creative Commons Attribution CC BY 3.0
|
| 567 |
+
IGO.
|
| 568 |
+
https://openknowledge.worldbank.org/handle/10986/44608
|
| 569 |
+
|
| 570 |
+
A31. Kose, M. Ayhan, Peter Nagle, Franziska Ohnsorge, and Naotaka Sugawara. 2021. Global
|
| 571 |
+
Waves of Debt: Causes and Consequences. Washington, DC: World Bank.
|
| 572 |
+
doi:10.1596/978-1-4648-1544-7. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 573 |
+
https://openknowledge.worldbank.org/handle/10986/32809
|
| 574 |
+
|
| 575 |
+
A32. Lang, Megan, Jonah Rexer, Siddharth Sharma, and Margaret Triyana, eds. 2025. From
|
| 576 |
+
Risk to Resilience: Helping People and Firms Adapt in South Asia. South Asia
|
| 577 |
+
Development Matters. Washington, DC: World Bank. doi:10.1596/978-1-4648-2152-3.
|
| 578 |
+
License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 579 |
+
https://openknowledge.worldbank.org/handle/10986/43230
|
| 580 |
+
|
| 581 |
+
A33. Le, Sang Minh, Eric Hahn, Tu Anh Tran, Selin Mavituna, and Tam Minh Thi Ta. 2024.
|
| 582 |
+
Human Resources for Mental Health Service Delivery in Viet Nam: Toward Achieving
|
| 583 |
+
Universal Health Coverage. International Development in Focus. Washington, DC:
|
| 584 |
+
World Bank. doi:10.1596/978-1-4648-2122-6. License: Creative Commons Attribution CC
|
| 585 |
+
BY 3.0 IGO.
|
| 586 |
+
https://openknowledge.worldbank.org/handle/10986/41623
|
| 587 |
+
|
| 588 |
+
A34. Li, Yue, and Martin Rama, eds. 2023. Private Cities: Outstanding Examples from
|
| 589 |
+
Developing Countries and Their Implications for Urban Policy. Urban Development
|
| 590 |
+
Series. Washington, DC: World Bank. doi:10.1596/978-1-4648-1833-2. License:
|
| 591 |
+
Creative Commons Attribution CC BY 3.0 IGO.
|
| 592 |
+
https://openknowledge.worldbank.org/handle/10986/39847
|
| 593 |
+
|
| 594 |
+
A35. Mawejje, Joseph. 2025. Fiscal Vulnerabilities in Low-Income Countries: Evolution,
|
| 595 |
+
Drivers, and Policies. Washington, DC: World Bank. doi: 10.1596/978-1-4648-1968-1.
|
| 596 |
+
License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 597 |
+
https://openknowledge.worldbank.org/handle/10986/42239
|
| 598 |
+
|
| 599 |
+
A36. Mensah, Julia, Stephen Kisembe Kiirya, Elizabeth Asege Ekochu, Rogers Ayiko,
|
| 600 |
+
Brendan Michael Hayes, Collins Chansa, Grace Murindwa, Richard Crabbe, and Marc
|
| 601 |
+
DeFrancis, eds. 2024. Investing in Reproductive, Maternal, Newborn, Child, and
|
| 602 |
+
Adolescent Health in Uganda: What Have We Learned, and Where Do We Go from Here?
|
| 603 |
+
International Development in Focus. Washington, DC: World Bank.
|
| 604 |
+
doi:10.1596/978-1-4648-1993-3. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 605 |
+
https://openknowledge.worldbank.org/handle/10986/41033
|
| 606 |
+
|
| 607 |
+
A37. Mukim, Megha, and Mark Roberts, editors. 2023. Thriving: Making Cities Green,
|
| 608 |
+
Resilient, and Inclusive in a Changing Climate. Washington, DC: World Bank.
|
| 609 |
+
doi:10.1596/978-1-4648-1935-3. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 610 |
+
https://openknowledge.worldbank.org/handle/10986/38295
|
| 611 |
+
|
| 612 |
+
A38. OECD/The World Bank (2026), Compendium of Good Practices on Quality Infrastructure
|
| 613 |
+
2026: Rebuilding for the Future, OECD Publishing, Paris,
|
| 614 |
+
https://doi.org/10.1787/6981eda5-en. © OECD and The World Bank 2026. License:
|
| 615 |
+
Creative Commons Attribution 3.0 IGO (CC BY 3.0 IGO). To the extent that Source-1
|
| 616 |
+
is regarded as an adaptation of this work: This is an adaptation of an original
|
| 617 |
+
work by the OECD and the World Bank. The opinions expressed and arguments employed
|
| 618 |
+
in this adaptation should not be reported as representing the official views of the
|
| 619 |
+
OECD, its Development Centre or of their Member countries or of the World Bank, its
|
| 620 |
+
Board of Executive Directors or the governments they represent.
|
| 621 |
+
https://openknowledge.worldbank.org/handle/10986/44596
|
| 622 |
+
|
| 623 |
+
A39. Peszko, Grzegorz, Markus Amann, Yewande Awe, Gary Kleiman, and Tamer Samah Rabie.
|
| 624 |
+
2022. Air Pollution and Climate Change: From Co-Benefits to Coherent Policies.
|
| 625 |
+
International Development in Focus. Washington, DC: World Bank. doi:
|
| 626 |
+
10.1596/978-1-4648-1835-6. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 627 |
+
https://openknowledge.worldbank.org/handle/10986/38524
|
| 628 |
+
|
| 629 |
+
A40. Riera-Crichton, Daniel, and Guillermo Vuletin. 2024. Public Spending Policies in
|
| 630 |
+
Latin America and the Caribbean: When Cyclicality Meets Rigidities. Latin American
|
| 631 |
+
Development Forum. Washington, DC: World Bank. doi: 10.1596/978-1-4648-2069-4.
|
| 632 |
+
License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 633 |
+
https://openknowledge.worldbank.org/handle/10986/42022
|
| 634 |
+
|
| 635 |
+
A41. Rocha, Nadia, and Michele Ruta. 2022. Deep Trade Agreements: Anchoring Global Value
|
| 636 |
+
Chains in Latin America and the Caribbean. Washington, DC: World Bank.
|
| 637 |
+
doi:10.1596/9781-4648-1824-0. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 638 |
+
https://openknowledge.worldbank.org/handle/10986/37655
|
| 639 |
+
|
| 640 |
+
A42. Rogger, Daniel, and Christian Schuster, eds. 2023. The Government Analytics
|
| 641 |
+
Handbook: Leveraging Data to Strengthen Public Administration. Washington, DC:
|
| 642 |
+
World Bank. doi:10.1596/978-1-4648-1957-5. License: Creative Commons Attribution CC
|
| 643 |
+
BY 3.0 IGO.
|
| 644 |
+
https://openknowledge.worldbank.org/handle/10986/39857
|
| 645 |
+
|
| 646 |
+
A43. Schady, Norbert, Alaka Holla, Shwetlena Sabarwal, Joana Silva, and Andres Yi Chang.
|
| 647 |
+
2023. Collapse and Recovery: How the COVID-19 Pandemic Eroded Human Capital and
|
| 648 |
+
What to Do about It. Washington, DC: World Bank. doi:10.1596/978-1-4648-1901-8.
|
| 649 |
+
License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 650 |
+
https://openknowledge.worldbank.org/handle/10986/39403
|
| 651 |
+
|
| 652 |
+
A44. Sharma, Siddanth, Stefano M. Bertozzi, Victoria Y. Fan, Dean T. Jamison, Ole F.
|
| 653 |
+
Norheim, Hitoshi Oshitani, and Muhammad Ali Pate, eds. 2026. Investing in Pandemic
|
| 654 |
+
Prevention, Preparedness, and Response. Disease Control Priorities, fourth edition,
|
| 655 |
+
volume 2. Washington, DC: World Bank. doi:10.1596/978-1-4648-2213-1. License:
|
| 656 |
+
Creative Commons Attribution CC BY 3.0 IGO.
|
| 657 |
+
https://openknowledge.worldbank.org/handle/10986/43718
|
| 658 |
+
|
| 659 |
+
A45. Shilpi, Forhad, Matthew E. Kahn, and Claudia Berg. 2025. Rethinking Resilience:
|
| 660 |
+
Adapting to a Changing Climate. Policy Research Report. Washington, DC: World Bank.
|
| 661 |
+
doi:10.1596/978-1-4648-2158-5. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 662 |
+
https://openknowledge.worldbank.org/handle/10986/43919
|
| 663 |
+
|
| 664 |
+
A46. Soh, Hoon Sahib, Youngsun Koh, and Anwar Aridi (eds.). 2023. Innovative Korea:
|
| 665 |
+
Leveraging Innovation and Technology for Development. Washington, DC: World Bank.
|
| 666 |
+
doi:10.1596/978-1-4648-1961-2. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 667 |
+
https://openknowledge.worldbank.org/handle/10986/40234
|
| 668 |
+
|
| 669 |
+
A47. Sosale, Shobhana, Seo Yeon Hong, Shalika Subasinghe, and Hiran Herat. 2023.
|
| 670 |
+
Enhancing Skills in Sri Lanka for Inclusion, Recovery, and Resilience.
|
| 671 |
+
International Development in Focus. Washington, DC: World Bank.
|
| 672 |
+
doi:10.1596/978-1-4648-2008-3. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 673 |
+
https://openknowledge.worldbank.org/handle/10986/40663
|
| 674 |
+
|
| 675 |
+
A48. Straub, Stéphane, He He, Yue Li, Xinxin Lyu, Jevgenijs Steinbuks, Estefanía Vergara
|
| 676 |
+
Cobos, Christopher Dann, Manuel García-Santana, and Harris Selod. 2026.
|
| 677 |
+
Infrastructure Foundations: From Current Assets to Future Growth. Sustainable
|
| 678 |
+
Infrastructure Series. Washington, DC: World Bank. doi:10.1596/978-1-4648-2311-4.
|
| 679 |
+
License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 680 |
+
https://openknowledge.worldbank.org/handle/10986/44671
|
| 681 |
+
|
| 682 |
+
A49. Sutton, William R., Alexander Lotsch, and Ashesh Prasann. 2024. Recipe for a
|
| 683 |
+
Livable Planet: Achieving Net Zero Emissions in the Agrifood System. Agriculture
|
| 684 |
+
and Food Series. Washington, DC: World Bank. doi:10.1596/978-1-4648-2093-9.
|
| 685 |
+
License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 686 |
+
https://openknowledge.worldbank.org/handle/10986/41468
|
| 687 |
+
|
| 688 |
+
A50. Tenenbaum, Bernard, Chris Greacen, and Ashish Shrestha. 2024. Mini Grid Solutions
|
| 689 |
+
for Underserved Customers: New Insights from Nigeria and India. International
|
| 690 |
+
Development in Focus. Washington, DC: World Bank. doi:10.1596/978-1-4648-2055-7.
|
| 691 |
+
License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 692 |
+
https://openknowledge.worldbank.org/handle/10986/40923
|
| 693 |
+
|
| 694 |
+
A51. Vasishtha, Garima, ed. 2026. Fiscal Policy in Commodity Exporters: A Balancing Act.
|
| 695 |
+
Advance Edition. Washington, DC: World Bank. License: Creative Commons Attribution
|
| 696 |
+
CC BY 3.0 IGO.
|
| 697 |
+
https://openknowledge.worldbank.org/handle/10986/45122
|
| 698 |
+
|
| 699 |
+
A52. Vostroknutova, Ekaterina, James Sampi, Charl Jooste, and Jorge Thompson Araujo.
|
| 700 |
+
2025. Competition and Productivity Growth in Latin America and the Caribbean.
|
| 701 |
+
Washington, DC: World Bank. doi:10.1596/978-1-4648-2081-6. License: Creative
|
| 702 |
+
Commons Attribution CC BY 3.0 IGO.
|
| 703 |
+
https://openknowledge.worldbank.org/handle/10986/42869
|
| 704 |
+
|
| 705 |
+
A53. Vuletin, Guillermo. 2025. Rethinking Taxation for Growth in Latin America and the
|
| 706 |
+
Caribbean: Objectives, Behavioral Responses, and Technological Advances.
|
| 707 |
+
Washington, DC: World Bank. doi:10.1596/978-1-4648-2253-7. License: Creative
|
| 708 |
+
Commons Attribution CC BY 3.0 IGO.
|
| 709 |
+
https://openknowledge.worldbank.org/handle/10986/43647
|
| 710 |
+
|
| 711 |
+
A54. World Bank and the Development Research Center of the State Council, the People’s
|
| 712 |
+
Republic of China. 2022. Four Decades of Poverty Reduction in China: Drivers,
|
| 713 |
+
Insights for the World, and the Way Ahead. Washington, DC: World Bank.
|
| 714 |
+
doi:10.1596/978-14648-1877-6. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 715 |
+
https://openknowledge.worldbank.org/handle/10986/37727
|
| 716 |
+
|
| 717 |
+
A55. World Bank and World Trade Organization. 2020. Women and Trade: The Role of Trade
|
| 718 |
+
in Promoting Gender Equality. Washington, DC: World Bank.
|
| 719 |
+
doi:10.1596/978-1-4648-1541-6. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 720 |
+
https://openknowledge.worldbank.org/handle/10986/34140
|
| 721 |
+
|
| 722 |
+
A56. World Bank. 2023. Planning National Telemedicine and Health Hotline Services: A
|
| 723 |
+
Toolkit for Service Providers Working with Governments. International Development
|
| 724 |
+
in Practice. Washington, DC: World Bank. doi:10.1596/978-1-4648-1956-8. License:
|
| 725 |
+
Creative Commons Attribution CC BY 3.0 IGO.
|
| 726 |
+
https://openknowledge.worldbank.org/handle/10986/39465
|
| 727 |
+
|
| 728 |
+
A57. World Bank. 2023. The Business of the State. Washington, DC: World Bank.
|
| 729 |
+
doi:10.1596/978-14648-1998-8. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 730 |
+
https://openknowledge.worldbank.org/handle/10986/40343
|
| 731 |
+
|
| 732 |
+
A58. World Bank. 2024. Digital Progress and Trends Report 2023. Washington, DC: World
|
| 733 |
+
Bank. doi:10.1596/978-1-4648-2049-6. License: Creative Commons Attribution CC BY
|
| 734 |
+
3.0 IGO.
|
| 735 |
+
https://openknowledge.worldbank.org/handle/10986/40970
|
| 736 |
+
|
| 737 |
+
A59. World Bank. 2024. Poverty, Prosperity, and Planet Report 2024: Pathways Out of the
|
| 738 |
+
Polycrisis. Washington, DC: World Bank. doi:10.1596/978-1-4648-2123-3. License:
|
| 739 |
+
Creative Commons Attribution CC BY 3.0 IGO.
|
| 740 |
+
https://openknowledge.worldbank.org/handle/10986/42211
|
| 741 |
+
|
| 742 |
+
A60. World Bank. 2024. Services Unbound: Digital Technologies and Policy Reform in East
|
| 743 |
+
Asia and Pacific. East Asia and Pacific Development Studies. Washington, DC: World
|
| 744 |
+
Bank. doi:10.1596/978-1-4648-2082-3. License: Creative Commons Attribution CC BY
|
| 745 |
+
3.0 IGO.
|
| 746 |
+
https://openknowledge.worldbank.org/handle/10986/42486
|
| 747 |
+
|
| 748 |
+
A61. World Bank. 2024. The Path to 5G in the Developing World: Planning Ahead for a
|
| 749 |
+
Smooth Transition. Sustainable Infrastructure Series. Washington, DC: World Bank.
|
| 750 |
+
doi:10.1596/9781-4648-1604-8. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 751 |
+
https://openknowledge.worldbank.org/handle/10986/41689
|
| 752 |
+
|
| 753 |
+
A62. World Bank. 2026. From Prospective to Prepared Teacher: A Global Study of Initial
|
| 754 |
+
Teacher Education. Human Development Perspectives. Washington, DC: World Bank.
|
| 755 |
+
doi:10.1596/978-1-4648-2201-8. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 756 |
+
https://openknowledge.worldbank.org/handle/10986/44498
|
| 757 |
+
|
| 758 |
+
A63. World Bank. 2026. Institutions and Prosperity: Public Institutions for Enabling the
|
| 759 |
+
Private Sector. World Bank. DOI: 10.1596/978-1-4648-2189-9. License: Creative
|
| 760 |
+
Commons Attribution CC BY 3.0 IGO.
|
| 761 |
+
https://openknowledge.worldbank.org/handle/10986/44937
|
| 762 |
+
|
| 763 |
+
A64. World Bank. 2026. Women, Business and the Law 2026: Benchmarking Laws for Jobs and
|
| 764 |
+
Inclusive Growth. World Bank. doi:10.1596/978-1-4648-2196-7. License: Creative
|
| 765 |
+
Commons Attribution CC BY 3.0 IGO.
|
| 766 |
+
https://openknowledge.worldbank.org/handle/10986/44315
|
| 767 |
+
|
| 768 |
+
A65. World Bank. 2026. World Development Report 2026: The Promise of Artificial
|
| 769 |
+
Intelligence. World Bank. doi:10.1596/978-1-4648-2331-2. License: Creative Commons
|
| 770 |
+
Attribution CC BY 3.0 IGO.
|
| 771 |
+
https://openknowledge.worldbank.org/handle/10986/45296
|
| 772 |
+
|
| 773 |
+
A66. World Bank. Poverty and Shared Prosperity 2022: Correcting Course. Washington, DC:
|
| 774 |
+
World Bank. doi:10.1596/978-1-4648-1893-6. License: Creative Commons Attribution CC
|
| 775 |
+
BY 3.0 IGO.
|
| 776 |
+
https://openknowledge.worldbank.org/handle/10986/37739
|
| 777 |
+
|
| 778 |
+
A67. World Bank. Women, Business and the Law 2023. Washington, DC: World Bank.
|
| 779 |
+
doi:10.1596/978-1-4648-1944-5. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 780 |
+
https://openknowledge.worldbank.org/handle/10986/39462
|
| 781 |
+
|
| 782 |
+
A68. Zhang, Fan, Christian Borja-Vega, Hrishikesh Arvind Chandanpurkar, James
|
| 783 |
+
Famiglietti, Rick Hogeboom, Regassa Namara, Zarif Rasul, Pavel Luengas-Sierra, and
|
| 784 |
+
Deyu Rao. 2025. Continental Drying: A Threat to Our Common Future. Global Water
|
| 785 |
+
Monitoring. Washington, DC: World Bank. doi:10.1596/978-1-4648-2269-8. License:
|
| 786 |
+
Creative Commons Attribution CC BY 3.0 IGO.
|
| 787 |
+
https://openknowledge.worldbank.org/handle/10986/43683
|
| 788 |
+
|
| 789 |
+
A69. Zhao, Feng, Rialda Kovacevic, David Bishai, and Jeff Weintraub, eds. 2024.
|
| 790 |
+
Strategic Investment for Health System Resilience: A Three-Layer Framework. Human
|
| 791 |
+
Development Perspectives. Washington, DC: World Bank.
|
| 792 |
+
doi:10.1596/978-1-4648-2116-5. License: Creative Commons Attribution CC BY 3.0 IGO.
|
| 793 |
+
https://openknowledge.worldbank.org/handle/10986/42085
|
| 794 |
+
|
| 795 |
+
A70. Левитанская Екатерина, Мартин Мелецкий и Андрей Милютин. 2026. Дома и надежды:
|
| 796 |
+
мобилизация частного капитала для повышения доступности жилья и семейных инвестиций
|
| 797 |
+
в жильё в Казахстане. Международное развитие в центре внимания. Всемирный банк.
|
| 798 |
+
doi: 10.1596/978-1-4648-2346-6. Лицензия: Creative Commons Attribution CC BY 3.0
|
| 799 |
+
IGO. (English title: Homes and Hopes: Mobilizing Private Capital for Affordable
|
| 800 |
+
Housing and Family Investment in Kazakhstan.)
|
| 801 |
+
https://openknowledge.worldbank.org/handle/10986/44936
|
README.md
ADDED
|
@@ -0,0 +1,409 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: apache-2.0
|
| 3 |
+
base_model: jhu-clsp/mmBERT-base
|
| 4 |
+
base_model_relation: finetune
|
| 5 |
+
# library_name is left out on purpose: Source-1 runs only through the included source1.py, and a
|
| 6 |
+
# "library_name: transformers" entry would offer AutoModel and pipeline snippets that load the backbone without
|
| 7 |
+
# the trained heads and return meaningless scores.
|
| 8 |
+
pipeline_tag: text-classification
|
| 9 |
+
language:
|
| 10 |
+
- en
|
| 11 |
+
- ar
|
| 12 |
+
- az
|
| 13 |
+
- bg
|
| 14 |
+
- bn
|
| 15 |
+
- ca
|
| 16 |
+
- cs
|
| 17 |
+
- da
|
| 18 |
+
- de
|
| 19 |
+
- el
|
| 20 |
+
- es
|
| 21 |
+
- et
|
| 22 |
+
- fa
|
| 23 |
+
- fi
|
| 24 |
+
- fil
|
| 25 |
+
- fr
|
| 26 |
+
- gu
|
| 27 |
+
- he
|
| 28 |
+
- hi
|
| 29 |
+
- hr
|
| 30 |
+
- hu
|
| 31 |
+
- id
|
| 32 |
+
- it
|
| 33 |
+
- ja
|
| 34 |
+
- ka
|
| 35 |
+
- kk
|
| 36 |
+
- kn
|
| 37 |
+
- ko
|
| 38 |
+
- lt
|
| 39 |
+
- lv
|
| 40 |
+
- ml
|
| 41 |
+
- mr
|
| 42 |
+
- ms
|
| 43 |
+
- nl
|
| 44 |
+
- "no"
|
| 45 |
+
- pl
|
| 46 |
+
- pt
|
| 47 |
+
- ro
|
| 48 |
+
- ru
|
| 49 |
+
- sk
|
| 50 |
+
- sl
|
| 51 |
+
- sq
|
| 52 |
+
- sr
|
| 53 |
+
- sv
|
| 54 |
+
- sw
|
| 55 |
+
- ta
|
| 56 |
+
- te
|
| 57 |
+
- th
|
| 58 |
+
- tr
|
| 59 |
+
- uk
|
| 60 |
+
- ur
|
| 61 |
+
- vi
|
| 62 |
+
- zh
|
| 63 |
+
tags:
|
| 64 |
+
- data-filtering
|
| 65 |
+
- pretraining-data
|
| 66 |
+
- quality-classifier
|
| 67 |
+
- data-quality
|
| 68 |
+
- data-curation
|
| 69 |
+
- text-quality
|
| 70 |
+
- multilingual
|
| 71 |
+
- modernbert
|
| 72 |
+
- mmbert
|
| 73 |
+
datasets:
|
| 74 |
+
- HuggingFaceFW/fineweb-2
|
| 75 |
+
- HuggingFaceFW/fineweb
|
| 76 |
+
- HuggingFaceFW/finepdfs
|
| 77 |
+
- HuggingFaceFW/finewiki
|
| 78 |
+
- HPLT/HPLT2.0_cleaned
|
| 79 |
+
- allenai/c4
|
| 80 |
+
- wikimedia/wikipedia
|
| 81 |
+
- wikimedia/wikisource
|
| 82 |
+
- HuggingFaceFW/fineweb-edu-llama3-annotations
|
| 83 |
+
- HuggingFaceTB/finemath
|
| 84 |
+
- open-web-math/open-web-math
|
| 85 |
+
- codeparrot/github-code-clean
|
| 86 |
+
- bigcode/commitpackft
|
| 87 |
+
- HuggingFaceTB/cosmopedia
|
| 88 |
+
- google/civil_comments
|
| 89 |
+
- PleIAs/common_corpus
|
| 90 |
+
- OpenAssistant/oasst2
|
| 91 |
+
- CohereLabs/aya_dataset
|
| 92 |
+
- HuggingFaceFW/finetranslations
|
| 93 |
+
- HuggingFaceTB/smollm-corpus
|
| 94 |
+
- HuggingFaceTB/issues-kaggle-notebooks
|
| 95 |
+
- allenai/WildChat-1M
|
| 96 |
+
- codeparrot/github-code
|
| 97 |
+
- bigcode/commitpack
|
| 98 |
+
- OxAISH-AL-LLM/wiki_toxic
|
| 99 |
+
- databricks/databricks-dolly-15k
|
| 100 |
+
- Hello-SimpleAI/HC3
|
| 101 |
+
- ai4bharat/sangraha
|
| 102 |
+
- liamdugan/raid
|
| 103 |
+
- PleIAs/Post-OCR-Correction
|
| 104 |
+
- joelniklaus/eurlex_resources
|
| 105 |
+
- common-pile/arxiv_abstracts_filtered
|
| 106 |
+
- common-pile/arxiv_papers_filtered
|
| 107 |
+
- common-pile/biodiversity_heritage_library
|
| 108 |
+
- common-pile/biodiversity_heritage_library_filtered
|
| 109 |
+
- common-pile/caselaw_access_project
|
| 110 |
+
- common-pile/data_provenance_initiative_filtered
|
| 111 |
+
- common-pile/doab_filtered
|
| 112 |
+
- common-pile/foodista_filtered
|
| 113 |
+
- common-pile/github_archive
|
| 114 |
+
- common-pile/libretexts_filtered
|
| 115 |
+
- common-pile/library_of_congress
|
| 116 |
+
- common-pile/library_of_congress_filtered
|
| 117 |
+
- common-pile/news_filtered
|
| 118 |
+
- common-pile/oercommons_filtered
|
| 119 |
+
- common-pile/peS2o_filtered
|
| 120 |
+
- common-pile/pre_1929_books
|
| 121 |
+
- common-pile/pre_1929_books_filtered
|
| 122 |
+
- common-pile/pressbooks_filtered
|
| 123 |
+
- common-pile/project_gutenberg_filtered
|
| 124 |
+
- common-pile/public_domain_review_filtered
|
| 125 |
+
- common-pile/python_enhancement_proposals_filtered
|
| 126 |
+
- common-pile/regulations_filtered
|
| 127 |
+
- common-pile/stackexchange
|
| 128 |
+
- common-pile/ubuntu_irc_filtered
|
| 129 |
+
- common-pile/uk_hansard_filtered
|
| 130 |
+
- common-pile/usgpo
|
| 131 |
+
- common-pile/usgpo_filtered
|
| 132 |
+
- common-pile/uspto_filtered
|
| 133 |
+
- common-pile/wikiteam
|
| 134 |
+
- common-pile/wikiteam_filtered
|
| 135 |
+
---
|
| 136 |
+
|
| 137 |
+
# Source-1
|
| 138 |
+
|
| 139 |
+
Source-1 scores text as pretraining data for language models. It reads up to 8,192 tokens at a time, in 53 languages.
|
| 140 |
+
For each chunk it returns 13 fields (labels, quality scores and red flags), an overall 0-5 score and a keep/drop
|
| 141 |
+
decision. 307M parameters, Apache-2.0.
|
| 142 |
+
|
| 143 |
+

|
| 144 |
+
|
| 145 |
+
## Highlights
|
| 146 |
+
|
| 147 |
+
- **Outperforms every public quality scorer we tested, on all three test sets**, measured as agreement with an
|
| 148 |
+
independent LLM grader using Source-1's rubric. On the held-out set: 0.90 against 0.76 for propella-1 4B; on its 159
|
| 149 |
+
English chunks, 0.86 against 0.53 for the FineWeb-Edu classifier.
|
| 150 |
+
- **About 13x fewer parameters than propella-1 4B, and still closer to the grader.** Source-1 has 307M parameters,
|
| 151 |
+
propella-1 4B about 4.0B, and Source-1 agrees with the grader more closely on all three sets.
|
| 152 |
+
- **Within 0.012 of its 27B teacher at about 1/88 the size.** On the held-out set Source-1 scores 0.900; the
|
| 153 |
+
open-weight 27B LLM teacher it learned from scores 0.912.
|
| 154 |
+
- **Ahead on educational value alone, too.** On the English exam its `educational_value` reaches 0.90 rank agreement
|
| 155 |
+
with the grader's, against 0.61 for the FineWeb-Edu classifier and 0.88 for propella-1 4B.
|
| 156 |
+
|
| 157 |
+
*Measured as agreement with an independent LLM grader applying Source-1's own rubric, on held-out data from the same
|
| 158 |
+
kinds of sources; see [EVALUATION.md](EVALUATION.md) for methods, intervals and limits.*
|
| 159 |
+
|
| 160 |
+
## Quick start
|
| 161 |
+
|
| 162 |
+
Source-1 runs through the included `source1.py`, which needs only `torch`, `transformers`, `safetensors` and
|
| 163 |
+
`tokenizers` (no `trust_remote_code`). Do not load it with transformers' `pipeline` or `AutoModel` classes: they load
|
| 164 |
+
the backbone without the trained heads in `heads.safetensors` and return meaningless scores. The model is published on
|
| 165 |
+
the Hugging Face Hub as `msmth/Source-1`.
|
| 166 |
+
|
| 167 |
+
```bash
|
| 168 |
+
pip install -U huggingface_hub # provides the hf command
|
| 169 |
+
hf download msmth/Source-1 --local-dir Source-1 --exclude "model.fp32.safetensors"
|
| 170 |
+
cd Source-1
|
| 171 |
+
pip install -r requirements.txt
|
| 172 |
+
python source1.py --model . --input examples/sample.jsonl --output scores.jsonl --device cpu # reproduces examples/expected_output.jsonl
|
| 173 |
+
```
|
| 174 |
+
|
| 175 |
+
```python
|
| 176 |
+
from source1 import Source1
|
| 177 |
+
|
| 178 |
+
model = Source1.from_pretrained(".") # a local directory or a Hub repo id; bfloat16 weights by default
|
| 179 |
+
|
| 180 |
+
doc = model.score(open("article.txt", encoding="utf-8").read(), title="Optional title")
|
| 181 |
+
print(doc["overall"], doc["keep"]) # 0-5 score and the keep/drop decision
|
| 182 |
+
print(doc["educational_value"], doc["spam_seo"], doc["format"])
|
| 183 |
+
|
| 184 |
+
# Many documents at once: plain strings, or dicts with "text" and optionally "title".
|
| 185 |
+
results = model.score_batch(["First document ...", {"text": "Second document ...", "title": "A title"}])
|
| 186 |
+
|
| 187 |
+
# The full-precision copy of the weights (model.fp32.safetensors), computing in float32:
|
| 188 |
+
model_fp32 = Source1.from_pretrained(".", precision="fp32", dtype="fp32")
|
| 189 |
+
```
|
| 190 |
+
|
| 191 |
+
```bash
|
| 192 |
+
python source1.py --model . --input docs.jsonl --text-field text --output scores.jsonl --device cuda
|
| 193 |
+
python source1.py --model . --input page.txt --device cpu --precision fp32 --dtype fp32
|
| 194 |
+
```
|
| 195 |
+
|
| 196 |
+
- **Output.** One flat dict per document: the 13 fields, `overall`, `keep`, `drop_reasons`, `parts` (chunks), the
|
| 197 |
+
length in tokens, whether a chunk was cut, and per-chunk results (`chunks`) for split documents.
|
| 198 |
+
- **Precision.** `precision="bf16"` (default, `model.safetensors`, 0.6 GB) or `"fp32"` (`model.fp32.safetensors`,
|
| 199 |
+
1.2 GB, as trained; drop the `--exclude` above to get it, or load by Hub repo id, which downloads only the file you
|
| 200 |
+
ask for). It computes in bfloat16 on GPUs with native bfloat16 (NVIDIA Ampere or newer), as in the evaluation, and
|
| 201 |
+
in float32 elsewhere; float16 is refused. In bfloat16 scores move slightly with batch composition (up to about 0.04
|
| 202 |
+
on `overall`); use `dtype="fp32"` or `batch_tokens=1` to avoid it.
|
| 203 |
+
- **Options.** `drop_line` (`"calibrated"` default, `"default"` for the rubric's hard filters, or an expression such
|
| 204 |
+
as `"toxicity >= 4 or spam_seo >= 3 or boilerplate >= 4.5"`), `apply_offsets`, `max_chunks`, `revision`;
|
| 205 |
+
`python source1.py --help` lists every command-line flag (JSONL, JSON or plain-text input).
|
| 206 |
+
- **Examples and speed.** `examples/` holds six documents and their expected CPU output (a GPU can differ by a few
|
| 207 |
+
hundredths). Source-1 scores about 30 chunks (50,000 tokens) per second on one RTX 3090 in bfloat16.
|
| 208 |
+
|
| 209 |
+
## What it returns
|
| 210 |
+
|
| 211 |
+
| field | kind | scale or values |
|
| 212 |
+
|---|---|---|
|
| 213 |
+
| `format` | label | tutorial, reference, news, forum_qa, academic, fiction, code_file, product_page, blog_opinion, other |
|
| 214 |
+
| `topic` | label | science, technology, programming, math, health, finance, history, politics_law, society, philosophy_religion, arts_entertainment, literature, sports, lifestyle, other |
|
| 215 |
+
| `content_type` | label | plain_text, text_with_code, code_only, math_heavy |
|
| 216 |
+
| `educational_value` | quality, 0-5, higher is better | Does it teach something useful? |
|
| 217 |
+
| `reasoning_depth` | quality, 0-5 | Does it explain why and walk through steps, or just state facts? |
|
| 218 |
+
| `writing_quality` | quality, 0-5 | Is it clear, well organized and coherent? (for code: naming, structure, comments) |
|
| 219 |
+
| `information_density` | quality, 0-5 | How much real content per word, versus padding and filler? |
|
| 220 |
+
| `reliability` | quality, 0-5 | Does it look careful and trustworthy, or sloppy and made up? (fiction is judged on care and consistency) |
|
| 221 |
+
| `spam_seo` | red flag, 0-5, higher is worse | Text written to rank or sell rather than to inform: ads and pages that mainly promote score 3; keyword stuffing, clickbait and thin affiliate or doorway pages 4; auto-generated SEO text and scams 5 |
|
| 222 |
+
| `boilerplate` | red flag, 0-5 | Templates, auto-generated pages, menus, link lists, cookie banners and other non-content text |
|
| 223 |
+
| `toxicity` | red flag, 0-5 | Hate, harassment, explicit content (the text's own toxicity, not its subject) |
|
| 224 |
+
| `code_quality` | gated, 0-5 or null | Is the code readable, correct-looking and documented, and written by a person rather than generated? Applies when `content_type` is text_with_code or code_only, or `format` is code_file |
|
| 225 |
+
| `math_quality` | gated, 0-5 or null | Is the notation correct, are the steps shown, and do the solutions follow logically? Applies when `content_type` is math_heavy, or `topic` is math |
|
| 226 |
+
|
| 227 |
+
Each 0-5 field is the expected level of a six-way head, a continuous number such as 2.73. Gated scores are null when
|
| 228 |
+
the predicted labels say they do not apply. The anchors for every level are in `source1.json` and in
|
| 229 |
+
[EVALUATION.md](EVALUATION.md#appendix-a-rubric-anchors), with the teacher's extra rules.
|
| 230 |
+
|
| 231 |
+
```
|
| 232 |
+
quality = 0.30*educational_value + 0.20*reasoning_depth + 0.15*writing_quality
|
| 233 |
+
+ 0.20*information_density + 0.15*reliability
|
| 234 |
+
if code_quality or math_quality applies:
|
| 235 |
+
quality = 0.8*quality + 0.2*mean(the gated scores that apply)
|
| 236 |
+
penalty = 0.5*max(0, spam_seo - 1) + 0.4*max(0, boilerplate - 1) + 0.8*max(0, toxicity - 1)
|
| 237 |
+
overall = clip(quality - penalty, 0, 5)
|
| 238 |
+
|
| 239 |
+
drop if toxicity >= 4 or spam_seo >= 3.5 or boilerplate >= 4.5 (keep = false)
|
| 240 |
+
```
|
| 241 |
+
|
| 242 |
+
The drop line in `calibration.json` is slightly stricter on spam than the rubric's own hard filters
|
| 243 |
+
(`spam_seo >= 4`); both keep ads and promotional pages, which the rubric puts at `spam_seo` 3. It was picked on the
|
| 244 |
+
teacher's validation labels by a fixed rule; stricter lines and their trade-offs are in
|
| 245 |
+
[EVALUATION.md](EVALUATION.md#the-drop-line). You do not have to use `keep`: ranking on `overall`, or your own rules on
|
| 246 |
+
single fields, may serve you better.
|
| 247 |
+
|
| 248 |
+
A document longer than about 7,800 tokens is split into balanced chunks at natural boundaries. Each chunk is scored
|
| 249 |
+
after a one-line header, as in training (source type, title if given, "Part i of n"). The document then takes the
|
| 250 |
+
label that covers the most tokens and the token-weighted mean of each score (the maximum for `toxicity`; gated scores
|
| 251 |
+
over the chunks where they apply), and `overall` and `keep` are recomputed from those.
|
| 252 |
+
|
| 253 |
+
## Intended use
|
| 254 |
+
|
| 255 |
+
- Filtering, ranking and weighting text for language-model pretraining in the 53 training languages.
|
| 256 |
+
- Building data mixtures from the labels and the individual scores, not just one number.
|
| 257 |
+
- Auditing a corpus: how much of it is boilerplate, spam, code, math, fiction, and so on.
|
| 258 |
+
- Research on data quality and on distilling LLM judgments into small encoders.
|
| 259 |
+
|
| 260 |
+
Out of scope:
|
| 261 |
+
|
| 262 |
+
- Judging people, applications or student work. The scores describe text as training data.
|
| 263 |
+
- Fact checking: `reliability` judges care and plausibility; it does not verify claims.
|
| 264 |
+
- Content moderation or safety decisions: `toxicity` is a coarse flag for data filtering.
|
| 265 |
+
- Deciding whether text is licensed, copyrighted or legal to use.
|
| 266 |
+
- Languages outside the 53 listed (never trained or tested on), non-text inputs, and generating text.
|
| 267 |
+
|
| 268 |
+
## Training
|
| 269 |
+
|
| 270 |
+
- **Base model.** [mmBERT-base](https://huggingface.co/jhu-clsp/mmBERT-base) (ModernBERT architecture, 8,192-token
|
| 271 |
+
context), all weights fine-tuned, with 13 linear heads on the mean-pooled hidden states: 307M parameters in total.
|
| 272 |
+
- **Data.** 172,895 chunks (350M tokens, 151,281 documents) in 53 languages, 38.9% English: filtered and unfiltered
|
| 273 |
+
web text, PDFs, wikis, math, permissively licensed code, conversations, synthetic text and openly licensed books,
|
| 274 |
+
after license filtering and a safety filter. 9,744 chunks for validation and 9,553 for testing, split by document.
|
| 275 |
+
- **Labels.** Every training label comes from an open-weight 27B LLM teacher scoring each chunk against the 13-field
|
| 276 |
+
rubric. No human labels were used, and no output of a proprietary model was used as a label or training target.
|
| 277 |
+
- **Recipe.** 2 epochs (5,320 steps of 131,072 tokens), AdamW, learning rate 5e-5 (heads 10x), bfloat16 mixed
|
| 278 |
+
precision, about 1.9 hours on one H100 NVL. The learning rate and language mix came from a short sweep read on
|
| 279 |
+
validation agreement with the teacher; the checkpoint is the final step, which also had the best validation score.
|
| 280 |
+
- **Development note.** AI assistants helped draft the rubric and write the code. Grades from proprietary LLM graders,
|
| 281 |
+
among them models of the evaluation grader's family, were used to choose between rubric revisions, to vet the
|
| 282 |
+
teacher and choose its prompt setup (on the samples that became the exam sets), and to set the candidate drop lines
|
| 283 |
+
and the 0.95 floor of the drop-line rule; LLM reviews also informed the license filtering. None of these grades or
|
| 284 |
+
reviews was used as a label or training example.
|
| 285 |
+
|
| 286 |
+
Data sources, filtering, recipe and development details:
|
| 287 |
+
[EVALUATION.md](EVALUATION.md#appendix-b-training-in-detail) and
|
| 288 |
+
[Independence from development](EVALUATION.md#independence-from-development).
|
| 289 |
+
|
| 290 |
+
## Evaluation
|
| 291 |
+
|
| 292 |
+
Source-1 was compared with 16 public quality scorers on three test sets, each graded by an independent proprietary
|
| 293 |
+
LLM grader applying Source-1's rubric; the grader's labels were never trained on. Each cell is the rank agreement
|
| 294 |
+
(Spearman) between each model's main score and the grader's overall score, on the chunks that model scored.
|
| 295 |
+
|
| 296 |
+
| test set | chunks | Source-1 | propella-1 4B | FineWeb-Edu classifier |
|
| 297 |
+
|---|---|---|---|---|
|
| 298 |
+
| held-out set, 53 languages (main result) | 495 | **0.900** | 0.756 | 0.529 (159 English chunks; Source-1 0.864) |
|
| 299 |
+
| English exam | 412 to 413 | **0.921** | 0.820 | 0.453 |
|
| 300 |
+
| 12-language exam (web text) | 350 to 352 | **0.895** | 0.637 | English only |
|
| 301 |
+
|
| 302 |
+
The open-weight 27B LLM teacher scores 0.912, 0.912 and 0.880 on the same sets. All 39 public-scorer comparisons
|
| 303 |
+
favor Source-1 with 95% intervals clear of zero. The held-out set is the main result because, unlike the exams, it
|
| 304 |
+
played no part in choosing the teacher. Scoring each held-out chunk in full rather than its first 512 tokens raises
|
| 305 |
+
agreement from 0.85 to 0.90 ([details](EVALUATION.md#reading-the-whole-chunk)).
|
| 306 |
+
|
| 307 |
+
propella-1 has no single score: we read it through a composite of four of its quality ratings. Counting its own
|
| 308 |
+
commercial-bias, content-ratio, integrity and safety ratings as well narrows its held-out gap from 0.14 to 0.06-0.08,
|
| 309 |
+
depending on the weighting and languages, still in Source-1's favor
|
| 310 |
+
([details](EVALUATION.md#how-propella-1-is-read)). The public scorers also read the text without Source-1's one-line
|
| 311 |
+
header ([details](EVALUATION.md#the-one-line-header)).
|
| 312 |
+
|
| 313 |
+
Every scorer, interval, drop-line result and caveat is in [EVALUATION.md](EVALUATION.md).
|
| 314 |
+
|
| 315 |
+
## Limitations
|
| 316 |
+
|
| 317 |
+
- **Home-ground benchmark.** The grader applies Source-1's own rubric, and grades from its model family steered rubric
|
| 318 |
+
revisions, the choice of teacher and the drop-line design. The public scorers were built for their own definitions
|
| 319 |
+
of quality. Whether filtering with Source-1 trains better language models has not been tested.
|
| 320 |
+
- **The exams helped choose the teacher.** The two exam sets are the samples on which the teacher and its prompt setup
|
| 321 |
+
were chosen, against the exam grader's labels, so they are not independent of the grader.
|
| 322 |
+
- **It rates some qualities higher than the grader.** On the English exam's random sample, `educational_value` is 0.28
|
| 323 |
+
levels above the grader on average, and `reasoning_depth`, `writing_quality` and `reliability` 0.26 to 0.39 (the
|
| 324 |
+
teacher: 0.31, and 0.26 to 0.42).
|
| 325 |
+
- **It misses a third of the grader's drops at the shipped line.** It catches 43 of 64 (0.67) on the held-out set and
|
| 326 |
+
the teacher 46; 17 of Source-1's 21 misses are also missed by the teacher. A stricter `drop_line` catches more.
|
| 327 |
+
- **Weaker on some kinds of text.** Held-out rank is 0.68 for books and 0.73 for conversations, code and synthetic text,
|
| 328 |
+
against about 0.90 for web text; `code_quality` and `math_quality` are the least reliable fields.
|
| 329 |
+
- **Not a fact checker or a moderation tool.** `reliability` is a surface judgment; `toxicity` was trained on data where
|
| 330 |
+
toxic text is rare.
|
| 331 |
+
- **One chunk at a time, and less data for some languages.** Nothing outside a chunk of up to 8,192 tokens is visible to
|
| 332 |
+
it; the 16 languages with the fewest training chunks have 1,249 to 1,470 each.
|
| 333 |
+
- **License screening has limits.** Notices worded in ways the patterns miss, and opt-outs outside the text (such as
|
| 334 |
+
robots.txt), were not caught. If you find such a document, tell us (see below).
|
| 335 |
+
- **No reproduction kit.** The evaluation chunks, the grader's labels, per-chunk scores, the metrics script and the
|
| 336 |
+
teacher's prompt are not included, so the numbers cannot be recomputed from this repository.
|
| 337 |
+
|
| 338 |
+
More: [Limitations in detail](EVALUATION.md#limitations-in-detail).
|
| 339 |
+
|
| 340 |
+
## Files
|
| 341 |
+
|
| 342 |
+
| file | contents |
|
| 343 |
+
|---|---|
|
| 344 |
+
| `model.safetensors` | the fine-tuned mmBERT-base backbone in bfloat16 (default) |
|
| 345 |
+
| `model.fp32.safetensors` | the same backbone in float32 (load with `precision="fp32"`) |
|
| 346 |
+
| `config.json` | backbone configuration (ModernBERT) |
|
| 347 |
+
| `heads.safetensors` | the 13 scoring heads |
|
| 348 |
+
| `source1.json` | rubric, head layout, pooling, maximum length and text normalization |
|
| 349 |
+
| `calibration.json` | the drop line and the quality-score offsets, with how they were chosen |
|
| 350 |
+
| `tokenizer.json`, `tokenizer_config.json` | mmBERT's tokenizer (same vocabulary and merges, re-saved) |
|
| 351 |
+
| `source1.py` | standalone loader, Python API and command line |
|
| 352 |
+
| `requirements.txt` | `torch`, `transformers`, `safetensors`, `tokenizers` |
|
| 353 |
+
| `examples/` | six sample inputs (`sample.jsonl`) and their expected command-line output (`expected_output.jsonl`) |
|
| 354 |
+
| `EVALUATION.md` | the full evaluation, the rubric anchors and the training details |
|
| 355 |
+
| `images/` | the benchmark chart above |
|
| 356 |
+
| `LICENSE`, `NOTICE`, `AUTHORS` | license text, third-party notices and credits, authors |
|
| 357 |
+
| `CREDITS_BOOKS.tsv` | per-work credits for the training books that are not public domain (part of `NOTICE`) |
|
| 358 |
+
|
| 359 |
+
## License and credits
|
| 360 |
+
|
| 361 |
+
Source-1 is released under the [Apache License 2.0](LICENSE). Copyright 2026 The Source-1 Authors (see `AUTHORS`).
|
| 362 |
+
[`NOTICE`](NOTICE) holds the full third-party notices and data credits. In short:
|
| 363 |
+
|
| 364 |
+
- **Base model.** Fine-tuned from [mmBERT-base](https://huggingface.co/jhu-clsp/mmBERT-base) by the mmBERT authors at
|
| 365 |
+
Johns Hopkins University (Marone et al., 2025), MIT License; all encoder weights were further trained and 13 scoring
|
| 366 |
+
heads added. mmBERT's tokenizer is based on the Gemma 2 tokenizer by Google.
|
| 367 |
+
- **Labels.** Produced by an open-weight 27B LLM released under Apache-2.0, self-hosted; no teacher weights are here.
|
| 368 |
+
- **Training data.** Public sources under their own terms, credited in `NOTICE` and `CREDITS_BOOKS.tsv`: among them
|
| 369 |
+
data under the ODC Attribution License (FineWeb, FineWeb-2, FinePDFs, C4, FineMath and others; Common Crawl data
|
| 370 |
+
through them was subject to the Common Crawl Terms of Use), Wikipedia-family text (CC BY-SA, GFDL or CC BY; with
|
| 371 |
+
thanks to its volunteer editors), Common Pile v0.1, HPLT 2.0, EU publications, and Parliamentary information
|
| 372 |
+
licensed under the Open Parliament Licence v3.0. Books include World Bank publications under CC BY 3.0 IGO; the World
|
| 373 |
+
Bank and the other publishers do not endorse this model. Code is limited to permissive licenses.
|
| 374 |
+
- **Share-alike text.** About 12.6% of training documents carry CC BY-SA or GFDL licenses. Source-1 is a classifier:
|
| 375 |
+
it outputs scores, not text, and no training text is distributed with it. The weights are released under Apache-2.0
|
| 376 |
+
with attribution, as comparable quality classifiers are, on the view that such a scorer is not an adaptation of the
|
| 377 |
+
text it was trained on. Copyleft code was removed anyway.
|
| 378 |
+
- Upstream license metadata can be wrong. If you find a source that should not be here, please tell us (below).
|
| 379 |
+
|
| 380 |
+
## Contact and takedown
|
| 381 |
+
|
| 382 |
+
Questions, corrections and removal requests: the [Community tab](https://huggingface.co/msmth/Source-1/discussions). If you believe your content was used to train Source-1 and
|
| 383 |
+
you want it excluded from future versions, tell us the URL or dataset and we will remove it from the training data of
|
| 384 |
+
the next release.
|
| 385 |
+
|
| 386 |
+
## Citation
|
| 387 |
+
|
| 388 |
+
```bibtex
|
| 389 |
+
@misc{source1_2026,
|
| 390 |
+
title = {Source-1: a multilingual 13-field scorer for pretraining data},
|
| 391 |
+
author = {{The Source-1 Authors}},
|
| 392 |
+
year = {2026},
|
| 393 |
+
howpublished = {\url{https://huggingface.co/msmth/Source-1}}
|
| 394 |
+
}
|
| 395 |
+
```
|
| 396 |
+
|
| 397 |
+
Please also cite mmBERT:
|
| 398 |
+
|
| 399 |
+
```bibtex
|
| 400 |
+
@misc{marone2025mmbertmodernmultilingualencoder,
|
| 401 |
+
title = {mmBERT: A Modern Multilingual Encoder with Annealed Language Learning},
|
| 402 |
+
author = {Marc Marone and Orion Weller and William Fleshman and Eugene Yang and Dawn Lawrie and Benjamin Van Durme},
|
| 403 |
+
year = {2025},
|
| 404 |
+
eprint = {2509.06888},
|
| 405 |
+
archivePrefix = {arXiv},
|
| 406 |
+
primaryClass = {cs.CL},
|
| 407 |
+
url = {https://arxiv.org/abs/2509.06888}
|
| 408 |
+
}
|
| 409 |
+
```
|
calibration.json
ADDED
|
@@ -0,0 +1,106 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"version": 1,
|
| 3 |
+
"model": "Source-1",
|
| 4 |
+
"fitted_on": "a held-out validation split of 9,744 chunks (documents never trained on), against the labels of the teacher, an open-weight 27B LLM scoring the 13-field rubric; nothing here was fitted on evaluation labels",
|
| 5 |
+
"drop_line": {
|
| 6 |
+
"line": "toxicity >= 4 or spam_seo >= 3.5 or boilerplate >= 4.5",
|
| 7 |
+
"meaning": "keep is false when this expression is true on a chunk's or a document's scores. It is the schema's default hard filters (schema_default) with the spam_seo threshold lowered from 4 to 3.5. source1.py applies it by default (drop_line=\"default\" applies schema_default instead).",
|
| 8 |
+
"schema_default": "toxicity >= 4 or spam_seo >= 4 or boilerplate >= 4.5",
|
| 9 |
+
"rule": "the candidate with the highest drop recall among those whose keep agreement with the teacher's keep flags is at least 0.95",
|
| 10 |
+
"reference": "the teacher's keep flags: schema_default applied to the teacher's labels of each validation chunk",
|
| 11 |
+
"columns": {
|
| 12 |
+
"keep_agreement": "share of chunks where the line and the teacher make the same keep decision",
|
| 13 |
+
"drop_recall": "share of the teacher's drops the line also drops",
|
| 14 |
+
"teacher_drops": "chunks the teacher drops",
|
| 15 |
+
"caught": "of those, chunks the line drops too",
|
| 16 |
+
"wrong_drops": "chunks the line drops but the teacher keeps",
|
| 17 |
+
"line_drops": "chunks the line drops",
|
| 18 |
+
"drop_share": "share of all chunks the line drops"
|
| 19 |
+
},
|
| 20 |
+
"candidates": [
|
| 21 |
+
{
|
| 22 |
+
"line": "toxicity >= 4 or spam_seo >= 4 or boilerplate >= 4.5",
|
| 23 |
+
"keep_agreement": 0.96,
|
| 24 |
+
"drop_recall": 0.7111,
|
| 25 |
+
"teacher_drops": 1042,
|
| 26 |
+
"caught": 741,
|
| 27 |
+
"wrong_drops": 89,
|
| 28 |
+
"line_drops": 830,
|
| 29 |
+
"drop_share": 0.0852
|
| 30 |
+
},
|
| 31 |
+
{
|
| 32 |
+
"line": "toxicity >= 4 or spam_seo >= 3.5 or boilerplate >= 4.5",
|
| 33 |
+
"keep_agreement": 0.9638,
|
| 34 |
+
"drop_recall": 0.7706,
|
| 35 |
+
"teacher_drops": 1042,
|
| 36 |
+
"caught": 803,
|
| 37 |
+
"wrong_drops": 114,
|
| 38 |
+
"line_drops": 917,
|
| 39 |
+
"drop_share": 0.0941
|
| 40 |
+
},
|
| 41 |
+
{
|
| 42 |
+
"line": "toxicity >= 4 or spam_seo >= 3 or boilerplate >= 4.5",
|
| 43 |
+
"keep_agreement": 0.929,
|
| 44 |
+
"drop_recall": 0.8301,
|
| 45 |
+
"teacher_drops": 1042,
|
| 46 |
+
"caught": 865,
|
| 47 |
+
"wrong_drops": 515,
|
| 48 |
+
"line_drops": 1380,
|
| 49 |
+
"drop_share": 0.1416
|
| 50 |
+
},
|
| 51 |
+
{
|
| 52 |
+
"line": "toxicity >= 4 or spam_seo >= 2.5 or boilerplate >= 4.5",
|
| 53 |
+
"keep_agreement": 0.8559,
|
| 54 |
+
"drop_recall": 0.8724,
|
| 55 |
+
"teacher_drops": 1042,
|
| 56 |
+
"caught": 909,
|
| 57 |
+
"wrong_drops": 1271,
|
| 58 |
+
"line_drops": 2180,
|
| 59 |
+
"drop_share": 0.2237
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"line": "toxicity >= 4 or spam_seo >= 3.5 or boilerplate >= 4",
|
| 63 |
+
"keep_agreement": 0.9453,
|
| 64 |
+
"drop_recall": 0.88,
|
| 65 |
+
"teacher_drops": 1042,
|
| 66 |
+
"caught": 917,
|
| 67 |
+
"wrong_drops": 408,
|
| 68 |
+
"line_drops": 1325,
|
| 69 |
+
"drop_share": 0.136
|
| 70 |
+
}
|
| 71 |
+
]
|
| 72 |
+
},
|
| 73 |
+
"offsets_meaning": "per quality score: mean teacher label minus mean Source-1 score over the validation chunks; added to Source-1's score, an offset removes its average lean against the teacher. All are below 0.02 in absolute value, so source1.py reports the model's own scores unless apply_offsets=True.",
|
| 74 |
+
"offsets": {
|
| 75 |
+
"educational_value": {
|
| 76 |
+
"offset": -0.00264963054187195,
|
| 77 |
+
"chunks": 9744,
|
| 78 |
+
"mean_teacher": 1.9859,
|
| 79 |
+
"mean_source1": 1.9886
|
| 80 |
+
},
|
| 81 |
+
"reasoning_depth": {
|
| 82 |
+
"offset": -0.005093390804597586,
|
| 83 |
+
"chunks": 9744,
|
| 84 |
+
"mean_teacher": 1.5808,
|
| 85 |
+
"mean_source1": 1.5859
|
| 86 |
+
},
|
| 87 |
+
"writing_quality": {
|
| 88 |
+
"offset": -0.009194376026272266,
|
| 89 |
+
"chunks": 9744,
|
| 90 |
+
"mean_teacher": 2.9367,
|
| 91 |
+
"mean_source1": 2.9459
|
| 92 |
+
},
|
| 93 |
+
"information_density": {
|
| 94 |
+
"offset": -0.005678571428571644,
|
| 95 |
+
"chunks": 9744,
|
| 96 |
+
"mean_teacher": 2.485,
|
| 97 |
+
"mean_source1": 2.4907
|
| 98 |
+
},
|
| 99 |
+
"reliability": {
|
| 100 |
+
"offset": -0.01612294745484366,
|
| 101 |
+
"chunks": 9744,
|
| 102 |
+
"mean_teacher": 3.1512,
|
| 103 |
+
"mean_source1": 3.1673
|
| 104 |
+
}
|
| 105 |
+
}
|
| 106 |
+
}
|
config.json
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"ModernBertModel"
|
| 4 |
+
],
|
| 5 |
+
"attention_bias": false,
|
| 6 |
+
"attention_dropout": 0.0,
|
| 7 |
+
"bos_token_id": 2,
|
| 8 |
+
"classifier_activation": "gelu",
|
| 9 |
+
"classifier_bias": false,
|
| 10 |
+
"classifier_dropout": 0.0,
|
| 11 |
+
"classifier_pooling": "mean",
|
| 12 |
+
"cls_token_id": 1,
|
| 13 |
+
"decoder_bias": true,
|
| 14 |
+
"deterministic_flash_attn": false,
|
| 15 |
+
"dtype": "bfloat16",
|
| 16 |
+
"embedding_dropout": 0.0,
|
| 17 |
+
"eos_token_id": 1,
|
| 18 |
+
"global_attn_every_n_layers": 3,
|
| 19 |
+
"global_rope_theta": 160000,
|
| 20 |
+
"gradient_checkpointing": false,
|
| 21 |
+
"hidden_activation": "gelu",
|
| 22 |
+
"hidden_size": 768,
|
| 23 |
+
"initializer_cutoff_factor": 2.0,
|
| 24 |
+
"initializer_range": 0.02,
|
| 25 |
+
"intermediate_size": 1152,
|
| 26 |
+
"layer_norm_eps": 1e-05,
|
| 27 |
+
"layer_types": [
|
| 28 |
+
"full_attention",
|
| 29 |
+
"sliding_attention",
|
| 30 |
+
"sliding_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"sliding_attention",
|
| 33 |
+
"sliding_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"sliding_attention",
|
| 36 |
+
"sliding_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"sliding_attention",
|
| 39 |
+
"sliding_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"sliding_attention",
|
| 42 |
+
"sliding_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"sliding_attention",
|
| 45 |
+
"sliding_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"sliding_attention",
|
| 48 |
+
"sliding_attention",
|
| 49 |
+
"full_attention"
|
| 50 |
+
],
|
| 51 |
+
"local_attention": 128,
|
| 52 |
+
"local_rope_theta": 160000,
|
| 53 |
+
"mask_token_id": 4,
|
| 54 |
+
"max_position_embeddings": 8192,
|
| 55 |
+
"mlp_bias": false,
|
| 56 |
+
"mlp_dropout": 0.0,
|
| 57 |
+
"model_type": "modernbert",
|
| 58 |
+
"norm_bias": false,
|
| 59 |
+
"norm_eps": 1e-05,
|
| 60 |
+
"num_attention_heads": 12,
|
| 61 |
+
"num_hidden_layers": 22,
|
| 62 |
+
"pad_token_id": 0,
|
| 63 |
+
"position_embedding_type": "sans_pos",
|
| 64 |
+
"rope_parameters": {
|
| 65 |
+
"full_attention": {
|
| 66 |
+
"rope_theta": 160000,
|
| 67 |
+
"rope_type": "default"
|
| 68 |
+
},
|
| 69 |
+
"sliding_attention": {
|
| 70 |
+
"rope_theta": 160000,
|
| 71 |
+
"rope_type": "default"
|
| 72 |
+
}
|
| 73 |
+
},
|
| 74 |
+
"sep_token_id": 1,
|
| 75 |
+
"sparse_pred_ignore_index": -100,
|
| 76 |
+
"sparse_prediction": false,
|
| 77 |
+
"tie_word_embeddings": true,
|
| 78 |
+
"transformers_version": "5.17.0",
|
| 79 |
+
"vocab_size": 256000
|
| 80 |
+
}
|
examples/expected_output.jsonl
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"id": "tutorial-binary-search", "format": "tutorial", "topic": "programming", "content_type": "plain_text", "educational_value": 4.035, "reasoning_depth": 3.852, "writing_quality": 4.17, "information_density": 4.034, "reliability": 4.135, "spam_seo": 0.009, "boilerplate": 0.008, "toxicity": 0.004, "code_quality": null, "math_quality": null, "overall": 4.033, "keep": true, "drop_reasons": [], "parts": 1, "tokens": 367, "truncated": false}
|
| 2 |
+
{"id": "seo-spam", "format": "product_page", "topic": "lifestyle", "content_type": "plain_text", "educational_value": 0.0, "reasoning_depth": 0.0, "writing_quality": 1.002, "information_density": 0.002, "reliability": 0.899, "spam_seo": 4.94, "boilerplate": 4.905, "toxicity": 0.002, "code_quality": null, "math_quality": null, "overall": 0.0, "keep": false, "drop_reasons": ["spam_seo >= 3.5", "boilerplate >= 4.5"], "parts": 1, "tokens": 115, "truncated": false}
|
| 3 |
+
{"id": "python-code", "format": "code_file", "topic": "programming", "content_type": "code_only", "educational_value": 3.021, "reasoning_depth": 2.315, "writing_quality": 3.955, "information_density": 3.987, "reliability": 3.981, "spam_seo": 0.007, "boilerplate": 0.016, "toxicity": 0.009, "code_quality": 3.982, "math_quality": null, "overall": 3.482, "keep": true, "drop_reasons": [], "parts": 1, "tokens": 235, "truncated": false}
|
| 4 |
+
{"id": "math-worked-example", "format": "tutorial", "topic": "math", "content_type": "plain_text", "educational_value": 4.165, "reasoning_depth": 4.093, "writing_quality": 3.869, "information_density": 4.119, "reliability": 4.101, "spam_seo": 0.483, "boilerplate": 0.105, "toxicity": 0.037, "code_quality": null, "math_quality": 4.109, "overall": 4.092, "keep": true, "drop_reasons": [], "parts": 1, "tokens": 194, "truncated": false}
|
| 5 |
+
{"id": "spanish-water-cycle", "format": "reference", "topic": "science", "content_type": "plain_text", "educational_value": 3.109, "reasoning_depth": 2.381, "writing_quality": 4.006, "information_density": 3.98, "reliability": 3.996, "spam_seo": 0.001, "boilerplate": 0.001, "toxicity": 0.0, "code_quality": null, "math_quality": null, "overall": 3.405, "keep": true, "drop_reasons": [], "parts": 1, "tokens": 131, "truncated": false}
|
| 6 |
+
{"id": "cookie-banner", "format": "other", "topic": "other", "content_type": "plain_text", "educational_value": 0.0, "reasoning_depth": 0.0, "writing_quality": 2.656, "information_density": 0.001, "reliability": 2.992, "spam_seo": 0.114, "boilerplate": 5.0, "toxicity": 0.001, "code_quality": null, "math_quality": null, "overall": 0.0, "keep": false, "drop_reasons": ["boilerplate >= 4.5"], "parts": 1, "tokens": 74, "truncated": false}
|
examples/sample.jsonl
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"id": "tutorial-binary-search", "title": "Binary search, step by step", "text": "Binary search finds a value in a sorted list by halving the part of the list that can still contain it.\n\n1. Start with two markers: low at the first position and high at the last.\n2. Look at the middle element, at position (low + high) // 2.\n3. If it equals the target, you are done. If it is smaller, the target can only be to its right, so move low to the middle plus one. If it is larger, move high to the middle minus one.\n4. Repeat until low passes high; then the value is not in the list.\n\nWhy does this work? Because the list is sorted, one comparison with the middle element rules out half of the remaining candidates. A list of 1,000 items therefore needs at most 10 comparisons, since 2^10 = 1,024. A linear scan may need all 1,000.\n\nExample: search for 23 in [2, 5, 8, 12, 16, 23, 38, 56, 72, 91]. The middle element is 16, which is smaller, so we keep [23, 38, 56, 72, 91]. Its middle is 56, which is larger, so we keep [23, 38]. The middle of that is 23: found after three comparisons.\n\nA common bug is computing the middle as (low + high) / 2 in languages with fixed-size integers, where low + high can overflow; low + (high - low) // 2 avoids it."}
|
| 2 |
+
{"id": "seo-spam", "text": "Best cheap running shoes 2026 - buy cheap running shoes online! Cheap running shoes sale, running shoes discount, best running shoes cheap price. Click here for the best cheap running shoes deals!!! Limited offer: cheap running shoes free shipping. Running shoes cheap, shoes running cheap, cheap shoes for running. Don't miss the best cheap running shoes - BUY NOW and save 70% on cheap running shoes. Top 10 cheap running shoes, cheap running shoes reviews, where to buy cheap running shoes. Visit our store today for cheap running shoes!"}
|
| 3 |
+
{"id": "python-code", "code_language": "Python", "text": "def moving_average(values, window):\n \"\"\"Return the simple moving averages of `values` over `window` consecutive items.\n\n >>> moving_average([1, 2, 3, 4, 5], 2)\n [1.5, 2.5, 3.5, 4.5]\n \"\"\"\n if window <= 0:\n raise ValueError(\"window must be positive\")\n if window > len(values):\n return []\n total = sum(values[:window])\n out = [total / window]\n for i in range(window, len(values)):\n # slide the window: add the new item, drop the oldest one\n total += values[i] - values[i - window]\n out.append(total / window)\n return out\n"}
|
| 4 |
+
{"id": "math-worked-example", "title": "Solving a quadratic equation", "text": "Solve x^2 - 5x + 6 = 0.\n\nWe look for two numbers whose product is 6 and whose sum is 5: these are 2 and 3. Hence x^2 - 5x + 6 = (x - 2)(x - 3).\nA product is zero exactly when one of its factors is zero, so x - 2 = 0 or x - 3 = 0, which gives x = 2 or x = 3.\n\nCheck with the quadratic formula: x = (5 ± sqrt(25 - 24)) / 2 = (5 ± 1) / 2, so x = 3 or x = 2. Substituting x = 2 gives 4 - 10 + 6 = 0, and x = 3 gives 9 - 15 + 6 = 0, as required."}
|
| 5 |
+
{"id": "spanish-water-cycle", "title": "El ciclo del agua", "text": "El ciclo del agua describe cómo el agua se mueve entre los océanos, la atmósfera y la tierra. El calor del sol evapora el agua de los mares y lagos; el vapor sube, se enfría y se condensa en pequeñas gotas que forman las nubes. Cuando las gotas crecen lo suficiente, caen como lluvia o nieve. Parte de esa agua corre por los ríos de vuelta al mar, otra parte se infiltra en el suelo y alimenta los acuíferos, y las plantas devuelven una fracción a la atmósfera por transpiración. Así, la misma agua puede recorrer el ciclo muchas veces a lo largo de los siglos."}
|
| 6 |
+
{"id": "cookie-banner", "text": "Home | About us | Products | Contact | Login\n\nWe use cookies to improve your experience. By continuing to browse this site you agree to our use of cookies. Accept all | Reject | Settings\n\nShare on Facebook | Share on X | Share by email\n\nPrivacy policy | Terms of use | Sitemap | Copyright 2026 Example Store. All rights reserved."}
|
heads.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:63f3a0a283d878dce4b172545b54d83e75c5aca91f2c944247b2413fee32c0f9
|
| 3 |
+
size 275868
|
images/source1_benchmark.png
ADDED
|
Git LFS Details
|
model.fp32.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d1f6ff1712c30b9059e37d202817d57d7164678a2ac01292dd4f4904612a2454
|
| 3 |
+
size 1227771776
|
model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8f33cf7774481046edc9fb1159ee9748c34aaa5532232fa8f5f2d6daf142e640
|
| 3 |
+
size 613892480
|
requirements.txt
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# source1.py needs Python >= 3.10 and these packages. The minimum versions are the oldest this release was tested
|
| 2 |
+
# with (torch 2.11 and 2.14, transformers 5.17), with both weight files: model.safetensors (bfloat16, the default)
|
| 3 |
+
# and model.fp32.safetensors (float32). Older versions may work but were not tested. No GPU is needed: on a CPU, or a
|
| 4 |
+
# GPU without native bfloat16, the bfloat16 weights are upcast to float32 when loaded.
|
| 5 |
+
torch>=2.11
|
| 6 |
+
transformers>=5.17
|
| 7 |
+
safetensors>=0.8
|
| 8 |
+
tokenizers>=0.23
|
source1.json
ADDED
|
@@ -0,0 +1,302 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architecture": "Source1ForScoring",
|
| 3 |
+
"harness_version": "0.1.0",
|
| 4 |
+
"pooling": "mean",
|
| 5 |
+
"max_length": 8192,
|
| 6 |
+
"input_format": "classifier",
|
| 7 |
+
"layout": [
|
| 8 |
+
[
|
| 9 |
+
"format",
|
| 10 |
+
"label_single",
|
| 11 |
+
10
|
| 12 |
+
],
|
| 13 |
+
[
|
| 14 |
+
"topic",
|
| 15 |
+
"label_single",
|
| 16 |
+
15
|
| 17 |
+
],
|
| 18 |
+
[
|
| 19 |
+
"content_type",
|
| 20 |
+
"label_single",
|
| 21 |
+
4
|
| 22 |
+
],
|
| 23 |
+
[
|
| 24 |
+
"educational_value",
|
| 25 |
+
"score",
|
| 26 |
+
6
|
| 27 |
+
],
|
| 28 |
+
[
|
| 29 |
+
"reasoning_depth",
|
| 30 |
+
"score",
|
| 31 |
+
6
|
| 32 |
+
],
|
| 33 |
+
[
|
| 34 |
+
"writing_quality",
|
| 35 |
+
"score",
|
| 36 |
+
6
|
| 37 |
+
],
|
| 38 |
+
[
|
| 39 |
+
"information_density",
|
| 40 |
+
"score",
|
| 41 |
+
6
|
| 42 |
+
],
|
| 43 |
+
[
|
| 44 |
+
"reliability",
|
| 45 |
+
"score",
|
| 46 |
+
6
|
| 47 |
+
],
|
| 48 |
+
[
|
| 49 |
+
"spam_seo",
|
| 50 |
+
"score",
|
| 51 |
+
6
|
| 52 |
+
],
|
| 53 |
+
[
|
| 54 |
+
"boilerplate",
|
| 55 |
+
"score",
|
| 56 |
+
6
|
| 57 |
+
],
|
| 58 |
+
[
|
| 59 |
+
"toxicity",
|
| 60 |
+
"score",
|
| 61 |
+
6
|
| 62 |
+
],
|
| 63 |
+
[
|
| 64 |
+
"code_quality",
|
| 65 |
+
"score",
|
| 66 |
+
6
|
| 67 |
+
],
|
| 68 |
+
[
|
| 69 |
+
"math_quality",
|
| 70 |
+
"score",
|
| 71 |
+
6
|
| 72 |
+
]
|
| 73 |
+
],
|
| 74 |
+
"schema_fingerprint": "7c8ff0235cbc7416",
|
| 75 |
+
"layout_fingerprint": "3df647d7f4bd5ee0",
|
| 76 |
+
"schema": {
|
| 77 |
+
"name": "source-1",
|
| 78 |
+
"version": 1,
|
| 79 |
+
"labels": {
|
| 80 |
+
"format": {
|
| 81 |
+
"short": "fmt",
|
| 82 |
+
"kind": "single",
|
| 83 |
+
"description": "What kind of document this is.",
|
| 84 |
+
"values": {
|
| 85 |
+
"tutorial": "Step-by-step instruction, how-to guides, lessons, worked walkthroughs.",
|
| 86 |
+
"reference": "Documentation, manuals, API references, specifications, encyclopedia or wiki entries.",
|
| 87 |
+
"news": "News reporting, press releases, announcements of current events.",
|
| 88 |
+
"forum_qa": "Forum threads, Q&A pages, comment sections, mailing-list discussions.",
|
| 89 |
+
"academic": "Research papers, theses, textbooks, lecture notes, scholarly writing.",
|
| 90 |
+
"fiction": "Stories, novels, poetry, screenplays, fan fiction and other creative writing.",
|
| 91 |
+
"code_file": "A source-code file or notebook - code with or without comments.",
|
| 92 |
+
"product_page": "Product listings, shopping pages, marketing or landing pages, ads.",
|
| 93 |
+
"blog_opinion": "Personal blogs, essays, opinion pieces, newsletters, reviews.",
|
| 94 |
+
"other": "Anything else - legal text, transcripts, lists, records, fragments."
|
| 95 |
+
}
|
| 96 |
+
},
|
| 97 |
+
"topic": {
|
| 98 |
+
"short": "top",
|
| 99 |
+
"kind": "single",
|
| 100 |
+
"description": "The main subject of the document.",
|
| 101 |
+
"values": {
|
| 102 |
+
"science": "Natural sciences - physics, chemistry, biology, earth and space science.",
|
| 103 |
+
"technology": "Engineering, hardware, electronics, IT operations, applied technology.",
|
| 104 |
+
"programming": "Software development, computer science, code and developer tooling.",
|
| 105 |
+
"math": "Mathematics, statistics and formal logic.",
|
| 106 |
+
"health": "Medicine, health, nutrition, fitness and mental health.",
|
| 107 |
+
"finance": "Finance, economics, business, investing, accounting and careers.",
|
| 108 |
+
"history": "History, archaeology and historical biography.",
|
| 109 |
+
"politics_law": "Politics, government, law, public policy and current affairs.",
|
| 110 |
+
"society": "Social science, education, culture, relationships and psychology.",
|
| 111 |
+
"philosophy_religion": "Philosophy, ethics, religion and spirituality.",
|
| 112 |
+
"arts_entertainment": "Art, music, film, television, games, celebrities and pop culture.",
|
| 113 |
+
"literature": "Literature, language, linguistics, and writing about books and writing.",
|
| 114 |
+
"sports": "Sports and athletics.",
|
| 115 |
+
"lifestyle": "Food, travel, home, fashion, hobbies, parenting and pets.",
|
| 116 |
+
"other": "Anything else, or no clear subject."
|
| 117 |
+
}
|
| 118 |
+
},
|
| 119 |
+
"content_type": {
|
| 120 |
+
"short": "ctype",
|
| 121 |
+
"kind": "single",
|
| 122 |
+
"description": "What the text is made of.",
|
| 123 |
+
"values": {
|
| 124 |
+
"plain_text": "Prose or other natural-language text with no significant code or math.",
|
| 125 |
+
"text_with_code": "Natural-language text that includes code snippets or code blocks.",
|
| 126 |
+
"code_only": "Source code with, at most, comments and docstrings.",
|
| 127 |
+
"math_heavy": "Text dominated by equations, formulas, proofs or worked calculations."
|
| 128 |
+
}
|
| 129 |
+
}
|
| 130 |
+
},
|
| 131 |
+
"quality": {
|
| 132 |
+
"educational_value": {
|
| 133 |
+
"short": "edu",
|
| 134 |
+
"weight": 0.3,
|
| 135 |
+
"description": "Does it teach something useful?",
|
| 136 |
+
"rubric": {
|
| 137 |
+
"0": "Teaches nothing - spam, ads, navigation, gibberish, or text with no informational purpose.",
|
| 138 |
+
"1": "Almost nothing to learn - a few incidental facts buried in promotional, personal or trivial text.",
|
| 139 |
+
"2": "Some useful information, but superficial, fragmentary, or mixed with a lot of irrelevant material.",
|
| 140 |
+
"3": "Useful and coherent - conveys real knowledge or skills, though without much depth or completeness.",
|
| 141 |
+
"4": "Clearly educational - explains concepts, methods or knowledge well enough to learn from, with minor gaps.",
|
| 142 |
+
"5": "Outstanding teaching material - deep, clear and well structured, with examples or worked solutions; comparable to an excellent textbook or expert tutorial."
|
| 143 |
+
}
|
| 144 |
+
},
|
| 145 |
+
"reasoning_depth": {
|
| 146 |
+
"short": "rsn",
|
| 147 |
+
"weight": 0.2,
|
| 148 |
+
"description": "Does it explain why and walk through steps, or just state facts?",
|
| 149 |
+
"rubric": {
|
| 150 |
+
"0": "No reasoning at all - fragments, lists, boilerplate or non-content.",
|
| 151 |
+
"1": "Bare assertions or opinions with no explanation.",
|
| 152 |
+
"2": "Occasional explanation, but mostly unsupported statements; steps are skipped.",
|
| 153 |
+
"3": "Explains the why behind key points, with some step-by-step structure or argument.",
|
| 154 |
+
"4": "Consistent, explicit reasoning - derivations, cause and effect, justified arguments, worked examples.",
|
| 155 |
+
"5": "Rigorous multi-step reasoning throughout - proofs, careful derivations, or thorough analysis that motivates every step and weighs alternatives."
|
| 156 |
+
}
|
| 157 |
+
},
|
| 158 |
+
"writing_quality": {
|
| 159 |
+
"short": "wrt",
|
| 160 |
+
"weight": 0.15,
|
| 161 |
+
"description": "Is it clear, well organized and coherent? For code, judge naming, structure and comments.",
|
| 162 |
+
"rubric": {
|
| 163 |
+
"0": "Unreadable - garbled, machine-generated nonsense, broken encoding, keyword soup.",
|
| 164 |
+
"1": "Very poor - frequent errors, incoherent structure, fragments; hard to follow.",
|
| 165 |
+
"2": "Below average - understandable but disorganized, repetitive or error-prone.",
|
| 166 |
+
"3": "Adequate - clear and coherent with minor issues.",
|
| 167 |
+
"4": "Good - well organized, fluent, precise and sensibly structured.",
|
| 168 |
+
"5": "Excellent - exemplary, publication-quality writing."
|
| 169 |
+
}
|
| 170 |
+
},
|
| 171 |
+
"information_density": {
|
| 172 |
+
"short": "dns",
|
| 173 |
+
"weight": 0.2,
|
| 174 |
+
"description": "How much real content per word, versus padding and filler?",
|
| 175 |
+
"rubric": {
|
| 176 |
+
"0": "No real content - filler, repetition, boilerplate.",
|
| 177 |
+
"1": "Mostly padding - long intros, fluff, repeated phrases or SEO filler around a little content.",
|
| 178 |
+
"2": "Noticeable padding or digressions; it could be far shorter without losing anything.",
|
| 179 |
+
"3": "Reasonable - mostly on point, with some filler.",
|
| 180 |
+
"4": "Dense - little wasted text; most sentences carry information.",
|
| 181 |
+
"5": "Very dense yet readable - every sentence adds substance."
|
| 182 |
+
}
|
| 183 |
+
},
|
| 184 |
+
"reliability": {
|
| 185 |
+
"short": "rel",
|
| 186 |
+
"weight": 0.15,
|
| 187 |
+
"description": "Does it look careful and trustworthy, or sloppy and made up? For fiction and other creative writing, judge care and internal consistency; invented events are not a reliability problem.",
|
| 188 |
+
"rubric": {
|
| 189 |
+
"0": "Fabricated, nonsensical or deceptive - scams, fake news, conspiracy content, generated nonsense.",
|
| 190 |
+
"1": "Largely unreliable - many errors, sensational or unsupported claims, misleading framing.",
|
| 191 |
+
"2": "Questionable - some errors or unsupported claims; careless.",
|
| 192 |
+
"3": "Generally plausible and consistent; no obvious errors, but informal or unverifiable.",
|
| 193 |
+
"4": "Careful and accurate - precise, consistent claims; shows its work or cites sources.",
|
| 194 |
+
"5": "Authoritative - expert-level accuracy, well sourced, carefully qualified claims."
|
| 195 |
+
}
|
| 196 |
+
}
|
| 197 |
+
},
|
| 198 |
+
"red_flags": {
|
| 199 |
+
"spam_seo": {
|
| 200 |
+
"short": "spam",
|
| 201 |
+
"description": "Keyword stuffing, clickbait, affiliate filler, and other text written to rank or sell rather than to inform.",
|
| 202 |
+
"penalty": {
|
| 203 |
+
"threshold": 1,
|
| 204 |
+
"weight": 0.5
|
| 205 |
+
},
|
| 206 |
+
"rubric": {
|
| 207 |
+
"0": "None.",
|
| 208 |
+
"1": "Minor promotion - a call to action or a brief ad in otherwise genuine content.",
|
| 209 |
+
"2": "Noticeable promotion - repeated calls to action, affiliate links, marketing tone.",
|
| 210 |
+
"3": "Substantial - the text mainly exists to promote, sell or rank; visible keyword repetition.",
|
| 211 |
+
"4": "Mostly spam - keyword stuffing, clickbait, thin affiliate or doorway content.",
|
| 212 |
+
"5": "Pure spam - auto-generated SEO text, scams, keyword lists."
|
| 213 |
+
}
|
| 214 |
+
},
|
| 215 |
+
"boilerplate": {
|
| 216 |
+
"short": "boil",
|
| 217 |
+
"description": "Templates, auto-generated pages, lists of links, cookie banners, menus and other non-content text.",
|
| 218 |
+
"penalty": {
|
| 219 |
+
"threshold": 1,
|
| 220 |
+
"weight": 0.4
|
| 221 |
+
},
|
| 222 |
+
"rubric": {
|
| 223 |
+
"0": "None - all of the text is real content.",
|
| 224 |
+
"1": "A little - a stray header, footer or copyright line.",
|
| 225 |
+
"2": "Noticeable - navigation, cookie or legal text, or share-button text mixed in (roughly 10-25% of the text).",
|
| 226 |
+
"3": "Substantial - roughly 25-50% templates, menus, link lists or legal notices.",
|
| 227 |
+
"4": "Mostly boilerplate - over half of the text is templated or navigational.",
|
| 228 |
+
"5": "Entirely boilerplate - auto-generated pages, link lists, cookie banners, error or index pages."
|
| 229 |
+
}
|
| 230 |
+
},
|
| 231 |
+
"toxicity": {
|
| 232 |
+
"short": "tox",
|
| 233 |
+
"aggregate": "max",
|
| 234 |
+
"description": "Hate, harassment and explicit content. Rate the text's own toxicity, not its subject: scholarly, journalistic or educational discussion of hate, violence or sexuality is not toxic.",
|
| 235 |
+
"penalty": {
|
| 236 |
+
"threshold": 1,
|
| 237 |
+
"weight": 0.8
|
| 238 |
+
},
|
| 239 |
+
"rubric": {
|
| 240 |
+
"0": "None.",
|
| 241 |
+
"1": "Mild - occasional profanity or rudeness, not aimed at anyone.",
|
| 242 |
+
"2": "Moderate - insults, crude humor, mild sexual references, heated hostility.",
|
| 243 |
+
"3": "Significant - harassment, demeaning stereotypes, graphic violence, or partly explicit sexual content.",
|
| 244 |
+
"4": "Severe - hate speech, targeted harassment, or sexually explicit material as the main content.",
|
| 245 |
+
"5": "Extreme - violent extremism, dehumanizing hate, incitement, or sexual content involving minors."
|
| 246 |
+
}
|
| 247 |
+
}
|
| 248 |
+
},
|
| 249 |
+
"gated": {
|
| 250 |
+
"code_quality": {
|
| 251 |
+
"short": "code",
|
| 252 |
+
"description": "Is the code readable, correct-looking and documented, and written by a person rather than generated?",
|
| 253 |
+
"applies_when": {
|
| 254 |
+
"content_type": [
|
| 255 |
+
"text_with_code",
|
| 256 |
+
"code_only"
|
| 257 |
+
],
|
| 258 |
+
"format": [
|
| 259 |
+
"code_file"
|
| 260 |
+
]
|
| 261 |
+
},
|
| 262 |
+
"rubric": {
|
| 263 |
+
"0": "Not usable code - garbled, minified, obfuscated or broken beyond repair.",
|
| 264 |
+
"1": "Very poor - likely non-functional fragments, no structure, or auto-generated boilerplate.",
|
| 265 |
+
"2": "Poor - may work but messy; unclear names, no comments, hard-coded values.",
|
| 266 |
+
"3": "Acceptable - readable and plausibly correct, with some structure and minimal documentation.",
|
| 267 |
+
"4": "Good - clean, idiomatic, well named and documented; plausibly correct, with edge cases handled.",
|
| 268 |
+
"5": "Excellent - exemplary, production-quality, well-documented and instructive code."
|
| 269 |
+
}
|
| 270 |
+
},
|
| 271 |
+
"math_quality": {
|
| 272 |
+
"short": "math",
|
| 273 |
+
"description": "Is the notation correct, are the steps shown, and do the solutions follow logically?",
|
| 274 |
+
"applies_when": {
|
| 275 |
+
"content_type": [
|
| 276 |
+
"math_heavy"
|
| 277 |
+
],
|
| 278 |
+
"topic": [
|
| 279 |
+
"math"
|
| 280 |
+
]
|
| 281 |
+
},
|
| 282 |
+
"rubric": {
|
| 283 |
+
"0": "Garbled math - broken notation or nonsense.",
|
| 284 |
+
"1": "Mostly wrong or incoherent; results asserted without work.",
|
| 285 |
+
"2": "Some correct math, but errors, sloppy notation, or skipped steps that break the logic.",
|
| 286 |
+
"3": "Generally correct, with standard notation and the key steps shown.",
|
| 287 |
+
"4": "Correct, clean notation, complete steps and a clear logical flow.",
|
| 288 |
+
"5": "Rigorous and elegant - precise notation, every step justified, solutions easy to verify."
|
| 289 |
+
}
|
| 290 |
+
}
|
| 291 |
+
},
|
| 292 |
+
"composite": {
|
| 293 |
+
"gated_weight": 0.2,
|
| 294 |
+
"hard_filters": [
|
| 295 |
+
"toxicity >= 4",
|
| 296 |
+
"spam_seo >= 4",
|
| 297 |
+
"boilerplate >= 4.5"
|
| 298 |
+
]
|
| 299 |
+
}
|
| 300 |
+
},
|
| 301 |
+
"text_normalize": "collapse_spaces"
|
| 302 |
+
}
|
source1.py
ADDED
|
@@ -0,0 +1,1011 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
# Copyright 2026 The Source-1 Authors
|
| 3 |
+
# SPDX-License-Identifier: Apache-2.0
|
| 4 |
+
"""Source-1: score text as language-model pretraining data.
|
| 5 |
+
|
| 6 |
+
A standalone loader and scorer for Source-1, an mmBERT-base encoder fine-tuned with 13 scoring heads. It needs
|
| 7 |
+
only torch, transformers, safetensors and tokenizers (no remote code, no other project code), and runs on a GPU or a
|
| 8 |
+
CPU.
|
| 9 |
+
|
| 10 |
+
from source1 import Source1
|
| 11 |
+
|
| 12 |
+
model = Source1.from_pretrained("path/to/Source-1") # a local directory or a Hugging Face repo id
|
| 13 |
+
model = Source1.from_pretrained("path/to/Source-1", precision="fp32") # the full-precision weights instead
|
| 14 |
+
doc = model.score("Photosynthesis is how plants ...", title="Photosynthesis")
|
| 15 |
+
doc["overall"], doc["keep"], doc["educational_value"]
|
| 16 |
+
docs = model.score_batch(["first document", {"text": "second document", "title": "A title"}])
|
| 17 |
+
|
| 18 |
+
python source1.py --input docs.jsonl --output scores.jsonl # one JSON object per line, "text" field
|
| 19 |
+
python source1.py --input page.txt # a .txt file is one document
|
| 20 |
+
|
| 21 |
+
Output: one flat dict per document
|
| 22 |
+
----------------------------------
|
| 23 |
+
format, topic, content_type labels: the most likely value (10, 15 and 4 values; see source1.json)
|
| 24 |
+
educational_value, reasoning_depth, writing_quality, information_density, reliability
|
| 25 |
+
quality scores 0-5, higher is better
|
| 26 |
+
spam_seo, boilerplate, toxicity
|
| 27 |
+
red flags 0-5, higher is worse
|
| 28 |
+
code_quality, math_quality gated scores 0-5, or None when the document is not code / not math
|
| 29 |
+
overall composite 0-5: weighted quality, blended with the gated scores that apply, minus
|
| 30 |
+
red-flag penalties (see ``composite``)
|
| 31 |
+
keep False when the drop line of calibration.json matches (see ``DropLine``)
|
| 32 |
+
drop_reasons the drop-line conditions that matched ([] when kept)
|
| 33 |
+
parts, tokens, truncated chunks the document was split into, its length in tokens, whether any chunk was
|
| 34 |
+
longer than the model's 8,192 tokens and was cut
|
| 35 |
+
chunks only when parts > 1: the same fields per chunk, with its part number, its character
|
| 36 |
+
span in the cleaned text (``clean_text``) and its token count
|
| 37 |
+
label_dist, ranges only when parts > 1: each label value's token share; each score's [min, max]
|
| 38 |
+
|
| 39 |
+
Scores are expected values (the six levels 0-5 weighted by their probabilities), so they are fractional, rounded
|
| 40 |
+
to 3 decimals like every other number here. The scores are the model's own; ``apply_offsets=True`` adds the small
|
| 41 |
+
calibration offsets of calibration.json to the quality scores (about 0.02 at most; off by default).
|
| 42 |
+
|
| 43 |
+
How a document is scored (the same steps the model was trained and evaluated with)
|
| 44 |
+
------------------------------------------------------------------------------------
|
| 45 |
+
1. ``clean_text``: line endings to "\\n", control characters dropped, Unicode NFC, trailing spaces dropped, at most
|
| 46 |
+
two blank lines in a row.
|
| 47 |
+
2. ``split_text``: a document longer than 7,808 tokens is split into balanced chunks, each ending at the most
|
| 48 |
+
natural boundary near its ideal end (headings, then paragraphs, lines, sentences, spaces; definitions in code).
|
| 49 |
+
3. ``build_input``: each chunk gets a one-line header, a blank line, then the chunk text:
|
| 50 |
+
Source: dataset record | Title: <title> | Part 2 of 3 of a longer document
|
| 51 |
+
Every training input had a Source line, almost always "dataset record", so that is the default; code files had
|
| 52 |
+
"<language> source file" (pass ``code_language="Python"``). Title is added when given, Part when the document has
|
| 53 |
+
more than one chunk. ``url`` is accepted but not shown to the model unless ``show_url=True``: no training input had
|
| 54 |
+
one. On 1,261 held-out benchmark chunks (495 graded blind-set chunks from the test split and 766 exam chunks),
|
| 55 |
+
dropping the whole header moved overall by 0.03 on average (at most 0.655), dropping a title by 0.05 on the chunks
|
| 56 |
+
that had one; adding a URL moved it by up to 0.7 and did not improve its rank agreement with the independent
|
| 57 |
+
graders of the model card's evaluation.
|
| 58 |
+
4. ``collapse_spaces``: runs of spaces and tabs become one space; at most two empty lines in a row.
|
| 59 |
+
5. Tokenized with <bos> and <eos>, at most 8,192 tokens (a longer input is cut at its end).
|
| 60 |
+
6. The final hidden states are mean-pooled over the tokens; one linear head per field.
|
| 61 |
+
7. A document's chunks are combined by token-weighted vote (labels) and token-weighted mean (scores; toxicity takes
|
| 62 |
+
the maximum; gated scores average over the chunks where they apply). ``overall`` and ``keep`` are then computed
|
| 63 |
+
on the combined scores.
|
| 64 |
+
|
| 65 |
+
Weights and precision
|
| 66 |
+
---------------------
|
| 67 |
+
Two copies of the backbone weights: ``model.safetensors`` in bfloat16 (the default, ``precision="bf16"``, half the
|
| 68 |
+
size) and ``model.fp32.safetensors`` in float32 (``precision="fp32"``, the full-precision copy). ``precision`` picks
|
| 69 |
+
the file; ``dtype`` picks what the model computes in. By default (``dtype="auto"``) it computes in bfloat16 on a GPU
|
| 70 |
+
with native bfloat16 (NVIDIA Ampere and newer), as in the project's own evaluation, and in float32 on older GPUs and on
|
| 71 |
+
a CPU, where bfloat16 weights are upcast to float32 (pass ``dtype="bf16"`` to compute in bfloat16 there too). Computing
|
| 72 |
+
in bfloat16, both files give the same scores, because the bfloat16 file holds exactly the float32 weights rounded to
|
| 73 |
+
bfloat16. Computing in float32, the bfloat16 weights move scores slightly away from the float32 weights' (on the same
|
| 74 |
+
1,261 benchmark chunks: up to 0.03 on overall and 0.08 on a single field, 3 labels and 1 keep decision changed); use
|
| 75 |
+
``precision="fp32", dtype="fp32"`` for full float32. float16 is refused: the
|
| 76 |
+
mean pooling overflows its range on long inputs and gives NaN scores. In bfloat16 a chunk's scores depend slightly on
|
| 77 |
+
which other inputs share its batch (scoring each benchmark chunk alone instead of in the default batches moved overall
|
| 78 |
+
by up to 0.04 and a single field by up to 0.10; 3 labels changed, no keep decision), and float32 differs from bfloat16
|
| 79 |
+
by a similar amount (up to 0.03 on overall and 0.09 on a single field; 4 labels and 1 keep decision changed); for
|
| 80 |
+
scores that do not depend on the batch, compute in float32 (``dtype="fp32"``) or use ``batch_tokens=1`` (one input per
|
| 81 |
+
forward pass).
|
| 82 |
+
"""
|
| 83 |
+
|
| 84 |
+
from __future__ import annotations
|
| 85 |
+
|
| 86 |
+
import argparse
|
| 87 |
+
import ast
|
| 88 |
+
import json
|
| 89 |
+
import math
|
| 90 |
+
import operator
|
| 91 |
+
import re
|
| 92 |
+
import sys
|
| 93 |
+
import time
|
| 94 |
+
import unicodedata
|
| 95 |
+
from bisect import bisect_left
|
| 96 |
+
from collections.abc import Iterable, Iterator, Mapping
|
| 97 |
+
from pathlib import Path
|
| 98 |
+
from typing import Any
|
| 99 |
+
|
| 100 |
+
import torch
|
| 101 |
+
import torch.nn as nn
|
| 102 |
+
|
| 103 |
+
__version__ = "1.0.0"
|
| 104 |
+
|
| 105 |
+
MAX_LENGTH = 8192 # tokens per model input, <bos> and <eos> included
|
| 106 |
+
CHUNK_TOKENS = 7808 # document tokens per chunk: 8,192 minus 384 kept for the header (as in training)
|
| 107 |
+
DEFAULT_SOURCE = "dataset record"
|
| 108 |
+
DEFAULT_BATCH_TOKENS = 65536 # padded tokens per forward pass
|
| 109 |
+
LEVELS = (0, 1, 2, 3, 4, 5)
|
| 110 |
+
# The backbone weights of each precision: bfloat16 (the default) and the full-precision float32 copy. The fp32 file
|
| 111 |
+
# follows transformers' variant naming (model.<variant>.safetensors), so AutoModel loads it with variant="fp32".
|
| 112 |
+
WEIGHTS = {"bf16": "model.safetensors", "fp32": "model.fp32.safetensors"}
|
| 113 |
+
DEFAULT_PRECISION = "bf16"
|
| 114 |
+
FILES = ("config.json", "heads.safetensors", "source1.json", "tokenizer.json") # needed besides the weights
|
| 115 |
+
CALIBRATION = "calibration.json" # the calibrated drop line and offsets; required unless drop_line is given
|
| 116 |
+
# What from_pretrained also downloads for a Hub repo id (with the weights of the chosen precision only): the
|
| 117 |
+
# calibration and the license files.
|
| 118 |
+
HUB_EXTRA = (CALIBRATION, "tokenizer_config.json", "LICENSE", "NOTICE", "AUTHORS", "CREDITS_BOOKS.tsv")
|
| 119 |
+
|
| 120 |
+
|
| 121 |
+
def hub_files(precision: str = DEFAULT_PRECISION) -> tuple[str, ...]:
|
| 122 |
+
"""The files from_pretrained downloads from a Hugging Face repo for ``precision``."""
|
| 123 |
+
return (WEIGHTS[resolve_precision(precision)], *FILES, *HUB_EXTRA)
|
| 124 |
+
|
| 125 |
+
# --------------------------------------------------------------------------------------------- text
|
| 126 |
+
|
| 127 |
+
|
| 128 |
+
_JUNK = re.compile("[\x00-\x08\x0b\x0e-\x1f\x7f\ud800-\udfff\ufeff\u200b\ufffe\uffff]")
|
| 129 |
+
_TRAILING_WS = re.compile(r"[ \t]+\n")
|
| 130 |
+
_BLANK_LINES = re.compile(r"\n{4,}")
|
| 131 |
+
_SPACE_RUN = re.compile(r"[ \t]{2,}")
|
| 132 |
+
_BLANK_RUN = re.compile(r"\n(?:[ \t]*\n){3,}")
|
| 133 |
+
|
| 134 |
+
|
| 135 |
+
def clean_text(text: str) -> str:
|
| 136 |
+
"""The document as the scorer sees it before chunking: "\\n" line endings (a form feed counts as a paragraph
|
| 137 |
+
break), control characters, zero-width spaces and byte-order marks dropped, Unicode NFC, no trailing spaces,
|
| 138 |
+
at most two blank lines in a row, no blank lines at either end."""
|
| 139 |
+
text = text.replace("\r\n", "\n").replace("\r", "\n").replace("\x0c", "\n\n")
|
| 140 |
+
text = _JUNK.sub("", text)
|
| 141 |
+
text = unicodedata.normalize("NFC", text)
|
| 142 |
+
text = _TRAILING_WS.sub("\n", text)
|
| 143 |
+
text = _BLANK_LINES.sub("\n\n\n", text)
|
| 144 |
+
return text.strip("\n").rstrip()
|
| 145 |
+
|
| 146 |
+
|
| 147 |
+
def collapse_spaces(text: str) -> str:
|
| 148 |
+
"""The model's input normalization: runs of 2+ spaces/tabs become one space, 3+ blank lines in a row (lines
|
| 149 |
+
holding only spaces or tabs) become two empty lines; newlines are kept."""
|
| 150 |
+
return _BLANK_RUN.sub("\n\n\n", _SPACE_RUN.sub(" ", text))
|
| 151 |
+
|
| 152 |
+
|
| 153 |
+
_SURROGATES = re.compile("[\ud800-\udfff]")
|
| 154 |
+
|
| 155 |
+
|
| 156 |
+
def _from_bytes(x: Any) -> Any:
|
| 157 |
+
"""bytes and bytearray decoded as UTF-8 (invalid bytes become U+FFFD); anything else unchanged."""
|
| 158 |
+
return x.decode("utf-8", errors="replace") if isinstance(x, (bytes, bytearray)) else x
|
| 159 |
+
|
| 160 |
+
|
| 161 |
+
def _one_line(s: Any, limit: int) -> str:
|
| 162 |
+
"""A header value on one line: lone surrogates (which cannot be tokenized) dropped, every run of whitespace
|
| 163 |
+
(newlines included) made one space, cut to ``limit`` characters."""
|
| 164 |
+
s = re.sub(r"\s+", " ", _SURROGATES.sub("", str(_from_bytes(s)))).strip()
|
| 165 |
+
return s if len(s) <= limit else s[: limit - 1] + "\u2026"
|
| 166 |
+
|
| 167 |
+
|
| 168 |
+
def source_description(source_type: str | None = None, code_language: str | None = None) -> str:
|
| 169 |
+
"""The header's Source value: ``source_type`` when given ("" leaves the Source line out), else
|
| 170 |
+
"<code_language> source file" for code, else "dataset record"."""
|
| 171 |
+
if source_type is not None:
|
| 172 |
+
return str(source_type)
|
| 173 |
+
if code_language:
|
| 174 |
+
return f"{code_language} source file"
|
| 175 |
+
return DEFAULT_SOURCE
|
| 176 |
+
|
| 177 |
+
|
| 178 |
+
def build_input(chunk: str, *, source_type: str | None = DEFAULT_SOURCE, title: str | None = None,
|
| 179 |
+
url: str | None = None, part: int = 1, parts: int = 1) -> str:
|
| 180 |
+
"""One chunk as the model reads it (before ``collapse_spaces``): a header line, a blank line, the chunk.
|
| 181 |
+
|
| 182 |
+
>>> build_input("Text.", title="On rivers", part=2, parts=3)
|
| 183 |
+
'Source: dataset record | Title: On rivers | Part 2 of 3 of a longer document\\n\\nText.'
|
| 184 |
+
"""
|
| 185 |
+
head = []
|
| 186 |
+
if source_type:
|
| 187 |
+
head.append(f"Source: {_one_line(source_type, 200)}")
|
| 188 |
+
if title:
|
| 189 |
+
head.append(f"Title: {_one_line(title, 200)}")
|
| 190 |
+
if url:
|
| 191 |
+
head.append(f"URL: {_one_line(url, 300)}")
|
| 192 |
+
if parts > 1:
|
| 193 |
+
head.append(f"Part {part} of {parts} of a longer document")
|
| 194 |
+
line = " | ".join(head)
|
| 195 |
+
return f"{line}\n\n{chunk}" if line else chunk
|
| 196 |
+
|
| 197 |
+
|
| 198 |
+
# --------------------------------------------------------------------------------------------- chunking
|
| 199 |
+
|
| 200 |
+
# Cost of splitting at each boundary level; the distance from the ideal point (as a fraction of the target
|
| 201 |
+
# chunk size) is added to it.
|
| 202 |
+
_LEVEL_COST = (0.0, 0.12, 0.3, 0.5, 0.8)
|
| 203 |
+
_HARD_CUT_COST = 2.0
|
| 204 |
+
_HEADING = re.compile(
|
| 205 |
+
r"\n(?=#{1,6} |(?:chapter|CHAPTER|Chapter|PART|Part|BOOK|Book|SECTION|Section|ACT|Act"
|
| 206 |
+
r"|Kapitel|KAPITEL|Chapitre|CHAPITRE|Cap[ií]tulo|CAP[IÍ]TULO|Capitolo|CAPITOLO|Глава|ГЛАВА|Rozdział|ROZDZIAŁ)\b[^\n]{0,80}\n"
|
| 207 |
+
r"|第[一二三四五六七八九十百千〇零0-9]+[章节節回卷部篇][^\n]{0,80}\n"
|
| 208 |
+
r"|[=\-*_]{3,}[ \t]*\n|\x0c|\\(?:chapter|section|subsection)\b)"
|
| 209 |
+
)
|
| 210 |
+
_CODE_DEF = re.compile(
|
| 211 |
+
r"\n(?=(?:def |async def |class |function |func |fn |pub |impl |struct |enum |interface |trait |type |"
|
| 212 |
+
r"module |package |public |private |protected |internal |static |export |const |let |var |@|#include|# ?%%))"
|
| 213 |
+
)
|
| 214 |
+
_PARAGRAPH = re.compile(r"\n[ \t]*\n+")
|
| 215 |
+
_LINE = re.compile(r"\n")
|
| 216 |
+
_SENTENCE = re.compile(r"[.!?…।॥۔؟։።။។៕][\"'”’)\]»]*\s+|[。!?][」』”’)\]]*")
|
| 217 |
+
_SPACE = re.compile(r"\s+")
|
| 218 |
+
|
| 219 |
+
|
| 220 |
+
def _joins_previous(ch: str) -> bool:
|
| 221 |
+
"""Characters that belong to the one before them: combining marks, ZWJ, variation selectors, skin tones."""
|
| 222 |
+
o = ord(ch)
|
| 223 |
+
return (unicodedata.category(ch) in ("Mn", "Mc", "Me") or o == 0x200D or 0xFE00 <= o <= 0xFE0F
|
| 224 |
+
or 0xE0100 <= o <= 0xE01EF or 0x1F3FB <= o <= 0x1F3FF)
|
| 225 |
+
|
| 226 |
+
|
| 227 |
+
def split_text(text: str, offsets: list[int], max_tokens: int = CHUNK_TOKENS,
|
| 228 |
+
is_code: bool = False) -> list[tuple[int, int, int]]:
|
| 229 |
+
"""Balanced chunks of ``text``: ``(start_char, end_char, tokens)`` each, ``offsets`` being the start character
|
| 230 |
+
of every token. A 9k-token document becomes two ~4.5k chunks, not 7.8k plus 1.2k; each chunk ends at the
|
| 231 |
+
cheapest boundary near its ideal end (headings or code definitions, paragraphs, lines, sentences, spaces), with
|
| 232 |
+
a hard cut only as a last resort, never inside a character's combining marks or an emoji sequence."""
|
| 233 |
+
n = len(offsets)
|
| 234 |
+
if not text:
|
| 235 |
+
return []
|
| 236 |
+
if n <= max_tokens:
|
| 237 |
+
return [(0, len(text), n)]
|
| 238 |
+
levels = [_CODE_DEF if is_code else _HEADING, _PARAGRAPH, _LINE, None if is_code else _SENTENCE, _SPACE]
|
| 239 |
+
spans: list[tuple[int, int, int]] = []
|
| 240 |
+
s_tok, s_char = 0, 0
|
| 241 |
+
while n - s_tok > max_tokens:
|
| 242 |
+
remaining = n - s_tok
|
| 243 |
+
k = math.ceil(remaining / max_tokens)
|
| 244 |
+
target = remaining / k
|
| 245 |
+
ideal = s_tok + target
|
| 246 |
+
hard = s_tok + max_tokens # the chunk ends at or before token `hard`
|
| 247 |
+
lo_tok = max(s_tok + max(1, int(target * 0.5)), n - (k - 1) * max_tokens)
|
| 248 |
+
lo_char, hi_char = int(offsets[lo_tok]), int(offsets[hard])
|
| 249 |
+
best: tuple[float, int, int] | None = None
|
| 250 |
+
for level, pattern in enumerate(levels):
|
| 251 |
+
if best is not None and _LEVEL_COST[level] >= best[0]:
|
| 252 |
+
break # nothing at this level or later can beat the current best
|
| 253 |
+
if pattern is None:
|
| 254 |
+
continue
|
| 255 |
+
for m in pattern.finditer(text, max(lo_char - 1, 0), min(len(text), hi_char + 256)):
|
| 256 |
+
pos = m.end()
|
| 257 |
+
if not lo_char <= pos <= hi_char:
|
| 258 |
+
continue
|
| 259 |
+
tok = bisect_left(offsets, pos)
|
| 260 |
+
if not s_tok < tok <= hard:
|
| 261 |
+
continue
|
| 262 |
+
cost = _LEVEL_COST[level] + abs(tok - ideal) / target
|
| 263 |
+
if best is None or cost < best[0]:
|
| 264 |
+
best = (cost, pos, tok)
|
| 265 |
+
if best is None or best[0] >= _HARD_CUT_COST:
|
| 266 |
+
c, t = hi_char, hard
|
| 267 |
+
while c > s_char + 1 and c < len(text) and (_joins_previous(text[c]) or text[c - 1] == "\u200d"):
|
| 268 |
+
c -= 1
|
| 269 |
+
if c != hi_char:
|
| 270 |
+
t2 = bisect_left(offsets, c)
|
| 271 |
+
if t2 > s_tok:
|
| 272 |
+
t = t2
|
| 273 |
+
else:
|
| 274 |
+
c = hi_char
|
| 275 |
+
best = (_HARD_CUT_COST, c, t)
|
| 276 |
+
_, split_char, split_tok = best
|
| 277 |
+
spans.append((s_char, split_char, split_tok - s_tok))
|
| 278 |
+
s_tok, s_char = split_tok, split_char
|
| 279 |
+
spans.append((s_char, len(text), n - s_tok))
|
| 280 |
+
return spans
|
| 281 |
+
|
| 282 |
+
|
| 283 |
+
def select_chunks(n: int, max_chunks: int) -> list[int]:
|
| 284 |
+
"""Indices of at most ``max_chunks`` evenly spaced chunks out of ``n`` (0 keeps all)."""
|
| 285 |
+
if max_chunks <= 0 or n <= max_chunks:
|
| 286 |
+
return list(range(n))
|
| 287 |
+
if max_chunks == 1:
|
| 288 |
+
return [n // 2]
|
| 289 |
+
last = n - 1
|
| 290 |
+
return sorted({round(i * last / (max_chunks - 1)) for i in range(max_chunks)})
|
| 291 |
+
|
| 292 |
+
|
| 293 |
+
# --------------------------------------------------------------------------------------------- scores
|
| 294 |
+
|
| 295 |
+
_CMP = {ast.Eq: operator.eq, ast.NotEq: operator.ne, ast.Lt: operator.lt, ast.LtE: operator.le,
|
| 296 |
+
ast.Gt: operator.gt, ast.GtE: operator.ge}
|
| 297 |
+
|
| 298 |
+
|
| 299 |
+
class DropLine:
|
| 300 |
+
"""A drop line such as ``toxicity >= 4 or spam_seo >= 3.5 or boilerplate >= 4.5``: field names, numbers,
|
| 301 |
+
comparisons, ``and`` / ``or`` / ``not`` and parentheses. A comparison with a missing score (None) is false.
|
| 302 |
+
A chunk or document is kept when the line does not match."""
|
| 303 |
+
|
| 304 |
+
_NODES = (ast.Expression, ast.BoolOp, ast.And, ast.Or, ast.UnaryOp, ast.Not, ast.USub, ast.Compare, ast.Name,
|
| 305 |
+
ast.Load, ast.Constant, *_CMP)
|
| 306 |
+
|
| 307 |
+
def __init__(self, source: str, names: Iterable[str]):
|
| 308 |
+
self.source = source.strip()
|
| 309 |
+
tree = ast.parse(self.source, mode="eval")
|
| 310 |
+
for node in ast.walk(tree):
|
| 311 |
+
if not isinstance(node, self._NODES):
|
| 312 |
+
raise ValueError(f"{type(node).__name__} is not allowed in a drop line: {source!r}")
|
| 313 |
+
if isinstance(node, ast.Name) and node.id not in set(names):
|
| 314 |
+
raise ValueError(f"unknown name {node.id!r} in drop line {source!r}")
|
| 315 |
+
self.tree = tree.body
|
| 316 |
+
top_or = isinstance(self.tree, ast.BoolOp) and isinstance(self.tree.op, ast.Or)
|
| 317 |
+
self.terms = list(self.tree.values) if top_or else [self.tree]
|
| 318 |
+
|
| 319 |
+
def reasons(self, env: dict) -> list[str]:
|
| 320 |
+
"""The terms of the line that match ``env`` (each ``or`` branch on its own); [] means keep."""
|
| 321 |
+
return [ast.unparse(t) for t in self.terms if self._eval(t, env)]
|
| 322 |
+
|
| 323 |
+
def _eval(self, node: ast.AST, env: dict) -> Any:
|
| 324 |
+
if isinstance(node, ast.Constant):
|
| 325 |
+
return node.value
|
| 326 |
+
if isinstance(node, ast.Name):
|
| 327 |
+
return env.get(node.id)
|
| 328 |
+
if isinstance(node, ast.BoolOp):
|
| 329 |
+
value: Any = isinstance(node.op, ast.And)
|
| 330 |
+
for v in node.values:
|
| 331 |
+
value = self._eval(v, env)
|
| 332 |
+
if bool(value) != isinstance(node.op, ast.And):
|
| 333 |
+
return value
|
| 334 |
+
return value
|
| 335 |
+
if isinstance(node, ast.UnaryOp):
|
| 336 |
+
v = self._eval(node.operand, env)
|
| 337 |
+
if isinstance(node.op, ast.Not):
|
| 338 |
+
return not v
|
| 339 |
+
return None if v is None else -v
|
| 340 |
+
if isinstance(node, ast.Compare):
|
| 341 |
+
left = self._eval(node.left, env)
|
| 342 |
+
for op, comp in zip(node.ops, node.comparators):
|
| 343 |
+
right = self._eval(comp, env)
|
| 344 |
+
if not isinstance(op, (ast.Eq, ast.NotEq)) and (left is None or right is None):
|
| 345 |
+
return False
|
| 346 |
+
try:
|
| 347 |
+
if not _CMP[type(op)](left, right):
|
| 348 |
+
return False
|
| 349 |
+
except TypeError:
|
| 350 |
+
return False
|
| 351 |
+
left = right
|
| 352 |
+
return True
|
| 353 |
+
raise ValueError(f"unsupported drop-line node {type(node).__name__}")
|
| 354 |
+
|
| 355 |
+
|
| 356 |
+
def _r3(x: float | None) -> float | None:
|
| 357 |
+
return None if x is None or (isinstance(x, float) and math.isnan(x)) else round(float(x), 3)
|
| 358 |
+
|
| 359 |
+
|
| 360 |
+
def composite(schema: dict, scores: dict) -> float | None:
|
| 361 |
+
"""The overall score (0-5) of a flat score dict, by the schema in source1.json: the weighted mean of the quality
|
| 362 |
+
scores; when gated scores apply, 80% of that plus 20% of their mean; minus, for each red flag, its penalty
|
| 363 |
+
weight times how far it is above its threshold (spam_seo 0.5, boilerplate 0.4, toxicity 0.8, each above 1);
|
| 364 |
+
clipped to 0-5."""
|
| 365 |
+
q_num = q_den = 0.0
|
| 366 |
+
for name, spec in schema["quality"].items():
|
| 367 |
+
v, w = scores.get(name), float(spec.get("weight", 1.0))
|
| 368 |
+
if v is not None and w > 0:
|
| 369 |
+
q_num += w * float(v)
|
| 370 |
+
q_den += w
|
| 371 |
+
if q_den <= 0:
|
| 372 |
+
return None
|
| 373 |
+
quality = q_num / q_den
|
| 374 |
+
gated = [float(scores[n]) for n in schema.get("gated", {}) if scores.get(n) is not None]
|
| 375 |
+
gw = float((schema.get("composite") or {}).get("gated_weight", 0.2))
|
| 376 |
+
if gated and gw > 0:
|
| 377 |
+
quality = (1 - gw) * quality + gw * (sum(gated) / len(gated))
|
| 378 |
+
penalty = 0.0
|
| 379 |
+
for name, spec in (schema.get("red_flags") or {}).items():
|
| 380 |
+
v, pen = scores.get(name), spec.get("penalty") or {}
|
| 381 |
+
if v is not None and float(pen.get("weight", 0.0)) > 0:
|
| 382 |
+
penalty += float(pen["weight"]) * max(0.0, float(v) - float(pen.get("threshold", 0.0)))
|
| 383 |
+
return round(min(5.0, max(0.0, quality - penalty)), 3)
|
| 384 |
+
|
| 385 |
+
|
| 386 |
+
def gate_applies(schema: dict, name: str, labels: dict) -> bool:
|
| 387 |
+
"""Whether gated score ``name`` applies, given the labels: any of its ``applies_when`` labels has one of the
|
| 388 |
+
listed values (code_quality: code content or a code file; math_quality: math-heavy content or the math topic)."""
|
| 389 |
+
for label, values in schema["gated"][name]["applies_when"].items():
|
| 390 |
+
values = [values] if isinstance(values, str) else values
|
| 391 |
+
if labels.get(label) in values:
|
| 392 |
+
return True
|
| 393 |
+
return False
|
| 394 |
+
|
| 395 |
+
|
| 396 |
+
def aggregate(schema: dict, items: list[tuple[int, dict]]) -> dict:
|
| 397 |
+
"""Document scores from ``(tokens, chunk_scores)`` pairs: labels by token-weighted vote, scores by token-weighted
|
| 398 |
+
mean (or max / min where the schema says so: toxicity uses max), gated scores over the chunks where they apply.
|
| 399 |
+
With several chunks also ``label_dist`` (token share of each label value) and ``ranges`` ([min, max] per score)."""
|
| 400 |
+
out: dict = {}
|
| 401 |
+
dist_out: dict = {}
|
| 402 |
+
ranges: dict = {}
|
| 403 |
+
total_w = sum(max(w, 1) for w, _ in items)
|
| 404 |
+
for name in schema["labels"]:
|
| 405 |
+
weights: dict[str, float] = {}
|
| 406 |
+
for w, sc in items:
|
| 407 |
+
v = sc.get(name)
|
| 408 |
+
if v is not None:
|
| 409 |
+
weights[v] = weights.get(v, 0.0) + max(w, 1)
|
| 410 |
+
if not weights:
|
| 411 |
+
out[name] = None
|
| 412 |
+
continue
|
| 413 |
+
dist = {k: round(v / total_w, 3) for k, v in sorted(weights.items(), key=lambda kv: -kv[1])}
|
| 414 |
+
out[name] = next(iter(dist))
|
| 415 |
+
if len(items) > 1:
|
| 416 |
+
dist_out[name] = dist
|
| 417 |
+
for group in ("quality", "red_flags", "gated"):
|
| 418 |
+
for name, spec in (schema.get(group) or {}).items():
|
| 419 |
+
vals = [(float(v), max(w, 1)) for w, sc in items if (v := sc.get(name)) is not None]
|
| 420 |
+
if not vals:
|
| 421 |
+
out[name] = None
|
| 422 |
+
continue
|
| 423 |
+
how = spec.get("aggregate", "mean")
|
| 424 |
+
if how == "max":
|
| 425 |
+
agg = max(v for v, _ in vals)
|
| 426 |
+
elif how == "min":
|
| 427 |
+
agg = min(v for v, _ in vals)
|
| 428 |
+
else:
|
| 429 |
+
agg = sum(v * w for v, w in vals) / sum(w for _, w in vals)
|
| 430 |
+
out[name] = _r3(agg)
|
| 431 |
+
if len(vals) > 1:
|
| 432 |
+
ranges[name] = [_r3(min(v for v, _ in vals)), _r3(max(v for v, _ in vals))]
|
| 433 |
+
if dist_out:
|
| 434 |
+
out["label_dist"] = dist_out
|
| 435 |
+
if ranges:
|
| 436 |
+
out["ranges"] = ranges
|
| 437 |
+
return out
|
| 438 |
+
|
| 439 |
+
|
| 440 |
+
# --------------------------------------------------------------------------------------------- model
|
| 441 |
+
|
| 442 |
+
_DTYPES = {"bfloat16": torch.bfloat16, "bf16": torch.bfloat16, "float32": torch.float32, "fp32": torch.float32}
|
| 443 |
+
DTYPE_CHOICES = ("auto", "bf16", "bfloat16", "fp32", "float32")
|
| 444 |
+
PRECISION_CHOICES = ("bf16", "fp32")
|
| 445 |
+
_FP16 = ("float16 is not supported: summing the hidden states for mean pooling overflows float16's range on long "
|
| 446 |
+
"inputs and gives NaN scores. Use dtype='bf16' (GPUs from NVIDIA Ampere on) or dtype='fp32' (any device).")
|
| 447 |
+
_FP16_WEIGHTS = ("there are no float16 weights, and float16 is not supported (summing the hidden states for mean "
|
| 448 |
+
"pooling overflows its range on long inputs and gives NaN scores). Use precision='bf16' (the "
|
| 449 |
+
"default, model.safetensors) or precision='fp32' (model.fp32.safetensors).")
|
| 450 |
+
|
| 451 |
+
|
| 452 |
+
def resolve_precision(precision: Any = None) -> str:
|
| 453 |
+
"""Which weights to load, "bf16" (model.safetensors, the default) or "fp32" (model.fp32.safetensors): None,
|
| 454 |
+
"bf16" / "bfloat16" / torch.bfloat16, or "fp32" / "float32" / torch.float32. float16 raises ValueError."""
|
| 455 |
+
if precision is None:
|
| 456 |
+
return DEFAULT_PRECISION
|
| 457 |
+
if isinstance(precision, str):
|
| 458 |
+
key = precision.lower().removeprefix("torch.")
|
| 459 |
+
if key in ("float16", "fp16", "half"):
|
| 460 |
+
raise ValueError(_FP16_WEIGHTS)
|
| 461 |
+
if key in ("bf16", "bfloat16"):
|
| 462 |
+
return "bf16"
|
| 463 |
+
if key in ("fp32", "float32"):
|
| 464 |
+
return "fp32"
|
| 465 |
+
raise ValueError(f"unknown precision {precision!r}: use 'bf16' (the default) or 'fp32'")
|
| 466 |
+
if precision == torch.float16:
|
| 467 |
+
raise ValueError(_FP16_WEIGHTS)
|
| 468 |
+
if precision == torch.bfloat16:
|
| 469 |
+
return "bf16"
|
| 470 |
+
if precision == torch.float32:
|
| 471 |
+
return "fp32"
|
| 472 |
+
raise ValueError(f"unknown precision {precision!r}: use 'bf16' (the default) or 'fp32'")
|
| 473 |
+
|
| 474 |
+
|
| 475 |
+
def stored_dtype(path: str | Path) -> str | None:
|
| 476 |
+
"""The dtype a safetensors file stores its tensors in ("BF16", "F32", ...; "mixed" when several), read from its
|
| 477 |
+
header; None when the header cannot be read."""
|
| 478 |
+
try:
|
| 479 |
+
with open(path, "rb") as f:
|
| 480 |
+
n = int.from_bytes(f.read(8), "little")
|
| 481 |
+
if not 0 < n <= 100_000_000:
|
| 482 |
+
return None
|
| 483 |
+
header = json.loads(f.read(n))
|
| 484 |
+
except (OSError, ValueError):
|
| 485 |
+
return None
|
| 486 |
+
kinds = {v.get("dtype") for k, v in header.items() if k != "__metadata__" and isinstance(v, dict)}
|
| 487 |
+
return None if not kinds else kinds.pop() if len(kinds) == 1 else "mixed"
|
| 488 |
+
|
| 489 |
+
|
| 490 |
+
def _native_bf16(device: torch.device) -> bool:
|
| 491 |
+
"""Whether ``device`` is a GPU that runs bfloat16 natively (not emulated)."""
|
| 492 |
+
if device.type != "cuda":
|
| 493 |
+
return False
|
| 494 |
+
try:
|
| 495 |
+
with torch.cuda.device(device):
|
| 496 |
+
return bool(torch.cuda.is_bf16_supported(including_emulation=False))
|
| 497 |
+
except TypeError: # a torch without including_emulation
|
| 498 |
+
return torch.cuda.get_device_capability(device)[0] >= 8
|
| 499 |
+
|
| 500 |
+
|
| 501 |
+
def resolve_dtype(dtype: Any, device: str | torch.device) -> torch.dtype:
|
| 502 |
+
"""The dtype to run in: None or "auto" = bfloat16 on a GPU with native bfloat16, float32 everywhere else;
|
| 503 |
+
"bf16" / "bfloat16" / torch.bfloat16 or "fp32" / "float32" / torch.float32 as given. float16 raises ValueError."""
|
| 504 |
+
device = torch.device(device)
|
| 505 |
+
if dtype is None or (isinstance(dtype, str) and dtype.lower() == "auto"):
|
| 506 |
+
return torch.bfloat16 if _native_bf16(device) else torch.float32
|
| 507 |
+
if isinstance(dtype, str):
|
| 508 |
+
key = dtype.lower().removeprefix("torch.")
|
| 509 |
+
if key in ("float16", "fp16", "half"):
|
| 510 |
+
raise ValueError(_FP16)
|
| 511 |
+
if key not in _DTYPES:
|
| 512 |
+
raise ValueError(f"unknown dtype {dtype!r}: use 'auto', 'bf16' or 'fp32'")
|
| 513 |
+
return _DTYPES[key]
|
| 514 |
+
if dtype == torch.float16:
|
| 515 |
+
raise ValueError(_FP16)
|
| 516 |
+
if dtype not in (torch.bfloat16, torch.float32):
|
| 517 |
+
raise ValueError(f"unsupported dtype {dtype}: use torch.bfloat16 or torch.float32")
|
| 518 |
+
return dtype
|
| 519 |
+
|
| 520 |
+
|
| 521 |
+
def _dtype_kwarg() -> str:
|
| 522 |
+
"""transformers 4.56 renamed from_pretrained(torch_dtype=...) to dtype=..."""
|
| 523 |
+
import transformers
|
| 524 |
+
|
| 525 |
+
major, minor = (int("".join(c for c in p if c.isdigit()) or 0) for p in transformers.__version__.split(".")[:2])
|
| 526 |
+
return "dtype" if (major, minor) >= (4, 56) else "torch_dtype"
|
| 527 |
+
|
| 528 |
+
|
| 529 |
+
_REPO_ID = re.compile(r"[A-Za-z0-9][\w.-]*/[\w.-]+", re.ASCII)
|
| 530 |
+
|
| 531 |
+
|
| 532 |
+
def _resolve_dir(path_or_repo: str | Path, revision: str | None = None,
|
| 533 |
+
precision: str = DEFAULT_PRECISION) -> Path:
|
| 534 |
+
"""A local directory as is; a Hugging Face repo id ("owner/name") downloaded with huggingface_hub (only the files
|
| 535 |
+
source1.py needs, with the weights of ``precision`` only, at ``revision`` when given). Anything else raises
|
| 536 |
+
FileNotFoundError, never a download."""
|
| 537 |
+
s = str(path_or_repo)
|
| 538 |
+
path = Path(s).expanduser()
|
| 539 |
+
if path.is_dir():
|
| 540 |
+
if revision is not None:
|
| 541 |
+
raise ValueError("revision= applies to a Hugging Face repo id, not to a local directory")
|
| 542 |
+
return path
|
| 543 |
+
if path.exists():
|
| 544 |
+
raise FileNotFoundError(f"{s} is a file; pass the Source-1 directory that holds it")
|
| 545 |
+
is_repo_id = (isinstance(path_or_repo, str) and _REPO_ID.fullmatch(s) is not None
|
| 546 |
+
and not s.startswith((".", "~", "/")) and not Path(s.split("/")[0]).exists())
|
| 547 |
+
if not is_repo_id:
|
| 548 |
+
raise FileNotFoundError(f"{s}: no such Source-1 directory")
|
| 549 |
+
from huggingface_hub import snapshot_download # installed with transformers
|
| 550 |
+
|
| 551 |
+
return Path(snapshot_download(repo_id=s, revision=revision, allow_patterns=list(hub_files(precision))))
|
| 552 |
+
|
| 553 |
+
|
| 554 |
+
def _load_calibration(path: Path, drop_line: str | None, apply_offsets: bool) -> dict | None:
|
| 555 |
+
"""calibration.json of a Source-1 directory. It is required for the calibrated drop line (the default) and for
|
| 556 |
+
``apply_offsets``; a missing or incomplete file then raises FileNotFoundError instead of silently falling back."""
|
| 557 |
+
cal_path = path / CALIBRATION
|
| 558 |
+
calibration = json.loads(cal_path.read_text(encoding="utf-8")) if cal_path.exists() else None
|
| 559 |
+
has_line = bool(((calibration or {}).get("drop_line") or {}).get("line"))
|
| 560 |
+
if drop_line in (None, "calibrated") and not has_line:
|
| 561 |
+
what = "has no drop line" if calibration is not None else "is missing"
|
| 562 |
+
raise FileNotFoundError(
|
| 563 |
+
f"{cal_path} {what}, and the default drop line comes from it: download it again, or pass "
|
| 564 |
+
"drop_line='default' (the rubric's own hard filters) or your own drop line to load without it")
|
| 565 |
+
if apply_offsets and not (calibration or {}).get("offsets"):
|
| 566 |
+
raise FileNotFoundError(f"{cal_path} is missing or has no offsets, which apply_offsets=True needs")
|
| 567 |
+
return calibration
|
| 568 |
+
|
| 569 |
+
|
| 570 |
+
class Source1(nn.Module):
|
| 571 |
+
"""Source-1: the mmBERT-base encoder, mean pooling and one linear head per field. Build it with
|
| 572 |
+
``Source1.from_pretrained``; score documents with ``score`` / ``score_batch``."""
|
| 573 |
+
|
| 574 |
+
def __init__(self, backbone: nn.Module, tokenizer: Any, config: dict, calibration: dict | None = None,
|
| 575 |
+
drop_line: str | None = None, apply_offsets: bool = False, show_url: bool = False):
|
| 576 |
+
super().__init__()
|
| 577 |
+
self.backbone = backbone
|
| 578 |
+
self.tok = tokenizer
|
| 579 |
+
self.config = config
|
| 580 |
+
self.schema = config["schema"]
|
| 581 |
+
self.layout = [tuple(x) for x in config["layout"]]
|
| 582 |
+
if any(kind not in ("label_single", "score") for _, kind, _ in self.layout):
|
| 583 |
+
raise ValueError("this scorer handles single-choice labels and 0-5 scores only")
|
| 584 |
+
if config.get("pooling", "mean") != "mean" or config.get("text_normalize") not in (None, "collapse_spaces"):
|
| 585 |
+
raise ValueError("unexpected source1.json: this scorer expects mean pooling and collapse_spaces")
|
| 586 |
+
self.normalize = collapse_spaces if config.get("text_normalize") else (lambda s: s)
|
| 587 |
+
hidden = int(backbone.config.hidden_size)
|
| 588 |
+
self.heads = nn.ModuleDict({name: nn.Linear(hidden, size) for name, _, size in self.layout})
|
| 589 |
+
self.max_length = int(config.get("max_length") or MAX_LENGTH)
|
| 590 |
+
self.labels = {name: list(spec["values"]) for name, spec in self.schema["labels"].items()}
|
| 591 |
+
self.fields = list(self.labels) + [n for g in ("quality", "red_flags", "gated") for n in self.schema.get(g, {})]
|
| 592 |
+
self.calibration = calibration or {}
|
| 593 |
+
default_line = " or ".join((self.schema.get("composite") or {}).get("hard_filters") or []) or "False"
|
| 594 |
+
if drop_line in (None, "calibrated"):
|
| 595 |
+
drop_line = (self.calibration.get("drop_line") or {}).get("line")
|
| 596 |
+
if not drop_line:
|
| 597 |
+
raise ValueError("no calibrated drop line (calibration.json missing or incomplete); pass "
|
| 598 |
+
"drop_line='default' or your own drop line")
|
| 599 |
+
elif drop_line == "default":
|
| 600 |
+
drop_line = default_line
|
| 601 |
+
self.drop_line = DropLine(drop_line, set(self.fields) | {"overall", "tokens", "parts"})
|
| 602 |
+
self.offsets = {k: float(v["offset"]) for k, v in (self.calibration.get("offsets") or {}).items()}
|
| 603 |
+
self.apply_offsets = apply_offsets
|
| 604 |
+
self.show_url = show_url
|
| 605 |
+
self.precision: str | None = None # set by from_pretrained: "bf16" or "fp32", the weights it loaded
|
| 606 |
+
self.weights_file: str | None = None
|
| 607 |
+
self.weights_dtype: str | None = None # how that file stores its tensors ("BF16", "F32")
|
| 608 |
+
ids = tokenizer.encode("", add_special_tokens=True).ids
|
| 609 |
+
if len(ids) != 2:
|
| 610 |
+
raise ValueError("expected the tokenizer to add exactly <bos> and <eos>")
|
| 611 |
+
self.bos_id, self.eos_id = ids
|
| 612 |
+
self.pad_id = int(tokenizer.token_to_id("<pad>") if tokenizer.token_to_id("<pad>") is not None else 0)
|
| 613 |
+
|
| 614 |
+
# ----------------------------------------------------------------------------------------- loading
|
| 615 |
+
|
| 616 |
+
@classmethod
|
| 617 |
+
def from_pretrained(cls, path_or_dir: str | Path, device: str | None = None, dtype: Any = None, *,
|
| 618 |
+
precision: Any = DEFAULT_PRECISION, drop_line: str | None = None,
|
| 619 |
+
apply_offsets: bool = False, show_url: bool = False, revision: str | None = None,
|
| 620 |
+
**backbone_kwargs: Any) -> "Source1":
|
| 621 |
+
"""Load Source-1 from a directory, or from a Hugging Face repo id ("owner/Source-1", downloaded with
|
| 622 |
+
huggingface_hub; ``revision`` pins a branch, tag or commit).
|
| 623 |
+
|
| 624 |
+
device: "cuda", "cuda:1", "cpu", ...; default the GPU when there is one, else the CPU.
|
| 625 |
+
precision: which weights to load: "bf16" (default) = model.safetensors, bfloat16; "fp32" =
|
| 626 |
+
model.fp32.safetensors, the full-precision float32 copy (a Hub repo id downloads only the chosen file).
|
| 627 |
+
float16 raises ValueError.
|
| 628 |
+
dtype: what the model computes in. None or "auto" (default) = bfloat16 on a GPU with native bfloat16
|
| 629 |
+
(Ampere and newer), float32 elsewhere (bfloat16 weights are then upcast to float32); or "bf16" / "fp32" /
|
| 630 |
+
torch.bfloat16 / torch.float32. float16 raises ValueError (it overflows).
|
| 631 |
+
drop_line: None or "calibrated" = calibration.json's line (the file is then required); "default" = the
|
| 632 |
+
schema's own hard filters (``toxicity >= 4 or spam_seo >= 4 or boilerplate >= 4.5``); or any expression
|
| 633 |
+
``DropLine`` accepts.
|
| 634 |
+
apply_offsets: add calibration.json's offsets to the quality scores (tiny; off by default).
|
| 635 |
+
show_url: show ``url`` to the model in the header (it was never trained with one; off by default).
|
| 636 |
+
backbone_kwargs: passed to transformers' AutoModel.from_pretrained (e.g. attn_implementation="sdpa")."""
|
| 637 |
+
from safetensors.torch import load_file
|
| 638 |
+
from tokenizers import Tokenizer
|
| 639 |
+
from transformers import AutoModel
|
| 640 |
+
|
| 641 |
+
precision = resolve_precision(precision)
|
| 642 |
+
if device is None:
|
| 643 |
+
device = "cuda" if torch.cuda.is_available() else "cpu"
|
| 644 |
+
dtype = resolve_dtype(dtype, device)
|
| 645 |
+
if "variant" in backbone_kwargs:
|
| 646 |
+
raise TypeError("pass precision='bf16' or precision='fp32' instead of variant=")
|
| 647 |
+
path = _resolve_dir(path_or_dir, revision, precision)
|
| 648 |
+
weights = WEIGHTS[precision]
|
| 649 |
+
if not (path / weights).exists():
|
| 650 |
+
other = "bf16" if precision == "fp32" else "fp32"
|
| 651 |
+
have_other = (path / WEIGHTS[other]).exists()
|
| 652 |
+
raise FileNotFoundError(
|
| 653 |
+
f"{path} lacks {weights}, the {'float32' if precision == 'fp32' else 'bfloat16'} weights that "
|
| 654 |
+
f"precision={precision!r} loads" + (f"; precision={other!r} loads {WEIGHTS[other]}, which is there"
|
| 655 |
+
if have_other else ""))
|
| 656 |
+
missing = [f for f in FILES if not (path / f).exists()]
|
| 657 |
+
if missing:
|
| 658 |
+
raise FileNotFoundError(f"{path} lacks {', '.join(missing)}")
|
| 659 |
+
config = json.loads((path / "source1.json").read_text(encoding="utf-8"))
|
| 660 |
+
calibration = _load_calibration(path, drop_line, apply_offsets)
|
| 661 |
+
tok = Tokenizer.from_file(str(path / "tokenizer.json"))
|
| 662 |
+
tok.no_truncation()
|
| 663 |
+
tok.no_padding()
|
| 664 |
+
variant = {"variant": "fp32"} if precision == "fp32" else {}
|
| 665 |
+
backbone = AutoModel.from_pretrained(str(path), **{_dtype_kwarg(): dtype}, **variant, **backbone_kwargs)
|
| 666 |
+
model = cls(backbone, tok, config, calibration, drop_line, apply_offsets, show_url)
|
| 667 |
+
model.heads.load_state_dict(load_file(str(path / "heads.safetensors")))
|
| 668 |
+
model.heads.to(dtype)
|
| 669 |
+
model.precision = precision
|
| 670 |
+
model.weights_file = weights
|
| 671 |
+
model.weights_dtype = stored_dtype(path / weights)
|
| 672 |
+
return model.to(device).eval()
|
| 673 |
+
|
| 674 |
+
@property
|
| 675 |
+
def device(self) -> torch.device:
|
| 676 |
+
return next(self.backbone.parameters()).device
|
| 677 |
+
|
| 678 |
+
# ----------------------------------------------------------------------------------------- inputs
|
| 679 |
+
|
| 680 |
+
def chunk(self, text: str, is_code: bool = False, max_tokens: int = CHUNK_TOKENS) -> list[tuple[int, int, int]]:
|
| 681 |
+
"""``split_text`` with this model's tokenizer: (start_char, end_char, tokens) per chunk of a cleaned text."""
|
| 682 |
+
if not text:
|
| 683 |
+
return []
|
| 684 |
+
offsets = [s for s, _ in self.tok.encode(text, add_special_tokens=False).offsets]
|
| 685 |
+
return split_text(text, offsets, max_tokens, is_code)
|
| 686 |
+
|
| 687 |
+
def encode(self, texts: list[str]) -> list[tuple[list[int], bool]]:
|
| 688 |
+
"""Token ids of model inputs (``collapse_spaces`` applied, <bos> ... <eos>, at most max_length tokens: a
|
| 689 |
+
longer input keeps its first max_length - 1 tokens and its <eos>), each with whether it had to be cut."""
|
| 690 |
+
out = []
|
| 691 |
+
for enc in self.tok.encode_batch([self.normalize(t) for t in texts], add_special_tokens=True):
|
| 692 |
+
ids = enc.ids
|
| 693 |
+
out.append((ids, False) if len(ids) <= self.max_length else (ids[: self.max_length - 1] + [self.eos_id], True))
|
| 694 |
+
return out
|
| 695 |
+
|
| 696 |
+
# ----------------------------------------------------------------------------------------- scoring
|
| 697 |
+
|
| 698 |
+
@torch.inference_mode()
|
| 699 |
+
def _logits(self, batch: list[list[int]]) -> dict[str, torch.Tensor]:
|
| 700 |
+
width = max(len(ids) for ids in batch)
|
| 701 |
+
x = torch.full((len(batch), width), self.pad_id, dtype=torch.long)
|
| 702 |
+
m = torch.zeros((len(batch), width), dtype=torch.long)
|
| 703 |
+
for row, ids in enumerate(batch):
|
| 704 |
+
x[row, : len(ids)] = torch.tensor(ids, dtype=torch.long)
|
| 705 |
+
m[row, : len(ids)] = 1
|
| 706 |
+
x, m = x.to(self.device), m.to(self.device)
|
| 707 |
+
out = self.backbone(input_ids=x, attention_mask=m)
|
| 708 |
+
hidden = out.last_hidden_state if hasattr(out, "last_hidden_state") else out[0]
|
| 709 |
+
mask = m.unsqueeze(-1).to(hidden.dtype)
|
| 710 |
+
pooled = (hidden * mask).sum(dim=1) / mask.sum(dim=1).clamp(min=1.0) # mean over the real tokens
|
| 711 |
+
pooled = pooled.to(next(iter(self.heads.values())).weight.dtype)
|
| 712 |
+
return {name: head(pooled) for name, head in self.heads.items()}
|
| 713 |
+
|
| 714 |
+
@torch.inference_mode()
|
| 715 |
+
def _decode(self, logits: dict[str, torch.Tensor]) -> list[dict]:
|
| 716 |
+
n = next(iter(logits.values())).shape[0]
|
| 717 |
+
for name, x in logits.items(): # NaN would silently keep a document and break the JSON output
|
| 718 |
+
finite = torch.isfinite(x).all(dim=-1)
|
| 719 |
+
if not bool(finite.all()):
|
| 720 |
+
raise FloatingPointError(
|
| 721 |
+
f"head {name!r} gave NaN or infinite outputs for {int((~finite).sum())} of {n} inputs; "
|
| 722 |
+
"this should not happen in bfloat16 or float32")
|
| 723 |
+
rows: list[dict] = [{} for _ in range(n)]
|
| 724 |
+
levels = torch.tensor(LEVELS, dtype=torch.float32, device=next(iter(logits.values())).device)
|
| 725 |
+
for name, kind, _ in self.layout:
|
| 726 |
+
x = logits[name].float()
|
| 727 |
+
if kind == "label_single":
|
| 728 |
+
for row, i in enumerate(x.argmax(dim=-1).tolist()):
|
| 729 |
+
rows[row][name] = self.labels[name][i]
|
| 730 |
+
else:
|
| 731 |
+
expected = (torch.softmax(x, dim=-1) * levels).sum(dim=-1).tolist()
|
| 732 |
+
for row, e in enumerate(expected):
|
| 733 |
+
rows[row][name] = round(e, 3)
|
| 734 |
+
return rows
|
| 735 |
+
|
| 736 |
+
def _finish(self, scores: dict) -> dict:
|
| 737 |
+
"""Gating, optional offsets, overall and keep for one chunk's raw head outputs (a flat dict)."""
|
| 738 |
+
labels = {k: scores[k] for k in self.labels}
|
| 739 |
+
out = {**labels}
|
| 740 |
+
for group in ("quality", "red_flags", "gated"):
|
| 741 |
+
for name in self.schema.get(group, {}):
|
| 742 |
+
v = scores[name]
|
| 743 |
+
if group == "gated" and not gate_applies(self.schema, name, labels):
|
| 744 |
+
v = None
|
| 745 |
+
elif group == "quality" and self.apply_offsets and name in self.offsets:
|
| 746 |
+
v = round(min(5.0, max(0.0, v + self.offsets[name])), 3)
|
| 747 |
+
out[name] = v
|
| 748 |
+
out["overall"] = composite(self.schema, out)
|
| 749 |
+
return out
|
| 750 |
+
|
| 751 |
+
def _judge(self, rec: dict) -> dict:
|
| 752 |
+
reasons = self.drop_line.reasons(rec)
|
| 753 |
+
rec["keep"] = not reasons
|
| 754 |
+
rec["drop_reasons"] = reasons
|
| 755 |
+
return rec
|
| 756 |
+
|
| 757 |
+
def score_inputs(self, inputs: list[str], batch_tokens: int = DEFAULT_BATCH_TOKENS) -> list[dict]:
|
| 758 |
+
"""Score ready-made model inputs (``build_input`` output: header, blank line, chunk text), one dict per
|
| 759 |
+
input with the 13 fields, overall, keep, drop_reasons, input_tokens and truncated.
|
| 760 |
+
|
| 761 |
+
Inputs are sorted by length and batched with at most ``batch_tokens`` padded tokens per forward pass;
|
| 762 |
+
a batch that runs out of GPU memory is split in half and retried."""
|
| 763 |
+
encoded = self.encode(inputs)
|
| 764 |
+
ids = [e[0] for e in encoded]
|
| 765 |
+
results: list[dict | None] = [None] * len(ids)
|
| 766 |
+
|
| 767 |
+
def run(batch: list[int]) -> None:
|
| 768 |
+
try:
|
| 769 |
+
rows = self._decode(self._logits([ids[i] for i in batch]))
|
| 770 |
+
except torch.cuda.OutOfMemoryError:
|
| 771 |
+
torch.cuda.empty_cache()
|
| 772 |
+
if len(batch) == 1:
|
| 773 |
+
raise
|
| 774 |
+
run(batch[: len(batch) // 2])
|
| 775 |
+
run(batch[len(batch) // 2:])
|
| 776 |
+
return
|
| 777 |
+
for i, raw in zip(batch, rows):
|
| 778 |
+
rec = self._judge(self._finish(raw))
|
| 779 |
+
rec["input_tokens"] = len(ids[i])
|
| 780 |
+
rec["truncated"] = encoded[i][1]
|
| 781 |
+
results[i] = rec
|
| 782 |
+
|
| 783 |
+
order = sorted(range(len(ids)), key=lambda i: -len(ids[i]))
|
| 784 |
+
start = 0
|
| 785 |
+
while start < len(order):
|
| 786 |
+
width = len(ids[order[start]])
|
| 787 |
+
batch = order[start: start + max(1, batch_tokens // max(width, 1))]
|
| 788 |
+
start += len(batch)
|
| 789 |
+
run(batch)
|
| 790 |
+
return results # type: ignore[return-value]
|
| 791 |
+
|
| 792 |
+
def score_batch(self, docs: Iterable[str | bytes | dict], batch_tokens: int = DEFAULT_BATCH_TOKENS, *,
|
| 793 |
+
text_field: str = "text", max_chunks: int = 0) -> list[dict]:
|
| 794 |
+
"""Score a list of documents: strings (bytes are decoded as UTF-8), or dicts with the text under
|
| 795 |
+
``text_field`` and optionally ``title``, ``url``, ``source_type`` and ``code_language`` (see ``score``).
|
| 796 |
+
A None text is scored as an empty document (keep False, drop_reasons ["empty text"]). Chunks of all
|
| 797 |
+
documents are batched together. ``max_chunks`` > 0 scores only that many evenly spaced chunks of a long
|
| 798 |
+
document (the Part numbers still count every chunk).
|
| 799 |
+
|
| 800 |
+
In bfloat16 a document's scores can shift slightly (up to about 0.04 on overall, 0.10 on a single field)
|
| 801 |
+
depending on which other documents share its batch, because the batch shape changes the kernels' rounding;
|
| 802 |
+
use dtype="fp32" or batch_tokens=1 when a document's scores must not depend on its neighbours."""
|
| 803 |
+
if isinstance(docs, (str, bytes, bytearray, Mapping)):
|
| 804 |
+
raise TypeError("score_batch takes a list of documents; use score() for one document")
|
| 805 |
+
plans: list[dict] = []
|
| 806 |
+
inputs: list[str] = []
|
| 807 |
+
for n, doc in enumerate(docs):
|
| 808 |
+
if isinstance(doc, Mapping):
|
| 809 |
+
if text_field not in doc:
|
| 810 |
+
raise KeyError(f"document {n} has no {text_field!r} field")
|
| 811 |
+
d = {k: _from_bytes(v) for k, v in doc.items()}
|
| 812 |
+
else:
|
| 813 |
+
d = {text_field: _from_bytes(doc)}
|
| 814 |
+
raw = d[text_field]
|
| 815 |
+
if raw is not None and not isinstance(raw, str):
|
| 816 |
+
raise TypeError(f"document {n}: the text must be a str, bytes or None, not {type(raw).__name__}")
|
| 817 |
+
text = clean_text(raw or "")
|
| 818 |
+
code_language = d.get("code_language")
|
| 819 |
+
source = source_description(d.get("source_type"), code_language)
|
| 820 |
+
spans = self.chunk(text, is_code=bool(code_language))
|
| 821 |
+
chosen = select_chunks(len(spans), max_chunks)
|
| 822 |
+
first = len(inputs)
|
| 823 |
+
for i in chosen:
|
| 824 |
+
s, e, _ = spans[i]
|
| 825 |
+
inputs.append(build_input(text[s:e].strip(), source_type=source, title=d.get("title"),
|
| 826 |
+
url=d.get("url") if self.show_url else None, part=i + 1,
|
| 827 |
+
parts=len(spans)))
|
| 828 |
+
plans.append({"spans": spans, "chosen": chosen, "first": first})
|
| 829 |
+
scored = self.score_inputs(inputs, batch_tokens)
|
| 830 |
+
return [self._document(p, scored[p["first"]: p["first"] + len(p["chosen"])]) for p in plans]
|
| 831 |
+
|
| 832 |
+
def score(self, text: str | bytes, title: str | None = None, url: str | None = None, *,
|
| 833 |
+
source_type: str | None = None, code_language: str | None = None, max_chunks: int = 0,
|
| 834 |
+
batch_tokens: int = DEFAULT_BATCH_TOKENS) -> dict:
|
| 835 |
+
"""Score one document (a str; bytes are decoded as UTF-8; anything else raises TypeError).
|
| 836 |
+
|
| 837 |
+
title: shown to the model in the header when given (as in training, where about a quarter of inputs had one).
|
| 838 |
+
url: kept out of the model's input unless the model was loaded with show_url=True.
|
| 839 |
+
source_type: the header's Source value; default "dataset record" ("" leaves the Source line out).
|
| 840 |
+
code_language: for source code, e.g. "Python": Source becomes "Python source file" and long files are split
|
| 841 |
+
at definitions."""
|
| 842 |
+
doc = {"text": text, "title": title, "url": url, "source_type": source_type, "code_language": code_language}
|
| 843 |
+
return self.score_batch([doc], batch_tokens, max_chunks=max_chunks)[0]
|
| 844 |
+
|
| 845 |
+
def _document(self, plan: dict, chunks: list[dict]) -> dict:
|
| 846 |
+
spans, chosen = plan["spans"], plan["chosen"]
|
| 847 |
+
if not spans:
|
| 848 |
+
out = {name: None for name in self.fields}
|
| 849 |
+
out.update(overall=None, keep=False, drop_reasons=["empty text"], parts=0, tokens=0, truncated=False)
|
| 850 |
+
return out
|
| 851 |
+
items = [(spans[i][2], c) for i, c in zip(chosen, chunks)]
|
| 852 |
+
agg = aggregate(self.schema, items)
|
| 853 |
+
out = {name: agg[name] for name in self.fields}
|
| 854 |
+
out["overall"] = composite(self.schema, out)
|
| 855 |
+
parts, tokens = len(spans), sum(s[2] for s in spans)
|
| 856 |
+
reasons = self.drop_line.reasons({**out, "parts": parts, "tokens": tokens})
|
| 857 |
+
out.update(keep=not reasons, drop_reasons=reasons, parts=parts, tokens=tokens,
|
| 858 |
+
truncated=any(c["truncated"] for c in chunks))
|
| 859 |
+
if len(spans) > 1:
|
| 860 |
+
out["chunks"] = [{"part": i + 1, "start": spans[i][0], "end": spans[i][1], "tokens": spans[i][2], **c}
|
| 861 |
+
for i, c in zip(chosen, chunks)]
|
| 862 |
+
for key in ("label_dist", "ranges"):
|
| 863 |
+
if key in agg:
|
| 864 |
+
out[key] = agg[key]
|
| 865 |
+
return out
|
| 866 |
+
|
| 867 |
+
|
| 868 |
+
# --------------------------------------------------------------------------------------------- CLI
|
| 869 |
+
|
| 870 |
+
|
| 871 |
+
class BadInput(ValueError):
|
| 872 |
+
"""A record of the input file that cannot be scored (the message starts with file:line)."""
|
| 873 |
+
|
| 874 |
+
|
| 875 |
+
def _read_docs(path: Path, text_field: str, skip_bad: bool = False) -> Iterator[dict]:
|
| 876 |
+
"""Documents of an input file: .jsonl / .ndjson (one JSON object per line), .json (a JSON array of objects, one
|
| 877 |
+
object, or JSON Lines), or any other file as one plain-text document. Invalid UTF-8 is replaced (with a warning
|
| 878 |
+
for JSON). A bad record raises BadInput, or with ``skip_bad`` is reported on stderr and skipped. A null text is
|
| 879 |
+
kept and scored as an empty document."""
|
| 880 |
+
|
| 881 |
+
def usable(rec: Any, where: str) -> bool:
|
| 882 |
+
if not isinstance(rec, dict):
|
| 883 |
+
problem = f"expected a JSON object, got {type(rec).__name__}"
|
| 884 |
+
elif text_field not in rec:
|
| 885 |
+
problem = f"no {text_field!r} field"
|
| 886 |
+
elif rec[text_field] is not None and not isinstance(rec[text_field], str):
|
| 887 |
+
problem = f"{text_field!r} is a {type(rec[text_field]).__name__}, not a string"
|
| 888 |
+
else:
|
| 889 |
+
return True
|
| 890 |
+
if not skip_bad:
|
| 891 |
+
raise BadInput(f"{where}: {problem}")
|
| 892 |
+
print(f"source1: skipped {where}: {problem}", file=sys.stderr)
|
| 893 |
+
return False
|
| 894 |
+
|
| 895 |
+
def decode(raw: bytes, where: str) -> str:
|
| 896 |
+
try:
|
| 897 |
+
return raw.decode("utf-8")
|
| 898 |
+
except UnicodeDecodeError as e:
|
| 899 |
+
print(f"source1: {where}: invalid UTF-8 at byte {e.start}; invalid bytes replaced with U+FFFD",
|
| 900 |
+
file=sys.stderr)
|
| 901 |
+
return raw.decode("utf-8", errors="replace")
|
| 902 |
+
|
| 903 |
+
suffix = path.suffix.lower()
|
| 904 |
+
if suffix not in (".jsonl", ".ndjson", ".json"):
|
| 905 |
+
yield {text_field: path.read_text(encoding="utf-8", errors="replace"), "id": path.name}
|
| 906 |
+
return
|
| 907 |
+
if suffix == ".json":
|
| 908 |
+
try:
|
| 909 |
+
data = json.loads(decode(path.read_bytes(), str(path)).lstrip(""))
|
| 910 |
+
except json.JSONDecodeError:
|
| 911 |
+
data = None # not a single JSON value: read the file as JSON Lines below
|
| 912 |
+
if data is not None:
|
| 913 |
+
for n, rec in enumerate(data if isinstance(data, list) else [data]):
|
| 914 |
+
if usable(rec, f"{path}[{n}]"):
|
| 915 |
+
yield rec
|
| 916 |
+
return
|
| 917 |
+
with open(path, "rb") as f:
|
| 918 |
+
for n, raw in enumerate(f, 1):
|
| 919 |
+
where = f"{path}:{n}"
|
| 920 |
+
line = decode(raw, where)
|
| 921 |
+
if n == 1:
|
| 922 |
+
line = line.lstrip("")
|
| 923 |
+
if not line.strip():
|
| 924 |
+
continue
|
| 925 |
+
try:
|
| 926 |
+
rec = json.loads(line)
|
| 927 |
+
except json.JSONDecodeError as e:
|
| 928 |
+
if not skip_bad:
|
| 929 |
+
raise BadInput(f"{where}: invalid JSON ({e.msg} at column {e.colno})") from None
|
| 930 |
+
print(f"source1: skipped {where}: invalid JSON ({e.msg} at column {e.colno})", file=sys.stderr)
|
| 931 |
+
continue
|
| 932 |
+
if usable(rec, where):
|
| 933 |
+
yield rec
|
| 934 |
+
|
| 935 |
+
|
| 936 |
+
def main(argv: list[str] | None = None) -> int:
|
| 937 |
+
p = argparse.ArgumentParser(description="Score documents with Source-1 (one JSON object per document).")
|
| 938 |
+
p.add_argument("--model", default=str(Path(__file__).resolve().parent),
|
| 939 |
+
help="Source-1 directory or Hugging Face repo id (default: this file's directory)")
|
| 940 |
+
p.add_argument("--input", required=True, help=".jsonl (one document per line), .json (an array of objects) or "
|
| 941 |
+
"a text file (one document)")
|
| 942 |
+
p.add_argument("--text-field", default="text", help="JSON field holding the text (default: text); "
|
| 943 |
+
"title, url, source_type and code_language fields are used when present")
|
| 944 |
+
p.add_argument("--output", help="output .jsonl (default: standard output)")
|
| 945 |
+
p.add_argument("--skip-bad", action="store_true", help="skip (and report on stderr) records that are not valid "
|
| 946 |
+
"JSON objects with a string text, instead of stopping")
|
| 947 |
+
p.add_argument("--revision", help="branch, tag or commit, when --model is a Hugging Face repo id")
|
| 948 |
+
p.add_argument("--device", help="cpu, cuda, cuda:1, ... (default: cuda when available)")
|
| 949 |
+
p.add_argument("--precision", choices=PRECISION_CHOICES, default=DEFAULT_PRECISION, help="weights to load: "
|
| 950 |
+
"bf16 (default, model.safetensors) or fp32 (model.fp32.safetensors, the full-precision copy)")
|
| 951 |
+
p.add_argument("--dtype", choices=DTYPE_CHOICES, default="auto", help="what to compute in; auto (default): "
|
| 952 |
+
"bfloat16 on a GPU with native bfloat16, else float32 (bf16 weights upcast); float16 is not "
|
| 953 |
+
"supported")
|
| 954 |
+
p.add_argument("--batch-tokens", type=int, default=DEFAULT_BATCH_TOKENS, help="padded tokens per forward pass")
|
| 955 |
+
p.add_argument("--max-chunks", type=int, default=0, help="score at most N evenly spaced chunks per document")
|
| 956 |
+
p.add_argument("--drop-line", help='"calibrated" (default), "default" (the schema\'s hard filters) or an '
|
| 957 |
+
"expression such as 'toxicity >= 4 or spam_seo >= 3'")
|
| 958 |
+
p.add_argument("--apply-offsets", action="store_true", help="add the calibration offsets to the quality scores")
|
| 959 |
+
p.add_argument("--show-url", action="store_true", help="show the url field to the model (untrained)")
|
| 960 |
+
p.add_argument("--no-chunks", action="store_true", help="leave out the per-chunk list of split documents")
|
| 961 |
+
p.add_argument("--group", type=int, default=256, help="documents scored together")
|
| 962 |
+
args = p.parse_args(argv)
|
| 963 |
+
if not Path(args.input).is_file():
|
| 964 |
+
p.error(f"--input {args.input}: no such file")
|
| 965 |
+
|
| 966 |
+
t0 = time.time()
|
| 967 |
+
model = Source1.from_pretrained(args.model, device=args.device, dtype=args.dtype, precision=args.precision,
|
| 968 |
+
drop_line=args.drop_line, apply_offsets=args.apply_offsets,
|
| 969 |
+
show_url=args.show_url, revision=args.revision)
|
| 970 |
+
compute = str(next(model.parameters()).dtype).replace("torch.", "")
|
| 971 |
+
print(f"source1: loaded {model.weights_file} ({model.weights_dtype or '?'} weights) on {model.device}, computing "
|
| 972 |
+
f"in {compute}, in {time.time() - t0:.1f} s; drop line: {model.drop_line.source}", file=sys.stderr)
|
| 973 |
+
out = open(args.output, "w", encoding="utf-8") if args.output else sys.stdout
|
| 974 |
+
done = 0
|
| 975 |
+
t0 = time.time()
|
| 976 |
+
|
| 977 |
+
def flush(group: list[dict]) -> None:
|
| 978 |
+
nonlocal done
|
| 979 |
+
for rec, res in zip(group, model.score_batch(group, args.batch_tokens, text_field=args.text_field,
|
| 980 |
+
max_chunks=args.max_chunks)):
|
| 981 |
+
if args.no_chunks:
|
| 982 |
+
res.pop("chunks", None)
|
| 983 |
+
if "id" in rec:
|
| 984 |
+
res = {"id": rec["id"], **res}
|
| 985 |
+
out.write(json.dumps(res, ensure_ascii=False, allow_nan=False) + "\n")
|
| 986 |
+
done += len(group)
|
| 987 |
+
print(f"source1: {done:,} documents, {done / max(time.time() - t0, 1e-9):.1f}/s", file=sys.stderr)
|
| 988 |
+
|
| 989 |
+
try:
|
| 990 |
+
group: list[dict] = []
|
| 991 |
+
try:
|
| 992 |
+
for rec in _read_docs(Path(args.input), args.text_field, args.skip_bad):
|
| 993 |
+
group.append(rec)
|
| 994 |
+
if len(group) >= args.group:
|
| 995 |
+
flush(group)
|
| 996 |
+
group = []
|
| 997 |
+
except BadInput as e:
|
| 998 |
+
if group:
|
| 999 |
+
flush(group) # the documents read before the bad record are still scored and written
|
| 1000 |
+
raise SystemExit(f"source1: {e}. The {done:,} documents before it were written; --skip-bad skips "
|
| 1001 |
+
"bad records") from None
|
| 1002 |
+
if group:
|
| 1003 |
+
flush(group)
|
| 1004 |
+
finally:
|
| 1005 |
+
if out is not sys.stdout:
|
| 1006 |
+
out.close()
|
| 1007 |
+
return 0
|
| 1008 |
+
|
| 1009 |
+
|
| 1010 |
+
if __name__ == "__main__":
|
| 1011 |
+
sys.exit(main())
|
tokenizer.json
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:609d8f4c067cd3950f88594c5a802616cea245823836ef5848ee4fc40aab5b6f
|
| 3 |
+
size 34363188
|
tokenizer_config.json
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"backend": "tokenizers",
|
| 3 |
+
"bos_token": "<bos>",
|
| 4 |
+
"clean_up_tokenization_spaces": false,
|
| 5 |
+
"cls_token": "<bos>",
|
| 6 |
+
"eos_token": "<eos>",
|
| 7 |
+
"extra_special_tokens": [
|
| 8 |
+
"<start_of_turn>",
|
| 9 |
+
"<end_of_turn>"
|
| 10 |
+
],
|
| 11 |
+
"mask_token": "<mask>",
|
| 12 |
+
"model_input_names": [
|
| 13 |
+
"input_ids",
|
| 14 |
+
"attention_mask"
|
| 15 |
+
],
|
| 16 |
+
"model_max_length": 8192,
|
| 17 |
+
"pad_token": "<pad>",
|
| 18 |
+
"padding_side": "right",
|
| 19 |
+
"sep_token": "<eos>",
|
| 20 |
+
"spaces_between_special_tokens": false,
|
| 21 |
+
"tokenizer_class": "PreTrainedTokenizerFast",
|
| 22 |
+
"unk_token": "<unk>"
|
| 23 |
+
}
|