Add lg tier: fa_ent_news_lg on 200k floret vectors
New table: 200k rows x 300d, full Persian Wikipedia dump, 5 epochs, vs md's 50k rows / 400k documents. - configs/fa_ner_lg.cfg, project.yml ent-lg workflow (vectors-lg through smoke-ent-lg) - scripts/compare_tiers.py: generalized sm/md pair to N tiers; ent NER test now includes lg; fixed sm baseline to the file that's actually scored (perdt-ner-test.json, not the missing ent-test.json) - scripts/finalize_pipeline.py: FLORET_LG source and vectors_note_lg corrected to full Wikipedia, 5 epochs (were a generic Wikipedia + OSCAR placeholder) - docs/MODELS.md §7: PerDT NER test ENTS_F 75.94 (sm 71.87, md 74.71), full per-label table, cost (217 MB wheel) Not built: fa_dep_news_lg / fa_core_news_lg.
This commit is contained in:
parent
c3cb02d9c3
commit
b89b01ceb6
|
|
@ -0,0 +1,154 @@
|
|||
# fa_ent_news_lg — Persian NER with the lg-tier static floret vectors.
|
||||
#
|
||||
# Identical to configs/fa_ner_md.cfg (which is identical to fa_ner_sm.cfg except
|
||||
# include_static_vectors: true). Only the vector table supplied at train time via
|
||||
# --paths.vectors differs: fa_floret, 200k rows x 300d, floret mode, trained on the full
|
||||
# Persian Wikipedia dump for 5 epochs (assets/vectors/fa_floret_lg), vs md's 50k rows x 300d
|
||||
# trained on 400k Persian documents.
|
||||
#
|
||||
# Same embedded-tok2vec design as sm/md (no Tok2VecListener), so the trained component stays
|
||||
# sourceable into a future fa_core_news_lg via `nlp.add_pipe("ner", source=...)`.
|
||||
|
||||
[paths]
|
||||
train = null
|
||||
dev = null
|
||||
vectors = null
|
||||
init_tok2vec = null
|
||||
|
||||
[system]
|
||||
gpu_allocator = null
|
||||
seed = 0
|
||||
|
||||
[nlp]
|
||||
lang = "fa"
|
||||
pipeline = ["ner"]
|
||||
batch_size = 1000
|
||||
disabled = []
|
||||
before_creation = null
|
||||
after_creation = null
|
||||
after_pipeline_creation = null
|
||||
|
||||
[nlp.tokenizer]
|
||||
@tokenizers = "spacy.Tokenizer.v1"
|
||||
|
||||
[nlp.vectors]
|
||||
@vectors = "spacy.Vectors.v1"
|
||||
|
||||
[components]
|
||||
|
||||
[components.ner]
|
||||
factory = "ner"
|
||||
moves = null
|
||||
update_with_oracle_cut_size = 100
|
||||
incorrect_spans_key = null
|
||||
|
||||
[components.ner.model]
|
||||
@architectures = "spacy.TransitionBasedParser.v2"
|
||||
state_type = "ner"
|
||||
extra_state_tokens = false
|
||||
hidden_width = 64
|
||||
maxout_pieces = 2
|
||||
use_upper = true
|
||||
nO = null
|
||||
|
||||
[components.ner.model.tok2vec]
|
||||
@architectures = "spacy.Tok2Vec.v2"
|
||||
|
||||
[components.ner.model.tok2vec.embed]
|
||||
@architectures = "spacy.MultiHashEmbed.v2"
|
||||
width = ${components.ner.model.tok2vec.encode.width}
|
||||
attrs = ["NORM", "PREFIX", "SUFFIX", "SHAPE"]
|
||||
rows = [5000, 1000, 2500, 2500]
|
||||
include_static_vectors = true
|
||||
|
||||
[components.ner.model.tok2vec.encode]
|
||||
@architectures = "spacy.MaxoutWindowEncoder.v2"
|
||||
width = 96
|
||||
depth = 4
|
||||
window_size = 1
|
||||
maxout_pieces = 3
|
||||
|
||||
[components.ner.scorer]
|
||||
@scorers = "spacy.ner_scorer.v1"
|
||||
|
||||
[corpora]
|
||||
|
||||
[corpora.train]
|
||||
@readers = "spacy.Corpus.v1"
|
||||
path = ${paths.train}
|
||||
max_length = 0
|
||||
gold_preproc = false
|
||||
limit = 0
|
||||
augmenter = null
|
||||
|
||||
[corpora.dev]
|
||||
@readers = "spacy.Corpus.v1"
|
||||
path = ${paths.dev}
|
||||
max_length = 0
|
||||
gold_preproc = false
|
||||
limit = 0
|
||||
augmenter = null
|
||||
|
||||
[training]
|
||||
dev_corpus = "corpora.dev"
|
||||
train_corpus = "corpora.train"
|
||||
seed = ${system.seed}
|
||||
gpu_allocator = ${system.gpu_allocator}
|
||||
dropout = 0.1
|
||||
accumulate_gradient = 1
|
||||
patience = 1600
|
||||
max_epochs = 0
|
||||
max_steps = 20000
|
||||
eval_frequency = 400
|
||||
frozen_components = []
|
||||
annotating_components = []
|
||||
before_to_disk = null
|
||||
before_update = null
|
||||
|
||||
[training.optimizer]
|
||||
@optimizers = "Adam.v1"
|
||||
beta1 = 0.9
|
||||
beta2 = 0.999
|
||||
L2_is_weight_decay = true
|
||||
L2 = 0.01
|
||||
grad_clip = 1.0
|
||||
use_averages = false
|
||||
eps = 1e-08
|
||||
learn_rate = 0.001
|
||||
|
||||
[training.batcher]
|
||||
@batchers = "spacy.batch_by_words.v1"
|
||||
discard_oversize = false
|
||||
tolerance = 0.2
|
||||
get_length = null
|
||||
|
||||
[training.batcher.size]
|
||||
@schedules = "compounding.v1"
|
||||
start = 100
|
||||
stop = 1000
|
||||
compound = 1.001
|
||||
t = 0.0
|
||||
|
||||
[training.logger]
|
||||
@loggers = "spacy.ConsoleLogger.v1"
|
||||
progress_bar = false
|
||||
|
||||
[training.score_weights]
|
||||
ents_f = 1.0
|
||||
ents_p = 0.0
|
||||
ents_r = 0.0
|
||||
ents_per_type = null
|
||||
|
||||
[initialize]
|
||||
vectors = ${paths.vectors}
|
||||
init_tok2vec = ${paths.init_tok2vec}
|
||||
vocab_data = null
|
||||
lookups = null
|
||||
before_init = null
|
||||
after_init = null
|
||||
|
||||
[initialize.tokenizer]
|
||||
|
||||
[initialize.components]
|
||||
|
||||
[pretraining]
|
||||
|
|
@ -44,6 +44,7 @@ pipelines such as `de_core_news_sm` as `news`.
|
|||
| `fa_core_web_sm` | same as core, mixed-genre training data | hash embeddings | not built; would add ParsTwiNER to cover social media |
|
||||
| `fa_dep_news_md` | same as `fa_dep_news_sm` | floret, 50k rows / 300d | built, shipping |
|
||||
| `fa_core_news_md` | same as `fa_core_news_sm` | floret, 50k rows / 300d | built, shipping |
|
||||
| `fa_ent_news_lg` | ner (own internal tok2vec) | floret, 200k rows / 300d, full-wiki 5 epochs | built, optional |
|
||||
| `fa_core_news_lg` | same | floret, 200k rows | not built; bigger table, same recipe as md |
|
||||
| `fa_core_news_trf` | transformer instead of tok2vec | `HooshvareLab/roberta-fa-zwnj-base` (Apache-2.0) | not on this hardware; 2 GB VRAM cannot fine-tune a 125M-param encoder |
|
||||
|
||||
|
|
@ -349,3 +350,59 @@ best checkpoint near 10,800).
|
|||
Whether that trade is worth it depends on deployment. For a 1.19 LAS and 2.85 NER F gain, a
|
||||
9x larger download and 16% slower parse is a good deal on a server and a bad one in a browser
|
||||
or a Lambda cold start. Both tiers ship; pick per target.
|
||||
|
||||
## 7. The `lg` tier: bigger floret table, `ent` only
|
||||
|
||||
Built after `md`, from a new `fa_floret` table — 200,000 rows x 300d, floret mode,
|
||||
`minn=maxn=5`, `hash_count=2`, trained on the full Persian Wikipedia dump for 5 epochs (4x
|
||||
the rows of `md`'s 50k-row table trained on 400k documents). Unpacked the same way as `md`
|
||||
via `scripts/unpack_vectors.py`, into `assets/vectors/fa_floret_lg`.
|
||||
|
||||
`configs/fa_ner_lg.cfg` is `fa_ner_md.cfg` unchanged except `--paths.vectors`. Only `ent` was
|
||||
trained at this tier (`fa_ent_news_lg`), not `dep`/`core`: the point of this run was to check
|
||||
whether a 4x larger table is worth it before spending the CPU time on `dep`/`core` too. Same
|
||||
seed, same corpus, same architecture as `sm`/`md`. Reproduce with `spacy project run ent-lg`,
|
||||
or the table alone with `python scripts/compare_tiers.py`.
|
||||
|
||||
### PerDT NER test split, `fa_ent_news_lg`
|
||||
|
||||
| Metric | `sm` | `md` | `lg` | Delta (lg vs sm) |
|
||||
| --- | --- | --- | --- | --- |
|
||||
| `ENTS_P` | 77.67 | 76.56 | 81.51 | +3.84 |
|
||||
| `ENTS_R` | 66.87 | 72.95 | 71.09 | +4.22 |
|
||||
| `ENTS_F` | 71.87 | 74.71 | 75.94 | +4.08 |
|
||||
|
||||
`lg` beats both `sm` and `md` on `ENTS_F`, and unlike `md`'s recall-only gain over `sm`, `lg`
|
||||
improves precision too (+3.84 over `sm`, whereas `md` cost -1.10). Consistent with a bigger,
|
||||
less collision-prone floret table giving both better recall on rare proper nouns and fewer
|
||||
false positives from hash collisions.
|
||||
|
||||
| Label | Gold in test | `sm` F | `md` F | `lg` F | Delta (lg vs sm) |
|
||||
| --- | --- | --- | --- | --- | --- |
|
||||
| `PER` | 297 | 65.29 | 68.18 | 72.63 | +7.33 |
|
||||
| `LOC` | 273 | 80.24 | 84.05 | 83.66 | +3.42 |
|
||||
| `ORG` | 144 | 68.77 | 70.25 | 71.01 | +2.24 |
|
||||
| `DAT` | 69 | 74.45 | 76.19 | 70.83 | -3.62 |
|
||||
| `MON` | 10 | 73.68 | 84.21 | 88.89 | +15.20 |
|
||||
| `TIM` | 9 | 66.67 | 66.67 | 61.54 | -5.13 |
|
||||
| `PCT` | 4 | 57.14 | 33.33 | 57.14 | +0.00 |
|
||||
|
||||
`PER`, `LOC` and `ORG` (714 entities, the labels with real support) all improve over both
|
||||
smaller tiers. `DAT` and `TIM` regress a few points against `md`; `MON`/`TIM`/`PCT` swings are
|
||||
one-or-two-entity noise, same caveat as §6.
|
||||
|
||||
### Cost
|
||||
|
||||
The bigger table dominates the artifact even more than `md`'s did: `fa_ent_news_lg` is a
|
||||
217 MB wheel against 5.6 MB for `sm` and 58 MB for `md` — the 200k x 300d float32 vector
|
||||
table alone is ~240 MB uncompressed. Training cost was comparable to `sm`/`md` (early stop
|
||||
at step 7,200 of 20,000, best checkpoint at step 5,600). The `spacy benchmark accuracy`
|
||||
words/s figures swung in `lg`'s favor in this run (15,614 vs 8,500 `sm` / 7,149 `md`); given
|
||||
`lg`'s tok2vec architecture is identical to `md`'s and only the static-vector table lookup
|
||||
differs, treat that as single-run CPU contention noise on shared hardware, not a real
|
||||
architectural speedup, and re-benchmark before citing a number.
|
||||
|
||||
For a 4x download over `md` (and 39x over `sm`) buying +4.08 ENTS_F over `sm` (+1.23 over
|
||||
`md`), `lg` is a server/offline-batch pipeline, not something to ship to a browser or a
|
||||
cold-start function. `dep`/`core` at this tier are not yet built; the `ent`-only result above
|
||||
is the signal for whether that investment is worth making.
|
||||
|
|
|
|||
90
project.yml
90
project.yml
|
|
@ -34,6 +34,12 @@ vars:
|
|||
core_md_package_name: "core_news_md"
|
||||
floret_wheel: "fa_floret-0.1.0-py3-none-any-400k-documents.whl"
|
||||
vectors_dir: "assets/vectors/fa_floret_400k"
|
||||
# lg tier. Same architecture as sm/md, but ner only (no dep/core trained yet), with a
|
||||
# larger floret table: 200k rows x 300d, trained on the full Persian Wikipedia dump for
|
||||
# 5 epochs (vs md's 50k rows / 400k documents).
|
||||
ent_lg_package_name: "ent_news_lg"
|
||||
floret_lg_wheel: "fa_floret-0.1.0-py3-none-any-full-wiki-200k-5epoch.whl"
|
||||
vectors_lg_dir: "assets/vectors/fa_floret_lg"
|
||||
|
||||
directories:
|
||||
- "assets"
|
||||
|
|
@ -93,6 +99,16 @@ workflows:
|
|||
- finalize-ent
|
||||
- evaluate-ent
|
||||
- package-ent
|
||||
# The lg tier: ner only, same corpus as sm/md, with a bigger floret table (200k rows,
|
||||
# full Persian Wikipedia, 5 epochs) than md's (50k rows, 400k documents).
|
||||
ent-lg:
|
||||
- vectors-lg
|
||||
- train-ner-lg
|
||||
- finalize-ent-lg
|
||||
- evaluate-ent-lg
|
||||
- compare-lg
|
||||
- package-ent-lg
|
||||
- smoke-ent-lg
|
||||
# The md tier: same corpus and architecture, plus the fa_floret static vectors.
|
||||
md:
|
||||
- vectors-md
|
||||
|
|
@ -314,6 +330,80 @@ commands:
|
|||
outputs:
|
||||
- "packages/${vars.lang}_${vars.ent_package_name}-${vars.package_version}"
|
||||
|
||||
# ---------------------------------------------------------------- lg tier (ner only)
|
||||
|
||||
- name: "vectors-lg"
|
||||
help: >
|
||||
Unpack the lg-tier fa_floret wheel into a plain spaCy model directory. 200k rows x
|
||||
300d in floret mode, trained on the full Persian Wikipedia dump for 5 epochs, vs
|
||||
vectors-md's 50k rows / 400k documents.
|
||||
script:
|
||||
- "python scripts/unpack_vectors.py ${vars.floret_lg_wheel} ${vars.vectors_lg_dir}"
|
||||
deps:
|
||||
- "${vars.floret_lg_wheel}"
|
||||
- "scripts/unpack_vectors.py"
|
||||
outputs:
|
||||
- "${vars.vectors_lg_dir}"
|
||||
|
||||
- name: "train-ner-lg"
|
||||
help: "Train the NER component with the lg-tier static floret vectors"
|
||||
script:
|
||||
- "python -m spacy train configs/fa_ner_lg.cfg --output training/perdt-ner-lg --paths.train corpus/perdt-ner/train.spacy --paths.dev corpus/perdt-ner/dev.spacy --paths.vectors ${vars.vectors_lg_dir} --gpu-id ${vars.gpu}"
|
||||
deps:
|
||||
- "corpus/perdt-ner/train.spacy"
|
||||
- "corpus/perdt-ner/dev.spacy"
|
||||
- "configs/fa_ner_lg.cfg"
|
||||
- "${vars.vectors_lg_dir}"
|
||||
outputs:
|
||||
- "training/perdt-ner-lg/model-best"
|
||||
|
||||
- name: "finalize-ent-lg"
|
||||
help: "Write fa_ent_news_lg metadata onto the trained lg model"
|
||||
script:
|
||||
- "python scripts/finalize_pipeline.py training/perdt-ner-lg/model-best training/fa_ent_news_lg --variant ent --size lg --version ${vars.package_version}"
|
||||
deps:
|
||||
- "training/perdt-ner-lg/model-best"
|
||||
- "scripts/finalize_pipeline.py"
|
||||
outputs:
|
||||
- "training/fa_ent_news_lg"
|
||||
|
||||
- name: "evaluate-ent-lg"
|
||||
help: "Score fa_ent_news_lg on the held-out PerDT NER test split"
|
||||
script:
|
||||
- "python -m spacy benchmark accuracy training/fa_ent_news_lg corpus/perdt-ner/test.spacy --output metrics/lg-perdt-ner-test.json --gpu-id ${vars.gpu}"
|
||||
- "python scripts/finalize_pipeline.py training/perdt-ner-lg/model-best training/fa_ent_news_lg --variant ent --size lg --version ${vars.package_version} --ner-metrics metrics/lg-perdt-ner-test.json"
|
||||
deps:
|
||||
- "training/fa_ent_news_lg"
|
||||
- "corpus/perdt-ner/test.spacy"
|
||||
outputs:
|
||||
- "metrics/lg-perdt-ner-test.json"
|
||||
|
||||
- name: "compare-lg"
|
||||
help: "Table the sm vs md vs lg ent NER deltas from the metrics/ JSON reports"
|
||||
script:
|
||||
- "python scripts/compare_tiers.py"
|
||||
deps:
|
||||
- "metrics/perdt-ner-test.json"
|
||||
- "metrics/md-perdt-ner-test.json"
|
||||
- "metrics/lg-perdt-ner-test.json"
|
||||
- "scripts/compare_tiers.py"
|
||||
|
||||
- name: "package-ent-lg"
|
||||
help: "Build the installable fa_ent_news_lg wheel + sdist"
|
||||
script:
|
||||
- "python -m spacy package training/fa_ent_news_lg packages --name ${vars.ent_lg_package_name} --version ${vars.package_version} --build sdist,wheel --force"
|
||||
deps:
|
||||
- "training/fa_ent_news_lg"
|
||||
outputs:
|
||||
- "packages/${vars.lang}_${vars.ent_lg_package_name}-${vars.package_version}"
|
||||
|
||||
- name: "smoke-ent-lg"
|
||||
help: "Load fa_ent_news_lg and run it over real Persian text"
|
||||
script:
|
||||
- "python scripts/smoke_test.py training/fa_ent_news_lg"
|
||||
deps:
|
||||
- "training/fa_ent_news_lg"
|
||||
|
||||
# ---------------------------------------------------------------- md tier
|
||||
|
||||
- name: "vectors-md"
|
||||
|
|
|
|||
|
|
@ -1,8 +1,9 @@
|
|||
"""Table the sm vs md test-set deltas.
|
||||
"""Table the sm vs md vs lg test-set deltas.
|
||||
|
||||
Both tiers are trained from the same corpus, the same seed and the same architecture; the
|
||||
only difference is `include_static_vectors`. So the delta printed here is attributable to the
|
||||
fa_floret vector table and nothing else.
|
||||
All tiers are trained from the same corpus, the same seed and the same architecture; the
|
||||
only difference is the static vector table (none for sm, fa_floret 50k rows for md, fa_floret
|
||||
200k rows for lg) via `include_static_vectors`. So the delta printed here is attributable to
|
||||
the vector table and nothing else.
|
||||
|
||||
Reads the `spacy benchmark accuracy` reports written by the `evaluate-*` targets. Missing
|
||||
files are reported rather than fatal, so this is runnable mid-build.
|
||||
|
|
@ -15,11 +16,18 @@ import argparse
|
|||
import json
|
||||
from pathlib import Path
|
||||
|
||||
# (label, sm report, md report)
|
||||
PAIRS = [
|
||||
("dep pipeline, UD test", "ud-test.json", "md-ud-test.json"),
|
||||
("core pipeline, UD test", "core-ud-test.json", "md-core-ud-test.json"),
|
||||
("core pipeline, NER test", "perdt-ner-test.json", "md-perdt-ner-test.json"),
|
||||
# (label, {tier_label: report_filename})
|
||||
GROUPS = [
|
||||
("dep pipeline, UD test", {"sm": "ud-test.json", "md": "md-ud-test.json"}),
|
||||
("core pipeline, UD test", {"sm": "core-ud-test.json", "md": "md-core-ud-test.json"}),
|
||||
(
|
||||
"ent NER test",
|
||||
{
|
||||
"sm": "perdt-ner-test.json",
|
||||
"md": "md-perdt-ner-test.json",
|
||||
"lg": "lg-perdt-ner-test.json",
|
||||
},
|
||||
),
|
||||
]
|
||||
|
||||
SCALARS = [
|
||||
|
|
@ -40,41 +48,81 @@ def load(path):
|
|||
return json.loads(path.read_text(encoding="utf8")) if path.exists() else None
|
||||
|
||||
|
||||
def table(title, sm, md, rows):
|
||||
def table(title, tiers, rows):
|
||||
"""tiers: list of (label, data-dict-or-None), first tier is the baseline for deltas."""
|
||||
labels = [label for label, _ in tiers]
|
||||
base_label, base = tiers[0]
|
||||
print(f"\n## {title}\n")
|
||||
print(f"| {'metric':<12} | {'sm':>7} | {'md':>7} | {'delta':>7} |")
|
||||
print(f"| {'-' * 12} | {'-' * 7} | {'-' * 7} | {'-' * 7} |")
|
||||
header = " | ".join(f"{label:>7}" for label in labels)
|
||||
delta_header = " | ".join(f"{'d(' + label + ')':>9}" for label, _ in tiers[1:])
|
||||
print(f"| {'metric':<12} | {header} | {delta_header} |")
|
||||
sep = " | ".join("-" * 7 for _ in labels)
|
||||
delta_sep = " | ".join("-" * 9 for _ in tiers[1:])
|
||||
print(f"| {'-' * 12} | {sep} | {delta_sep} |")
|
||||
for key, label in rows:
|
||||
a, b = sm.get(key), md.get(key)
|
||||
if a is None and b is None:
|
||||
values = [d.get(key) if d is not None else None for _, d in tiers]
|
||||
if all(v is None for v in values):
|
||||
continue
|
||||
# The NER report scores tag_acc 0.0 because its corpus has no gold tags.
|
||||
if a == 0.0 and b == 0.0:
|
||||
if all(v == 0.0 for v in values):
|
||||
continue
|
||||
cells = [f"{v * 100:.2f}" if isinstance(v, float) else "-" for v in (a, b)]
|
||||
delta = f"{(b - a) * 100:+.2f}" if isinstance(a, float) and isinstance(b, float) else "-"
|
||||
print(f"| {label:<12} | {cells[0]:>7} | {cells[1]:>7} | {delta:>7} |")
|
||||
for key, label in (("speed", "words/s"),):
|
||||
a, b = sm.get(key), md.get(key)
|
||||
if isinstance(a, float) and isinstance(b, float):
|
||||
print(f"| {label:<12} | {a:>7.0f} | {b:>7.0f} | {b / a - 1:>+6.1%} |")
|
||||
cells = [f"{v * 100:.2f}" if isinstance(v, float) else "-" for v in values]
|
||||
deltas = []
|
||||
for v in values[1:]:
|
||||
a, b = values[0], v
|
||||
deltas.append(
|
||||
f"{(b - a) * 100:+.2f}" if isinstance(a, float) and isinstance(b, float) else "-"
|
||||
)
|
||||
row = " | ".join(f"{c:>7}" for c in cells)
|
||||
drow = " | ".join(f"{d:>9}" for d in deltas)
|
||||
print(f"| {label:<12} | {row} | {drow} |")
|
||||
speeds = [d.get("speed") if d is not None else None for _, d in tiers]
|
||||
if isinstance(speeds[0], float):
|
||||
cells = [f"{s:.0f}" if isinstance(s, float) else "-" for s in speeds]
|
||||
deltas = [
|
||||
f"{s / speeds[0] - 1:+.1%}" if isinstance(s, float) else "-" for s in speeds[1:]
|
||||
]
|
||||
row = " | ".join(f"{c:>7}" for c in cells)
|
||||
drow = " | ".join(f"{d:>9}" for d in deltas)
|
||||
print(f"| {'words/s':<12} | {row} | {drow} |")
|
||||
|
||||
|
||||
def per_type(title, sm, md):
|
||||
a, b = sm.get("ents_per_type"), md.get("ents_per_type")
|
||||
if not a or not b:
|
||||
def per_type(title, tiers):
|
||||
per_types = [(label, (d or {}).get("ents_per_type")) for label, d in tiers]
|
||||
if not any(pt for _, pt in per_types):
|
||||
return
|
||||
labels = [label for label, _ in tiers]
|
||||
|
||||
def pct(v):
|
||||
return f"{v * 100:.2f}" if v is not None else "-"
|
||||
|
||||
all_labels = set()
|
||||
for _, pt in per_types:
|
||||
if pt:
|
||||
all_labels |= set(pt)
|
||||
|
||||
print(f"\n### {title}, per label\n")
|
||||
print(f"| {'label':<6} | {'sm F':>7} | {'md F':>7} | {'delta':>7} |")
|
||||
print(f"| {'-' * 6} | {'-' * 7} | {'-' * 7} | {'-' * 7} |")
|
||||
for label in sorted(set(a) | set(b), key=lambda k: -b.get(k, {}).get("f", 0)):
|
||||
fa, fb = a.get(label, {}).get("f"), b.get(label, {}).get("f")
|
||||
delta = f"{(fb - fa) * 100:+.2f}" if fa is not None and fb is not None else "-"
|
||||
print(f"| {label:<6} | {pct(fa):>7} | {pct(fb):>7} | {delta:>7} |")
|
||||
header = " | ".join(f"{label + ' F':>7}" for label in labels)
|
||||
delta_header = " | ".join(f"{'d(' + label + ')':>9}" for label in labels[1:])
|
||||
print(f"| {'label':<6} | {header} | {delta_header} |")
|
||||
sep = " | ".join("-" * 7 for _ in labels)
|
||||
delta_sep = " | ".join("-" * 9 for _ in labels[1:])
|
||||
print(f"| {'-' * 6} | {sep} | {delta_sep} |")
|
||||
|
||||
def sort_key(entity_label):
|
||||
last_pt = per_types[-1][1] or {}
|
||||
return -last_pt.get(entity_label, {}).get("f", 0)
|
||||
|
||||
for entity_label in sorted(all_labels, key=sort_key):
|
||||
fs = [(pt or {}).get(entity_label, {}).get("f") for _, pt in per_types]
|
||||
cells = [pct(f) for f in fs]
|
||||
deltas = []
|
||||
for f in fs[1:]:
|
||||
a = fs[0]
|
||||
deltas.append(f"{(f - a) * 100:+.2f}" if a is not None and f is not None else "-")
|
||||
row = " | ".join(f"{c:>7}" for c in cells)
|
||||
drow = " | ".join(f"{d:>9}" for d in deltas)
|
||||
print(f"| {entity_label:<6} | {row} | {drow} |")
|
||||
|
||||
|
||||
def main():
|
||||
|
|
@ -82,19 +130,26 @@ def main():
|
|||
ap.add_argument("--metrics-dir", type=Path, default=Path("metrics"))
|
||||
args = ap.parse_args()
|
||||
|
||||
print("# sm vs md (fa_floret 400k static vectors)")
|
||||
print("\nSame corpus, same seed, same architecture. Only difference:")
|
||||
print("`include_static_vectors = false -> true`.")
|
||||
print("# sm vs md vs lg (fa_floret static vectors)")
|
||||
print("\nSame corpus, same seed, same architecture per group. Only difference:")
|
||||
print("`include_static_vectors = false -> true`, and which floret table (md: 50k rows,")
|
||||
print("400k documents; lg: 200k rows, full Persian Wikipedia, 5 epochs).")
|
||||
|
||||
for title, sm_name, md_name in PAIRS:
|
||||
sm = load(args.metrics_dir / sm_name)
|
||||
md = load(args.metrics_dir / md_name)
|
||||
if sm is None or md is None:
|
||||
missing = [n for n, d in ((sm_name, sm), (md_name, md)) if d is None]
|
||||
print(f"\n## {title}\n\n (skipped, missing {', '.join(missing)})")
|
||||
for title, reports in GROUPS:
|
||||
tiers = []
|
||||
missing = []
|
||||
for label, fname in reports.items():
|
||||
data = load(args.metrics_dir / fname)
|
||||
if data is None:
|
||||
missing.append(fname)
|
||||
tiers.append((label, data))
|
||||
if tiers[0][1] is None:
|
||||
print(f"\n## {title}\n\n (skipped, missing baseline {reports[list(reports)[0]]})")
|
||||
continue
|
||||
table(title, sm, md, SCALARS)
|
||||
per_type(title, sm, md)
|
||||
table(title, tiers, SCALARS)
|
||||
per_type(title, tiers)
|
||||
if missing:
|
||||
print(f"\n (missing: {', '.join(missing)})")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
|
|
|||
|
|
@ -67,8 +67,8 @@ FLORET = {
|
|||
"license": "CC BY-SA 4.0",
|
||||
}
|
||||
FLORET_LG = {
|
||||
"name": "fa_floret static vectors (lg tier: larger floret table trained on fa Wikipedia + "
|
||||
"OSCAR via spacy-vectors-builder)",
|
||||
"name": "fa_floret static vectors (lg tier: 200k rows x 300d floret table trained on "
|
||||
"the full Persian Wikipedia dump, 5 epochs, via spacy-vectors-builder)",
|
||||
"url": PROJECT_URL,
|
||||
"author": "Kiyarash Fazeli",
|
||||
"license": "CC BY-SA 4.0",
|
||||
|
|
@ -119,9 +119,10 @@ def vectors_note_lg(nlp):
|
|||
return (
|
||||
f"This is the `lg` tier: identical architecture to `sm`/`md` but a larger static "
|
||||
f"floret vector table ({rows:,} rows x {dim} dimensions, minn=maxn=5, hash_count=2) "
|
||||
f"trained on Persian Wikipedia + OSCAR via spacy-vectors-builder. Same zero-OOV "
|
||||
f"rationale as `md` (see docs/MODELS.md): floret hashes subwords into a fixed table, "
|
||||
f"so `token.has_vector` is always True despite Persian's ZWNJ (U+200C) inconsistency."
|
||||
f"trained on the full Persian Wikipedia dump for 5 epochs via spacy-vectors-builder. "
|
||||
f"Same zero-OOV rationale as `md` (see docs/MODELS.md): floret hashes subwords into "
|
||||
f"a fixed table, so `token.has_vector` is always True despite Persian's ZWNJ "
|
||||
f"(U+200C) inconsistency."
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
Loading…
Reference in New Issue