Add Colab GPU training notebook for lg/trf tiers
- notebooks/fa_lg_trf_training.ipynb: trains fa_dep_news_lg/fa_core_news_lg (floret vectors, byte-identical recipe to md, only --paths.vectors differs) and fa_dep_news_trf/fa_core_news_trf (HooshvareLab/roberta-fa-zwnj-base, fine-tuned via spacy-transformers; not ParsBERT, whose model card carries no licence). Self-contained: downloads UD_Persian-PerDT from public GitHub URLs, expects the repo source (git archive) and an lg floret wheel uploaded by hand since Gitea is LAN-only and floret training happens elsewhere. - scripts/finalize_pipeline.py: extend --size to lg/trf. lg's vectors note is generated from the trained model's actual vector table shape rather than hardcoded, since that table is still being iterated on; trf records the transformer source instead of a floret one. - configs/fa_dep_news_md.cfg, configs/fa_ner_md.cfg, scripts/unpack_vectors.py: existing dependencies the notebook calls directly, committed so a git-archive-based upload is self-sufficient.
This commit is contained in:
parent
079663083a
commit
6abc333078
|
|
@ -0,0 +1,235 @@
|
||||||
|
# fa_dep_news_md — tagger, morphologizer, trainable_lemmatizer, parser, WITH static vectors.
|
||||||
|
#
|
||||||
|
# Byte-identical to configs/fa_dep_news_sm.cfg except:
|
||||||
|
# - [components.tok2vec.model.embed] include_static_vectors: false -> true
|
||||||
|
#
|
||||||
|
# Everything else (seed, widths, rows, batcher, patience, eval_frequency) is held constant so
|
||||||
|
# the sm/md delta measures the floret vectors and nothing else.
|
||||||
|
#
|
||||||
|
# Vectors are supplied at train time via --paths.vectors, pointing at the fa_floret table
|
||||||
|
# (50k rows x 300d, floret mode, minn=maxn=5, hash_count=2) trained on 400k Persian documents.
|
||||||
|
# floret has no OOV: every string hashes into the table, which is the point for Persian, where
|
||||||
|
# ZWNJ inconsistency (میرود / میرود / می رود) would shatter a classic word table.
|
||||||
|
#
|
||||||
|
# No `ner` here by design; see configs/fa_ner_md.cfg and project.yml.
|
||||||
|
|
||||||
|
[paths]
|
||||||
|
train = null
|
||||||
|
dev = null
|
||||||
|
vectors = null
|
||||||
|
init_tok2vec = null
|
||||||
|
|
||||||
|
[system]
|
||||||
|
gpu_allocator = null
|
||||||
|
seed = 0
|
||||||
|
|
||||||
|
[nlp]
|
||||||
|
lang = "fa"
|
||||||
|
pipeline = ["tok2vec", "tagger", "morphologizer", "trainable_lemmatizer", "parser"]
|
||||||
|
batch_size = 1000
|
||||||
|
disabled = []
|
||||||
|
before_creation = null
|
||||||
|
after_creation = null
|
||||||
|
after_pipeline_creation = null
|
||||||
|
|
||||||
|
[corpora]
|
||||||
|
|
||||||
|
[training]
|
||||||
|
dev_corpus = "corpora.dev"
|
||||||
|
train_corpus = "corpora.train"
|
||||||
|
seed = ${system.seed}
|
||||||
|
gpu_allocator = ${system.gpu_allocator}
|
||||||
|
dropout = 0.1
|
||||||
|
accumulate_gradient = 1
|
||||||
|
patience = 1600
|
||||||
|
max_epochs = 0
|
||||||
|
max_steps = 20000
|
||||||
|
eval_frequency = 400
|
||||||
|
frozen_components = []
|
||||||
|
annotating_components = []
|
||||||
|
before_to_disk = null
|
||||||
|
before_update = null
|
||||||
|
|
||||||
|
[initialize]
|
||||||
|
vectors = ${paths.vectors}
|
||||||
|
init_tok2vec = ${paths.init_tok2vec}
|
||||||
|
vocab_data = null
|
||||||
|
lookups = null
|
||||||
|
before_init = null
|
||||||
|
after_init = null
|
||||||
|
|
||||||
|
[components]
|
||||||
|
|
||||||
|
[pretraining]
|
||||||
|
|
||||||
|
[nlp.tokenizer]
|
||||||
|
@tokenizers = "spacy.Tokenizer.v1"
|
||||||
|
|
||||||
|
[nlp.vectors]
|
||||||
|
@vectors = "spacy.Vectors.v1"
|
||||||
|
|
||||||
|
[corpora.train]
|
||||||
|
@readers = "spacy.Corpus.v1"
|
||||||
|
path = ${paths.train}
|
||||||
|
max_length = 0
|
||||||
|
gold_preproc = false
|
||||||
|
limit = 0
|
||||||
|
augmenter = null
|
||||||
|
|
||||||
|
[corpora.dev]
|
||||||
|
@readers = "spacy.Corpus.v1"
|
||||||
|
path = ${paths.dev}
|
||||||
|
max_length = 0
|
||||||
|
gold_preproc = false
|
||||||
|
limit = 0
|
||||||
|
augmenter = null
|
||||||
|
|
||||||
|
[training.optimizer]
|
||||||
|
@optimizers = "Adam.v1"
|
||||||
|
beta1 = 0.9
|
||||||
|
beta2 = 0.999
|
||||||
|
L2_is_weight_decay = true
|
||||||
|
L2 = 0.01
|
||||||
|
grad_clip = 1.0
|
||||||
|
use_averages = false
|
||||||
|
eps = 1e-08
|
||||||
|
learn_rate = 0.001
|
||||||
|
|
||||||
|
[training.batcher]
|
||||||
|
@batchers = "spacy.batch_by_words.v1"
|
||||||
|
discard_oversize = false
|
||||||
|
tolerance = 0.2
|
||||||
|
get_length = null
|
||||||
|
|
||||||
|
[training.logger]
|
||||||
|
@loggers = "spacy.ConsoleLogger.v1"
|
||||||
|
progress_bar = false
|
||||||
|
|
||||||
|
[training.score_weights]
|
||||||
|
tag_acc = 0.25
|
||||||
|
pos_acc = 0.12
|
||||||
|
tag_micro_p = null
|
||||||
|
tag_micro_r = null
|
||||||
|
tag_micro_f = null
|
||||||
|
morph_acc = 0.12
|
||||||
|
morph_per_feat = null
|
||||||
|
lemma_acc = 0.25
|
||||||
|
dep_uas = 0.12
|
||||||
|
dep_las = 0.12
|
||||||
|
dep_las_per_type = null
|
||||||
|
sents_p = null
|
||||||
|
sents_r = null
|
||||||
|
sents_f = 0.0
|
||||||
|
|
||||||
|
[initialize.tokenizer]
|
||||||
|
|
||||||
|
[initialize.components]
|
||||||
|
|
||||||
|
[components.tok2vec]
|
||||||
|
factory = "tok2vec"
|
||||||
|
|
||||||
|
[components.tagger]
|
||||||
|
factory = "tagger"
|
||||||
|
label_smoothing = 0.05
|
||||||
|
overwrite = false
|
||||||
|
neg_prefix = "!"
|
||||||
|
|
||||||
|
[components.morphologizer]
|
||||||
|
factory = "morphologizer"
|
||||||
|
label_smoothing = 0.05
|
||||||
|
overwrite = true
|
||||||
|
extend = false
|
||||||
|
|
||||||
|
[components.trainable_lemmatizer]
|
||||||
|
factory = "trainable_lemmatizer"
|
||||||
|
backoff = "orth"
|
||||||
|
min_tree_freq = 3
|
||||||
|
overwrite = false
|
||||||
|
top_k = 1
|
||||||
|
|
||||||
|
[components.parser]
|
||||||
|
factory = "parser"
|
||||||
|
moves = null
|
||||||
|
update_with_oracle_cut_size = 100
|
||||||
|
learn_tokens = false
|
||||||
|
min_action_freq = 30
|
||||||
|
|
||||||
|
[training.batcher.size]
|
||||||
|
@schedules = "compounding.v1"
|
||||||
|
start = 100
|
||||||
|
stop = 1000
|
||||||
|
compound = 1.001
|
||||||
|
t = 0.0
|
||||||
|
|
||||||
|
[components.tok2vec.model]
|
||||||
|
@architectures = "spacy.Tok2Vec.v2"
|
||||||
|
|
||||||
|
[components.tagger.model]
|
||||||
|
@architectures = "spacy.Tagger.v2"
|
||||||
|
nO = null
|
||||||
|
normalize = false
|
||||||
|
|
||||||
|
[components.tagger.scorer]
|
||||||
|
@scorers = "spacy.tagger_scorer.v1"
|
||||||
|
|
||||||
|
[components.morphologizer.model]
|
||||||
|
@architectures = "spacy.Tagger.v2"
|
||||||
|
nO = null
|
||||||
|
normalize = false
|
||||||
|
|
||||||
|
[components.morphologizer.scorer]
|
||||||
|
@scorers = "spacy.morphologizer_scorer.v1"
|
||||||
|
|
||||||
|
[components.trainable_lemmatizer.model]
|
||||||
|
@architectures = "spacy.Tagger.v2"
|
||||||
|
nO = null
|
||||||
|
normalize = false
|
||||||
|
|
||||||
|
[components.trainable_lemmatizer.scorer]
|
||||||
|
@scorers = "spacy.lemmatizer_scorer.v1"
|
||||||
|
|
||||||
|
[components.parser.model]
|
||||||
|
@architectures = "spacy.TransitionBasedParser.v2"
|
||||||
|
state_type = "parser"
|
||||||
|
extra_state_tokens = false
|
||||||
|
hidden_width = 128
|
||||||
|
maxout_pieces = 3
|
||||||
|
use_upper = true
|
||||||
|
nO = null
|
||||||
|
|
||||||
|
[components.parser.scorer]
|
||||||
|
@scorers = "spacy.parser_scorer.v1"
|
||||||
|
|
||||||
|
[components.tok2vec.model.embed]
|
||||||
|
@architectures = "spacy.MultiHashEmbed.v2"
|
||||||
|
width = ${components.tok2vec.model.encode.width}
|
||||||
|
attrs = ["NORM", "PREFIX", "SUFFIX", "SHAPE"]
|
||||||
|
rows = [5000, 1000, 2500, 2500]
|
||||||
|
include_static_vectors = true
|
||||||
|
|
||||||
|
[components.tok2vec.model.encode]
|
||||||
|
@architectures = "spacy.MaxoutWindowEncoder.v2"
|
||||||
|
width = 96
|
||||||
|
depth = 4
|
||||||
|
window_size = 1
|
||||||
|
maxout_pieces = 3
|
||||||
|
|
||||||
|
[components.tagger.model.tok2vec]
|
||||||
|
@architectures = "spacy.Tok2VecListener.v1"
|
||||||
|
width = ${components.tok2vec.model.encode.width}
|
||||||
|
upstream = "*"
|
||||||
|
|
||||||
|
[components.morphologizer.model.tok2vec]
|
||||||
|
@architectures = "spacy.Tok2VecListener.v1"
|
||||||
|
width = ${components.tok2vec.model.encode.width}
|
||||||
|
upstream = "*"
|
||||||
|
|
||||||
|
[components.trainable_lemmatizer.model.tok2vec]
|
||||||
|
@architectures = "spacy.Tok2VecListener.v1"
|
||||||
|
width = ${components.tok2vec.model.encode.width}
|
||||||
|
upstream = "*"
|
||||||
|
|
||||||
|
[components.parser.model.tok2vec]
|
||||||
|
@architectures = "spacy.Tok2VecListener.v1"
|
||||||
|
width = ${components.tok2vec.model.encode.width}
|
||||||
|
upstream = "*"
|
||||||
|
|
@ -0,0 +1,154 @@
|
||||||
|
# fa_ent_news_md — Persian NER with static floret vectors.
|
||||||
|
#
|
||||||
|
# Identical to configs/fa_ner_sm.cfg except:
|
||||||
|
# - [components.ner.model.tok2vec.embed] include_static_vectors: false -> true
|
||||||
|
#
|
||||||
|
# Same embedded-tok2vec design as the sm variant (no Tok2VecListener), so the trained
|
||||||
|
# component stays sourceable into fa_core_news_md via `nlp.add_pipe("ner", source=...)`.
|
||||||
|
#
|
||||||
|
# Vectors supplied at train time via --paths.vectors (fa_floret, 50k rows x 300d,
|
||||||
|
# trained on 400k Persian documents).
|
||||||
|
|
||||||
|
[paths]
|
||||||
|
train = null
|
||||||
|
dev = null
|
||||||
|
vectors = null
|
||||||
|
init_tok2vec = null
|
||||||
|
|
||||||
|
[system]
|
||||||
|
gpu_allocator = null
|
||||||
|
seed = 0
|
||||||
|
|
||||||
|
[nlp]
|
||||||
|
lang = "fa"
|
||||||
|
pipeline = ["ner"]
|
||||||
|
batch_size = 1000
|
||||||
|
disabled = []
|
||||||
|
before_creation = null
|
||||||
|
after_creation = null
|
||||||
|
after_pipeline_creation = null
|
||||||
|
|
||||||
|
[nlp.tokenizer]
|
||||||
|
@tokenizers = "spacy.Tokenizer.v1"
|
||||||
|
|
||||||
|
[nlp.vectors]
|
||||||
|
@vectors = "spacy.Vectors.v1"
|
||||||
|
|
||||||
|
[components]
|
||||||
|
|
||||||
|
[components.ner]
|
||||||
|
factory = "ner"
|
||||||
|
moves = null
|
||||||
|
update_with_oracle_cut_size = 100
|
||||||
|
incorrect_spans_key = null
|
||||||
|
|
||||||
|
[components.ner.model]
|
||||||
|
@architectures = "spacy.TransitionBasedParser.v2"
|
||||||
|
state_type = "ner"
|
||||||
|
extra_state_tokens = false
|
||||||
|
hidden_width = 64
|
||||||
|
maxout_pieces = 2
|
||||||
|
use_upper = true
|
||||||
|
nO = null
|
||||||
|
|
||||||
|
[components.ner.model.tok2vec]
|
||||||
|
@architectures = "spacy.Tok2Vec.v2"
|
||||||
|
|
||||||
|
[components.ner.model.tok2vec.embed]
|
||||||
|
@architectures = "spacy.MultiHashEmbed.v2"
|
||||||
|
width = ${components.ner.model.tok2vec.encode.width}
|
||||||
|
attrs = ["NORM", "PREFIX", "SUFFIX", "SHAPE"]
|
||||||
|
rows = [5000, 1000, 2500, 2500]
|
||||||
|
include_static_vectors = true
|
||||||
|
|
||||||
|
[components.ner.model.tok2vec.encode]
|
||||||
|
@architectures = "spacy.MaxoutWindowEncoder.v2"
|
||||||
|
width = 96
|
||||||
|
depth = 4
|
||||||
|
window_size = 1
|
||||||
|
maxout_pieces = 3
|
||||||
|
|
||||||
|
[components.ner.scorer]
|
||||||
|
@scorers = "spacy.ner_scorer.v1"
|
||||||
|
|
||||||
|
[corpora]
|
||||||
|
|
||||||
|
[corpora.train]
|
||||||
|
@readers = "spacy.Corpus.v1"
|
||||||
|
path = ${paths.train}
|
||||||
|
max_length = 0
|
||||||
|
gold_preproc = false
|
||||||
|
limit = 0
|
||||||
|
augmenter = null
|
||||||
|
|
||||||
|
[corpora.dev]
|
||||||
|
@readers = "spacy.Corpus.v1"
|
||||||
|
path = ${paths.dev}
|
||||||
|
max_length = 0
|
||||||
|
gold_preproc = false
|
||||||
|
limit = 0
|
||||||
|
augmenter = null
|
||||||
|
|
||||||
|
[training]
|
||||||
|
dev_corpus = "corpora.dev"
|
||||||
|
train_corpus = "corpora.train"
|
||||||
|
seed = ${system.seed}
|
||||||
|
gpu_allocator = ${system.gpu_allocator}
|
||||||
|
dropout = 0.1
|
||||||
|
accumulate_gradient = 1
|
||||||
|
patience = 1600
|
||||||
|
max_epochs = 0
|
||||||
|
max_steps = 20000
|
||||||
|
eval_frequency = 400
|
||||||
|
frozen_components = []
|
||||||
|
annotating_components = []
|
||||||
|
before_to_disk = null
|
||||||
|
before_update = null
|
||||||
|
|
||||||
|
[training.optimizer]
|
||||||
|
@optimizers = "Adam.v1"
|
||||||
|
beta1 = 0.9
|
||||||
|
beta2 = 0.999
|
||||||
|
L2_is_weight_decay = true
|
||||||
|
L2 = 0.01
|
||||||
|
grad_clip = 1.0
|
||||||
|
use_averages = false
|
||||||
|
eps = 1e-08
|
||||||
|
learn_rate = 0.001
|
||||||
|
|
||||||
|
[training.batcher]
|
||||||
|
@batchers = "spacy.batch_by_words.v1"
|
||||||
|
discard_oversize = false
|
||||||
|
tolerance = 0.2
|
||||||
|
get_length = null
|
||||||
|
|
||||||
|
[training.batcher.size]
|
||||||
|
@schedules = "compounding.v1"
|
||||||
|
start = 100
|
||||||
|
stop = 1000
|
||||||
|
compound = 1.001
|
||||||
|
t = 0.0
|
||||||
|
|
||||||
|
[training.logger]
|
||||||
|
@loggers = "spacy.ConsoleLogger.v1"
|
||||||
|
progress_bar = false
|
||||||
|
|
||||||
|
[training.score_weights]
|
||||||
|
ents_f = 1.0
|
||||||
|
ents_p = 0.0
|
||||||
|
ents_r = 0.0
|
||||||
|
ents_per_type = null
|
||||||
|
|
||||||
|
[initialize]
|
||||||
|
vectors = ${paths.vectors}
|
||||||
|
init_tok2vec = ${paths.init_tok2vec}
|
||||||
|
vocab_data = null
|
||||||
|
lookups = null
|
||||||
|
before_init = null
|
||||||
|
after_init = null
|
||||||
|
|
||||||
|
[initialize.tokenizer]
|
||||||
|
|
||||||
|
[initialize.components]
|
||||||
|
|
||||||
|
[pretraining]
|
||||||
|
|
@ -0,0 +1,523 @@
|
||||||
|
{
|
||||||
|
"cells": [
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"# fa_dep_news_lg / fa_core_news_lg / fa_dep_news_trf / fa_core_news_trf — Colab training\n",
|
||||||
|
"\n",
|
||||||
|
"Trains the `lg` (bigger floret vectors) and `trf` (fine-tuned transformer) tiers of the\n",
|
||||||
|
"Persian `spacy-fa-pipeline` project on a Colab GPU. `sm`/`md` are already built on CPU\n",
|
||||||
|
"locally — this notebook only adds the two tiers that need real GPU memory.\n",
|
||||||
|
"\n",
|
||||||
|
"**`trf` uses `HooshvareLab/roberta-fa-zwnj-base` (Apache-2.0), not ParsBERT** — ParsBERT's\n",
|
||||||
|
"model card carries no explicit licence, which is disqualifying for a package meant to be\n",
|
||||||
|
"redistributed. See `TODO.md` in the repo.\n",
|
||||||
|
"\n",
|
||||||
|
"Floret vector *training* itself (the actual `lg`-tier 200k-row Wikipedia+OSCAR table) is\n",
|
||||||
|
"not part of this notebook — that happens elsewhere (CPU-days, `spacy-vectors-builder`).\n",
|
||||||
|
"This notebook only trains spaCy pipelines against whatever floret wheel you upload in\n",
|
||||||
|
"step 6.\n",
|
||||||
|
"\n",
|
||||||
|
"## Before you run this\n",
|
||||||
|
"\n",
|
||||||
|
"1. **Runtime -> Change runtime type -> GPU** (a 16 GB T4/A10 is plenty for a base-size\n",
|
||||||
|
" transformer; no need for A100).\n",
|
||||||
|
"2. Have ready, to upload when asked:\n",
|
||||||
|
" - A zip of the repo's **source only** (`git archive -o repo.zip HEAD` from the repo\n",
|
||||||
|
" root -- this naturally excludes everything `.gitignore` excludes: `assets/ corpus/\n",
|
||||||
|
" training/ metrics/ packages/ .venv/`). The self-hosted Gitea remote is LAN-only and\n",
|
||||||
|
" unreachable from Colab, so this notebook cannot `git clone` it directly.\n",
|
||||||
|
" - An `lg`-tier floret wheel (`fa_floret-0.1.0-py3-none-any-*.whl`) once it's built\n",
|
||||||
|
" elsewhere. If you don't have one yet, upload whatever `md`-tier wheel you have as a\n",
|
||||||
|
" stand-in -- the run will still be valid, just not the final `lg` numbers.\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 1. Confirm the GPU"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"!nvidia-smi\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 2. Upload the repo source\n",
|
||||||
|
"\n",
|
||||||
|
"Upload the `repo.zip` produced by `git archive -o repo.zip HEAD` (run locally, in the repo\n",
|
||||||
|
"root, before starting this notebook).\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"from google.colab import files\n",
|
||||||
|
"import zipfile, pathlib\n",
|
||||||
|
"\n",
|
||||||
|
"REPO = pathlib.Path(\"/content/repo\")\n",
|
||||||
|
"REPO.mkdir(parents=True, exist_ok=True)\n",
|
||||||
|
"\n",
|
||||||
|
"uploaded = files.upload()\n",
|
||||||
|
"(zip_name,) = uploaded.keys()\n",
|
||||||
|
"with zipfile.ZipFile(zip_name) as z:\n",
|
||||||
|
" z.extractall(REPO)\n",
|
||||||
|
"\n",
|
||||||
|
"%cd {REPO}\n",
|
||||||
|
"!ls\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 3. Install dependencies"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"# Colab ships a CUDA-enabled torch already; spacy[transformers] pulls in spacy-transformers +\n",
|
||||||
|
"# a matching transformers/tokenizers. Installing spacy[cuda-autodetect] too is cheap insurance\n",
|
||||||
|
"# for the GPU allocator path (unlike the local 940MX box, this doesn't need a manual cupy[ctk]\n",
|
||||||
|
"# CUDA-toolkit install -- Colab's base image already has the CUDA libs on the system path).\n",
|
||||||
|
"!pip install -q -U pip\n",
|
||||||
|
"!pip install -q \"spacy[transformers,cuda-autodetect]\" spacy-transformers spacy-lookups-data\n",
|
||||||
|
"\n",
|
||||||
|
"import spacy, torch, spacy_transformers\n",
|
||||||
|
"print(\"spacy\", spacy.__version__)\n",
|
||||||
|
"print(\"spacy-transformers\", spacy_transformers.__version__)\n",
|
||||||
|
"print(\"torch\", torch.__version__, \"cuda available:\", torch.cuda.is_available())\n",
|
||||||
|
"\n",
|
||||||
|
"from thinc.api import prefer_gpu\n",
|
||||||
|
"print(\"thinc prefer_gpu:\", prefer_gpu())\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 4. Download the UD_Persian-PerDT assets\n",
|
||||||
|
"\n",
|
||||||
|
"Same public GitHub URLs and checksums as `project.yml` -- no private infrastructure needed.\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import hashlib, urllib.request, pathlib\n",
|
||||||
|
"\n",
|
||||||
|
"ASSETS = [\n",
|
||||||
|
" (\"assets/ud/fa_perdt-ud-train.conllu\",\n",
|
||||||
|
" \"https://raw.githubusercontent.com/UniversalDependencies/UD_Persian-PerDT/master/fa_perdt-ud-train.conllu\",\n",
|
||||||
|
" \"f5a8ba901a776b4fd1941ecadcc6d506\"),\n",
|
||||||
|
" (\"assets/ud/fa_perdt-ud-dev.conllu\",\n",
|
||||||
|
" \"https://raw.githubusercontent.com/UniversalDependencies/UD_Persian-PerDT/master/fa_perdt-ud-dev.conllu\",\n",
|
||||||
|
" \"f103020da7c1e917aafb8a8321f4cb84\"),\n",
|
||||||
|
" (\"assets/ud/fa_perdt-ud-test.conllu\",\n",
|
||||||
|
" \"https://raw.githubusercontent.com/UniversalDependencies/UD_Persian-PerDT/master/fa_perdt-ud-test.conllu\",\n",
|
||||||
|
" \"b62a66994cef2c50f7e524a1471102d8\"),\n",
|
||||||
|
" (\"assets/ud-ner/train_with_NER_tag.txt\",\n",
|
||||||
|
" \"https://raw.githubusercontent.com/UniversalDependencies/UD_Persian-PerDT/master/not-to-release/Dadegan%20with%20NER%20tag/train_with_NER_tag.txt\",\n",
|
||||||
|
" \"ecb96cf99b38bc485cac21d22914e413\"),\n",
|
||||||
|
" (\"assets/ud-ner/dev_with_NER_tag.txt\",\n",
|
||||||
|
" \"https://raw.githubusercontent.com/UniversalDependencies/UD_Persian-PerDT/master/not-to-release/Dadegan%20with%20NER%20tag/dev_with_NER_tag.txt\",\n",
|
||||||
|
" \"2a56ef7eb2e3732e221317af457d1c09\"),\n",
|
||||||
|
" (\"assets/ud-ner/test_with_NER_tag.txt\",\n",
|
||||||
|
" \"https://raw.githubusercontent.com/UniversalDependencies/UD_Persian-PerDT/master/not-to-release/Dadegan%20with%20NER%20tag/test_with_NER_tag.txt\",\n",
|
||||||
|
" \"6d80dd783527562c2ea5189f218a12b5\"),\n",
|
||||||
|
"]\n",
|
||||||
|
"\n",
|
||||||
|
"for dest, url, checksum in ASSETS:\n",
|
||||||
|
" dest = pathlib.Path(dest)\n",
|
||||||
|
" dest.parent.mkdir(parents=True, exist_ok=True)\n",
|
||||||
|
" urllib.request.urlretrieve(url, dest)\n",
|
||||||
|
" got = hashlib.md5(dest.read_bytes()).hexdigest()\n",
|
||||||
|
" status = \"OK\" if got == checksum else f\"MISMATCH (got {got})\"\n",
|
||||||
|
" print(f\"{dest}: {status}\")\n",
|
||||||
|
" assert got == checksum, f\"checksum mismatch on {dest}\"\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 5. Build the UD + NER corpora (mirrors `project.yml`'s `convert-ud`/`transfer-ner`/`convert-ner`)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"!python -m spacy convert assets/ud/fa_perdt-ud-train.conllu corpus/merged --converter conllu --n-sents 10 --merge-subtokens\n",
|
||||||
|
"!python -m spacy convert assets/ud/fa_perdt-ud-dev.conllu corpus/merged --converter conllu --n-sents 10 --merge-subtokens\n",
|
||||||
|
"!python -m spacy convert assets/ud/fa_perdt-ud-test.conllu corpus/merged --converter conllu --n-sents 10 --merge-subtokens\n",
|
||||||
|
"\n",
|
||||||
|
"!python scripts/transfer_perdt_ner.py --conllu-dir assets/ud --ner-dir assets/ud-ner --out corpus/perdt-ner-iob\n",
|
||||||
|
"\n",
|
||||||
|
"!python -m spacy convert corpus/perdt-ner-iob/train.txt corpus/perdt-ner --converter ner --n-sents 10 --lang fa\n",
|
||||||
|
"!python -m spacy convert corpus/perdt-ner-iob/dev.txt corpus/perdt-ner --converter ner --n-sents 10 --lang fa\n",
|
||||||
|
"!python -m spacy convert corpus/perdt-ner-iob/test.txt corpus/perdt-ner --converter ner --n-sents 10 --lang fa\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 6. `lg`-tier floret vectors\n",
|
||||||
|
"\n",
|
||||||
|
"Upload a floret wheel (`fa_floret-0.1.0-py3-none-any-*.whl`), built elsewhere. Use the real\n",
|
||||||
|
"200k-row Wikipedia+OSCAR table if you have one; otherwise upload whatever `md`-tier wheel\n",
|
||||||
|
"you have as a stand-in -- the run will still be valid, just not the final `lg` numbers.\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"from google.colab import files\n",
|
||||||
|
"\n",
|
||||||
|
"uploaded = files.upload()\n",
|
||||||
|
"(floret_wheel,) = uploaded.keys()\n",
|
||||||
|
"\n",
|
||||||
|
"!python scripts/unpack_vectors.py {floret_wheel} assets/vectors/fa_floret_lg\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 7. Train the `lg` tier\n",
|
||||||
|
"\n",
|
||||||
|
"Byte-identical to the `md` recipe (`configs/fa_dep_news_md.cfg` / `configs/fa_ner_md.cfg`) --\n",
|
||||||
|
"only `--paths.vectors` changes, isolating the effect of the bigger table exactly the way\n",
|
||||||
|
"`md` isolated the effect of adding vectors over `sm`. No new config file needed.\n",
|
||||||
|
"\n",
|
||||||
|
"`--gpu-id 0` for both here: the earlier CPU-vs-GPU timing experiment ran on a 2 GB GTX 940MX,\n",
|
||||||
|
"where the small NER architecture's transfer/launch overhead beat its GPU compute win. A 16 GB\n",
|
||||||
|
"Colab GPU has far more bandwidth/compute headroom, so that conclusion may not hold here --\n",
|
||||||
|
"worth timing both `--gpu-id 0` and `--gpu-id -1` yourself if you want to confirm.\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"!python -m spacy train configs/fa_dep_news_md.cfg --output training/dep-lg \\\n",
|
||||||
|
" --paths.train corpus/merged/fa_perdt-ud-train.spacy \\\n",
|
||||||
|
" --paths.dev corpus/merged/fa_perdt-ud-dev.spacy \\\n",
|
||||||
|
" --paths.vectors assets/vectors/fa_floret_lg \\\n",
|
||||||
|
" --gpu-id 0\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"!python -m spacy train configs/fa_ner_md.cfg --output training/perdt-ner-lg \\\n",
|
||||||
|
" --paths.train corpus/perdt-ner/train.spacy \\\n",
|
||||||
|
" --paths.dev corpus/perdt-ner/dev.spacy \\\n",
|
||||||
|
" --paths.vectors assets/vectors/fa_floret_lg \\\n",
|
||||||
|
" --gpu-id 0\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 8. Assemble + evaluate `lg`"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"!python -m spacy benchmark accuracy training/dep-lg/model-best corpus/merged/fa_perdt-ud-test.spacy \\\n",
|
||||||
|
" --output metrics/lg-ud-test.json --gpu-id 0\n",
|
||||||
|
"\n",
|
||||||
|
"!python scripts/finalize_pipeline.py training/dep-lg/model-best training/fa_dep_news_lg \\\n",
|
||||||
|
" --variant dep --size lg --version 3.8.0 --ud-metrics metrics/lg-ud-test.json\n",
|
||||||
|
"\n",
|
||||||
|
"!python scripts/finalize_pipeline.py training/dep-lg/model-best training/fa_core_news_lg \\\n",
|
||||||
|
" --variant core --size lg --version 3.8.0 --add-ner training/perdt-ner-lg/model-best\n",
|
||||||
|
"\n",
|
||||||
|
"!python -m spacy benchmark accuracy training/fa_core_news_lg corpus/merged/fa_perdt-ud-test.spacy \\\n",
|
||||||
|
" --output metrics/lg-core-ud-test.json --gpu-id 0\n",
|
||||||
|
"!python -m spacy benchmark accuracy training/fa_core_news_lg corpus/perdt-ner/test.spacy \\\n",
|
||||||
|
" --output metrics/lg-perdt-ner-test.json --gpu-id 0\n",
|
||||||
|
"\n",
|
||||||
|
"!python scripts/finalize_pipeline.py training/dep-lg/model-best training/fa_core_news_lg \\\n",
|
||||||
|
" --variant core --size lg --version 3.8.0 --add-ner training/perdt-ner-lg/model-best \\\n",
|
||||||
|
" --ud-metrics metrics/lg-core-ud-test.json --ner-metrics metrics/lg-perdt-ner-test.json\n",
|
||||||
|
"\n",
|
||||||
|
"!python scripts/smoke_test.py training/fa_core_news_lg\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 9. Generate the `trf` configs\n",
|
||||||
|
"\n",
|
||||||
|
"`spacy init config --optimize accuracy -G` fills in a valid `spacy-transformers`\n",
|
||||||
|
"architecture automatically (letting spaCy own the schema instead of hand-writing one).\n",
|
||||||
|
"The only edit afterward is swapping the default transformer name for\n",
|
||||||
|
"`HooshvareLab/roberta-fa-zwnj-base` and turning on mixed precision, since 16 GB has room\n",
|
||||||
|
"for it.\n",
|
||||||
|
"\n",
|
||||||
|
"Same split as `sm`/`md`/`lg`: `dep` (tagger/morphologizer/lemmatizer/parser) and `ner`\n",
|
||||||
|
"trained as separate pipelines, each with its own transformer, so `ner` can be re-sourced\n",
|
||||||
|
"into `core` afterward exactly like the CPU tiers.\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"!python -m spacy init config configs/fa_dep_news_trf.cfg --lang fa \\\n",
|
||||||
|
" --pipeline tagger,morphologizer,trainable_lemmatizer,parser \\\n",
|
||||||
|
" --optimize accuracy -G --force\n",
|
||||||
|
"\n",
|
||||||
|
"!python -m spacy init config configs/fa_ner_trf.cfg --lang fa \\\n",
|
||||||
|
" --pipeline ner \\\n",
|
||||||
|
" --optimize accuracy -G --force\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"\n",
|
||||||
|
"import re\n",
|
||||||
|
"\n",
|
||||||
|
"TRANSFORMER_NAME = \"HooshvareLab/roberta-fa-zwnj-base\"\n",
|
||||||
|
"\n",
|
||||||
|
"for path in [\"configs/fa_dep_news_trf.cfg\", \"configs/fa_ner_trf.cfg\"]:\n",
|
||||||
|
" text = open(path, encoding=\"utf8\").read()\n",
|
||||||
|
" # Swap whatever default transformer `init config` picked for roberta-fa-zwnj-base.\n",
|
||||||
|
" text = re.sub(\n",
|
||||||
|
" r'(\\[components\\.transformer\\.model\\]\\nname = )\"[^\"]+\"',\n",
|
||||||
|
" lambda m: m.group(1) + '\"' + TRANSFORMER_NAME + '\"',\n",
|
||||||
|
" text,\n",
|
||||||
|
" )\n",
|
||||||
|
" # 16 GB has room for mixed precision; halves activation memory, meaningfully faster.\n",
|
||||||
|
" if \"mixed_precision\" in text:\n",
|
||||||
|
" text = text.replace(\"mixed_precision = false\", \"mixed_precision = true\")\n",
|
||||||
|
" else:\n",
|
||||||
|
" text = text.replace(\"[training]\\n\", \"[training]\\nmixed_precision = true\\n\", 1)\n",
|
||||||
|
" open(path, \"w\", encoding=\"utf8\").write(text)\n",
|
||||||
|
" print(\"patched\", path, \"transformer =\", TRANSFORMER_NAME)\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"!python -m spacy debug config configs/fa_dep_news_trf.cfg \\\n",
|
||||||
|
" --paths.train corpus/merged/fa_perdt-ud-train.spacy \\\n",
|
||||||
|
" --paths.dev corpus/merged/fa_perdt-ud-dev.spacy\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 10. Train the `trf` tier"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"!python -m spacy train configs/fa_dep_news_trf.cfg --output training/dep-trf \\\n",
|
||||||
|
" --paths.train corpus/merged/fa_perdt-ud-train.spacy \\\n",
|
||||||
|
" --paths.dev corpus/merged/fa_perdt-ud-dev.spacy \\\n",
|
||||||
|
" --gpu-id 0\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"!python -m spacy train configs/fa_ner_trf.cfg --output training/perdt-ner-trf \\\n",
|
||||||
|
" --paths.train corpus/perdt-ner/train.spacy \\\n",
|
||||||
|
" --paths.dev corpus/perdt-ner/dev.spacy \\\n",
|
||||||
|
" --gpu-id 0\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 11. Assemble + evaluate `trf`"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"!python -m spacy benchmark accuracy training/dep-trf/model-best corpus/merged/fa_perdt-ud-test.spacy \\\n",
|
||||||
|
" --output metrics/trf-ud-test.json --gpu-id 0\n",
|
||||||
|
"\n",
|
||||||
|
"!python scripts/finalize_pipeline.py training/dep-trf/model-best training/fa_dep_news_trf \\\n",
|
||||||
|
" --variant dep --size trf --version 3.8.0 --ud-metrics metrics/trf-ud-test.json\n",
|
||||||
|
"\n",
|
||||||
|
"!python scripts/finalize_pipeline.py training/dep-trf/model-best training/fa_core_news_trf \\\n",
|
||||||
|
" --variant core --size trf --version 3.8.0 --add-ner training/perdt-ner-trf/model-best\n",
|
||||||
|
"\n",
|
||||||
|
"!python -m spacy benchmark accuracy training/fa_core_news_trf corpus/merged/fa_perdt-ud-test.spacy \\\n",
|
||||||
|
" --output metrics/trf-core-ud-test.json --gpu-id 0\n",
|
||||||
|
"!python -m spacy benchmark accuracy training/fa_core_news_trf corpus/perdt-ner/test.spacy \\\n",
|
||||||
|
" --output metrics/trf-perdt-ner-test.json --gpu-id 0\n",
|
||||||
|
"\n",
|
||||||
|
"!python scripts/finalize_pipeline.py training/dep-trf/model-best training/fa_core_news_trf \\\n",
|
||||||
|
" --variant core --size trf --version 3.8.0 --add-ner training/perdt-ner-trf/model-best \\\n",
|
||||||
|
" --ud-metrics metrics/trf-core-ud-test.json --ner-metrics metrics/trf-perdt-ner-test.json\n",
|
||||||
|
"\n",
|
||||||
|
"!python scripts/smoke_test.py training/fa_core_news_trf\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 12. Compare every tier\n",
|
||||||
|
"\n",
|
||||||
|
"Reads whichever `metrics/*-ud-test.json` / `metrics/*-perdt-ner-test.json` files exist in\n",
|
||||||
|
"this Colab session (only `lg` and `trf`, produced above). To compare against the local\n",
|
||||||
|
"`sm`/`md` numbers, upload `metrics/core-ud-test.json`, `metrics/perdt-ner-test.json`,\n",
|
||||||
|
"`metrics/md-core-ud-test.json`, `metrics/md-perdt-ner-test.json` from the repo first.\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import json, pathlib\n",
|
||||||
|
"\n",
|
||||||
|
"ROWS = [\n",
|
||||||
|
" (\"sm\", \"metrics/core-ud-test.json\", \"metrics/perdt-ner-test.json\"),\n",
|
||||||
|
" (\"md\", \"metrics/md-core-ud-test.json\", \"metrics/md-perdt-ner-test.json\"),\n",
|
||||||
|
" (\"lg\", \"metrics/lg-core-ud-test.json\", \"metrics/lg-perdt-ner-test.json\"),\n",
|
||||||
|
" (\"trf\", \"metrics/trf-core-ud-test.json\", \"metrics/trf-perdt-ner-test.json\"),\n",
|
||||||
|
"]\n",
|
||||||
|
"\n",
|
||||||
|
"def load(path):\n",
|
||||||
|
" p = pathlib.Path(path)\n",
|
||||||
|
" return json.loads(p.read_text()) if p.exists() else None\n",
|
||||||
|
"\n",
|
||||||
|
"print(f\"{'tier':<5}{'tag_acc':>9}{'dep_las':>9}{'lemma_acc':>11}{'ents_f':>9}\")\n",
|
||||||
|
"for tier, ud_path, ner_path in ROWS:\n",
|
||||||
|
" ud, ner = load(ud_path), load(ner_path)\n",
|
||||||
|
" tag = f\"{ud['tag_acc']*100:.2f}\" if ud and ud.get('tag_acc') is not None else \"-\"\n",
|
||||||
|
" las = f\"{ud['dep_las']*100:.2f}\" if ud and ud.get('dep_las') is not None else \"-\"\n",
|
||||||
|
" lem = f\"{ud['lemma_acc']*100:.2f}\" if ud and ud.get('lemma_acc') is not None else \"-\"\n",
|
||||||
|
" entf = f\"{ner['ents_f']*100:.2f}\" if ner and ner.get('ents_f') is not None else \"-\"\n",
|
||||||
|
" print(f\"{tier:<5}{tag:>9}{las:>9}{lem:>11}{entf:>9}\")\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 13. Download the results"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"from google.colab import files\n",
|
||||||
|
"\n",
|
||||||
|
"!zip -r /content/lg_trf_results.zip training/fa_dep_news_lg training/fa_core_news_lg \\\n",
|
||||||
|
" training/fa_dep_news_trf training/fa_core_news_trf metrics assets/vectors/fa_floret_lg \\\n",
|
||||||
|
" configs/fa_dep_news_trf.cfg configs/fa_ner_trf.cfg\n",
|
||||||
|
"\n",
|
||||||
|
"files.download(\"/content/lg_trf_results.zip\")\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## 14. (Optional) package as installable wheels"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"!python -m spacy package training/fa_dep_news_lg packages --name dep_news_lg --version 3.8.0 --build sdist,wheel --force\n",
|
||||||
|
"!python -m spacy package training/fa_core_news_lg packages --name core_news_lg --version 3.8.0 --build sdist,wheel --force\n",
|
||||||
|
"!python -m spacy package training/fa_dep_news_trf packages --name dep_news_trf --version 3.8.0 --build sdist,wheel --force\n",
|
||||||
|
"!python -m spacy package training/fa_core_news_trf packages --name core_news_trf --version 3.8.0 --build sdist,wheel --force\n",
|
||||||
|
"\n",
|
||||||
|
"!zip -r /content/packages.zip packages\n",
|
||||||
|
"files.download(\"/content/packages.zip\")\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"metadata": {
|
||||||
|
"accelerator": "GPU",
|
||||||
|
"colab": {
|
||||||
|
"name": "fa_lg_trf_training.ipynb",
|
||||||
|
"provenance": []
|
||||||
|
},
|
||||||
|
"kernelspec": {
|
||||||
|
"display_name": "Python 3",
|
||||||
|
"name": "python3"
|
||||||
|
},
|
||||||
|
"language_info": {
|
||||||
|
"name": "python"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"nbformat": 4,
|
||||||
|
"nbformat_minor": 5
|
||||||
|
}
|
||||||
|
|
@ -3,9 +3,13 @@
|
||||||
Three variants, following spaCy's `[lang]_[type]_[genre]_[size]` naming
|
Three variants, following spaCy's `[lang]_[type]_[genre]_[size]` naming
|
||||||
(https://spacy.io/models#conventions):
|
(https://spacy.io/models#conventions):
|
||||||
|
|
||||||
dep -> fa_dep_news_sm tagger + morphologizer + trainable_lemmatizer + parser
|
dep -> fa_dep_news_<size> tagger + morphologizer + trainable_lemmatizer + parser
|
||||||
core -> fa_core_news_sm the above plus ner
|
core -> fa_core_news_<size> the above plus ner
|
||||||
ent -> fa_ent_news_sm ner only
|
ent -> fa_ent_news_<size> ner only
|
||||||
|
|
||||||
|
`--size` fills the size slot: `sm` (hash embeddings only, the default) or `md` (the same
|
||||||
|
architecture plus the fa_floret static vector table). It is metadata only; which vectors a
|
||||||
|
model actually carries is decided at train time by `--paths.vectors`.
|
||||||
|
|
||||||
All three are built from UD_Persian-PerDT alone, including the NER, which comes from that
|
All three are built from UD_Persian-PerDT alone, including the NER, which comes from that
|
||||||
treebank's own `not-to-release/Dadegan with NER tag/` layer. That is what makes `core`
|
treebank's own `not-to-release/Dadegan with NER tag/` layer. That is what makes `core`
|
||||||
|
|
@ -56,6 +60,25 @@ LANG_DATA = {
|
||||||
"author": "Explosion and spaCy contributors",
|
"author": "Explosion and spaCy contributors",
|
||||||
"license": "MIT",
|
"license": "MIT",
|
||||||
}
|
}
|
||||||
|
FLORET = {
|
||||||
|
"name": "fa_floret static vectors (50k rows x 300d, floret mode, 400k Persian documents)",
|
||||||
|
"url": PROJECT_URL,
|
||||||
|
"author": "Kiyarash Fazeli",
|
||||||
|
"license": "CC BY-SA 4.0",
|
||||||
|
}
|
||||||
|
FLORET_LG = {
|
||||||
|
"name": "fa_floret static vectors (lg tier: larger floret table trained on fa Wikipedia + "
|
||||||
|
"OSCAR via spacy-vectors-builder)",
|
||||||
|
"url": PROJECT_URL,
|
||||||
|
"author": "Kiyarash Fazeli",
|
||||||
|
"license": "CC BY-SA 4.0",
|
||||||
|
}
|
||||||
|
TRANSFORMER = {
|
||||||
|
"name": "HooshvareLab/roberta-fa-zwnj-base",
|
||||||
|
"url": "https://huggingface.co/HooshvareLab/roberta-fa-zwnj-base",
|
||||||
|
"author": "Hooshvare Team",
|
||||||
|
"license": "Apache-2.0",
|
||||||
|
}
|
||||||
|
|
||||||
NER_NOTE = (
|
NER_NOTE = (
|
||||||
"The ner component is trained on the NER layer shipped in UD_Persian-PerDT's "
|
"The ner component is trained on the NER layer shipped in UD_Persian-PerDT's "
|
||||||
|
|
@ -78,6 +101,36 @@ CHUNK_NOTE = (
|
||||||
"ClearNLP labels that do not exist in Universal Dependencies, see "
|
"ClearNLP labels that do not exist in Universal Dependencies, see "
|
||||||
"docs/upstream/fa-noun-chunks.md."
|
"docs/upstream/fa-noun-chunks.md."
|
||||||
)
|
)
|
||||||
|
VECTORS_NOTE = (
|
||||||
|
"This is the `md` tier: identical architecture to the `sm` pipeline plus static floret "
|
||||||
|
"vectors (50,000 rows x 300 dimensions, minn=maxn=5, hash_count=2) trained on 400,000 "
|
||||||
|
"Persian documents. floret hashes subwords into a fixed table, so there are no "
|
||||||
|
"out-of-vocabulary tokens and `token.has_vector` is always True. That matters for "
|
||||||
|
"Persian, where inconsistent ZWNJ (U+200C) usage splits one word across several surface "
|
||||||
|
"forms (mi-ravad written joined, with ZWNJ, or with a space) that a classic word-vector "
|
||||||
|
"table would miss."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def vectors_note_lg(nlp):
|
||||||
|
"""Row/dim counts come from the trained model, not a hardcoded description, because the
|
||||||
|
lg-tier floret table is still being iterated on (unlike md's fixed, shipped table)."""
|
||||||
|
rows, dim = nlp.vocab.vectors.shape
|
||||||
|
return (
|
||||||
|
f"This is the `lg` tier: identical architecture to `sm`/`md` but a larger static "
|
||||||
|
f"floret vector table ({rows:,} rows x {dim} dimensions, minn=maxn=5, hash_count=2) "
|
||||||
|
f"trained on Persian Wikipedia + OSCAR via spacy-vectors-builder. Same zero-OOV "
|
||||||
|
f"rationale as `md` (see docs/MODELS.md): floret hashes subwords into a fixed table, "
|
||||||
|
f"so `token.has_vector` is always True despite Persian's ZWNJ (U+200C) inconsistency."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
TRANSFORMER_NOTE = (
|
||||||
|
"This is the `trf` tier: no static vectors; contextual embeddings instead come from a "
|
||||||
|
"fine-tuned HooshvareLab/roberta-fa-zwnj-base (Apache-2.0) transformer via "
|
||||||
|
"spacy-transformers. Not ParsBERT: its model card carries no licence. GPU is recommended "
|
||||||
|
"for both training and inference."
|
||||||
|
)
|
||||||
# CC BY-SA 4.0 on the treebank propagates to anything derived from it.
|
# CC BY-SA 4.0 on the treebank propagates to anything derived from it.
|
||||||
PERDT_LICENSE = "CC BY-SA 4.0"
|
PERDT_LICENSE = "CC BY-SA 4.0"
|
||||||
ATTRIBUTION = (
|
ATTRIBUTION = (
|
||||||
|
|
@ -92,10 +145,10 @@ NER_KEYS = ("ents_p", "ents_r", "ents_f", "ents_per_type")
|
||||||
|
|
||||||
VARIANTS = {
|
VARIANTS = {
|
||||||
"dep": {
|
"dep": {
|
||||||
"name": "dep_news_sm",
|
"name": "dep_news_{size}",
|
||||||
"description": (
|
"description": (
|
||||||
"Persian dependency pipeline optimized for CPU. Components: tok2vec, tagger, "
|
"Persian dependency pipeline optimized for CPU. Components: tok2vec, tagger, "
|
||||||
"morphologizer, trainable_lemmatizer, parser. No NER, see fa_core_news_sm."
|
"morphologizer, trainable_lemmatizer, parser. No NER, see fa_core_news_{size}."
|
||||||
),
|
),
|
||||||
"license": PERDT_LICENSE,
|
"license": PERDT_LICENSE,
|
||||||
"sources": [PERDT, LANG_DATA],
|
"sources": [PERDT, LANG_DATA],
|
||||||
|
|
@ -105,7 +158,7 @@ VARIANTS = {
|
||||||
"require_msg": "a 'dep' pipeline must not contain an ner component",
|
"require_msg": "a 'dep' pipeline must not contain an ner component",
|
||||||
},
|
},
|
||||||
"core": {
|
"core": {
|
||||||
"name": "core_news_sm",
|
"name": "core_news_{size}",
|
||||||
"description": (
|
"description": (
|
||||||
"Persian pipeline optimized for CPU. Components: tok2vec, tagger, morphologizer, "
|
"Persian pipeline optimized for CPU. Components: tok2vec, tagger, morphologizer, "
|
||||||
"trainable_lemmatizer, parser, ner. Entity labels: PER, LOC, ORG, DAT, MON, TIM, "
|
"trainable_lemmatizer, parser, ner. Entity labels: PER, LOC, ORG, DAT, MON, TIM, "
|
||||||
|
|
@ -119,7 +172,7 @@ VARIANTS = {
|
||||||
"require_msg": "a 'core' pipeline must contain both parser and ner",
|
"require_msg": "a 'core' pipeline must contain both parser and ner",
|
||||||
},
|
},
|
||||||
"ent": {
|
"ent": {
|
||||||
"name": "ent_news_sm",
|
"name": "ent_news_{size}",
|
||||||
"description": (
|
"description": (
|
||||||
"Persian named entity recognizer optimized for CPU, with its own internal "
|
"Persian named entity recognizer optimized for CPU, with its own internal "
|
||||||
"tok2vec. Labels: PER, LOC, ORG, DAT, MON, TIM, PCT."
|
"tok2vec. Labels: PER, LOC, ORG, DAT, MON, TIM, PCT."
|
||||||
|
|
@ -140,6 +193,10 @@ def main():
|
||||||
ap.add_argument("output", help="destination directory")
|
ap.add_argument("output", help="destination directory")
|
||||||
ap.add_argument("--variant", choices=sorted(VARIANTS), required=True)
|
ap.add_argument("--variant", choices=sorted(VARIANTS), required=True)
|
||||||
ap.add_argument("--version", default="3.8.0")
|
ap.add_argument("--version", default="3.8.0")
|
||||||
|
ap.add_argument("--size", choices=("sm", "md", "lg", "trf"), default="sm",
|
||||||
|
help="size slot in the package name. 'md'/'lg' additionally record the "
|
||||||
|
"floret vector table as a source and append a vectors note; 'trf' "
|
||||||
|
"records the transformer source and appends a transformer note.")
|
||||||
ap.add_argument("--ud-metrics", default=None,
|
ap.add_argument("--ud-metrics", default=None,
|
||||||
help="benchmark accuracy JSON scored on the UD test split; supplies the "
|
help="benchmark accuracy JSON scored on the UD test split; supplies the "
|
||||||
"tagger/morph/lemma/parser keys only")
|
"tagger/morph/lemma/parser keys only")
|
||||||
|
|
@ -152,7 +209,20 @@ def main():
|
||||||
args = ap.parse_args()
|
args = ap.parse_args()
|
||||||
|
|
||||||
spec = VARIANTS[args.variant]
|
spec = VARIANTS[args.variant]
|
||||||
|
name = spec["name"].format(size=args.size)
|
||||||
|
description = spec["description"].format(size=args.size)
|
||||||
|
sources = list(spec["sources"])
|
||||||
|
notes = spec["notes"]
|
||||||
nlp = spacy.load(args.model)
|
nlp = spacy.load(args.model)
|
||||||
|
if args.size == "md":
|
||||||
|
sources.append(FLORET)
|
||||||
|
notes = " ".join([notes, VECTORS_NOTE])
|
||||||
|
elif args.size == "lg":
|
||||||
|
sources.append(FLORET_LG)
|
||||||
|
notes = " ".join([notes, vectors_note_lg(nlp)])
|
||||||
|
elif args.size == "trf":
|
||||||
|
sources.append(TRANSFORMER)
|
||||||
|
notes = " ".join([notes, TRANSFORMER_NOTE])
|
||||||
if args.add_ner:
|
if args.add_ner:
|
||||||
ner_nlp = spacy.load(args.add_ner)
|
ner_nlp = spacy.load(args.add_ner)
|
||||||
if ner_nlp.pipe_names != ["ner"]:
|
if ner_nlp.pipe_names != ["ner"]:
|
||||||
|
|
@ -202,22 +272,22 @@ def main():
|
||||||
nlp.meta.update(
|
nlp.meta.update(
|
||||||
{
|
{
|
||||||
"lang": "fa",
|
"lang": "fa",
|
||||||
"name": spec["name"],
|
"name": name,
|
||||||
"version": args.version,
|
"version": args.version,
|
||||||
"description": spec["description"],
|
"description": description,
|
||||||
"author": AUTHOR,
|
"author": AUTHOR,
|
||||||
"email": EMAIL,
|
"email": EMAIL,
|
||||||
"url": PROJECT_URL,
|
"url": PROJECT_URL,
|
||||||
"license": spec["license"],
|
"license": spec["license"],
|
||||||
"sources": spec["sources"],
|
"sources": sources,
|
||||||
"notes": spec["notes"],
|
"notes": notes,
|
||||||
"performance": performance,
|
"performance": performance,
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
|
|
||||||
out = Path(args.output)
|
out = Path(args.output)
|
||||||
nlp.to_disk(out)
|
nlp.to_disk(out)
|
||||||
print(f"wrote {out} as fa_{spec['name']} {args.version} ({spec['license']})")
|
print(f"wrote {out} as fa_{name} {args.version} ({spec['license']})")
|
||||||
scalars = {k: round(v * 100, 2) for k, v in performance.items() if isinstance(v, float)}
|
scalars = {k: round(v * 100, 2) for k, v in performance.items() if isinstance(v, float)}
|
||||||
print(json.dumps(scalars, indent=2))
|
print(json.dumps(scalars, indent=2))
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,53 @@
|
||||||
|
"""Unpack a spaCy vectors-only wheel into a plain model directory.
|
||||||
|
|
||||||
|
`spacy train --paths.vectors` wants a directory it can `spacy.load()`. The fa_floret wheel
|
||||||
|
already contains exactly that (an empty pipeline carrying only vocab/vectors), it is just
|
||||||
|
buried under the wheel's package layout, so this unwraps it rather than pip-installing a
|
||||||
|
package whose only job is to hold a 57 MB array.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
python scripts/unpack_vectors.py fa_floret-0.1.0-py3-none-any-400k-documents.whl \\
|
||||||
|
assets/vectors/fa_floret_400k
|
||||||
|
"""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import shutil
|
||||||
|
import sys
|
||||||
|
import tempfile
|
||||||
|
import zipfile
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
ap = argparse.ArgumentParser()
|
||||||
|
ap.add_argument("wheel", type=Path)
|
||||||
|
ap.add_argument("output", type=Path)
|
||||||
|
args = ap.parse_args()
|
||||||
|
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
tmp = Path(tmp)
|
||||||
|
with zipfile.ZipFile(args.wheel) as z:
|
||||||
|
z.extractall(tmp)
|
||||||
|
# The model directory is the one holding config.cfg, e.g. fa_floret/fa_floret-0.1.0/.
|
||||||
|
models = sorted(p.parent for p in tmp.rglob("config.cfg"))
|
||||||
|
if len(models) != 1:
|
||||||
|
sys.exit(f"expected exactly one config.cfg in {args.wheel}, found {len(models)}")
|
||||||
|
if args.output.exists():
|
||||||
|
shutil.rmtree(args.output)
|
||||||
|
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
shutil.move(str(models[0]), str(args.output))
|
||||||
|
|
||||||
|
import spacy
|
||||||
|
|
||||||
|
nlp = spacy.load(args.output)
|
||||||
|
vectors = nlp.vocab.vectors
|
||||||
|
if vectors.shape[0] == 0:
|
||||||
|
sys.exit(f"{args.output} has no vectors")
|
||||||
|
print(
|
||||||
|
f"{args.output}: mode={vectors.mode} shape={vectors.shape} "
|
||||||
|
f"n_keys={vectors.n_keys} pipeline={nlp.pipe_names}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Loading…
Reference in New Issue