spaCy/bin/parser/train_ud.py

import plac
import json
from os import path
import shutil
import os
import random
import io

from spacy.tokens import Doc
from spacy.syntax.nonproj import PseudoProjectivity
from spacy.language import Language
from spacy.gold import GoldParse
from spacy.vocab import Vocab
from spacy.tagger import Tagger
from spacy.pipeline import DependencyParser
from spacy.syntax.parser import get_templates
from spacy.syntax.arc_eager import ArcEager
from spacy.scorer import Scorer
import spacy.attrs

try:
    from codecs import open
except ImportError:
    pass


def read_conllx(loc):
    with open(loc, 'r', 'utf8') as file_:
        text = file_.read()
    for sent in text.strip().split('\n\n'):
        lines = sent.strip().split('\n')
        if lines:
            while lines[0].startswith('#'):
                lines.pop(0)
            tokens = []
            for line in lines:
                id_, word, lemma, tag, pos, morph, head, dep, _1, _2 = line.split()
                if '-' in id_:
                    continue
                try:
                    id_ = int(id_) - 1
                    head = (int(head) - 1) if head != '0' else id_
                    dep = 'ROOT' if dep == 'root' else dep
                    tokens.append((id_, word, tag, head, dep, 'O'))
                except:
                    print(line)
                    raise
            tuples = [list(t) for t in zip(*tokens)]
            yield (None, [[tuples, []]])


def score_model(vocab, tagger, parser, gold_docs, verbose=False):
    scorer = Scorer()
    for _, gold_doc in gold_docs:
        for (ids, words, tags, heads, deps, entities), _ in gold_doc:
            doc = Doc(vocab, words=words)
            tagger(doc)
            parser(doc)
            gold = GoldParse(doc, tags=tags, heads=heads, deps=deps)
            scorer.score(doc, gold, verbose=verbose)
    return scorer


def main(train_loc, dev_loc, model_dir, tag_map_loc):
    with open(tag_map_loc) as file_:
        tag_map = json.loads(file_.read())
    train_sents = list(read_conllx(train_loc))
    train_sents = PseudoProjectivity.preprocess_training_data(train_sents)
    actions = ArcEager.get_actions(gold_parses=train_sents)
    features = get_templates('basic')

    vocab = Vocab(lex_attr_getters=Language.Defaults.lex_attr_getters, tag_map=tag_map)
    # Populate vocab
    for _, doc_sents in train_sents:
        for (ids, words, tags, heads, deps, ner), _ in doc_sents:
            for word in words:
                _ = vocab[word]
            for tag in tags:
                assert tag in tag_map, repr(tag)
    print(tags)
    tagger = Tagger(vocab, tag_map=tag_map)
    parser = DependencyParser(vocab, actions=actions, features=features)
    
    for itn in range(15):
        for _, doc_sents in train_sents:
            for (ids, words, tags, heads, deps, ner), _ in doc_sents:
                doc = Doc(vocab, words=words)
                gold = GoldParse(doc, tags=tags, heads=heads, deps=deps)
                tagger(doc)
                parser.update(doc, gold)
                doc = Doc(vocab, words=words)
                tagger.update(doc, gold)
        random.shuffle(train_sents)
        scorer = score_model(vocab, tagger, parser, read_conllx(dev_loc))
        print('%d:\t%.3f\t%.3f' % (itn, scorer.uas, scorer.tags_acc))
    nlp = Language(vocab=vocab, tagger=tagger, parser=parser)
    nlp.end_training(model_dir)
    scorer = score_model(vocab, tagger, parser, read_conllx(dev_loc))
    print('%d:\t%.3f\t%.3f\t%.3f' % (itn, scorer.uas, scorer.las, scorer.tags_acc))
 

if __name__ == '__main__':
    plac.call(main)
* Add script to train models off the UD treebanks. Note that the UD data is restricted to research purposes only, and should only be used to train models for academic experiments. 2015-10-08 04:00:11 +03:00			`import plac`
			`import json`
			`from os import path`
			`import shutil`
			`import os`
			`import random`
* Fix model saving 2016-05-23 15:01:46 +03:00			`import io`
* Add script to train models off the UD treebanks. Note that the UD data is restricted to research purposes only, and should only be used to train models for academic experiments. 2015-10-08 04:00:11 +03:00
Fix train_ud script, which trains models from the Universal Dependencies format. 2016-11-25 20:19:33 +03:00			`from spacy.tokens import Doc`
			`from spacy.syntax.nonproj import PseudoProjectivity`
			`from spacy.language import Language`
* Add script to train models off the UD treebanks. Note that the UD data is restricted to research purposes only, and should only be used to train models for academic experiments. 2015-10-08 04:00:11 +03:00			`from spacy.gold import GoldParse`
			`from spacy.vocab import Vocab`
			`from spacy.tagger import Tagger`
Fix train_ud script, which trains models from the Universal Dependencies format. 2016-11-25 20:19:33 +03:00			`from spacy.pipeline import DependencyParser`
* Add script to train models off the UD treebanks. Note that the UD data is restricted to research purposes only, and should only be used to train models for academic experiments. 2015-10-08 04:00:11 +03:00			`from spacy.syntax.parser import get_templates`
Fix train_ud script, which trains models from the Universal Dependencies format. 2016-11-25 20:19:33 +03:00			`from spacy.syntax.arc_eager import ArcEager`
* Add script to train models off the UD treebanks. Note that the UD data is restricted to research purposes only, and should only be used to train models for academic experiments. 2015-10-08 04:00:11 +03:00			`from spacy.scorer import Scorer`
* Work around get_lex_attr bug introduced during German parsing 2016-05-23 13:53:00 +03:00			`import spacy.attrs`
* Add script to train models off the UD treebanks. Note that the UD data is restricted to research purposes only, and should only be used to train models for academic experiments. 2015-10-08 04:00:11 +03:00
			`try:`
			`from codecs import open`
			`except ImportError:`
			`pass`


			`def read_conllx(loc):`
			`with open(loc, 'r', 'utf8') as file_:`
			`text = file_.read()`
			`for sent in text.strip().split('\n\n'):`
			`lines = sent.strip().split('\n')`
			`if lines:`
* Work around get_lex_attr bug introduced during German parsing 2016-05-23 13:53:00 +03:00			`while lines[0].startswith('#'):`
* Add script to train models off the UD treebanks. Note that the UD data is restricted to research purposes only, and should only be used to train models for academic experiments. 2015-10-08 04:00:11 +03:00			`lines.pop(0)`
			`tokens = []`
			`for line in lines:`
Fix train_ud script, which trains models from the Universal Dependencies format. 2016-11-25 20:19:33 +03:00			`id_, word, lemma, tag, pos, morph, head, dep, _1, _2 = line.split()`
* Add script to train models off the UD treebanks. Note that the UD data is restricted to research purposes only, and should only be used to train models for academic experiments. 2015-10-08 04:00:11 +03:00			`if '-' in id_:`
			`continue`
Fix train_ud script, which trains models from the Universal Dependencies format. 2016-11-25 20:19:33 +03:00			`try:`
			`id_ = int(id_) - 1`
			`head = (int(head) - 1) if head != '0' else id_`
			`dep = 'ROOT' if dep == 'root' else dep`
			`tokens.append((id_, word, tag, head, dep, 'O'))`
			`except:`
			`print(line)`
			`raise`
			`tuples = [list(t) for t in zip(*tokens)]`
			`yield (None, [[tuples, []]])`


			`def score_model(vocab, tagger, parser, gold_docs, verbose=False):`
* Add script to train models off the UD treebanks. Note that the UD data is restricted to research purposes only, and should only be used to train models for academic experiments. 2015-10-08 04:00:11 +03:00			`scorer = Scorer()`
			`for _, gold_doc in gold_docs:`
Fix train_ud script, which trains models from the Universal Dependencies format. 2016-11-25 20:19:33 +03:00			`for (ids, words, tags, heads, deps, entities), _ in gold_doc:`
			`doc = Doc(vocab, words=words)`
			`tagger(doc)`
			`parser(doc)`
			`gold = GoldParse(doc, tags=tags, heads=heads, deps=deps)`
			`scorer.score(doc, gold, verbose=verbose)`
* Add script to train models off the UD treebanks. Note that the UD data is restricted to research purposes only, and should only be used to train models for academic experiments. 2015-10-08 04:00:11 +03:00			`return scorer`


			`def main(train_loc, dev_loc, model_dir, tag_map_loc):`
			`with open(tag_map_loc) as file_:`
			`tag_map = json.loads(file_.read())`
			`train_sents = list(read_conllx(train_loc))`
Fix train_ud script, which trains models from the Universal Dependencies format. 2016-11-25 20:19:33 +03:00			`train_sents = PseudoProjectivity.preprocess_training_data(train_sents)`
			`actions = ArcEager.get_actions(gold_parses=train_sents)`
			`features = get_templates('basic')`

			`vocab = Vocab(lex_attr_getters=Language.Defaults.lex_attr_getters, tag_map=tag_map)`
			`# Populate vocab`
			`for _, doc_sents in train_sents:`
			`for (ids, words, tags, heads, deps, ner), _ in doc_sents:`
			`for word in words:`
			`_ = vocab[word]`
			`for tag in tags:`
			`assert tag in tag_map, repr(tag)`
			`print(tags)`
			`tagger = Tagger(vocab, tag_map=tag_map)`
			`parser = DependencyParser(vocab, actions=actions, features=features)`
* Add script to train models off the UD treebanks. Note that the UD data is restricted to research purposes only, and should only be used to train models for academic experiments. 2015-10-08 04:00:11 +03:00
			`for itn in range(15):`
			`for _, doc_sents in train_sents:`
			`for (ids, words, tags, heads, deps, ner), _ in doc_sents:`
Fix train_ud script, which trains models from the Universal Dependencies format. 2016-11-25 20:19:33 +03:00			`doc = Doc(vocab, words=words)`
			`gold = GoldParse(doc, tags=tags, heads=heads, deps=deps)`
			`tagger(doc)`
			`parser.update(doc, gold)`
			`doc = Doc(vocab, words=words)`
			`tagger.update(doc, gold)`
* Add script to train models off the UD treebanks. Note that the UD data is restricted to research purposes only, and should only be used to train models for academic experiments. 2015-10-08 04:00:11 +03:00			`random.shuffle(train_sents)`
Fix train_ud script, which trains models from the Universal Dependencies format. 2016-11-25 20:19:33 +03:00			`scorer = score_model(vocab, tagger, parser, read_conllx(dev_loc))`
* Add script to train models off the UD treebanks. Note that the UD data is restricted to research purposes only, and should only be used to train models for academic experiments. 2015-10-08 04:00:11 +03:00			`print('%d:\t%.3f\t%.3f' % (itn, scorer.uas, scorer.tags_acc))`
Fix train_ud script, which trains models from the Universal Dependencies format. 2016-11-25 20:19:33 +03:00			`nlp = Language(vocab=vocab, tagger=tagger, parser=parser)`
* Add script to train models off the UD treebanks. Note that the UD data is restricted to research purposes only, and should only be used to train models for academic experiments. 2015-10-08 04:00:11 +03:00			`nlp.end_training(model_dir)`
Fix train_ud script, which trains models from the Universal Dependencies format. 2016-11-25 20:19:33 +03:00			`scorer = score_model(vocab, tagger, parser, read_conllx(dev_loc))`
* Add script to train models off the UD treebanks. Note that the UD data is restricted to research purposes only, and should only be used to train models for academic experiments. 2015-10-08 04:00:11 +03:00			`print('%d:\t%.3f\t%.3f\t%.3f' % (itn, scorer.uas, scorer.las, scorer.tags_acc))`


			`if __name__ == '__main__':`
			`plac.call(main)`