mirror of
https://github.com/explosion/spaCy.git
synced 2024-12-26 18:06:29 +03:00
8656a08777
* Get basic beam tests working * Get basic beam tests working * Compile _beam_utils * Remove prints * Test beam density * Beam parser seems to train * Draft beam NER * Upd beam * Add hypothesis as dev dependency * Implement missing is-gold-parse method * Implement early update * Fix state hashing * Fix test * Fix test * Default to non-beam in parser constructor * Improve oracle for beam * Start refactoring beam * Update test * Refactor beam * Update nn * Refactor beam and weight by cost * Update ner beam settings * Update test * Add __init__.pxd * Upd test * Fix test * Upd test * Fix test * Remove ring buffer history from StateC * WIP change arc-eager transitions * Add state tests * Support ternary sent start values * Fix arc eager * Fix NER * Pass oracle cut size for beam * Fix ner test * Fix beam * Improve StateC.clone * Improve StateClass.borrow * Work directly with StateC, not StateClass * Remove print statements * Fix state copy * Improve state class * Refactor parser oracles * Fix arc eager oracle * Fix arc eager oracle * Use a vector to implement the stack * Refactor state data structure * Fix alignment of sent start * Add get_aligned_sent_starts method * Add test for ae oracle when bad sentence starts * Fix sentence segment handling * Avoid Reduce that inserts illegal sentence * Update preset SBD test * Fix test * Remove prints * Fix sent starts in Example * Improve python API of StateClass * Tweak comments and debug output of arc eager * Upd test * Fix state test * Fix state test
93 lines
2.6 KiB
Python
93 lines
2.6 KiB
Python
import pytest
|
|
from thinc.api import Adam
|
|
from spacy.attrs import NORM
|
|
from spacy.vocab import Vocab
|
|
from spacy import registry
|
|
from spacy.training import Example
|
|
from spacy.pipeline.dep_parser import DEFAULT_PARSER_MODEL
|
|
from spacy.tokens import Doc
|
|
from spacy.pipeline import DependencyParser
|
|
|
|
|
|
@pytest.fixture
|
|
def vocab():
|
|
return Vocab(lex_attr_getters={NORM: lambda s: s})
|
|
|
|
|
|
def _parser_example(parser):
|
|
doc = Doc(parser.vocab, words=["a", "b", "c", "d"])
|
|
gold = {"heads": [1, 1, 3, 3], "deps": ["right", "ROOT", "left", "ROOT"]}
|
|
return Example.from_dict(doc, gold)
|
|
|
|
|
|
@pytest.fixture
|
|
def parser(vocab):
|
|
vocab.strings.add("ROOT")
|
|
config = {
|
|
"learn_tokens": False,
|
|
"min_action_freq": 30,
|
|
"update_with_oracle_cut_size": 100,
|
|
}
|
|
cfg = {"model": DEFAULT_PARSER_MODEL}
|
|
model = registry.resolve(cfg, validate=True)["model"]
|
|
parser = DependencyParser(vocab, model, **config)
|
|
parser.cfg["token_vector_width"] = 4
|
|
parser.cfg["hidden_width"] = 32
|
|
# parser.add_label('right')
|
|
parser.add_label("left")
|
|
parser.initialize(lambda: [_parser_example(parser)])
|
|
sgd = Adam(0.001)
|
|
|
|
for i in range(10):
|
|
losses = {}
|
|
doc = Doc(vocab, words=["a", "b", "c", "d"])
|
|
example = Example.from_dict(
|
|
doc, {"heads": [1, 1, 3, 3], "deps": ["left", "ROOT", "left", "ROOT"]}
|
|
)
|
|
parser.update([example], sgd=sgd, losses=losses)
|
|
return parser
|
|
|
|
|
|
def test_no_sentences(parser):
|
|
doc = Doc(parser.vocab, words=["a", "b", "c", "d"])
|
|
doc = parser(doc)
|
|
assert len(list(doc.sents)) >= 1
|
|
|
|
|
|
def test_sents_1(parser):
|
|
doc = Doc(parser.vocab, words=["a", "b", "c", "d"])
|
|
doc[2].sent_start = True
|
|
doc = parser(doc)
|
|
assert len(list(doc.sents)) >= 2
|
|
doc = Doc(parser.vocab, words=["a", "b", "c", "d"])
|
|
doc[1].sent_start = False
|
|
doc[2].sent_start = True
|
|
doc[3].sent_start = False
|
|
doc = parser(doc)
|
|
assert len(list(doc.sents)) == 2
|
|
|
|
|
|
def test_sents_1_2(parser):
|
|
doc = Doc(parser.vocab, words=["a", "b", "c", "d"])
|
|
doc[1].sent_start = True
|
|
doc[2].sent_start = True
|
|
doc = parser(doc)
|
|
assert len(list(doc.sents)) >= 3
|
|
|
|
|
|
def test_sents_1_3(parser):
|
|
doc = Doc(parser.vocab, words=["a", "b", "c", "d"])
|
|
doc[0].is_sent_start = True
|
|
doc[1].is_sent_start = True
|
|
doc[2].is_sent_start = None
|
|
doc[3].is_sent_start = True
|
|
doc = parser(doc)
|
|
assert len(list(doc.sents)) >= 3
|
|
doc = Doc(parser.vocab, words=["a", "b", "c", "d"])
|
|
doc[0].is_sent_start = True
|
|
doc[1].is_sent_start = True
|
|
doc[2].is_sent_start = False
|
|
doc[3].is_sent_start = True
|
|
doc = parser(doc)
|
|
assert len(list(doc.sents)) == 3
|