mirror of
				https://github.com/explosion/spaCy.git
				synced 2025-10-26 21:51:24 +03:00 
			
		
		
		
	* Get basic beam tests working * Get basic beam tests working * Compile _beam_utils * Remove prints * Test beam density * Beam parser seems to train * Draft beam NER * Upd beam * Add hypothesis as dev dependency * Implement missing is-gold-parse method * Implement early update * Fix state hashing * Fix test * Fix test * Default to non-beam in parser constructor * Improve oracle for beam * Start refactoring beam * Update test * Refactor beam * Update nn * Refactor beam and weight by cost * Update ner beam settings * Update test * Add __init__.pxd * Upd test * Fix test * Upd test * Fix test * Remove ring buffer history from StateC * WIP change arc-eager transitions * Add state tests * Support ternary sent start values * Fix arc eager * Fix NER * Pass oracle cut size for beam * Fix ner test * Fix beam * Improve StateC.clone * Improve StateClass.borrow * Work directly with StateC, not StateClass * Remove print statements * Fix state copy * Improve state class * Refactor parser oracles * Fix arc eager oracle * Fix arc eager oracle * Use a vector to implement the stack * Refactor state data structure * Fix alignment of sent start * Add get_aligned_sent_starts method * Add test for ae oracle when bad sentence starts * Fix sentence segment handling * Avoid Reduce that inserts illegal sentence * Update preset SBD test * Fix test * Remove prints * Fix sent starts in Example * Improve python API of StateClass * Tweak comments and debug output of arc eager * Upd test * Fix state test * Fix state test
		
			
				
	
	
		
			93 lines
		
	
	
		
			2.6 KiB
		
	
	
	
		
			Python
		
	
	
	
	
	
			
		
		
	
	
			93 lines
		
	
	
		
			2.6 KiB
		
	
	
	
		
			Python
		
	
	
	
	
	
| import pytest
 | |
| from thinc.api import Adam
 | |
| from spacy.attrs import NORM
 | |
| from spacy.vocab import Vocab
 | |
| from spacy import registry
 | |
| from spacy.training import Example
 | |
| from spacy.pipeline.dep_parser import DEFAULT_PARSER_MODEL
 | |
| from spacy.tokens import Doc
 | |
| from spacy.pipeline import DependencyParser
 | |
| 
 | |
| 
 | |
| @pytest.fixture
 | |
| def vocab():
 | |
|     return Vocab(lex_attr_getters={NORM: lambda s: s})
 | |
| 
 | |
| 
 | |
| def _parser_example(parser):
 | |
|     doc = Doc(parser.vocab, words=["a", "b", "c", "d"])
 | |
|     gold = {"heads": [1, 1, 3, 3], "deps": ["right", "ROOT", "left", "ROOT"]}
 | |
|     return Example.from_dict(doc, gold)
 | |
| 
 | |
| 
 | |
| @pytest.fixture
 | |
| def parser(vocab):
 | |
|     vocab.strings.add("ROOT")
 | |
|     config = {
 | |
|         "learn_tokens": False,
 | |
|         "min_action_freq": 30,
 | |
|         "update_with_oracle_cut_size": 100,
 | |
|     }
 | |
|     cfg = {"model": DEFAULT_PARSER_MODEL}
 | |
|     model = registry.resolve(cfg, validate=True)["model"]
 | |
|     parser = DependencyParser(vocab, model, **config)
 | |
|     parser.cfg["token_vector_width"] = 4
 | |
|     parser.cfg["hidden_width"] = 32
 | |
|     # parser.add_label('right')
 | |
|     parser.add_label("left")
 | |
|     parser.initialize(lambda: [_parser_example(parser)])
 | |
|     sgd = Adam(0.001)
 | |
| 
 | |
|     for i in range(10):
 | |
|         losses = {}
 | |
|         doc = Doc(vocab, words=["a", "b", "c", "d"])
 | |
|         example = Example.from_dict(
 | |
|             doc, {"heads": [1, 1, 3, 3], "deps": ["left", "ROOT", "left", "ROOT"]}
 | |
|         )
 | |
|         parser.update([example], sgd=sgd, losses=losses)
 | |
|     return parser
 | |
| 
 | |
| 
 | |
| def test_no_sentences(parser):
 | |
|     doc = Doc(parser.vocab, words=["a", "b", "c", "d"])
 | |
|     doc = parser(doc)
 | |
|     assert len(list(doc.sents)) >= 1
 | |
| 
 | |
| 
 | |
| def test_sents_1(parser):
 | |
|     doc = Doc(parser.vocab, words=["a", "b", "c", "d"])
 | |
|     doc[2].sent_start = True
 | |
|     doc = parser(doc)
 | |
|     assert len(list(doc.sents)) >= 2
 | |
|     doc = Doc(parser.vocab, words=["a", "b", "c", "d"])
 | |
|     doc[1].sent_start = False
 | |
|     doc[2].sent_start = True
 | |
|     doc[3].sent_start = False
 | |
|     doc = parser(doc)
 | |
|     assert len(list(doc.sents)) == 2
 | |
| 
 | |
| 
 | |
| def test_sents_1_2(parser):
 | |
|     doc = Doc(parser.vocab, words=["a", "b", "c", "d"])
 | |
|     doc[1].sent_start = True
 | |
|     doc[2].sent_start = True
 | |
|     doc = parser(doc)
 | |
|     assert len(list(doc.sents)) >= 3
 | |
| 
 | |
| 
 | |
| def test_sents_1_3(parser):
 | |
|     doc = Doc(parser.vocab, words=["a", "b", "c", "d"])
 | |
|     doc[0].is_sent_start = True
 | |
|     doc[1].is_sent_start = True
 | |
|     doc[2].is_sent_start = None
 | |
|     doc[3].is_sent_start = True
 | |
|     doc = parser(doc)
 | |
|     assert len(list(doc.sents)) >= 3
 | |
|     doc = Doc(parser.vocab, words=["a", "b", "c", "d"])
 | |
|     doc[0].is_sent_start = True
 | |
|     doc[1].is_sent_start = True
 | |
|     doc[2].is_sent_start = False
 | |
|     doc[3].is_sent_start = True
 | |
|     doc = parser(doc)
 | |
|     assert len(list(doc.sents)) == 3
 |