mirror of
https://github.com/explosion/spaCy.git
synced 2025-01-12 10:16:27 +03:00
db55577c45
* Remove unicode declarations * Remove Python 3.5 and 2.7 from CI * Don't require pathlib * Replace compat helpers * Remove OrderedDict * Use f-strings * Set Cython compiler language level * Fix typo * Re-add OrderedDict for Table * Update setup.cfg * Revert CONTRIBUTING.md * Revert lookups.md * Revert top-level.md * Small adjustments and docs [ci skip]
28 lines
835 B
Python
28 lines
835 B
Python
import pytest
|
||
|
||
|
||
@pytest.mark.parametrize("text", ["z.B.", "Jan."])
|
||
def test_lb_tokenizer_handles_abbr(lb_tokenizer, text):
|
||
tokens = lb_tokenizer(text)
|
||
assert len(tokens) == 1
|
||
|
||
|
||
@pytest.mark.parametrize("text", ["d'Saach", "d'Kanner", "d’Welt", "d’Suen"])
|
||
def test_lb_tokenizer_splits_contractions(lb_tokenizer, text):
|
||
tokens = lb_tokenizer(text)
|
||
assert len(tokens) == 2
|
||
|
||
|
||
def test_lb_tokenizer_handles_exc_in_text(lb_tokenizer):
|
||
text = "Mee 't ass net evident, d'Liewen."
|
||
tokens = lb_tokenizer(text)
|
||
assert len(tokens) == 9
|
||
assert tokens[1].text == "'t"
|
||
assert tokens[1].lemma_ == "et"
|
||
|
||
|
||
@pytest.mark.parametrize("text,norm", [("dass", "datt"), ("viläicht", "vläicht")])
|
||
def test_lb_norm_exceptions(lb_tokenizer, text, norm):
|
||
tokens = lb_tokenizer(text)
|
||
assert tokens[0].norm_ == norm
|