spaCy/spacy/lang/it/punctuation.py

13 lines
252 B
Python
Raw Normal View History

from ..punctuation import TOKENIZER_INFIXES
from ..char_classes import ALPHA
2019-11-15 18:19:01 +03:00
ELISION = " ' ".strip().replace(" ", "")
_infixes = TOKENIZER_INFIXES + [
r"(?<=[{a}][{el}])(?=[{a}])".format(a=ALPHA, el=ELISION)
]
TOKENIZER_INFIXES = _infixes