spaCy/spacy/lang/it/punctuation.py

16 lines
308 B
Python
Raw Normal View History

# coding: utf8
from __future__ import unicode_literals
from ..punctuation import TOKENIZER_INFIXES
from ..char_classes import ALPHA
2019-11-15 15:19:01 +00:00
ELISION = " ' ".strip().replace(" ", "")
_infixes = TOKENIZER_INFIXES + [
r"(?<=[{a}][{el}])(?=[{a}])".format(a=ALPHA, el=ELISION)
]
TOKENIZER_INFIXES = _infixes