spaCy/spacy/lang/lij/punctuation.py

12 lines
269 B
Python
Raw Normal View History

2020-03-20 04:20:17 +00:00
from ..char_classes import ALPHA
from ..punctuation import TOKENIZER_INFIXES
2020-03-20 04:20:17 +00:00
ELISION = " ' ".strip().replace(" ", "").replace("\n", "")
_infixes = TOKENIZER_INFIXES + [
r"(?<=[{a}][{el}])(?=[{a}])".format(a=ALPHA, el=ELISION)
]
TOKENIZER_INFIXES = _infixes