2017-05-08 13:52:01 +00:00
|
|
|
# coding: utf8
|
|
|
|
from __future__ import unicode_literals
|
|
|
|
|
2017-05-12 13:37:39 +00:00
|
|
|
from ...attrs import LIKE_NUM
|
2017-05-08 13:52:01 +00:00
|
|
|
|
|
|
|
|
2017-05-12 13:37:39 +00:00
|
|
|
_num_words = ['zero', 'um', 'dois', 'três', 'quatro', 'cinco', 'seis', 'sete',
|
|
|
|
'oito', 'nove', 'dez', 'onze', 'doze', 'treze', 'catorze',
|
2018-05-09 18:49:31 +00:00
|
|
|
'quinze', 'dezesseis', 'dezasseis', 'dezessete', 'dezassete', 'dezoito', 'dezenove', 'dezanove', 'vinte',
|
2017-05-12 13:37:39 +00:00
|
|
|
'trinta', 'quarenta', 'cinquenta', 'sessenta', 'setenta',
|
2018-05-09 18:49:31 +00:00
|
|
|
'oitenta', 'noventa', 'cem', 'mil', 'milhão', 'bilhão', 'bilião', 'trilhão', 'trilião',
|
|
|
|
'quatrilhão']
|
2017-05-08 13:52:01 +00:00
|
|
|
|
2018-01-08 02:38:44 +00:00
|
|
|
_ordinal_words = ['primeiro', 'segundo', 'terceiro', 'quarto', 'quinto', 'sexto',
|
|
|
|
'sétimo', 'oitavo', 'nono', 'décimo', 'vigésimo', 'trigésimo',
|
|
|
|
'quadragésimo', 'quinquagésimo', 'sexagésimo', 'septuagésimo',
|
|
|
|
'octogésimo', 'nonagésimo', 'centésimo', 'ducentésimo',
|
|
|
|
'trecentésimo', 'quadringentésimo', 'quingentésimo', 'sexcentésimo',
|
|
|
|
'septingentésimo', 'octingentésimo', 'nongentésimo', 'milésimo',
|
|
|
|
'milionésimo', 'bilionésimo']
|
2017-05-08 13:52:01 +00:00
|
|
|
|
2017-05-12 13:37:39 +00:00
|
|
|
|
|
|
|
def like_num(text):
|
2018-10-01 08:49:14 +00:00
|
|
|
if text.startswith(('+', '-', '±', '~')):
|
|
|
|
text = text[1:]
|
2017-05-12 13:37:39 +00:00
|
|
|
text = text.replace(',', '').replace('.', '')
|
|
|
|
if text.isdigit():
|
|
|
|
return True
|
|
|
|
if text.count('/') == 1:
|
|
|
|
num, denom = text.split('/')
|
|
|
|
if num.isdigit() and denom.isdigit():
|
|
|
|
return True
|
2018-01-08 02:25:08 +00:00
|
|
|
if text.lower() in _num_words:
|
2017-05-12 13:37:39 +00:00
|
|
|
return True
|
2018-01-08 02:28:50 +00:00
|
|
|
if text.lower() in _ordinal_words:
|
|
|
|
return True
|
2017-05-12 13:37:39 +00:00
|
|
|
return False
|
|
|
|
|
|
|
|
|
|
|
|
LEX_ATTRS = {
|
|
|
|
LIKE_NUM: like_num
|
|
|
|
}
|