spaCy/spacy/tests/matcher/test_matcher_bugfixes.py

import pytest
import numpy

from spacy.matcher import Matcher
from spacy.attrs import ORTH, LOWER, ENT_IOB, ENT_TYPE
from spacy.symbols import DATE


def test_overlap_issue118(EN):
    '''Test a bug that arose from having overlapping matches'''
    doc = EN.tokenizer(u'how many points did lebron james score against the boston celtics last night')
    ORG = doc.vocab.strings['ORG']
    matcher = Matcher(EN.vocab,
        {'BostonCeltics':
            ('ORG', {},
                [
                    [{LOWER: 'celtics'}],
                    [{LOWER: 'boston'}, {LOWER: 'celtics'}],
                ]
            )
        }
    )
    
    assert len(list(doc.ents)) == 0
    matches = matcher(doc)
    assert matches == [(ORG, 9, 11), (ORG, 10, 11)]
    ents = list(doc.ents)
    assert len(ents) == 1
    assert ents[0].label == ORG
    assert ents[0].start == 9
    assert ents[0].end == 11


def test_overlap_reorder(EN):
    '''Test order dependence'''
    doc = EN.tokenizer(u'how many points did lebron james score against the boston celtics last night')
    ORG = doc.vocab.strings['ORG']
    matcher = Matcher(EN.vocab,
        {'BostonCeltics':
            ('ORG', {},
                [
                    [{LOWER: 'boston'}, {LOWER: 'celtics'}],
                    [{LOWER: 'celtics'}],
                ]
            )
        }
    )
    
    assert len(list(doc.ents)) == 0
    matches = matcher(doc)
    assert matches == [(ORG, 9, 11), (ORG, 10, 11)]
    ents = list(doc.ents)
    assert len(ents) == 1
    assert ents[0].label == ORG
    assert ents[0].start == 9
    assert ents[0].end == 11


def test_overlap_prefix(EN):
    '''Test order dependence'''
    doc = EN.tokenizer(u'how many points did lebron james score against the boston celtics last night')
    ORG = doc.vocab.strings['ORG']
    matcher = Matcher(EN.vocab,
        {'BostonCeltics':
            ('ORG', {},
                [
                    [{LOWER: 'boston'}],
                    [{LOWER: 'boston'}, {LOWER: 'celtics'}],
                ]
            )
        }
    )
    
    assert len(list(doc.ents)) == 0
    matches = matcher(doc)
    assert matches == [(ORG, 9, 10), (ORG, 9, 11)]
    ents = list(doc.ents)
    assert len(ents) == 1
    assert ents[0].label == ORG
    assert ents[0].start == 9
    assert ents[0].end == 11


def test_overlap_prefix_reorder(EN):
    '''Test order dependence'''
    doc = EN.tokenizer(u'how many points did lebron james score against the boston celtics last night')
    ORG = doc.vocab.strings['ORG']
    matcher = Matcher(EN.vocab,
        {'BostonCeltics':
            ('ORG', {},
                [
                    [{LOWER: 'boston'}, {LOWER: 'celtics'}],
                    [{LOWER: 'boston'}],
                ]
            )
        }
    )
    
    assert len(list(doc.ents)) == 0
    matches = matcher(doc)
    assert matches == [(ORG, 9, 10), (ORG, 9, 11)]
    ents = list(doc.ents)
    assert len(ents) == 1
    assert ents[0].label == ORG
    assert ents[0].start == 9
    assert ents[0].end == 11


@pytest.mark.models
def test_ner_interaction(EN):
    EN.matcher.add('LAX_Airport', 'AIRPORT', {}, [[{ORTH: 'LAX'}]])
    EN.matcher.add('SFO_Airport', 'AIRPORT', {}, [[{ORTH: 'SFO'}]])
    doc = EN(u'get me a flight from SFO to LAX leaving 20 December and arriving on January 5th')

    ents = [(ent.label_, ent.text) for ent in doc.ents]
    assert ents[0] == ('AIRPORT', 'SFO')
    assert ents[1] == ('AIRPORT', 'LAX')
    assert ents[2] == ('DATE', '20 December')
    assert ents[3] == ('DATE', 'January 5th')
* Add a test for Issue #118: Matcher behaves unpredictably with overlapping entities 2015-10-01 06:21:00 +00:00			`import pytest`
* Add test for user NER classes in matcher blocking the NER model. Re Issue #178 and Issue #217 2016-01-19 18:23:16 +00:00			`import numpy`
* Add a test for Issue #118: Matcher behaves unpredictably with overlapping entities 2015-10-01 06:21:00 +00:00
			`from spacy.matcher import Matcher`
* Add test for user NER classes in matcher blocking the NER model. Re Issue #178 and Issue #217 2016-01-19 18:23:16 +00:00			`from spacy.attrs import ORTH, LOWER, ENT_IOB, ENT_TYPE`
			`from spacy.symbols import DATE`
* Fix Issue #118: Matcher behaves unpredictably when matches overlap. 2015-10-19 05:45:12 +00:00
* Add a test for Issue #118: Matcher behaves unpredictably with overlapping entities 2015-10-01 06:21:00 +00:00
			`def test_overlap_issue118(EN):`
			`'''Test a bug that arose from having overlapping matches'''`
			`doc = EN.tokenizer(u'how many points did lebron james score against the boston celtics last night')`
			`ORG = doc.vocab.strings['ORG']`
* Fix Issue #118: Matcher behaves unpredictably when matches overlap. 2015-10-19 05:45:12 +00:00			`matcher = Matcher(EN.vocab,`
			`{'BostonCeltics':`
			`('ORG', {},`
			`[`
			`[{LOWER: 'celtics'}],`
			`[{LOWER: 'boston'}, {LOWER: 'celtics'}],`
			`]`
			`)`
			`}`
			`)`

			`assert len(list(doc.ents)) == 0`
			`matches = matcher(doc)`
			`assert matches == [(ORG, 9, 11), (ORG, 10, 11)]`
			`ents = list(doc.ents)`
			`assert len(ents) == 1`
			`assert ents[0].label == ORG`
			`assert ents[0].start == 9`
			`assert ents[0].end == 11`


			`def test_overlap_reorder(EN):`
			`'''Test order dependence'''`
			`doc = EN.tokenizer(u'how many points did lebron james score against the boston celtics last night')`
			`ORG = doc.vocab.strings['ORG']`
			`matcher = Matcher(EN.vocab,`
			`{'BostonCeltics':`
			`('ORG', {},`
			`[`
			`[{LOWER: 'boston'}, {LOWER: 'celtics'}],`
			`[{LOWER: 'celtics'}],`
			`]`
			`)`
			`}`
			`)`

			`assert len(list(doc.ents)) == 0`
			`matches = matcher(doc)`
			`assert matches == [(ORG, 9, 11), (ORG, 10, 11)]`
			`ents = list(doc.ents)`
			`assert len(ents) == 1`
			`assert ents[0].label == ORG`
			`assert ents[0].start == 9`
			`assert ents[0].end == 11`


			`def test_overlap_prefix(EN):`
			`'''Test order dependence'''`
			`doc = EN.tokenizer(u'how many points did lebron james score against the boston celtics last night')`
			`ORG = doc.vocab.strings['ORG']`
			`matcher = Matcher(EN.vocab,`
			`{'BostonCeltics':`
			`('ORG', {},`
			`[`
			`[{LOWER: 'boston'}],`
			`[{LOWER: 'boston'}, {LOWER: 'celtics'}],`
			`]`
			`)`
			`}`
			`)`

			`assert len(list(doc.ents)) == 0`
			`matches = matcher(doc)`
			`assert matches == [(ORG, 9, 10), (ORG, 9, 11)]`
			`ents = list(doc.ents)`
			`assert len(ents) == 1`
			`assert ents[0].label == ORG`
			`assert ents[0].start == 9`
			`assert ents[0].end == 11`


			`def test_overlap_prefix_reorder(EN):`
			`'''Test order dependence'''`
			`doc = EN.tokenizer(u'how many points did lebron james score against the boston celtics last night')`
			`ORG = doc.vocab.strings['ORG']`
			`matcher = Matcher(EN.vocab,`
			`{'BostonCeltics':`
			`('ORG', {},`
			`[`
			`[{LOWER: 'boston'}, {LOWER: 'celtics'}],`
			`[{LOWER: 'boston'}],`
			`]`
			`)`
			`}`
			`)`
* Add a test for Issue #118: Matcher behaves unpredictably with overlapping entities 2015-10-01 06:21:00 +00:00
* Fix Issue #118: Matcher behaves unpredictably when matches overlap. 2015-10-19 05:45:12 +00:00			`assert len(list(doc.ents)) == 0`
* Add a test for Issue #118: Matcher behaves unpredictably with overlapping entities 2015-10-01 06:21:00 +00:00			`matches = matcher(doc)`
* Fix Issue #118: Matcher behaves unpredictably when matches overlap. 2015-10-19 05:45:12 +00:00			`assert matches == [(ORG, 9, 10), (ORG, 9, 11)]`
* Add a test for Issue #118: Matcher behaves unpredictably with overlapping entities 2015-10-01 06:21:00 +00:00			`ents = list(doc.ents)`
			`assert len(ents) == 1`
			`assert ents[0].label == ORG`
			`assert ents[0].start == 9`
			`assert ents[0].end == 11`

* Add test for user NER classes in matcher blocking the NER model. Re Issue #178 and Issue #217 2016-01-19 18:23:16 +00:00
			`@pytest.mark.models`
			`def test_ner_interaction(EN):`
			`EN.matcher.add('LAX_Airport', 'AIRPORT', {}, [[{ORTH: 'LAX'}]])`
			`EN.matcher.add('SFO_Airport', 'AIRPORT', {}, [[{ORTH: 'SFO'}]])`
* Fix matcher test 2016-01-19 19:24:01 +00:00			`doc = EN(u'get me a flight from SFO to LAX leaving 20 December and arriving on January 5th')`
* Add test for user NER classes in matcher blocking the NER model. Re Issue #178 and Issue #217 2016-01-19 18:23:16 +00:00
			`ents = [(ent.label_, ent.text) for ent in doc.ents]`
			`assert ents[0] == ('AIRPORT', 'SFO')`
			`assert ents[1] == ('AIRPORT', 'LAX')`
			`assert ents[2] == ('DATE', '20 December')`
			`assert ents[3] == ('DATE', 'January 5th')`