Merge pull request #2408 from lonvia/icu-change-word-table-layout

[nominatim.git] / test / bdd / steps / steps_db_ops.py
diff --git a/test/bdd/steps/steps_db_ops.py b/test/bdd/steps/steps_db_ops.py

index 8285f9389dfc8ee8f027950e76ce058ebf19fc14..ac61fc67356aa8ab04274fe69bf9b28f0eddeffd 100644 (file)
--- a/test/bdd/steps/steps_db_ops.py
+++ b/test/bdd/steps/steps_db_ops.py
@@ -214,7 +214,7 @@ def check_search_name_contents(context, exclude):
                      for name, value in zip(row.headings, row.cells):
                          if name in ('name_vector', 'nameaddress_vector'):
                              items = [x.strip() for x in value.split(',')]
                      for name, value in zip(row.headings, row.cells):
                          if name in ('name_vector', 'nameaddress_vector'):
                              items = [x.strip() for x in value.split(',')]
-                            tokens = analyzer.get_word_token_info(context.db, items)
+                            tokens = analyzer.get_word_token_info(items)
  
                              if not exclude:
                                  assert len(tokens) >= len(items), \
  
                              if not exclude:
                                  assert len(tokens) >= len(items), \
@@ -266,20 +266,36 @@ def check_location_postcode(context):
  
              db_row.assert_row(row, ('country', 'postcode'))
  
  
              db_row.assert_row(row, ('country', 'postcode'))
  
-@then("word contains(?P<exclude> not)?")
-def check_word_table(context, exclude):
-    """ Check the contents of the word table. Each row represents a table row
-        and all data must match. Data not present in the expected table, may
-        be arbitry. The rows are identified via all given columns.
+@then("there are(?P<exclude> no)? word tokens for postcodes (?P<postcodes>.*)")
+def check_word_table_for_postcodes(context, exclude, postcodes):
+    """ Check that the tokenizer produces postcode tokens for the given
+        postcodes. The postcodes are a comma-separated list of postcodes.
+        Whitespace matters.
      """
      """
+    nctx = context.nominatim
+    tokenizer = tokenizer_factory.get_tokenizer_for_db(nctx.get_test_config())
+    with tokenizer.name_analyzer() as ana:
+        plist = [ana.normalize_postcode(p) for p in postcodes.split(',')]
+
+    plist.sort()
+
      with context.db.cursor(cursor_factory=psycopg2.extras.DictCursor) as cur:
      with context.db.cursor(cursor_factory=psycopg2.extras.DictCursor) as cur:
-        for row in context.table:
-            wheres = ' AND '.join(["{} = %s".format(h) for h in row.headings])
-            cur.execute("SELECT * from word WHERE " + wheres, list(row.cells))
-            if exclude:
-                assert cur.rowcount == 0, "Row still in word table: %s" % '/'.join(values)
-            else:
-                assert cur.rowcount > 0, "Row not in word table: %s" % '/'.join(values)
+        if nctx.tokenizer == 'legacy_icu':
+            cur.execute("SELECT word FROM word WHERE type = 'P' and word = any(%s)",
+                        (plist,))
+        else:
+            cur.execute("""SELECT word FROM word WHERE word = any(%s)
+                             and class = 'place' and type = 'postcode'""",
+                        (plist,))
+
+        found = [row[0] for row in cur]
+        assert len(found) == len(set(found)), f"Duplicate rows for postcodes: {found}"
+
+    if exclude:
+        assert len(found) == 0, f"Unexpected postcodes: {found}"
+    else:
+        assert set(found) == set(plist), \
+        f"Missing postcodes {set(plist) - set(found)}. Found: {found}"
  
  @then("place_addressline contains")
  def check_place_addressline(context):
  
  @then("place_addressline contains")
  def check_place_addressline(context):