handle postcodes properly on word table updates

[nominatim.git] / test / bdd / steps / steps_db_ops.py
diff --git a/test/bdd/steps/steps_db_ops.py b/test/bdd/steps/steps_db_ops.py

index e02cad8f4ac92c46d17ab4a4e7c74c1a3016cefc..37d541533dcd76ab06b21381013eec60b87a4659 100644 (file)
--- a/test/bdd/steps/steps_db_ops.py
+++ b/test/bdd/steps/steps_db_ops.py
@@ -18,13 +18,18 @@ from nominatim.tokenizer import factory as tokenizer_factory
  def check_database_integrity(context):
      """ Check some generic constraints on the tables.
      """
-    # place_addressline should not have duplicate (place_id, address_place_id)
-    cur = context.db.cursor()
-    cur.execute("""SELECT count(*) FROM
-                    (SELECT place_id, address_place_id, count(*) as c
-                     FROM place_addressline GROUP BY place_id, address_place_id) x
-                   WHERE c > 1""")
-    assert cur.fetchone()[0] == 0, "Duplicates found in place_addressline"
+    with context.db.cursor() as cur:
+        # place_addressline should not have duplicate (place_id, address_place_id)
+        cur.execute("""SELECT count(*) FROM
+                        (SELECT place_id, address_place_id, count(*) as c
+                         FROM place_addressline GROUP BY place_id, address_place_id) x
+                       WHERE c > 1""")
+        assert cur.fetchone()[0] == 0, "Duplicates found in place_addressline"
+
+        # word table must not have empty word_tokens
+        cur.execute("SELECT count(*) FROM word WHERE word_token = ''")
+        assert cur.fetchone()[0] == 0, "Empty word tokens found in word table"
+
  
  
  ################################ GIVEN ##################################
@@ -93,12 +98,17 @@ def add_data_to_planet_ways(context):
  def import_and_index_data_from_place_table(context):
      """ Import data previously set up in the place table.
      """
-    context.nominatim.run_nominatim('refresh', '--functions')
      context.nominatim.run_nominatim('import', '--continue', 'load-data',
-                                              '--index-noanalyse', '-q')
+                                              '--index-noanalyse', '-q',
+                                              '--offline')
  
      check_database_integrity(context)
  
+    # Remove the output of the input, when all was right. Otherwise it will be
+    # output when there are errors that had nothing to do with the import
+    # itself.
+    context.log_capture.buffer.clear()
+
  @when("updating places")
  def update_place_table(context):
      """ Update the place table with the given data. Also runs all triggers
@@ -112,6 +122,12 @@ def update_place_table(context):
      context.nominatim.reindex_placex(context.db)
      check_database_integrity(context)
  
+    # Remove the output of the input, when all was right. Otherwise it will be
+    # output when there are errors that had nothing to do with the import
+    # itself.
+    context.log_capture.buffer.clear()
+
+
  @when("updating postcodes")
  def update_postcodes(context):
      """ Rerun the calculation of postcodes.
@@ -131,6 +147,11 @@ def delete_places(context, oids):
  
      context.nominatim.reindex_placex(context.db)
  
+    # Remove the output of the input, when all was right. Otherwise it will be
+    # output when there are errors that had nothing to do with the import
+    # itself.
+    context.log_capture.buffer.clear()
+
  ################################ THEN ##################################
  
  @then("(?P<table>placex|place) contains(?P<exact> exactly)?")
@@ -266,7 +287,7 @@ def check_word_table_for_postcodes(context, exclude, postcodes):
      plist.sort()
  
      with context.db.cursor(cursor_factory=psycopg2.extras.DictCursor) as cur:
-        if nctx.tokenizer == 'icu':
+        if nctx.tokenizer != 'legacy':
              cur.execute("SELECT word FROM word WHERE type = 'P' and word = any(%s)",
                          (plist,))
          else: