add actual removal of housenumber tokens

[nominatim.git] / nominatim / tokenizer / icu_tokenizer.py
diff --git a/nominatim/tokenizer/icu_tokenizer.py b/nominatim/tokenizer/icu_tokenizer.py

index 33f05cc4b21d463d22dfc675bab665c804fbc276..0841300a9b92421ac130ab827ff2a9935af507ed 100644 (file)
--- a/nominatim/tokenizer/icu_tokenizer.py
+++ b/nominatim/tokenizer/icu_tokenizer.py
@@ -1,3 +1,9 @@
+# SPDX-License-Identifier: GPL-2.0-only
+#
+# This file is part of Nominatim. (https://nominatim.org)
+#
+# Copyright (C) 2022 by the Nominatim developer community.
+# For a full list of authors see the git log.
  """
  Tokenizer implementing normalisation as used before Nominatim 4 but using
  libICU instead of the PostgreSQL module.
  """
  Tokenizer implementing normalisation as used before Nominatim 4 but using
  libICU instead of the PostgreSQL module.
@@ -106,6 +112,45 @@ class LegacyICUTokenizer(AbstractTokenizer):
              conn.commit()
  
  
              conn.commit()
  
  
+    def _cleanup_housenumbers(self):
+        """ Remove unused house numbers.
+        """
+        with connect(self.dsn) as conn:
+            with conn.cursor(name="hnr_counter") as cur:
+                cur.execute("""SELECT word_id, word_token FROM word
+                               WHERE type = 'H'
+                                 AND NOT EXISTS(SELECT * FROM search_name
+                                                WHERE ARRAY[word.word_id] && name_vector)
+                                 AND (char_length(word_token) > 6
+                                      OR word_token not similar to '\d+')
+                            """)
+                candidates = {token: wid for wid, token in cur}
+            with conn.cursor(name="hnr_counter") as cur:
+                cur.execute("""SELECT housenumber FROM placex
+                               WHERE housenumber is not null
+                                     AND (char_length(housenumber) > 6
+                                          OR housenumber not similar to '\d+')
+                            """)
+                for row in cur:
+                    for hnr in row[0].split(';'):
+                        candidates.pop(hnr, None)
+            LOG.info("There are %s outdated housenumbers.", len(candidates))
+            if candidates:
+                with conn.cursor() as cur:
+                    cur.execute("""DELETE FROM word WHERE word_id = any(%s)""",
+                                (list(candidates.values()), ))
+                conn.commit()
+
+
+
+    def update_word_tokens(self):
+        """ Remove unused tokens.
+        """
+        LOG.warn("Cleaning up housenumber tokens.")
+        self._cleanup_housenumbers()
+        LOG.warn("Tokenizer house-keeping done.")
+
+
      def name_analyzer(self):
          """ Create a new analyzer for tokenizing names and queries
              using this tokinzer. Analyzers are context managers and should
      def name_analyzer(self):
          """ Create a new analyzer for tokenizing names and queries
              using this tokinzer. Analyzers are context managers and should
@@ -407,18 +452,18 @@ class LegacyICUNameAnalyzer(AbstractAnalyzer):
  
  
      def _process_place_address(self, token_info, address):
  
  
      def _process_place_address(self, token_info, address):
-        hnrs = []
+        hnrs = set()
          addr_terms = []
          streets = []
          for item in address:
              if item.kind == 'postcode':
                  self._add_postcode(item.name)
          addr_terms = []
          streets = []
          for item in address:
              if item.kind == 'postcode':
                  self._add_postcode(item.name)
-            elif item.kind in ('housenumber', 'streetnumber', 'conscriptionnumber'):
-                hnrs.append(item.name)
+            elif item.kind == 'housenumber':
+                norm_name = self._make_standard_hnr(item.name)
+                if norm_name:
+                    hnrs.add(norm_name)
              elif item.kind == 'street':
              elif item.kind == 'street':
-                token = self._retrieve_full_token(item.name)
-                if token:
-                    streets.append(token)
+                streets.extend(self._retrieve_full_tokens(item.name))
              elif item.kind == 'place':
                  if not item.suffix:
                      token_info.add_place(self._compute_partial_tokens(item.name))
              elif item.kind == 'place':
                  if not item.suffix:
                      token_info.add_place(self._compute_partial_tokens(item.name))
@@ -427,8 +472,7 @@ class LegacyICUNameAnalyzer(AbstractAnalyzer):
                  addr_terms.append((item.kind, self._compute_partial_tokens(item.name)))
  
          if hnrs:
                  addr_terms.append((item.kind, self._compute_partial_tokens(item.name)))
  
          if hnrs:
-            hnrs = self._split_housenumbers(hnrs)
-            token_info.add_housenumbers(self.conn, [self._make_standard_hnr(n) for n in hnrs])
+            token_info.add_housenumbers(self.conn, hnrs)
  
          if addr_terms:
              token_info.add_address_terms(addr_terms)
  
          if addr_terms:
              token_info.add_address_terms(addr_terms)
@@ -465,25 +509,20 @@ class LegacyICUNameAnalyzer(AbstractAnalyzer):
          return tokens
  
  
          return tokens
  
  
-    def _retrieve_full_token(self, name):
+    def _retrieve_full_tokens(self, name):
          """ Get the full name token for the given name, if it exists.
              The name is only retrived for the standard analyser.
          """
          """ Get the full name token for the given name, if it exists.
              The name is only retrived for the standard analyser.
          """
-        norm_name = self._normalized(name)
+        norm_name = self._search_normalized(name)
  
          # return cached if possible
          if norm_name in self._cache.fulls:
              return self._cache.fulls[norm_name]
  
  
          # return cached if possible
          if norm_name in self._cache.fulls:
              return self._cache.fulls[norm_name]
  
-        # otherwise compute
-        full, _ = self._cache.names.get(norm_name, (None, None))
-
-        if full is None:
-            with self.conn.cursor() as cur:
-                cur.execute("SELECT word_id FROM word WHERE word = %s and type = 'W' LIMIT 1",
-                            (norm_name, ))
-                if cur.rowcount > 0:
-                    full = cur.fetchone()[0]
+        with self.conn.cursor() as cur:
+            cur.execute("SELECT word_id FROM word WHERE word_token = %s and type = 'W'",
+                        (norm_name, ))
+            full = [row[0] for row in cur]
  
          self._cache.fulls[norm_name] = full
  
  
          self._cache.fulls[norm_name] = full
  
@@ -546,24 +585,6 @@ class LegacyICUNameAnalyzer(AbstractAnalyzer):
                  self._cache.postcodes.add(postcode)
  
  
                  self._cache.postcodes.add(postcode)
  
  
-    @staticmethod
-    def _split_housenumbers(hnrs):
-        if len(hnrs) > 1 or ',' in hnrs[0] or ';' in hnrs[0]:
-            # split numbers if necessary
-            simple_list = []
-            for hnr in hnrs:
-                simple_list.extend((x.strip() for x in re.split(r'[;,]', hnr)))
-
-            if len(simple_list) > 1:
-                hnrs = list(set(simple_list))
-            else:
-                hnrs = simple_list
-
-        return hnrs
-
-
-
-
  class _TokenInfo:
      """ Collect token information to be sent back to the database.
      """
  class _TokenInfo:
      """ Collect token information to be sent back to the database.
      """