Merge remote-tracking branch 'upstream/master'

[nominatim.git] / nominatim / tokenizer / base.py
diff --git a/nominatim/tokenizer/base.py b/nominatim/tokenizer/base.py

index 1c1ca9f7bcfca3d7fa407504ac0b0c9d728191a1..12c826eb21b19da1b1ad989d4cc5ede9f0c699cf 100644 (file)
--- a/nominatim/tokenizer/base.py
+++ b/nominatim/tokenizer/base.py
@@ -5,17 +5,17 @@
  # Copyright (C) 2022 by the Nominatim developer community.
  # For a full list of authors see the git log.
  """
  # Copyright (C) 2022 by the Nominatim developer community.
  # For a full list of authors see the git log.
  """
-Abstract class defintions for tokenizers. These base classes are here
+Abstract class definitions for tokenizers. These base classes are here
  mainly for documentation purposes.
  """
  from abc import ABC, abstractmethod
  from typing import List, Tuple, Dict, Any, Optional, Iterable
  from pathlib import Path
  
  mainly for documentation purposes.
  """
  from abc import ABC, abstractmethod
  from typing import List, Tuple, Dict, Any, Optional, Iterable
  from pathlib import Path
  
-from typing_extensions import Protocol
-
  from nominatim.config import Configuration
  from nominatim.config import Configuration
+from nominatim.db.connection import Connection
  from nominatim.data.place_info import PlaceInfo
  from nominatim.data.place_info import PlaceInfo
+from nominatim.typing import Protocol
  
  class AbstractAnalyzer(ABC):
      """ The analyzer provides the functions for analysing names and building
  
  class AbstractAnalyzer(ABC):
      """ The analyzer provides the functions for analysing names and building
@@ -53,8 +53,8 @@ class AbstractAnalyzer(ABC):
  
              Returns:
                  The function returns the list of all tuples that could be
  
              Returns:
                  The function returns the list of all tuples that could be
-                found for the given words. Each list entry is a tuple of
-                (original word, word token, word id).
+                    found for the given words. Each list entry is a tuple of
+                    (original word, word token, word id).
          """
  
  
          """
  
  
@@ -114,11 +114,11 @@ class AbstractAnalyzer(ABC):
              the search index.
  
              Arguments:
              the search index.
  
              Arguments:
-                place: Place information retrived from the database.
+                place: Place information retrieved from the database.
  
              Returns:
                  A JSON-serialisable structure that will be handed into
  
              Returns:
                  A JSON-serialisable structure that will be handed into
-                the database via the `token_info` field.
+                    the database via the `token_info` field.
          """
  
  
          """
  
  
@@ -142,10 +142,8 @@ class AbstractTokenizer(ABC):
  
                init_db: When set to False, then initialisation of database
                  tables should be skipped. This option is only required for
  
                init_db: When set to False, then initialisation of database
                  tables should be skipped. This option is only required for
-                migration purposes and can be savely ignored by custom
+                migration purposes and can be safely ignored by custom
                  tokenizers.
                  tokenizers.
-
-            TODO: can we move the init_db parameter somewhere else?
          """
  
  
          """
  
  
@@ -197,13 +195,13 @@ class AbstractTokenizer(ABC):
  
              Returns:
                If an issue was found, return an error message with the
  
              Returns:
                If an issue was found, return an error message with the
-              description of the issue as well as hints for the user on
-              how to resolve the issue. If everything is okay, return `None`.
+                  description of the issue as well as hints for the user on
+                  how to resolve the issue. If everything is okay, return `None`.
          """
  
  
      @abstractmethod
          """
  
  
      @abstractmethod
-    def update_statistics(self) -> None:
+    def update_statistics(self, config: Configuration, threads: int = 1) -> None:
          """ Recompute any tokenizer statistics necessary for efficient lookup.
              This function is meant to be called from time to time by the user
              to improve performance. However, the tokenizer must not depend on
          """ Recompute any tokenizer statistics necessary for efficient lookup.
              This function is meant to be called from time to time by the user
              to improve performance. However, the tokenizer must not depend on
@@ -234,6 +232,17 @@ class AbstractTokenizer(ABC):
          """
  
  
          """
  
  
+    @abstractmethod
+    def most_frequent_words(self, conn: Connection, num: int) -> List[str]:
+        """ Return a list of the most frequent full words in the database.
+
+            Arguments:
+              conn: Open connection to the database which may be used to
+                    retrieve the words.
+              num: Maximum number of words to return.
+        """
+
+
  class TokenizerModule(Protocol):
      """ Interface that must be exported by modules that implement their
          own tokenizer.
  class TokenizerModule(Protocol):
      """ Interface that must be exported by modules that implement their
          own tokenizer.