X-Git-Url: https://git.openstreetmap.org./nominatim.git/blobdiff_plain/d711f5a81e6d02926cefe5c9c94bcf1d8f375b11..0ba93e5ba97c96aee788dccd20a6f7f6e308548f:/lib-sql/tokenizer/legacy_tokenizer.sql diff --git a/lib-sql/tokenizer/legacy_tokenizer.sql b/lib-sql/tokenizer/legacy_tokenizer.sql index d78c800c..5e80b414 100644 --- a/lib-sql/tokenizer/legacy_tokenizer.sql +++ b/lib-sql/tokenizer/legacy_tokenizer.sql @@ -7,6 +7,7 @@ AS $$ SELECT (info->>'names')::INTEGER[] $$ LANGUAGE SQL IMMUTABLE STRICT; + -- Get tokens for matching the place name against others. -- -- This should usually be restricted to full name tokens. @@ -17,6 +18,66 @@ AS $$ $$ LANGUAGE SQL IMMUTABLE STRICT; +-- Return the housenumber tokens applicable for the place. +CREATE OR REPLACE FUNCTION token_get_housenumber_search_tokens(info JSONB) + RETURNS INTEGER[] +AS $$ + SELECT (info->>'hnr_tokens')::INTEGER[] +$$ LANGUAGE SQL IMMUTABLE STRICT; + + +-- Return the housenumber in the form that it can be matched during search. +CREATE OR REPLACE FUNCTION token_normalized_housenumber(info JSONB) + RETURNS TEXT +AS $$ + SELECT info->>'hnr'; +$$ LANGUAGE SQL IMMUTABLE STRICT; + + +CREATE OR REPLACE FUNCTION token_addr_street_match_tokens(info JSONB) + RETURNS INTEGER[] +AS $$ + SELECT (info->>'street')::INTEGER[] +$$ LANGUAGE SQL IMMUTABLE STRICT; + + +CREATE OR REPLACE FUNCTION token_addr_place_match_tokens(info JSONB) + RETURNS INTEGER[] +AS $$ + SELECT (info->>'place_match')::INTEGER[] +$$ LANGUAGE SQL IMMUTABLE STRICT; + + +CREATE OR REPLACE FUNCTION token_addr_place_search_tokens(info JSONB) + RETURNS INTEGER[] +AS $$ + SELECT (info->>'place_search')::INTEGER[] +$$ LANGUAGE SQL IMMUTABLE STRICT; + + +DROP TYPE IF EXISTS token_addresstoken CASCADE; +CREATE TYPE token_addresstoken AS ( + key TEXT, + match_tokens INT[], + search_tokens INT[] +); + +CREATE OR REPLACE FUNCTION token_get_address_tokens(info JSONB) + RETURNS SETOF token_addresstoken +AS $$ + SELECT key, (value->>1)::int[] as match_tokens, + (value->>0)::int[] as search_tokens + FROM jsonb_each(info->'addr'); +$$ LANGUAGE SQL IMMUTABLE STRICT; + + +CREATE OR REPLACE FUNCTION token_normalized_postcode(postcode TEXT) + RETURNS TEXT +AS $$ + SELECT CASE WHEN postcode SIMILAR TO '%(,|;)%' THEN NULL ELSE upper(trim(postcode))END; +$$ LANGUAGE SQL IMMUTABLE STRICT; + + -- Return token info that should be saved permanently in the database. CREATE OR REPLACE FUNCTION token_strip_info(info JSONB) RETURNS JSONB @@ -75,26 +136,25 @@ END; $$ LANGUAGE plpgsql; + -- Create housenumber tokens from an OSM addr:housenumber. -- The housnumber is split at comma and semicolon as necessary. -- The function returns the normalized form of the housenumber suitable -- for comparison. -CREATE OR REPLACE FUNCTION create_housenumber_id(housenumber TEXT) - RETURNS TEXT +CREATE OR REPLACE FUNCTION create_housenumbers(housenumbers TEXT[], + OUT tokens TEXT, + OUT normtext TEXT) AS $$ -DECLARE - normtext TEXT; BEGIN - SELECT array_to_string(array_agg(trans), ';') - INTO normtext - FROM (SELECT lookup_word as trans, getorcreate_housenumber_id(lookup_word) + SELECT array_to_string(array_agg(trans), ';'), array_agg(tid)::TEXT + INTO normtext, tokens + FROM (SELECT lookup_word as trans, getorcreate_housenumber_id(lookup_word) as tid FROM (SELECT make_standard_name(h) as lookup_word - FROM regexp_split_to_table(housenumber, '[,;]') h) x) y; - - return normtext; + FROM unnest(housenumbers) h) x) y; END; $$ LANGUAGE plpgsql STABLE STRICT; + CREATE OR REPLACE FUNCTION getorcreate_housenumber_id(lookup_word TEXT) RETURNS INTEGER AS $$ @@ -117,26 +177,26 @@ $$ LANGUAGE plpgsql; -CREATE OR REPLACE FUNCTION getorcreate_postcode_id(postcode TEXT) - RETURNS INTEGER +CREATE OR REPLACE FUNCTION create_postcode_id(postcode TEXT) + RETURNS BOOLEAN AS $$ DECLARE + r RECORD; lookup_token TEXT; - lookup_word TEXT; return_word_id INTEGER; BEGIN - lookup_word := upper(trim(postcode)); - lookup_token := ' ' || make_standard_name(lookup_word); - SELECT min(word_id) FROM word - WHERE word_token = lookup_token and word = lookup_word + lookup_token := ' ' || make_standard_name(postcode); + FOR r IN + SELECT word_id FROM word + WHERE word_token = lookup_token and word = postcode and class='place' and type='postcode' - INTO return_word_id; - IF return_word_id IS NULL THEN - return_word_id := nextval('seq_word'); - INSERT INTO word VALUES (return_word_id, lookup_token, lookup_word, - 'place', 'postcode', null, 0); - END IF; - RETURN return_word_id; + LOOP + RETURN false; + END LOOP; + + INSERT INTO word VALUES (nextval('seq_word'), lookup_token, postcode, + 'place', 'postcode', null, 0); + RETURN true; END; $$ LANGUAGE plpgsql;