"""House cross-source matching — 3-tier algorithm. Tier 0 (confidence 1.0): cadastral_number exact match on houses table. Tier 1 (confidence 1.0): ext_source + ext_id already in house_sources. Tier 2 (confidence 0.9): address_fingerprint match in house_address_aliases. Tier 3 (confidence 0.7): geo-proximity within 30 m (PostGIS ST_DWithin). New (confidence 1.0): INSERT new canonical house. Algorithm reference: decisions/Cross_Source_Matching_Strategy.md sec 3 """ import logging from sqlalchemy import text from sqlalchemy.orm import Session from app.services.matching.normalize import address_fingerprint, normalize_address logger = logging.getLogger(__name__) def match_or_create_house( db: Session, ext_source: str, ext_id: str, address: str | None = None, lat: float | None = None, lon: float | None = None, *, year_built: int | None = None, building_cadastral_number: str | None = None, cadastral_number: str | None = None, source_url: str | None = None, ) -> tuple[int, float, str]: """Match existing house or create new canonical record. Concurrency-safe: serializes concurrent calls for the same address fingerprint via pg_advisory_xact_lock(42, hashtext(fp)). The lock is transaction-scoped, released on the caller's COMMIT/ROLLBACK. Without this, two scrapers calling for an unknown address could both miss Tier 0-3 and each INSERT a duplicate house row. Closes finding #1 from 2026-05-24 audit. Returns: (house_id, confidence ∈ [0.0, 1.0], method ∈ { 'cadastr_exact', 'source_exact', 'fingerprint', 'geo_proximity', 'new' }) Method values: 'cadastr_exact' — matched by cadastral number (confidence 1.0) 'source_exact' — already in house_sources for this source+ext_id (confidence 1.0) 'fingerprint' — matched by address fingerprint (confidence 0.9) 'geo_proximity' — matched by geo within 30 m (confidence 0.7) 'new' — new house created (confidence 1.0) """ cad = building_cadastral_number or cadastral_number # Compute fingerprint early so we can acquire the advisory lock before any tier reads. fp = address_fingerprint(address, lat, lon) # Serialize concurrent calls for the same address fingerprint to prevent # racing INSERTs into houses (finding #1 from 2026-05-24 audit). # pg_advisory_xact_lock takes (int4, int4) — namespace 42 = "tradein house matching". # Lock is released automatically on caller's COMMIT/ROLLBACK. db.execute( text("SELECT pg_advisory_xact_lock(42, hashtext(:fp))"), {"fp": fp}, ) # Tier 0: cadastral number exact match if cad: row = ( db.execute( text("SELECT id FROM houses WHERE cadastral_number = :cad ORDER BY id ASC LIMIT 1"), {"cad": cad}, ) .mappings() .first() ) if row: house_id = int(row["id"]) _upsert_house_source( db, house_id=house_id, ext_source=ext_source, ext_id=ext_id, method="cadastr_exact", confidence=1.0, ) # Register fingerprint/normalized_address alias so a later scrape of the same # house from a cadastr-less source (typical SERP) hits Tier 2a/2b instead of # falling through to a duplicate New INSERT when geo jitter exceeds 30 m. _insert_alias(db, house_id=house_id, address=address, fp=fp, source=ext_source) logger.info("house match cadastr_exact house_id=%s cad=%s", house_id, cad) return (house_id, 1.0, "cadastr_exact") # Tier 1: source+ext_id already registered in house_sources row = ( db.execute( text( "SELECT house_id FROM house_sources " "WHERE ext_source = :s AND ext_id = :e LIMIT 1" ), {"s": ext_source, "e": str(ext_id)}, ) .mappings() .first() ) if row: house_id = int(row["house_id"]) logger.info( "house match source_exact house_id=%s src=%s ext_id=%s", house_id, ext_source, ext_id ) return (house_id, 1.0, "source_exact") # Tier 2: address fingerprint lookup. # Two sub-tiers to handle coord drift across scrapers: # 2a. exact fingerprint match (address + rounded coords) # 2b. normalized_address-only match — same street/number, different provider coords # Without 2b, two scrapers for the same house with slightly different lat/lon (beyond # the 4-decimal rounding tolerance) would produce distinct fingerprints, miss Tier 2a, # and each potentially create a duplicate house row. norm_addr = normalize_address(address) row = ( db.execute( text("SELECT house_id FROM house_address_aliases " "WHERE fingerprint = :fp LIMIT 1"), {"fp": fp}, ) .mappings() .first() ) if row is None and norm_addr: # Tier 2b: same normalized address, possibly different coords fingerprint row = ( db.execute( text( "SELECT house_id FROM house_address_aliases " "WHERE normalized_address = :na LIMIT 1" ), {"na": norm_addr}, ) .mappings() .first() ) if row: house_id = int(row["house_id"]) _upsert_house_source( db, house_id=house_id, ext_source=ext_source, ext_id=ext_id, method="fingerprint", confidence=0.9, ) # Ensure this fingerprint is also registered so future calls hit Tier 2a directly. _insert_alias(db, house_id=house_id, address=address, fp=fp, source=ext_source) logger.info("house match fingerprint house_id=%s fp=%s", house_id, fp) return (house_id, 0.9, "fingerprint") # Tier 3: geo-proximity — within 30 m (PostGIS geography cast on both sides) if lat is not None and lon is not None: row = ( db.execute( text(""" SELECT id, ST_Distance(geom::geography, ST_MakePoint(:lon, :lat)::geography) AS dist FROM houses WHERE geom IS NOT NULL AND ST_DWithin(geom::geography, ST_MakePoint(:lon, :lat)::geography, 30) ORDER BY dist ASC LIMIT 1 """), {"lat": lat, "lon": lon}, ) .mappings() .first() ) if row: house_id = int(row["id"]) _upsert_house_source( db, house_id=house_id, ext_source=ext_source, ext_id=ext_id, method="geo_proximity", confidence=0.7, ) # Register fingerprint alias so future calls skip geo lookup _insert_alias(db, house_id=house_id, address=address, fp=fp, source=ext_source) logger.info( "house match geo_proximity house_id=%s dist=%.1f src=%s", house_id, float(row["dist"]), ext_source, ) return (house_id, 0.7, "geo_proximity") # New house — INSERT canonical record. # geom column is auto-populated by houses_set_geom_trg BEFORE INSERT trigger from lat/lon. # Do NOT include geom in the INSERT column list — trigger handles it. url = source_url or f"matching://{ext_source}/{ext_id}" row = ( db.execute( text(""" INSERT INTO houses (source, ext_house_id, url, address, lat, lon, year_built, cadastral_number) VALUES ( :src, :eid, :url, :addr, CAST(:lat AS double precision), CAST(:lon AS double precision), CAST(:yb AS integer), :cad ) ON CONFLICT (source, ext_house_id) DO UPDATE SET address = COALESCE(EXCLUDED.address, houses.address) RETURNING id """), { "src": ext_source, "eid": str(ext_id), "url": url, "addr": address, "lat": lat, "lon": lon, "yb": year_built, "cad": cad, }, ) .mappings() .one() ) house_id = int(row["id"]) _upsert_house_source( db, house_id=house_id, ext_source=ext_source, ext_id=ext_id, method="new", confidence=1.0, ) _insert_alias(db, house_id=house_id, address=address, fp=fp, source=ext_source) logger.info("house new house_id=%s addr=%r src=%s", house_id, address, ext_source) return (house_id, 1.0, "new") def match_house_readonly( db: Session, *, address: str | None = None, lat: float | None = None, lon: float | None = None, cadastral_number: str | None = None, ) -> tuple[int, float, str] | None: """Match a target address to an EXISTING canonical house WITHOUT creating one. Read-only counterpart of match_or_create_house() for the estimate-target flow: we must NOT pollute `houses` with user-queried addresses, so this never INSERTs, never writes house_sources / aliases, and takes no advisory lock (no write race to guard). Tiers (same lookups as match_or_create_house, minus source_exact which needs an ext_id): 1. cadastr_exact — houses.cadastral_number = :cad (confidence 1.0) 2. fingerprint — house_address_aliases.fingerprint (confidence 0.9) 3. geo_proximity — within 50 m of an existing house (confidence 0.7) Returns (house_id, confidence, method) or None if nothing confident matches. Uses 50 m for geo (vs 30 m in match_or_create_house) to absorb geocoder jitter on the user-entered target address. """ # Tier 0: cadastral number exact match if cadastral_number: row = ( db.execute( text("SELECT id FROM houses WHERE cadastral_number = :cad ORDER BY id ASC LIMIT 1"), {"cad": cadastral_number}, ) .mappings() .first() ) if row: house_id = int(row["id"]) logger.info( "house readonly match cadastr_exact house_id=%s cad=%s", house_id, cadastral_number ) return (house_id, 1.0, "cadastr_exact") # Tier 1: address fingerprint lookup fp = address_fingerprint(address, lat, lon) row = ( db.execute( text("SELECT house_id FROM house_address_aliases WHERE fingerprint = :fp LIMIT 1"), {"fp": fp}, ) .mappings() .first() ) if row: house_id = int(row["house_id"]) logger.info("house readonly match fingerprint house_id=%s fp=%s", house_id, fp) return (house_id, 0.9, "fingerprint") # Tier 2: geo-proximity — within 50 m if lat is not None and lon is not None: row = ( db.execute( text(""" SELECT id, ST_Distance(geom::geography, ST_MakePoint(:lon, :lat)::geography) AS dist FROM houses WHERE geom IS NOT NULL AND ST_DWithin(geom::geography, ST_MakePoint(:lon, :lat)::geography, 50) ORDER BY dist ASC LIMIT 1 """), {"lat": lat, "lon": lon}, ) .mappings() .first() ) if row: house_id = int(row["id"]) logger.info( "house readonly match geo_proximity house_id=%s dist=%.1f", house_id, float(row["dist"]), ) return (house_id, 0.7, "geo_proximity") return None def _upsert_house_source( db: Session, *, house_id: int, ext_source: str, ext_id: str, method: str, confidence: float, ) -> None: """Insert or refresh house_sources row for this source+ext_id.""" db.execute( text(""" INSERT INTO house_sources ( house_id, ext_source, ext_id, confidence, matched_method, matched_at, last_seen_at ) VALUES ( CAST(:hid AS bigint), :s, :e, CAST(:c AS real), :m, NOW(), NOW() ) ON CONFLICT (ext_source, ext_id) DO UPDATE SET confidence = GREATEST(EXCLUDED.confidence, house_sources.confidence), -- Keep provenance consistent with the winning confidence: only adopt the -- new method/timestamp when the incoming match is more confident than the -- stored one (e.g. geo_proximity/0.7 later upgraded to cadastr_exact/1.0). matched_method = CASE WHEN EXCLUDED.confidence > house_sources.confidence THEN EXCLUDED.matched_method ELSE house_sources.matched_method END, matched_at = CASE WHEN EXCLUDED.confidence > house_sources.confidence THEN EXCLUDED.matched_at ELSE house_sources.matched_at END, last_seen_at = NOW() """), {"hid": house_id, "s": ext_source, "e": str(ext_id), "c": confidence, "m": method}, ) def _insert_alias( db: Session, *, house_id: int, address: str | None, fp: str, source: str, ) -> None: """Register normalized_address alias for this house (idempotent on normalized_address). ON CONFLICT DO UPDATE keeps the fingerprint in sync with the most recent call so that two scrapers providing the same address with slightly different coords (beyond the 4-decimal rounding boundary) do NOT spawn duplicate alias rows — they converge to one row with the latest fingerprint, which is then found by Tier 2a on the next scrape. house_id is not updated on conflict: the first writer wins canonical ownership. """ db.execute( text(""" INSERT INTO house_address_aliases (house_id, normalized_address, fingerprint, source) VALUES (CAST(:hid AS bigint), :na, :fp, :src) ON CONFLICT (normalized_address) DO UPDATE SET fingerprint = EXCLUDED.fingerprint, source = EXCLUDED.source """), { "hid": house_id, "na": normalize_address(address), "fp": fp, "src": source, }, )