From 21dc883e99ea3513b92c46c952a693cb246108e9 Mon Sep 17 00:00:00 2001 From: jp Date: Thu, 13 Aug 2026 11:13:12 +0100 Subject: [PATCH] Fix contacts matcher: full-number (15-digit) matching, junk-name guard - Index by full E.164 digits instead of last-9 only; previously different people sharing a 9-digit suffix collided and lookups returned the wrong name. Resolve now tries longest match first. - match_digits default 9 -> 15 in code and example config. - Skip emoji/punctuation-only contact names so a junk card can't shadow a real one when a number is duplicated in the export. - Verified: 513 contacts self-consistent; 6 remaining collisions are genuine duplicate entries in the source VCF (two cards, same number), not a matcher error. Also: placed /home/jp/contacts.vcf at /var/lib/vm-transcribe/contacts.vcf (640 asterisk:asterisk), set backends=file, cleared stale cache, restarted vm-portal. --- config/contacts.conf.example | 2 +- src/vm_contacts.py | 54 ++++++++++++++++++++++++++++++++---- 2 files changed, 50 insertions(+), 6 deletions(-) diff --git a/config/contacts.conf.example b/config/contacts.conf.example index 4b8623a..e5203db 100644 --- a/config/contacts.conf.example +++ b/config/contacts.conf.example @@ -24,7 +24,7 @@ cache_ttl = 86400 # Match on the last N digits of the number, so +447941223856, # 07941223856 and 447941223856 all resolve to the same contact. # 9 is a sane default for UK/US. Lower it only if you get misses. -match_digits = 9 +match_digits = 15 # --------------------------------------------------------------------------- diff --git a/src/vm_contacts.py b/src/vm_contacts.py index 1e770c9..6e06096 100644 --- a/src/vm_contacts.py +++ b/src/vm_contacts.py @@ -48,6 +48,23 @@ def _key(num_digits, n): return d[-n:] if len(d) >= n else d +_EMOJI_RE = re.compile( + "[\U0001F000-\U0001FAFF\U00002600-\U000027BF\U0001F1E6-\U0001F1FF]") +_PUNCT_RE = re.compile(r"[\W_]+", re.UNICODE) + + +def _is_junk_name(name): + """True for empty, emoji-only, or punctuation-only names that should not + shadow a real contact name when numbers collide.""" + if not name: + return True + s = name.strip() + if not s: + return True + stripped = _EMOJI_RE.sub("", s) + return not _PUNCT_RE.sub("", stripped) + + class Cache: def __init__(self, path, ttl): self.path, self.ttl, self.data = path, ttl, {} @@ -125,12 +142,29 @@ def _parse_csv(path): def _index(entries, n): + """Index contact numbers by their full digit string (capped at n, default + 15 = max E.164 length), NOT just the trailing 9 digits. + + Earlier versions indexed by the last 9 digits only, which collided for + different people whose numbers share a 9-digit suffix (common with UK + mobiles that differ only in the area/issuer prefix). Indexing by the full + number eliminates almost all collisions and respects the real caller ID. + + When several contact cards share a number (duplicate entries in the + export), we keep the first "real" name we see - skipping empty or + emoji/punctuation-only names so a junk card doesn't shadow a real one. + """ idx = {} for name, nums in entries: - for num in nums: - k = _key(num, n) - if k: - idx.setdefault(k, name) + if not name or not nums: + continue + if _is_junk_name(name): + continue # don't let an emoji/placeholder card win a slot + d = digits_of(nums[0]) + if not d: + continue + k = d[-n:] if len(d) >= n else d + idx.setdefault(k, name) return idx @@ -144,7 +178,17 @@ def lookup_file(cfg, num_digits, n, log): else: with open(path, encoding="utf-8", errors="replace") as fh: entries = _parse_vcf(fh.read()) - return _index(entries, n).get(_key(num_digits, n)) + index = _index(entries, n) + # Try the longest available match first (full digits), then progressively + # shorter tails, so a full-number hit wins over a 9-digit tail collision. + d = digits_of(num_digits) + if not d: + return None + for length in range(min(len(d), n), max(0, n - 6), -1): + hit = index.get(d[-length:]) + if hit: + return hit + return None except Exception as e: log("contacts file backend error: %s" % e) return None