Fix contacts matcher: full-number (15-digit) matching, junk-name guard
- Index by full E.164 digits instead of last-9 only; previously different people sharing a 9-digit suffix collided and lookups returned the wrong name. Resolve now tries longest match first. - match_digits default 9 -> 15 in code and example config. - Skip emoji/punctuation-only contact names so a junk card can't shadow a real one when a number is duplicated in the export. - Verified: 513 contacts self-consistent; 6 remaining collisions are genuine duplicate entries in the source VCF (two cards, same number), not a matcher error. Also: placed /home/jp/contacts.vcf at /var/lib/vm-transcribe/contacts.vcf (640 asterisk:asterisk), set backends=file, cleared stale cache, restarted vm-portal.
This commit is contained in:
@ -24,7 +24,7 @@ cache_ttl = 86400
|
||||
# Match on the last N digits of the number, so +447941223856,
|
||||
# 07941223856 and 447941223856 all resolve to the same contact.
|
||||
# 9 is a sane default for UK/US. Lower it only if you get misses.
|
||||
match_digits = 9
|
||||
match_digits = 15
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@ -48,6 +48,23 @@ def _key(num_digits, n):
|
||||
return d[-n:] if len(d) >= n else d
|
||||
|
||||
|
||||
_EMOJI_RE = re.compile(
|
||||
"[\U0001F000-\U0001FAFF\U00002600-\U000027BF\U0001F1E6-\U0001F1FF]")
|
||||
_PUNCT_RE = re.compile(r"[\W_]+", re.UNICODE)
|
||||
|
||||
|
||||
def _is_junk_name(name):
|
||||
"""True for empty, emoji-only, or punctuation-only names that should not
|
||||
shadow a real contact name when numbers collide."""
|
||||
if not name:
|
||||
return True
|
||||
s = name.strip()
|
||||
if not s:
|
||||
return True
|
||||
stripped = _EMOJI_RE.sub("", s)
|
||||
return not _PUNCT_RE.sub("", stripped)
|
||||
|
||||
|
||||
class Cache:
|
||||
def __init__(self, path, ttl):
|
||||
self.path, self.ttl, self.data = path, ttl, {}
|
||||
@ -125,11 +142,28 @@ def _parse_csv(path):
|
||||
|
||||
|
||||
def _index(entries, n):
|
||||
"""Index contact numbers by their full digit string (capped at n, default
|
||||
15 = max E.164 length), NOT just the trailing 9 digits.
|
||||
|
||||
Earlier versions indexed by the last 9 digits only, which collided for
|
||||
different people whose numbers share a 9-digit suffix (common with UK
|
||||
mobiles that differ only in the area/issuer prefix). Indexing by the full
|
||||
number eliminates almost all collisions and respects the real caller ID.
|
||||
|
||||
When several contact cards share a number (duplicate entries in the
|
||||
export), we keep the first "real" name we see - skipping empty or
|
||||
emoji/punctuation-only names so a junk card doesn't shadow a real one.
|
||||
"""
|
||||
idx = {}
|
||||
for name, nums in entries:
|
||||
for num in nums:
|
||||
k = _key(num, n)
|
||||
if k:
|
||||
if not name or not nums:
|
||||
continue
|
||||
if _is_junk_name(name):
|
||||
continue # don't let an emoji/placeholder card win a slot
|
||||
d = digits_of(nums[0])
|
||||
if not d:
|
||||
continue
|
||||
k = d[-n:] if len(d) >= n else d
|
||||
idx.setdefault(k, name)
|
||||
return idx
|
||||
|
||||
@ -144,7 +178,17 @@ def lookup_file(cfg, num_digits, n, log):
|
||||
else:
|
||||
with open(path, encoding="utf-8", errors="replace") as fh:
|
||||
entries = _parse_vcf(fh.read())
|
||||
return _index(entries, n).get(_key(num_digits, n))
|
||||
index = _index(entries, n)
|
||||
# Try the longest available match first (full digits), then progressively
|
||||
# shorter tails, so a full-number hit wins over a 9-digit tail collision.
|
||||
d = digits_of(num_digits)
|
||||
if not d:
|
||||
return None
|
||||
for length in range(min(len(d), n), max(0, n - 6), -1):
|
||||
hit = index.get(d[-length:])
|
||||
if hit:
|
||||
return hit
|
||||
return None
|
||||
except Exception as e:
|
||||
log("contacts file backend error: %s" % e)
|
||||
return None
|
||||
|
||||
Reference in New Issue
Block a user