# -*- coding: utf-8 -*- """AI Visibility Index: did an AI answer name this Google Maps business? Matcher v2 (2026-09-17). Replaces the v1 rule inside ai_visibility_index.py, which counted a business as "mentioned" when generic words from its listing title appeared in the answer: the city ("Miami"), the trade ("Real Estate", "Personal Injury", "Plumber") or "NYC". Those words are in nearly every answer to "I need a [category] in [city]", so v1 overstated visibility. v2 rules, in order: 1. Split the Maps title into segments on " - ", " | ", ":", "," and "(". Titles are often keyword-stuffed ("Royal Lane Dental Center - Best Dentist in Dallas | Clear Aligners"), and the business name is the first segment that carries a distinctive word. 2. Distinctive words = words that are not trade, city, legal-suffix or filler words (GENERIC below), at least 2 characters long. 3. Distinctive words that are everyday English (COMMON: "one", "modern", "central") do not count on their own. Of the rest, one must appear as a whole word if the name has one, and at least two if it has two or more. 4. If every distinctive word is COMMON, a two-word phrase from the name must appear ("one real", "modern dentistry", "four seasons"). 5. If no segment has a distinctive word ("Philadelphia Dental"), the whole first segment must appear as an exact phrase, minus a trailing "Inc"/"LLC". 6. The business's own website domain appearing in the answer also counts. Text is lowercased, apostrophes dropped ("Nick's" -> "nicks") and every other non-alphanumeric character treated as a space, on both sides. Street addresses are removed from the answer first, so "Bellaire Air Conditioning" does not match a different company listed at "5933 Bellaire Blvd". Ambiguous near-names ("Dental Arts of Dallas" for "Dallas Dental Arts") are left counted as mentions: every judgement call errs toward visibility, so published invisibility rates can only be understated by the matcher, not inflated. The module is deliberately dependency-free so it can be published alongside the data and re-run by anyone. """ import re MATCHER_VERSION = "v2-2026-09-17" GENERIC = set(""" plumber plumbers plumbing drain drains drainage sewer sewers rooter rooters water heater heaters leak leaks pipe pipes repair repairs dentist dentists dental dentistry orthodontics orthodontist orthodontists implant implants cosmetic family pediatric general smile smiles teeth tooth dds dmd dr md lawyer lawyers attorney attorneys law legal injury injuries personal accident accidents car truck auto firm llp esq trial real estate realty realtor realtors agent agents agency broker brokers brokerage homes home properties property hvac heating cooling air conditioning ac furnace furnaces heat mechanical contractor contractors inc llc pllc pc co company corp corporation group team associates partners services service solutions pros pro professional professionals experts expert specialists specialist center centre clinic office offices care the and of in at by for a an to your my our near me best top local affordable quality emergency hour hours 24 7 247 same day fast son sons brothers bros new york nyc ny manhattan brooklyn queens bronx los angeles la socal dtla ca california chicago il illinois loop houston tx texas phoenix az arizona philadelphia philly pa pennsylvania san antonio diego dallas miami fl florida brickell downtown midtown uptown city usa us com st on with by brokered """.split()) # Everyday English words that appear in business names but also in ordinary # prose ("one of the best", "a modern office", "a central location"). Alone they # prove nothing, so a name whose only distinctive words are in this list must # appear as a two-word phrase from the name ("one real", "modern dentistry"). COMMON = set(""" one two three four first all same main central public direct green blue red white black gold golden silver modern american national united west east north south northwest southwest northeast southeast found night tech master rescue mission elite temp happy relax ultra king royal core champion village pride express star sky metro flow preferred heritage collective insider weather makers seasons coastal mesa tree heights oak broad walnut street avenue arts fifth premier choice smart bright prime premium advanced integrity trusted reliable honest precision perfect superior summit peak liberty freedom eagle lion diamond crystal pacific atlantic desert valley river lake park bay beach coast harbor sun sunset sunrise lane road way point square plaza hills hill grove place """.split()) _SEG_SPLIT = re.compile(r"\s+[-\u2013\u2014|]\s+|\s*[|:,(]\s*") def norm(s): s = (s or "").lower().replace("'", "").replace("’", "") s = re.sub(r"[^a-z0-9]+", " ", s) return " " + " ".join(s.split()) + " " def _words(seg): return norm(seg).split() def name_core(name): """(segment, distinctive words) for the segment used to match.""" segs = [s for s in _SEG_SPLIT.split(name or "") if s and s.strip()] for s in segs: d = [w for w in _words(s) if w not in GENERIC and len(w) >= 2] if d: return s.strip(), d return (segs[0].strip() if segs else (name or "")), [] def _root_domain(domain): if not domain: return None d = domain.lower().split("//")[-1].split("/")[0] if d.startswith("www."): d = d[4:] return d or None # Street addresses ("5933 Bellaire Blvd", "1601 Walnut St") appear in search-backed # answers next to OTHER businesses; a name that shares a street name must not match them. _ADDRESS = re.compile( r" \d+[a-z]? (?:[nsew] )?(?:[a-z0-9]+ ){1,4}?(?:st|street|ave|avenue|blvd|boulevard|rd|road|dr|drive|" r"ln|lane|pkwy|parkway|hwy|highway|way|ct|court|pl|place|ter|terrace|cir|circle|fwy|freeway|sq|square)" r"(?= )") def mentioned(business, answer, domain=None): """True/False, or None when there is no answer (engine excluded, never a miss).""" if not answer: return None text = _ADDRESS.sub(" ", norm(answer)) dom = _root_domain(domain) if dom and dom in (answer or "").lower(): return True seg, d = name_core(business) strong = list(dict.fromkeys(w for w in d if w not in COMMON)) if strong: hits = sum(1 for w in strong if f" {w} " in text) return hits >= min(2, len(strong)) phrase = _common_phrase(seg, d) if d else " ".join(_trim(_words(seg))) return bool(phrase) and f" {phrase} " in text _TRAIL = {"inc", "llc", "pllc", "pc", "co", "corp", "corporation", "company", "llp"} def _trim(words): while words and words[-1] in _TRAIL: words = words[:-1] return words def _common_phrase(seg, d): """Shortest phrase from the name that is specific enough: the run of distinctive words if it is 2+ words long, otherwise the distinctive word plus its neighbour in the name.""" w = _trim(_words(seg)) i = next(k for k, x in enumerate(w) if x in d) j = i while j + 1 < len(w) and w[j + 1] in d: j += 1 if j > i: return " ".join(w[i:j + 1]) if i + 1 < len(w): return " ".join(w[i:i + 2]) return " ".join(w[max(0, i - 1):i + 1]) def explain(business, domain=None): seg, d = name_core(business) strong = list(dict.fromkeys(w for w in d if w not in COMMON)) if strong: rule = "need %d of %s" % (min(2, len(strong)), sorted(strong)) elif d: rule = "phrase: %r" % _common_phrase(seg, d) else: rule = "phrase: %r" % " ".join(_trim(_words(seg))) return {"segment": seg, "distinctive": d, "domain": _root_domain(domain), "rule": rule}