#!/usr/bin/env python3 """ geo-score — score a site 0-100 on whether AI answer engines can find, parse, trust and cite it. Implements the open AIV rubric v1.1. python3 geo_score.py https://example.com # level 1: readiness, free python3 geo_score.py example.com --ask "best X for Y" # level 2: + a live citation check (your API keys) python3 geo_score.py watch run # level 3: track a question set weekly python3 geo_score.py mcp # all three levels over MCP (stdio) Levels 2 and 3 live in geo_watch.py next to this file. No dependencies. Python 3.8+. Reads only public URLs. Rubric: https://github.com/jianruntech/geo-score """ import argparse, concurrent.futures as cf, datetime, fnmatch, html, io, json, os, re, sys, textwrap, threading, time, zlib import http.client, ipaddress, socket, urllib.error, urllib.parse, urllib.request try: import ssl except ImportError: # a Python built without OpenSSL: https fails, and says so ssl = None __version__ = "1.8.0" RUBRIC = "v1.1" UA_BROWSER = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 " "(KHTML, like Gecko) Chrome/131.0 Safari/537.36") # The 10 retrieval crawlers of reference/ai-crawlers.md, the agents g.robots scores and g.reachable probes, # each with the user-agent string its vendor documents. Anthropic documents only the product tokens of # Claude-SearchBot and Claude-User; their strings follow the form of ClaudeBot's. RETRIEVAL_UAS = [ ("OAI-SearchBot", "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) " "Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot"), ("ChatGPT-User", "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; ChatGPT-User/1.0; +https://openai.com/bot"), ("Claude-SearchBot", "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Claude-SearchBot/1.0; " "+Claude-SearchBot@anthropic.com)"), ("Claude-User", "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Claude-User/1.0; " "+Claude-User@anthropic.com)"), ("PerplexityBot", "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; " "+https://perplexity.ai/perplexitybot)"), ("Perplexity-User", "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Perplexity-User/1.0; " "+https://perplexity.ai/perplexity-user)"), ("Googlebot", "Mozilla/5.0 (compatible; Googlebot/2.1; +http://www.google.com/bot.html)"), ("Bingbot", "Mozilla/5.0 (compatible; bingbot/2.0; +http://www.bing.com/bingbot.htm)"), ("Applebot", "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) " "Version/17.4 Safari/605.1.15 (Applebot/0.1; +http://www.apple.com/go/applebot)"), ("Amazonbot", "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Amazonbot/0.1) " "Chrome/119.0.6045.214 Safari/537.36"), ] # Training crawlers and opt-out tokens. They are never scored: a robots.txt group that disallows one by name is # reported in g.robots' evidence, and nothing else. Not scored is not free: Google-Extended also governs grounding in # Gemini Apps and Vertex AI, Meta documents Meta-ExternalAgent for indexing as well as training, and for the others # the cost is unknown; no check measures any of it (reference/ai-crawlers.md). NOT_SCORED_TOKENS = ["GPTBot", "ClaudeBot", "CCBot", "Google-Extended", "Applebot-Extended", "anthropic-ai", "cohere-ai", "Bytespider", "Meta-ExternalAgent"] CN_UAS = [("Baiduspider", "Mozilla/5.0 (compatible; Baiduspider/2.0; +http://www.baidu.com/search/spider.html)"), ("Sogou", "Sogou web spider/4.0"), ("PetalBot", "Mozilla/5.0 (compatible; PetalBot;+https://webmaster.petalsearch.com/site/petalbot)")] # A heading matches question intent if a person would phrase their question that way — a question, # a task, an explanation or a comparison — not merely if it carries a question mark. The English and the # Chinese forms match the same kinds of heading (test_geo_score.py holds them to it): "Create your first app" # and "创建你的第一个应用", "How workflows work" and "工作流的工作原理", "Plans compared" and "版本对比". # Chinese has no word boundary, so a task verb that opens a noun is excluded by what follows it: 管理团队 is a # team, 处理器 a processor, 集成电路 a chip, 使用条款 the terms of use. So are two product specs: 对比度 is a contrast # ratio, 原理图 a schematic. QP_INTENT = re.compile(r"\b(how|what|why|when|where|which|who)\b|[??]" r"|^(get|getting|set|setting|add|adding|build|building|create|creating|use|using|" r"install|installing|deploy|deploying|configure|connect|accept|send|manage|migrate|" r"write|writing|run|running|test|testing|choose|handle|customi[sz]e|compare|comparing)\b" r"|\b(vs|versus|compared)\b" r"|怎么|如何|什么|为什么|是否|多少|入门|教程|指南" r"|^(创建|新建|设置|配置|安装|部署|使用|添加|连接|接入|集成|管理|迁移|选择|处理|自定义|编写|运行|" r"测试|构建|搭建|发送|接收|开始|导入|导出|获取|开通|启用)(?!器|员|层|者|团队|电路|条款|协议)" r"|原理(?!图)|运作方式|对比(?!度)|区别|哪个好", re.I) # Short nav-like labels are not headings. \w matches CJK in Python, so the short-label branch must not # swallow real CJK headings: "怎么定价" (4 chars) is a question heading, "价格" is a nav label. NAV_LABEL = re.compile(r"^\s*(\w+\s*[||·]\s*\w+|[A-Za-z0-9_]{1,12}|[\u4e00-\u9fff]{1,3})\s*$") # ── colour ───────────────────────────────────────────────────────────────── class C: on = sys.stdout.isatty() and os.environ.get("NO_COLOR") is None def __getattr__(self, k): codes = dict(dim="\033[2m", b="\033[1m", r="\033[0m", green="\033[38;5;35m", amber="\033[38;5;179m", red="\033[38;5;167m", grey="\033[38;5;245m", brass="\033[38;5;137m", pine="\033[38;5;29m") return codes.get(k, "") if self.on else "" c = C() # ── fetching ─────────────────────────────────────────────────────────────── def text_file(r): """A file answer, not a site's catch-all page. Single-page apps answer every path with 200 and their HTML shell, which read as an llms.txt, an llms-full.txt and an ai.txt that do not exist.""" head = r.body[:2048].lower() if re.match(rb"\s*( or . A single-page app's catch-all answers /sitemap.xml with 200 and its HTML shell, which is not a sitemap.""" return r.ok and bool(re.search(r"<(urlset|sitemapindex)\b", r.text[:200000], re.I)) class Resp: __slots__ = ("url", "status", "body", "headers", "err", "elapsed", "_kind") def __init__(self, url, status=0, body=b"", headers=None, err=None, elapsed=0.0, kind=None): self.url, self.status, self.body = url, status, body self.headers, self.err, self.elapsed, self._kind = headers or {}, err, elapsed, kind @property def kind(self): """Why this answer is no page to score, in schema/error.v1.json's words: with no HTTP answer dns, refused, tls, timeout, dropped, no_response (or not_public under the address guard), or slow and deadline when fetch() stopped it for time; http_ for an HTTP error; no_content for an empty 2xx. None for a page with content.""" if self.status == 0: return self._kind or fault_kind(self.err) if not self.ok: return "http_%d" % self.status return None if self.body else "no_content" @property def text(self): cs = "utf-8" m = re.search(r'charset=["\']?([\w-]+)', self.headers.get("content-type", ""), re.I) if m: cs = m.group(1) try: return self.body.decode(cs, "replace") except LookupError: return self.body.decode("utf-8", "replace") @property def ok(self): return 200 <= self.status < 300 UNSAFE = ' <>"{}|\\^`' def idna(url): """A non-ASCII host has to go on the wire as punycode, and a non-ASCII path as percent-encoded UTF-8. urllib does neither, so a Chinese domain used to raise UnicodeEncodeError before a single request went out.""" try: u = urllib.parse.urlsplit(url) host = u.hostname or "" if any(ord(ch) > 127 for ch in host): enc = host.encode("idna").decode("ascii") netloc = enc + (":%d" % u.port if u.port else "") if u.username: netloc = "%s%s@%s" % (u.username, ":" + u.password if u.password else "", netloc) u = u._replace(netloc=netloc) if any(ord(ch) > 127 or ch in UNSAFE for ch in u.path + u.query): u = u._replace(path=urllib.parse.quote(u.path, safe="/~:@!$&'()*+,;="), query=urllib.parse.quote(u.query, safe="=&/~:@!$'()*+,;")) return urllib.parse.urlunsplit(u) except Exception: return url # Characters that are never legitimate in evidence and are the tools of choice for hiding or reordering # text an agent reads: C0/C1 controls (CR included, which would start a new CI workflow command), soft # hyphen, Arabic letter mark, Mongolian vowel separator, zero-width characters, bidi embeddings and # isolates, line and paragraph separators, interlinear annotation marks, the two non-characters XML # cannot carry, Unicode tag characters (invisible to people, read by language models), and lone surrogates, # which a JSON escape can produce and no UTF-8 output can carry. _UNSAFE = re.compile("[\x00-\x08\x0b-\x1f\x7f-\x9f\u00ad\u061c\u180e\u200b-\u200f\u2028-\u202e" "\u2060-\u2064\u2066-\u2069\ud800-\udfff\ufeff\ufff9-\ufffb\ufffe\uffff\U000e0000-\U000e007f]") # ── address guard ── # The CLI fetches whatever its user names, private hosts included. Under the MCP server an agent chooses # the URL, and the audited site chooses every URL after it (redirects, sameAs, logo, sitemap), so there # every connection must go to a public address. geo_watch's MCP server switches this on. ADDRESS_GUARD = False def public_ip(addr): """Globally routable unicast only: this refuses loopback, private, link-local, CGNAT (100.64/10, where cloud metadata services such as 100.100.100.200 live), reserved and multicast addresses.""" ip = ipaddress.ip_address(str(addr).split("%")[0]) if ip.version == 6: if ip.ipv4_mapped: ip = ip.ipv4_mapped # NAT64 and 6to4 prefixes carry an IPv4 address inside; on a DNS64 network they reach it elif any(ip in n for n in _EMBEDS_V4): return False return ip.is_global and not ip.is_multicast _EMBEDS_V4 = [ipaddress.ip_network(n) for n in ("64:ff9b::/96", "64:ff9b:1::/48", "2002::/16")] def resolve_public(host, port): """Resolve once; every answer must be public. Returns the addresses to connect to, in order.""" try: infos = socket.getaddrinfo(host, port, type=socket.SOCK_STREAM) except (socket.gaierror, UnicodeError) as e: raise OSError("cannot resolve %s: %s" % (host, e)) from None if not infos or not all(public_ip(i[4][0]) for i in infos): raise OSError("refused: %s does not resolve to a public address" % host) return [i[4][:2] for i in infos] def _guarded(conn_class, proxied=False): """An http.client connection factory whose TCP connect is checked. A direct connection goes to an address that passed the check, so a name cannot resolve to something else a moment later. Through a proxy the proxy resolves the name: the target was checked before the request (check_url, and on every redirect hop), and the proxy itself is the user's own, so its address is not refused.""" def make(host, **kw): until = getattr(_LIMIT, "until", None) if until is not None and isinstance(kw.get("timeout"), (int, float)): # the connect and the TLS handshake wait no longer than the fetch has left left = until - time.monotonic() if left <= 0: raise OutOfTime("out of time") kw["timeout"] = min(kw["timeout"], left) c = conn_class(host, **kw) c.response_class = _TimedResponse orig = c._create_connection def create(address, *a, **k): if not ADDRESS_GUARD or proxied: return orig(address, *a, **k) if getattr(c, "_tunnel_host", None): # HTTPS through a proxy: CONNECT to the target resolve_public(c._tunnel_host, c._tunnel_port or 443) return orig(address, *a, **k) err = None for addr in resolve_public(*address): # every answer was checked; try them in order try: return orig(addr, *a, **k) except OSError as e: err = e raise err c._create_connection = create return c return make class _GuardedHTTP(urllib.request.HTTPHandler): def http_open(self, req): # plain http through a proxy sends the absolute URL to the proxy, with no CONNECT tunnel return self.do_open(_guarded(http.client.HTTPConnection, proxied=req.has_proxy()), req) class _GuardedHTTPS(urllib.request.HTTPSHandler): def https_open(self, req): return self.do_open(_guarded(http.client.HTTPSConnection), req, context=self._context) def check_url(url): """Under the guard, a URL whose host does not resolve to a public address is refused before any request.""" if not ADDRESS_GUARD: return u = urllib.parse.urlsplit(url) if u.scheme not in ("http", "https") or not u.hostname: raise OSError("refused: %s is not an http(s) URL" % url[:80]) resolve_public(u.hostname, u.port or (443 if u.scheme == "https" else 80)) class _Redirects(urllib.request.HTTPRedirectHandler): """Follow 307 and 308 as well as 301/302. urllib below Python 3.11 does not treat 307/308 as redirects — it raises HTTPError instead. A site whose www-to-apex hop is a 308 then reads as unreachable, which dropped openai.com, runwayml.com, neon.tech and others out of the benchmark entirely. Worse, it made the score depend on which Python ran the tool: 3.11+ followed the hop and scored the site, 3.8-3.10 reported it as unfetchable. Real retrieval crawlers follow these hops, so the score has to as well. """ def redirect_request(self, req, fp, code, msg, headers, newurl): try: check_url(newurl) # every hop, not only the first URL except OSError as e: raise urllib.error.URLError(str(e)) from None # The base redirect_request in 3.9 rejects 308 outright, so aliasing # http_error_308 is not enough on its own — this was the incomplete first fix. if code in (307, 308) and req.get_method() in ("GET", "HEAD"): return urllib.request.Request( newurl, headers=req.headers, method=req.get_method(), origin_req_host=req.origin_req_host, unverifiable=True) return urllib.request.HTTPRedirectHandler.redirect_request( self, req, fp, code, msg, headers, newurl) http_error_307 = urllib.request.HTTPRedirectHandler.http_error_301 http_error_308 = urllib.request.HTTPRedirectHandler.http_error_301 _OPENER = urllib.request.build_opener(_Redirects, _GuardedHTTP, _GuardedHTTPS) MAX_BODY = 4_000_000 # bytes read off the wire MAX_INFLATED = 16_000_000 # bytes after gunzip def gunzip(raw, limit=MAX_INFLATED): """Inflate a gzip body, but never past `limit` bytes. The 4 MB read cap applies to the compressed stream. gzip.decompress has no output limit, so a few MB of crafted gzip inflates to gigabytes and takes the process down — on a GitHub Action runner that is an out-of-memory kill, not a score. Truncated output is still parsed; a real page never gets near 16 MB.""" try: d = zlib.decompressobj(16 + zlib.MAX_WBITS) return d.decompress(raw, limit) except zlib.error: return raw # Seconds a request may wait for the server, per read. main() sets it from --timeout; run(timeout=…) overrides it. DEFAULT_TIMEOUT = 15 TIMEOUT = DEFAULT_TIMEOUT MAX_TIMEOUT = 300 # The time a timeout cannot bound. TIMEOUT limits each read, not their sum: a server that sent one byte every few # seconds (a tarpit, which some sites keep for crawlers) held a fetch, and with it a CI job or an MCP worker, for # hours. One request, redirects included, takes at most max(FETCH_BUDGET, 2 × its timeout) seconds; past that it # is an answer of kind "slow", and it is not retried. One run takes at most RUN_DEADLINE seconds: no request starts # after it and one in flight stops there, as an answer of kind "deadline", so the checks that needed it are not # observed. Neither is a flag; the tests set them lower. FETCH_BUDGET = 30 RUN_DEADLINE = 300 READ_CHUNK = 64 * 1024 # Resolve the host before a run, so a mistyped name fails at once. main() switches this on for the CLI. It is a # module setting, not a run() parameter, so run() keeps the signature callers and test stubs rely on; the MCP # server resolves the host itself before it calls run(). PREFLIGHT = False class OutOfTime(socket.timeout): """A fetch whose time is up. fetch() answers it as kind "slow" or "deadline", never as a plain timeout.""" # The fetch this thread is making: when its time is up (a time.monotonic() value) and its wait per read. Set by # fetch() around the request, so the connection it opens and every read of the answer end by then. _LIMIT = threading.local() class _TimedReader(io.RawIOBase): """A response's socket reads, each cut to the time its fetch has left. The status line and the headers are read through it as well as the body, so a server that drips its headers a byte at a time is stopped too.""" def __init__(self, raw, sock, until, per_read): io.RawIOBase.__init__(self) self.raw, self.sock, self.until, self.per_read = raw, sock, until, per_read def readable(self): return True def fileno(self): return self.raw.fileno() def readinto(self, b): left = self.until - time.monotonic() if left <= 0: raise OutOfTime("out of time") self.sock.settimeout(min(self.per_read, left)) return self.raw.readinto(b) def close(self): if not self.closed: self.raw.close() # the socket's own reader: closing it lets the connection close io.RawIOBase.close(self) class _TimedResponse(http.client.HTTPResponse): """http.client's response, reading through _TimedReader while a fetch() has a time limit set.""" def __init__(self, sock, *a, **kw): http.client.HTTPResponse.__init__(self, sock, *a, **kw) until = getattr(_LIMIT, "until", None) if until is not None: self.fp = io.BufferedReader(_TimedReader(self.fp.detach(), sock, until, _LIMIT.timeout)) def read_body(r, limit, until): """Up to `limit` bytes of a response's body, READ_CHUNK at a time: raises OutOfTime once `until` has passed. read1 returns what one read brings, so a body sent a byte at a time is checked against the clock each time.""" read = getattr(r, "read1", None) or r.read parts, got = [], 0 while got < limit: if time.monotonic() >= until: raise OutOfTime("out of time") b = read(min(READ_CHUNK, limit - got)) if not b: break parts.append(b) got += len(b) return b"".join(parts) def header_map(msg): """A response's headers by lower-case name. A field sent on several lines keeps its last line, except Link, whose lines are joined with ", ", as RFC 9110 section 5.3 allows for a list field: a page's hreflang annotations may come on more than one.""" h = {} for k, v in (msg or {}).items(): k = k.lower() h[k] = h[k] + ", " + v if k == "link" and k in h else v return h def fetch(url, ua=UA_BROWSER, timeout=None, method="GET", *, deadline=None): """One request, never retried. It waits `timeout` seconds per read (default TIMEOUT) and max(FETCH_BUDGET, 2 × timeout) in all. `deadline`, a time.monotonic() value, is the run's: it stops the request sooner, and a request asked for after it is not sent.""" timeout = timeout or TIMEOUT t0, now = time.time(), time.monotonic() budget = max(FETCH_BUDGET, 2 * timeout) until, kind, why = now + budget, "slow", "exceeded %g s total" % budget if deadline is not None and deadline <= until: until, kind, why = deadline, "deadline", "run deadline reached" if now >= until: return Resp(str(url), 0, b"", err=why, kind=kind) _LIMIT.until, _LIMIT.timeout = until, timeout try: # inside the try: a URL the site wrote badly (relative, schemeless, empty) is a failed fetch, # never an exception that ends the run url = idna(str(url)) req = urllib.request.Request(url, method=method, headers={ "User-Agent": ua, "Accept": "text/html,application/xhtml+xml,*/*;q=0.8", "Accept-Encoding": "gzip", "Accept-Language": "en,zh;q=0.8"}) check_url(url) with _OPENER.open(req, timeout=timeout) as r: raw = read_body(r, MAX_BODY, until) if r.headers.get("Content-Encoding") == "gzip": raw = gunzip(raw) return Resp(r.geturl(), r.status, raw, header_map(r.headers), elapsed=time.time() - t0) except urllib.error.HTTPError as e: try: raw = read_body(e, 400_000, until) except Exception: raw = b"" h = header_map(e.headers) if h.get("content-encoding") == "gzip": # an error page is read too, for a bot challenge (Cloudflare's comes back 403, gzipped) raw = gunzip(raw) return Resp(url, e.code, raw, h, elapsed=time.time() - t0) except Exception as e: k = fault_kind(e) if k == "timeout" and time.monotonic() >= until: # the last wait was cut to what was left of the time: the answer ran out of it, not one read return Resp(url, 0, b"", err=why, elapsed=time.time() - t0, kind=kind) # a server's status line lands in this message: keep it to one clean line return Resp(url, 0, b"", err=" ".join(_UNSAFE.sub("", str(e)).split())[:120], elapsed=time.time() - t0, kind=k) finally: _LIMIT.until = None # Retries. A timeout, a reset, 408, 429 or a 5xx can change on the next try; any other answer is definite. # The wait grows by RETRY_BACKOFF seconds a try (the tests set it to 0), or follows a Retry-After of up to # RETRY_AFTER_MAX seconds. RETRY_BACKOFF, RETRY_AFTER_MAX = 1.5, 5 STEADY_TRIES = 3 # a crawler probe gets one retry: enough for a dropped connection or a 429 from the burst of probes, and a # server that drops every bot request costs one extra round, not two PROBE_TRIES = 2 def transient(status): return status in (0, 408, 429) or 500 <= status < 600 # faults the next try meets again at once: a name that does not resolve, a closed port, a failed handshake DEFINITE = {"dns", "refused", "not_public", "tls"} # answers that ran out of time: a retry would take as long again, or be refused by the run's deadline OUT_OF_TIME = {"slow", "deadline"} def retry_wait(r, i): if r.status == 0 and r.kind == "timeout": return 0 # a timeout has already waited ra = r.headers.get("retry-after", "").strip() if ra.isdigit(): return min(int(ra), RETRY_AFTER_MAX) return RETRY_BACKOFF * (i + 1) class FetchMemo: """One run's answers from fetch_steady, keyed by (url, user agent, method), so the homepage, robots.txt and the sitemap are fetched once per run. discover() used to read robots.txt with retries and run() read it again without: one dropped second request turned a blocking robots.txt into "No robots.txt". Made per run, never per module: the MCP server scores several sites at once in one process.""" def __init__(self): self.lock, self.got = threading.Lock(), {} def fetch_steady(url, tries=STEADY_TRIES, *, memo=None, **kw): """fetch() for the requests that decide which pages get sampled — the homepage, robots.txt, the sitemap, section indexes, and the sampled pages themselves. One dropped request there used to swap the whole sample: a sitemap that timed out once sent discovery to homepage links instead, a different eight pages were scored, and a site measured 71, 40 and 71 on three consecutive runs from one machine (the middle sample fell under the g.ssr gate). Timeouts, resets, 408, 429 and 5xx are retried with a short backoff; a definite answer, a DNS, refused or TLS fault, or an answer that ran out of time, is returned at once. With a memo, an answer already fetched in this run is returned without a request.""" key = (url, kw.get("ua", UA_BROWSER), kw.get("method", "GET")) if memo is not None: with memo.lock: if key in memo.got: return memo.got[key] for i in range(tries): r = fetch(url, **kw) if not transient(r.status) or r.kind in DEFINITE or r.kind in OUT_OF_TIME: break if i < tries - 1: time.sleep(retry_wait(r, i)) if memo is not None: with memo.lock: r = memo.got.setdefault(key, r) return r def fault_kind(e): """The kind of a fault that left no HTTP answer, from the exception fetch() caught, or from its one-line text when only that is left.""" if isinstance(e, BaseException): why = getattr(e, "reason", None) # URLError wraps the cause x = why if isinstance(why, BaseException) else e if isinstance(x, (socket.timeout, TimeoutError)): return "timeout" if isinstance(x, socket.gaierror): return "dns" if isinstance(x, ConnectionRefusedError): return "refused" if ssl is not None and isinstance(x, (ssl.SSLEOFError, ssl.SSLZeroReturnError)): return "dropped" # the connection was cut in the middle of the handshake: a drop, retried if ssl is not None and isinstance(x, ssl.SSLError): # certificate failures included return "tls" if "UNEXPECTED_EOF" not in str(x) else "dropped" if isinstance(x, (ConnectionError, http.client.IncompleteRead)): return "dropped" e = why if isinstance(why, str) else str(x) e = (e or "").lower() for keys, kind in ((("timed out", "timeout"), "timeout"), (("refused:",), "not_public"), (("cannot resolve", "getaddrinfo", "nodename", "name or service", "name resolution"), "dns"), (("refused",), "refused"), (("unexpected_eof", "eof occurred in violation"), "dropped"), (("ssl", "certificate", "tls"), "tls"), (("reset", "closed connection", "remote end closed", "broken pipe", "eof"), "dropped")): if any(k in e for k in keys): return kind return "no_response" class ScoreError(RuntimeError): """A site that cannot be scored. `kind` says why in schema/error.v1.json's words; the CLI prints the message, and under --json the JSON error as well.""" def __init__(self, message, kind="error", target=None): RuntimeError.__init__(self, message) self.kind, self.target = kind, target # What each kind of failed fetch means, in one plain sentence, and the section of guide/troubleshooting.md that # explains it (test_geo_score.py checks that every anchor is a heading there) FAULTS = { "dns": ("cannot resolve {host}: check the spelling, or your DNS/proxy settings (HTTPS_PROXY)", "dns-the-name-does-not-resolve"), "refused": ("{host} refused the connection: nothing accepts connections on that port, or a firewall turned " "the request away", "refused-the-connection-was-refused"), "not_public": ("the address guard refused {host}: it does not resolve to a public address", "could-not-fetch--and-exit-code-2"), "tls": ("the TLS handshake with {host} failed: its certificate is invalid, expired or for another name, or " "something between here and the site interferes with HTTPS", "tls-the-https-handshake-failed"), "timeout": ("{host} did not answer within {timeout:g} s on any try (--timeout sets the wait)", "timeout-no-answer-in-time"), "slow": ("{host} kept its answer coming too slowly to finish within {budget:g} s, the most one request may take", "slow-the-answer-never-finished"), "deadline": ("the run reached its {deadline:g} s limit before a page to score came back", "deadline-the-run-ran-out-of-time"), "dropped": ("{host} closed the connection without answering, on every try", "dropped-no_response-no-answer-came-back"), "no_response": ("{host} gave no usable HTTP answer", "dropped-no_response-no-answer-came-back"), "http": ("every page asked for answered with an HTTP error ({codes})", "http_nnn-the-pages-answered-with-an-http-error"), "no_content": ("{host} answered with an empty page", "no_content-the-page-came-back-empty"), } def troubleshooting(anchor): """A section of the troubleshooting guide, pinned to this release's tag.""" return "https://github.com/jianruntech/geo-score/blob/v%s/guide/troubleshooting.md#%s" % (__version__, anchor) def proxy_for(url): """The proxy a request to `url` goes through, as host:port (never its credentials), or None.""" u = urllib.parse.urlsplit(url) p = urllib.request.getproxies().get(u.scheme) if not p or urllib.request.proxy_bypass(u.hostname or ""): return None pu = urllib.parse.urlsplit(p if "://" in p else "http://" + p) return pu.hostname + (":%d" % pu.port if pu.port else "") if pu.hostname else None def fetch_failure(base, rs, timeout=None): """The error for a site that gave no page to score: the first answer's kind, in one plain sentence that names it, with the troubleshooting section for it.""" # pages the run's deadline stopped are why there is nothing to score, whatever the first page answered kind = "deadline" if any(r.kind == "deadline" for r in rs) else rs[0].kind or "no_content" codes = sorted({r.status for r in rs if r.status and not r.ok}) host = urllib.parse.urlsplit(base).hostname or base text, anchor = FAULTS["http" if kind.startswith("http_") else kind] t = timeout or TIMEOUT text = text.format(host=host, timeout=t, codes=", ".join("HTTP %d" % c for c in codes), budget=max(FETCH_BUDGET, 2 * t), deadline=RUN_DEADLINE) if kind == "tls" and ssl is not None and not getattr(ssl, "HAS_TLSv1_3", True): # macOS's /usr/bin/python3 is built on LibreSSL 2.8, which cannot speak TLS 1.3; many sites require it text += ("; this Python's TLS library (%s) cannot speak TLS 1.3, which many sites require, so run the CLI " "with a newer Python" % ssl.OPENSSL_VERSION) via = proxy_for(base) if rs[0].status == 0 and kind not in ("not_public", "deadline") else None return ScoreError("could not fetch %s (%s): %s%s; see %s" % ( base, kind, text, "; the request went through the proxy at %s, which may be where it failed" % via if via else "", troubleshooting(anchor)), kind, base) # Bot-challenge pages: what a bot-protection service serves, in place of the page, to a client it has not cleared. # A static fetch cannot pass one, so a challenge is no page to score: read as content, it scored g.ssr 0 and every # page check at the bottom, and published a site's bot protection as "content needs JavaScript". Each vendor's # markers are all ones its challenge page carries and its ordinary pages do not: an ordinary page can carry the # vendor's sensor script (Cloudflare's /cdn-cgi/challenge-platform/…/jsd/, PerimeterX's _pxAppId, Imperva's # _Incapsula_Resource script), and that is no challenge. A page with substantive body text is never one. BOT_CHALLENGE = "bot challenge served to a browser user-agent" # Akamai's reference number: "Reference #18.5a3c2e17.1790692071.9f00b2c", or the bare number on a newer page _AKAMAI_REF = r"Reference\s*#|errors\.edgesuite\.net|\b\d{1,3}\.[0-9a-f]{6,8}\.\d{10}\.[0-9a-f]{6,10}\b" CHALLENGES = [ # (vendor, the statuses its challenge answers with or None for any, patterns that must all match) ("AWS WAF", (202, 405), (re.compile(r"awswaf|reportChallengeError", re.I),)), ("Akamai", None, (re.compile(r"Access Denied"), re.compile(_AKAMAI_REF))), ("Cloudflare", None, (re.compile(r"cf[-_]chl|/cdn-cgi/challenge-platform/[^\s\"'<>]*orchestrate" r"|\s*Just a moment", re.I),)), ("Fastly", None, (re.compile(r"/_fs-ch-|<title>\s*Client Challenge", re.I),)), ("PerimeterX", None, (re.compile(r"px-captcha|_pxJsClientSrc"),)), ("DataDome", None, (re.compile(r"captcha-delivery\.com", re.I),)), ("Imperva", None, (re.compile(r"_Incapsula_Resource"), re.compile(r"incident_id|Incapsula incident", re.I))), ] # the most of a page the markers are looked for in: every challenge page is small, and its markers sit near the top CHALLENGE_SCAN = 64 * 1024 def challenge_kind(resp): """The vendor whose bot-challenge page this answer is ("AWS WAF", "Cloudflare", …), or None for a page, an error page or an empty answer. Any status: a 403 challenge is a bot challenge, not merely an HTTP error.""" if not resp.body: return None # Cloudflare says so in a header when it answered with a challenge vendor = "Cloudflare" if resp.headers.get("cf-mitigated", "").strip().lower() == "challenge" else None head = None for name, statuses, marks in CHALLENGES: if vendor: break if statuses and resp.status not in statuses: continue if head is None: # Akamai writes its reference number as HTML entities ("Reference #18.…") head = html.unescape(resp.body[:CHALLENGE_SCAN].decode("utf-8", "replace")) if all(m.search(head) for m in marks): vendor = name # the same measure as g.ssr: a page a reader could use is scored, whatever scripts it also loads return vendor if vendor and wc(visible_text(main_html(resp.text))) < 120 else None # getaddrinfo's answers for a name that does not exist (EAI_NODATA is missing on some platforms) NO_SUCH_NAME = {getattr(socket, n) for n in ("EAI_NONAME", "EAI_NODATA") if hasattr(socket, n)} def preflight_host(url): """Resolve the host once before anything is fetched, so a mistyped name fails at once, in plain words. Skipped when a proxy will carry the request: behind a proxy local DNS can fail while the proxy resolves the name. A temporary resolver failure (EAI_AGAIN) lets the run go ahead.""" u = urllib.parse.urlsplit(idna(url)) if not u.hostname or proxy_for(url): return try: socket.getaddrinfo(u.hostname, u.port or (443 if u.scheme == "https" else 80), type=socket.SOCK_STREAM) except socket.gaierror as e: if e.errno in NO_SUCH_NAME: raise ScoreError("cannot resolve %s: check the spelling, or your DNS/proxy settings (HTTPS_PROXY); see %s" % (u.hostname, troubleshooting(FAULTS["dns"][1])), "dns", url) from None except (UnicodeError, ValueError): pass def fault_name(err): """A request that got no answer, in the tool's words: the raw error can carry a server's status line.""" e = (err or "").lower() for keys, name in ((("timed out", "timeout"), "timed out"), (("refused",), "connection refused"), (("reset", "closed connection", "remote end closed", "broken pipe"), "connection dropped"), (("ssl", "certificate", "tls"), "TLS failure"), (("resolve", "getaddrinfo", "nodename", "name or service"), "DNS failure")): if any(k in e for k in keys): return name return "no response" def join_link(base, ref): """urljoin for a link the site wrote, which never raises: a malformed one ("http://[::1") is no link, where it used to end the run.""" try: return urllib.parse.urljoin(base, ref) except ValueError: return "" def pmap(fn, items, workers=8): with cf.ThreadPoolExecutor(max_workers=workers) as ex: return list(ex.map(fn, items)) # ── html helpers ─────────────────────────────────────────────────────────── # Every scanner here reads a page in one pass. The regexes they replace, such as # <(script|…)\b.*?</\1> and <p\b[^>]*>(.*?)</p>, retried each unclosed opening to the end of the page, so # 80 KB of "<script>x " took 1.8 s and the 4 MB a fetch may return would have taken hours: a hostile page # could hold a CI job or an MCP worker until it was killed. An opening is found with a compiled search from # the current position, and its close once, with str.find on a lower-cased copy. _ASCII_LOWER = {c: c + 32 for c in range(ord("A"), ord("Z") + 1)} def _lower(h): """h lower-cased, index for index. str.lower() lengthens a few characters (U+0130 becomes two), which would shift every close found in the copy; then only ASCII letters are lowered, which is all a tag name needs.""" low = h.lower() return low if len(low) == len(h) else h.translate(_ASCII_LOWER) def inner_html(h, opening, close, low=None): """The content of each element that `opening` finds (a compiled pattern for "<name\b"), from the end of its tag to the first `close` after it (a lower-case string such as "</p>", or a compiled pattern run on the lower-cased page), in document order: what re.finditer(r"<name\b[^>]*>(.*?)</name>", h, re.S | re.I) yields. An opening with no ">" or no close after it ends the scan, because no later one can have one.""" i = 0 while True: m = opening.search(h, i) if not m: return gt = h.find(">", m.end()) if gt < 0: return if low is None: low = _lower(h) if isinstance(close, str): end = low.find(close, gt + 1) after = end + len(close) else: c = close.search(low, gt + 1) end, after = (c.start(), c.end()) if c else (-1, -1) if end < 0: return yield h[gt + 1:end] i = after def in_tags(pattern, start, h): """pattern.finditer(h), for a pattern that looks inside one tag: it begins with what `start` finds (a compiled pattern) and reaches the rest of its match through [^>]* or [^>]+. The regex alone retries every start inside a tag that never closes, each time to the end of the page; here a start that does not match skips the rest of its tag, where no start can match either.""" i = 0 while True: a = start.search(h, i) if not a: return m = pattern.match(h, a.start()) if m: yield m i = max(m.end(), a.start() + 1) continue gt = h.find(">", a.end()) if gt < 0: return i = gt + 1 _HIDDEN = re.compile(r"<(script|style|noscript|template|svg)\b", re.I) _TAG = re.compile(r"<[^>]+>") def strip_hidden(h): """h with each script, style, noscript, template and svg element replaced by a space. A script, style, noscript or template that is never closed hides the rest of the page, as it does in a browser. An <svg> that is never closed hides nothing, as before: a browser shows the text after it.""" out, i, low, open_svg = [], 0, None, False while True: m = _HIDDEN.search(h, i) if not m: out.append(h[i:]) return "".join(out) if low is None: low = _lower(h) name = m.group(1).lower() close = "</%s>" % name # once one <svg> has no close, no later one has either: skip the search, which would scan to the end end = -1 if name == "svg" and open_svg else low.find(close, m.end()) if end < 0 and name == "svg": open_svg = True out.append(h[i:m.end()]) i = m.end() continue out.append(h[i:m.start()]) if end < 0: return "".join(out) out.append(" ") i = end + len(close) def visible_text(h): h = strip_hidden(h) # nothing after the last ">" can be a tag, and <[^>]+> would scan to the end of the page from every "<" there k = h.rfind(">") + 1 h = _TAG.sub(" ", h[:k]) + h[k:] return re.sub(r"\s+", " ", html.unescape(h)).strip() _MAIN, _ARTICLE = re.compile(r"<main\b", re.I), re.compile(r"<article\b", re.I) def main_html(h): """The first <main> element's content, else the first <article>'s, else the whole page.""" low = _lower(h) for opening, close in ((_MAIN, "</main>"), (_ARTICLE, "</article>")): body = next(inner_html(h, opening, close, low), None) if body is not None: return body return h _SURROGATE_ESC = re.compile(r"\\u[dD][89a-fA-F]") _SCRIPT = re.compile(r"<script", re.I) _LD_TYPE = re.compile(r'type=["\']application/ld\+json["\']', re.I) def ld_blocks(h): r"""The text of each JSON-LD block, as re.finditer(r'<script[^>]+type=["\']application/ld\+json["\'][^>]*> (.*?)</script>', h, re.S | re.I) found them, in one pass. A <script> tag that is not JSON-LD is skipped to its ">", and an unclosed block ends the scan.""" i, low = 0, None while True: m = _SCRIPT.search(h, i) if not m: return gt = h.find(">", m.end()) if gt < 0: return # the type attribute sits inside the tag, at least one character after "<script" if not _LD_TYPE.search(h, m.end() + 1, gt): i = gt + 1 continue if low is None: low = _lower(h) end = low.find("</script>", gt + 1) if end < 0: return yield h[gt + 1:end] i = end + len("</script>") def jsonld(h): out = [] for block in ld_blocks(h): raw = block.strip() try: d = json.loads(raw) except Exception: try: d = json.loads(re.sub(r",\s*([}\]])", r"\1", raw)) except Exception: continue if _SURROGATE_ESC.search(raw): # a lone \ud800 escape decodes to a character no UTF-8 output can carry, and the first print of a # name holding one ended the run: it becomes "?" here, while a pair (an emoji) comes through whole try: d = json.loads(json.dumps(d, ensure_ascii=False).encode("utf-8", "replace")) except Exception: continue out.extend(d if isinstance(d, list) else [d]) # flattened in document order with a stack, not by recursion: a page can nest lists or @graph a few # thousand deep, which json.loads accepts on newer Pythons and a recursive walk did not survive flat, stack = [], out[::-1] while stack: o = stack.pop() if isinstance(o, dict): if isinstance(o.get("@graph"), list): stack.extend(o["@graph"][::-1]) else: flat.append(o) elif isinstance(o, list): stack.extend(o[::-1]) return resolve_refs(flat) def resolve_refs(objs): """Follow {"@id": "..."} references inside a @graph. Sites that use @graph name an entity once and point at it everywhere else; a parser that does not follow the pointer reports a page with a named author as having none.""" by_id = {o["@id"]: o for o in objs if isinstance(o.get("@id"), str) and len(o) > 1} if not by_id: return objs def deref(v, depth=0): if depth > 3: return v if isinstance(v, dict): # an @id that is an object or a list names nothing, and cannot be looked up if set(v) == {"@id"} and isinstance(v["@id"], str) and v["@id"] in by_id: return deref({k: x for k, x in by_id[v["@id"]].items() if k != "@id"}, depth + 1) return {k: deref(x, depth + 1) for k, x in v.items()} if isinstance(v, list): return [deref(x, depth + 1) for x in v] return v return [deref(o) for o in objs] def types_of(objs): """Every type the objects declare. Only a string names a type: an @type that is an object, a number or null, alone or in a list, names none (one written as {"@id": "Organization"} used to end the run).""" t = set() for o in objs: v = o.get("@type") t.update(x for x in (v if isinstance(v, list) else [v]) if isinstance(x, str) and x) return t # The fields that make a page-type object carry information rather than a bare type: any one of them, # non-empty, is a "real field value" in the rubric's sense for p1.page-type. PAGE_TYPE_FIELDS = { "Product": ("name",), "Offer": ("price", "priceSpecification"), "FAQPage": ("mainEntity",), "HowTo": ("step",), "SoftwareApplication": ("name",), "Course": ("name",), "Recipe": ("name",), "Event": ("startDate",), "JobPosting": ("title",), "Dataset": ("name",), "Article": ("headline", "name"), "BlogPosting": ("headline", "name"), "NewsArticle": ("headline", "name"), "TechArticle": ("headline", "name")} def substantive(obj, types=PAGE_TYPE_FIELDS): """True when a JSON-LD object of one of `types` carries a non-empty value in one of its fields.""" t = obj.get("@type") for name in (t if isinstance(t, list) else [t]): if isinstance(name, str) and name in types and any(obj.get(f) not in (None, "", [], {}) for f in types[name]): return True return False _MONTHS = {m: i for i, m in enumerate(("jan", "feb", "mar", "apr", "may", "jun", "jul", "aug", "sep", "oct", "nov", "dec"), 1)} # numeric dates must not start inside a longer number; \b fails when a date sits flush against Chinese # text ("更新于2026年8月28日"), because Python counts CJK characters as word characters _NUM_DATE = re.compile(r"(?<!\d)(20[12]\d)\s*[-/.年]\s*(\d{1,2})\s*[-/.月]\s*(\d{1,2})(?!\d)") _DMY = re.compile(r"(?<!\d)(\d{1,2})\s+(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?,?\s+(20[12]\d)(?!\d)", re.I) _MDY = re.compile(r"\b(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?\s+(\d{1,2}),?\s+(20[12]\d)(?!\d)", re.I) def _date(y, m, d): try: return datetime.date(int(y), int(m), int(d)) except (ValueError, TypeError): return None def iso_date(v): """The calendar date at the start of an ISO 8601 string (dateModified, <time datetime>), or None.""" m = re.match(r"\s*(\d{4})-(\d{2})-(\d{2})", str(v or "")) return _date(*m.groups()) if m else None # tag-attribute patterns, read with in_tags(pattern, start, page) _TIME, _TIME_DATETIME = re.compile(r"<time", re.I), re.compile(r'<time[^>]+datetime=["\']([^"\']+)', re.I) _ARTICLE_TIME_PROP = re.compile(r'property=["\']article:(published|modified)_time["\']', re.I) _ARTICLE_TIME = re.compile(r'property=["\']article:(published|modified)_time["\'][^>]*content=["\']([^"\']+)', re.I) _A = re.compile(r"<a\b", re.I) _A_HREF = re.compile(r'<a\b[^>]*href=["\']([^"\'#]+)', re.I) _REL_AUTHOR = re.compile(r'rel=["\']author["\']') _REL_AUTHOR_NAME = re.compile(r'rel=["\']author["\'][^>]*>\s*([A-Z][\w.\-]+(?:\s+[A-Z][\w.\-]+){0,2})') _OG_SITE_PROP = re.compile(r'property=["\']og:site_name["\']', re.I) _OG_SITE_NAME = re.compile(r'property=["\']og:site_name["\'][^>]*content=["\']([^"\']+)', re.I) _CONTENT_VALUE = re.compile(r'content=["\']([^"\']+)["\']', re.I) _OG_SITE_NAME_TAIL = re.compile(r'property=["\']og:site_name', re.I) _LINK = re.compile(r"<link", re.I) _LINK_REL = re.compile(r'rel=["\'](?:llms|ai-content|alternate)["\']', re.I) _LINK_TYPE = re.compile(r'type=["\']text/(?:plain|markdown)', re.I) def geo_link(h): """re.search(r'<link[^>]+rel=["\'](?:llms|ai-content|alternate)["\'][^>]*type=["\']text/(?:plain|markdown)', h, re.I), tag by tag: the two [^>]* runs made the regex quadratic within one long tag, as well as across tags.""" i = 0 while True: a = _LINK.search(h, i) if not a: return False gt = h.find(">", a.end()) end = gt if gt >= 0 else len(h) rel = _LINK_REL.search(h, a.end() + 1, end) if rel and _LINK_TYPE.search(h, rel.end(), end): return True if gt < 0: return False i = gt + 1 def og_site_name(h): """What re.search(r'property=["\']og:site_name["\'][^>]*content=["\']([^"\']+)', h, re.I) captures, or else re.search(r'content=["\']([^"\']+)["\'][^>]*property=["\']og:site_name', h, re.I), in one pass each. In the second a content value may itself hold ">", so a tag cannot be skipped whole: a value matches when the first og:site_name after it comes before the first ">" after it.""" m = next(in_tags(_OG_SITE_NAME, _OG_SITE_PROP, h), None) if m: return m.group(1) i, gt, prop = 0, -1, -1 while True: a = _CONTENT_VALUE.search(h, i) if not a: return None e = a.end() # each value found ends after the one before it, so both positions only move forward if gt < e: gt = h.find(">", e) if gt < 0: gt = len(h) if prop < e: p = _OG_SITE_NAME_TAIL.search(h, e) if not p: return None prop = p.start() if prop < gt: return a.group(1) i = a.start() + 1 def has_md_link(t): r"""re.search(r"\[[^\]]+\]\([^)]+\)", t) in one pass: a "[" whose "]" is not followed by "(…)" skips to that "]", since every "[" before it meets the same "]".""" i = 0 while True: a = t.find("[", i) if a < 0: return False b = t.find("]", a + 1) if b < 0: return False if b > a + 1 and t.startswith("(", b + 1): c = t.find(")", b + 2) if c < 0: return False if c > b + 2: return True i = b + 1 # A Sitemap: line in robots.txt. [^\S\n]* rather than \s*: \s* crossed newlines, so the regex was tried at every # line start of a run of blank lines and scanned the whole run each time (100 KB of newlines took 45 s). The # matches are the same, because a match can only begin on the line its "sitemap:" is on or in the blank run # before it, and it captures the same URL either way. _SITEMAP_DECL = re.compile(r"(?im)^[^\S\n]*sitemap:\s*(\S+)") def visible_dates(page_html): """Dates a reader or a crawler can see on the page: <time datetime> and dates written in the text.""" out = [iso_date(m.group(1)) for m in in_tags(_TIME_DATETIME, _TIME, page_html)] text = visible_text(main_html(page_html))[:3000] out += [_date(*m.groups()) for m in _NUM_DATE.finditer(text)] out += [_date(m.group(3), _MONTHS[m.group(2).lower()[:3]], m.group(1)) for m in _DMY.finditer(text)] out += [_date(m.group(3), _MONTHS[m.group(1).lower()[:3]], m.group(2)) for m in _MDY.finditer(text)] return [d for d in out if d] def own_dates(page_html, objs): """The page's own dates, the ones a dateModified has to agree with: datePublished in its JSON-LD, article:published_time / modified_time, and the first date in the main content (a byline or a header). Not every date on the page: related-story rails, listings and comments carry dates of other content.""" out = [iso_date(x.get("datePublished")) for x in objs] out += [iso_date(m.group(2)) for m in in_tags(_ARTICLE_TIME, _ARTICLE_TIME_PROP, page_html)] first = first_date(main_html(page_html)[:20000]) return [d for d in out + [first] if d] def first_date(html_part): """The first date in document order: a <time> element stands in for its datetime, then the earliest written date wins.""" text = visible_text(re.sub(r'<time[^>]+datetime=["\']([^"\']+)["\'][^>]*>.*?</time>', lambda m: " %s " % (m.group(1)[:10]), html_part, flags=re.S | re.I)) found = [(m.start(), _date(*m.groups())) for m in _NUM_DATE.finditer(text)] found += [(m.start(), _date(m.group(3), _MONTHS[m.group(2).lower()[:3]], m.group(1))) for m in _DMY.finditer(text)] found += [(m.start(), _date(m.group(3), _MONTHS[m.group(1).lower()[:3]], m.group(2))) for m in _MDY.finditer(text)] found = [f for f in found if f[1]] return min(found)[1] if found else None # a word before "by" that makes it a credit, a sort order or a filter, not a byline _CREDIT = re.compile(r"(?:powered|built|made|hosted|designed|developed|sponsored|presented|backed|trusted|used|loved|" r"created|brought to you|followed|group|grouped|sort|sorted|filter|filtered|curated|operated|" r"owned|published|translated|funded|supported|inspired)\s*$", re.I) _BYLINE = re.compile(r"(?<![A-Za-z])[Bb][Yy]\s+((?:(?:Dr|Prof|Mr|Ms|Mrs)\.\s+)?(?:[A-Z]\.|[A-Z][A-Za-z'\u2019\-]+)" r"(?:\s+(?:[A-Z]\.|[A-Z][A-Za-z'\u2019\-]+|van|von|de|da|del|der|den|bin|al|le|la|di|du)){1,3})") def byline_name(text): """The first "By Jane Doe" in `text` that is a byline: two to four name words, not a credit such as "Powered by Hugo" or "Sorted by Date", cut before any trailing punctuation.""" for m in _BYLINE.finditer(text): if _CREDIT.search(text[max(0, m.start() - 40):m.start()]): continue name = m.group(1).rstrip(".,;:!?)'\u2019 ") if len(name.split()) >= 2: return name return None _HEADING, _HEADING_CLOSE = re.compile(r"<h[1-4]\b", re.I), re.compile(r"</h[1-4]>") _TITLE, _PARA = re.compile(r"<title\b", re.I), re.compile(r"<p\b", re.I) def title_html(h, low=None): """The first <title> element's content, or None.""" return next(inner_html(h, _TITLE, "", low), None) def headings(h): low = _lower(h) # an

–

runs to the first heading close of any level, as the pattern it replaces did out = [visible_text(x)[:200] for x in inner_html(h, _HEADING, _HEADING_CLOSE, low)] t = title_html(h, low) if t is not None: out.append(visible_text(t)[:200]) return [x for x in out if x] def paragraphs(h): return [p for p in (visible_text(x) for x in inner_html(main_html(h), _PARA, "

")) if p] # Chinese, Japanese and Korean characters: Han, kana (hiragana, katakana, half-width katakana) and hangul _KANA = "\u3040-\u30ff\u31f0-\u31ff\uff66-\uff9f" _HANGUL = "\uac00-\ud7af\u1100-\u11ff\u3130-\u318f" _CJK = re.compile("[\u4e00-\u9fff%s%s]" % (_KANA, _HANGUL)) def cjk_chars(s): return len(_CJK.findall(s)) def wc(s): """Words, or characters for text past 20 Chinese, Japanese or Korean characters that outnumber its words: an English page's language menu (日本語, 한국어, 简体中文 …) does not make its 1,600 words read as 24 characters.""" cjk, words = cjk_chars(s), len(s.split()) return cjk if cjk > 20 and cjk >= words else words def host_stem(root): parts = urllib.parse.urlsplit(root).netloc.replace("www.", "").split(".") return parts[0].lower() def is_cjk(s): """Is this page written in Chinese, Japanese or Korean? Judged on what a reader sees, not on raw markup — a Chinese page whose HTML is mostly CSS/JS would otherwise fall under the 8% threshold.""" if "<" in s and ">" in s: s = visible_text(s) return script_lang(s) is not None # ── page language ────────────────────────────────────────────────────────── # Each page is read in its own language: an English page on a site whose homepage came back in Chinese is # still read in words, and a Japanese or German page is not read with the English or Chinese lexicon. CJK_LANGS = ("zh", "ja", "ko") # code samples: in no natural language, so never read to tell a page's _CODE_TAGS = ("pre", "code") # the Chinese, Japanese or Korean characters a page declared in one of those languages keeps its declaration with CJK_FLOOR = 30 # languages this file has heading, source-phrase and byline lexicons for; "und" is a page whose language could # not be told, which is read with them as every page was before LEXICON_LANGS = ("en", "zh", "und") _HTML_OPEN = re.compile(r", else of its Content-Language header, lower-cased; or None.""" h = resp.text m = _HTML_OPEN.search(h) gt = h.find(">", m.end()) if m else -1 a = _LANG_ATTR.search(h, m.end(), gt) if gt >= 0 else None if a: return a.group(1).lower() c = _LANG_TAG.match(resp.headers.get("content-language", "")) return c.group(1).lower() if c else None def script_lang(text): """ja, ko or zh when the text is written in Chinese, Japanese or Korean (kana is Japanese, hangul Korean, Han alone Chinese); None otherwise.""" han = len(re.findall("[\u4e00-\u9fff]", text)) kana = len(re.findall("[%s]" % _KANA, text)) hangul = len(re.findall("[%s]" % _HANGUL, text)) total = han + kana + hangul if total <= max(30, len(text) * 0.08): return None if hangul * 2 >= total: return "ko" return "ja" if kana >= max(10, total * 0.05) else "zh" def latin_lang(text): """en, de, fr, es or pt, told by common words, when one language clearly leads; else "und".""" counts = dict.fromkeys(_STOPWORDS, 0) for w in _LATIN_WORD.findall(text[:20000].lower()): for k, words in _STOPWORDS.items(): if w in words: counts[k] += 1 ranked = sorted(counts.items(), key=lambda kv: -kv[1]) (best, n), (_, second) = ranked[0], ranked[1] return best if n >= 5 and n >= 2 * second else "und" def page_lang(resp): """The language a page is written in, as a lower-case primary tag (zh, ja, ko, en, de, fr, es, pt, or any other tag the page declares), or "und". , else the Content-Language header, says which, unless the text contradicts it: a page declared in a Latin-script language whose visible text is Chinese, Japanese or Korean is read in that language, and a page declared Chinese, Japanese or Korean with next to none of that text (under CJK_FLOOR characters) is told by its words. With no declaration the script decides, and Latin-script text is told apart by a few common words. Code (
, ) is in no natural language and
    is not read: a docs page's samples, with their English comments, made a Chinese page read as English."""
    text = visible_text(strip_chrome(strip_comments(strip_hidden(resp.text)), _CODE_TAGS, ()))
    script, declared = script_lang(text), declared_lang(resp)
    if declared and (declared in CJK_LANGS) == bool(script):
        return declared
    if declared in CJK_LANGS and cjk_chars(text) >= CJK_FLOOR:
        # a declared Chinese, Japanese or Korean page with that much of its text is one, whatever else it carries
        return declared
    if script:
        return script
    latin = latin_lang(text)
    return latin if latin != "und" else (declared or "und")


def audience_language(langs):
    """The report's audience_language: the most common page language, ties to the page sampled first, as a
    BCP 47 primary tag. Chinese stays "zh-CN", as it was when the field could only be that or "en". "und" when
    no page's language could be told."""
    known = [x for x in langs if x != "und"]
    if not known:
        return "und"
    best = max(dict.fromkeys(known), key=known.count)
    return "zh-CN" if best == "zh" else best


# Wikipedia editions, besides English, that a knowledge-graph lookup may go to: the larger ones
WIKI_LANGS = frozenset("ar bg ca cs da de el es et fa fi fr he hi hr hu id it ja ko lt lv ms nl no pl pt ro ru sk sl "
                       "sr sv th tr uk vi zh".split())


def lexicon_gap(langs):
    """The page languages, in order, that have no lexicon here."""
    return list(dict.fromkeys(x for x in langs if x not in LEXICON_LANGS))

# ── figures and their sources (p2.sourced-stats) ───────────────────────────
# The rubric's attributed figure is one whose source sits beside it. A link, a  or a source word anywhere
# on the page is not that: read page-wide, social icons, partner logos and a "Research" menu item in the chrome
# attributed every figure on the page. Only p2.sourced-stats reads a page this way, so no other check moves.
_CHROME_TAGS = ("header", "nav", "footer", "aside")
_CHROME_ROLES = ("banner", "navigation", "contentinfo", "complementary")
# a role is honoured on these containers only, so at most these names are ever paired open to close
_ROLE_HOSTS = {"div", "section", "ul", "ol", "menu", "span", "p", "form", "table"} | set(_CHROME_TAGS)
_OPEN_TAG = re.compile(r"<([A-Za-z][A-Za-z0-9-]*)")
_ROLE = re.compile(r"""\brole\s*=\s*["']?\s*(%s)\b""" % "|".join(_CHROME_ROLES), re.I)


def strip_comments(h):
    """h with each HTML comment replaced by a space. One that is never closed hides the rest of the page, as it
    does in a browser."""
    out, i = [], 0
    while True:
        a = h.find("", a + 4)
        if b < 0:
            return "".join(out)
        out.append(" ")
        i = b + 3


def _closes(low, name):
    """{where each  opening starts: where the name of its matching  ends}, in one pass over the
    lower-cased page with a stack, so that nested elements of one name pair up. An opening never closed is
    left out."""
    out, stack = {}, []
    for m in re.finditer(r"<(/?)%s\b" % re.escape(name), low):
        if not m.group(1):
            stack.append(m.start())
        elif stack:
            out[stack.pop()] = m.end()
    return out


def strip_chrome(h, tags=_CHROME_TAGS, roles=_CHROME_ROLES):
    """h without each element named in `tags`, or whose role is one of `roles`, and everything inside it. An
    element that is never closed stays, content and all: where a browser would end it is a guess."""
    low, cut, pairs, i = None, [], {}, 0
    while True:
        m = _OPEN_TAG.search(h, i)
        if not m:
            break
        gt = h.find(">", m.end())
        if gt < 0:
            break
        name = m.group(1).lower()
        chrome = name in tags
        if not chrome and name in _ROLE_HOSTS:
            r = _ROLE.search(h, m.end(), gt)
            chrome = bool(r) and r.group(1).lower() in roles
        if chrome:
            if low is None:
                low = _lower(h)
            if name not in pairs:
                pairs[name] = _closes(low, name)
            end = pairs[name].get(m.start())
            if end is not None:
                close = h.find(">", end)
                i = close + 1 if close >= 0 else len(h)
                cut.append((m.start(), i))
                continue
        i = gt + 1
    if not cut:
        return h
    out, at = [], 0
    for a, b in cut:
        out += [h[at:a], " "]
        at = b
    out.append(h[at:])
    return "".join(out)


def _content(clean):
    body = main_html(clean)
    if body is clean:
        return strip_chrome(clean)
    return strip_chrome(body, ("nav", "aside"), ("navigation", "complementary"))


def content_html(h):
    """What p2.sourced-stats reads, and no other check: the first 
, else the first
, without any nav or aside inside it; on a page with neither, the whole page without its header, nav, footer and aside and the elements whose role is banner, navigation, contentinfo or complementary. No script, style, noscript, template, svg or comment either way.""" return _content(strip_comments(strip_hidden(h))) # The edges of a block: the elements read as blocks (p, li, tr, td, dd, figcaption, blockquote, caption), and the # headings and containers around them, so that text outside all of those does not run on into one long block. _BLOCK_EDGE = re.compile(r"", m.end()) if m else -1 piece = h[start:m.start() if gt >= 0 else len(h)] t = visible_text(piece) if t: out.append((piece, t)) if gt < 0: return out start = i = gt + 1 # a figure: a percentage, a sum of money, or a number of three or more characters, less the prices and identifiers # figures() leaves out. Latin letters, digits, a point and a "#" before it bound it (so "v1.2.3" holds no "2.3", and # "#1928093" is a reference, not a figure); \w would also count Chinese characters, so "占用了61%的" would not be a # figure at all _FIGURE = re.compile(r"(? as often as in a