#!/usr/bin/env python3
# SPDX-License-Identifier: MIT OR Apache-2.0
"""Audit a page's search and sharing metadata, the way a crawler and a link preview read it. Standard library only.
Checks, each PASS, WARN or FAIL:
- the basics: charset, viewport, lang, a title and a description of a length search results show whole
- canonical: present, absolute, and (for a URL) answered 200 without a redirect, since a canonical that redirects
is ignored
- Open Graph: title, description, type, url (equal to the canonical), site_name, and every og:image with its width,
height, type and alt
- X/Twitter: card, title, description, image, image:alt (a large card needs a wide image)
- icons: rel=icon and apple-touch-icon present and reachable
- JSON-LD: every block parses, carries @context, and names the page's url somewhere
- each card image fetched: its real size read from the file (PNG, JPEG, GIF, WebP header) must equal the declared
og:image:width/height; at least 200×200 (Facebook's floor), wide cards near 1.91:1, under 5 MB (X's limit)
usage: python3 tools/seo.py https://deltaverse.pythai.net/bankml
python3 tools/seo.py page.html --base https://deltaverse.pythai.net/bankml (a local file; fetches its images)
python3 tools/seo.py page.html --offline (no network: tags only)
exit 0 when nothing FAILs. Pairs with tools/makecards.py, which draws the cards."""
import argparse, html.parser, json, struct, sys, urllib.error, urllib.parse, urllib.request
UA = "Mozilla/5.0 (compatible; bankml-seo/1.0; +https://github.com/cryptoAGI/bankml)"
class Head(html.parser.HTMLParser):
"""Collects ,
, every and , and the JSON-LD blocks."""
def __init__(self):
super().__init__(convert_charrefs=True)
self.lang, self.title, self.metas, self.links, self.ld = None, None, [], [], []
self._in_title = self._in_ld = False
self._buf = []
def handle_starttag(self, tag, attrs):
a = {k.lower(): (v or "") for k, v in attrs}
if tag == "html":
self.lang = a.get("lang")
elif tag == "title":
self._in_title, self._buf = True, []
elif tag == "meta":
self.metas.append(a)
elif tag == "link":
self.links.append(a)
elif tag == "script" and a.get("type", "").lower() == "application/ld+json":
self._in_ld, self._buf = True, []
def handle_endtag(self, tag):
if tag == "title" and self._in_title:
self.title, self._in_title = "".join(self._buf).strip(), False
elif tag == "script" and self._in_ld:
self.ld.append("".join(self._buf))
self._in_ld = False
def handle_data(self, data):
if self._in_title or self._in_ld:
self._buf.append(data)
def meta(self, key):
"""Every content for a name= or property= key, in page order."""
return [m.get("content", "") for m in self.metas if (m.get("property") or m.get("name") or "").lower() == key]
def first(self, key):
v = self.meta(key)
return v[0] if v else None
def link(self, rel):
return [l.get("href", "") for l in self.links if rel in l.get("rel", "").lower().split()]
def fetch(url, follow=True, limit=6 << 20):
"""(status, final url, headers, body) — `follow=False` reports a redirect instead of following it."""
class NoRedirect(urllib.request.HTTPRedirectHandler):
def redirect_request(self, *a, **k):
return None
opener = urllib.request.build_opener(*([] if follow else [NoRedirect()]))
req = urllib.request.Request(url, headers={"User-Agent": UA})
try:
with opener.open(req, timeout=30) as r:
return r.status, r.geturl(), r.headers, r.read(limit)
except urllib.error.HTTPError as e:
return e.code, url, e.headers, b""
def image_size(b):
"""(kind, width, height) from the first bytes of a PNG, JPEG, GIF or WebP; None when unknown."""
if b[:8] == b"\x89PNG\r\n\x1a\n":
return ("png",) + struct.unpack(">II", b[16:24])
if b[:6] in (b"GIF87a", b"GIF89a"):
return ("gif",) + struct.unpack("> 14) & 0x3FFF) + 1)
if b[:2] == b"\xff\xd8":
i = 2
while i + 9 < len(b):
if b[i] != 0xFF:
i += 1
continue
marker = b[i + 1]
if marker in (0xC0, 0xC1, 0xC2, 0xC3, 0xC5, 0xC6, 0xC7, 0xC9, 0xCA, 0xCB, 0xCD, 0xCE, 0xCF):
h, w = struct.unpack(">HH", b[i + 5:i + 9])
return ("jpeg", w, h)
if marker in (0xD8, 0x01) or 0xD0 <= marker <= 0xD7:
i += 2
continue
i += 2 + struct.unpack(">H", b[i + 2:i + 4])[0]
return None
class Report:
def __init__(self):
self.rows = []
def add(self, level, what, detail=""):
self.rows.append((level, what, detail))
def ok(self, cond, what, detail="", level="FAIL"):
self.add("PASS" if cond else level, what, detail)
return cond
def show(self):
for level, what, detail in self.rows:
print(f"{level:4} {what}" + (f" — {detail}" if detail else ""))
n = {k: sum(1 for r in self.rows if r[0] == k) for k in ("PASS", "WARN", "FAIL")}
print(f"\n{n['PASS']} pass, {n['WARN']} warn, {n['FAIL']} fail")
return n["FAIL"] == 0
def audit(text, page_url, online):
h = Head()
h.feed(text)
r = Report()
absu = lambda u: urllib.parse.urljoin(page_url or "", u) if u else u
# the basics
r.ok(any(m.get("charset") for m in h.metas), "charset declared")
r.ok(any((m.get("name") or "").lower() == "viewport" for m in h.metas), "viewport meta (mobile layout)")
r.ok(bool(h.lang), "", h.lang or "missing", level="WARN")
t = h.title or ""
r.ok(bool(t), "title", t[:80])
if t:
r.ok(15 <= len(t) <= 70, "title length 15–70 (shown whole in results)", f"{len(t)} characters", level="WARN")
desc = h.first("description") or ""
r.ok(bool(desc), "meta description")
if desc:
r.ok(70 <= len(desc) <= 160, "description length 70–160", f"{len(desc)} characters", level="WARN")
r.ok(bool(h.first("robots")), "robots meta", h.first("robots") or "absent (defaults to index, follow)", level="WARN")
r.ok(bool(h.first("theme-color")), "theme-color", level="WARN")
# canonical
canon = (h.link("canonical") or [None])[0]
r.ok(bool(canon), "canonical link", canon or "")
if canon:
r.ok(canon.startswith("https://"), "canonical is absolute https")
if online:
st, _, hd, _ = fetch(canon, follow=False, limit=1)
r.ok(st == 200, "canonical answers 200 without a redirect", f"{st} {hd.get('Location', '') if hd else ''}".strip())
# Open Graph
for k in ("og:title", "og:description", "og:type", "og:url", "og:site_name"):
r.ok(bool(h.first(k)), k, (h.first(k) or "")[:90])
if canon and h.first("og:url"):
r.ok(h.first("og:url") == canon, "og:url equals the canonical", h.first("og:url"))
imgs = h.meta("og:image")
r.ok(bool(imgs), "og:image", f"{len(imgs)} image(s)")
widths, heights, alts, types = h.meta("og:image:width"), h.meta("og:image:height"), h.meta("og:image:alt"), h.meta("og:image:type")
r.ok(len(widths) == len(imgs) and len(heights) == len(imgs), "og:image:width/height for every image", level="WARN")
r.ok(len(alts) >= len(imgs), "og:image:alt for every image", level="WARN")
# X / Twitter
card = h.first("twitter:card")
r.ok(card in ("summary", "summary_large_image", "player", "app"), "twitter:card", card or "missing")
for k in ("twitter:title", "twitter:description", "twitter:image"):
r.ok(bool(h.first(k)) or bool(h.first(k.replace("twitter:", "og:"))), k, (h.first(k) or "(falls back to og)")[:90], level="WARN")
r.ok(bool(h.first("twitter:image:alt")), "twitter:image:alt", level="WARN")
# icons
icons, touch = h.link("icon"), h.link("apple-touch-icon")
r.ok(bool(icons), "rel=icon", ", ".join(icons))
r.ok(bool(touch), "apple-touch-icon", ", ".join(touch), level="WARN")
# JSON-LD
r.ok(bool(h.ld), "JSON-LD present", f"{len(h.ld)} block(s)", level="WARN")
for i, block in enumerate(h.ld):
try:
d = json.loads(block)
r.ok("@context" in d, f"JSON-LD {i + 1} parses and has @context")
if canon:
r.ok(canon in block, f"JSON-LD {i + 1} names the canonical url", level="WARN")
except ValueError as e:
r.add("FAIL", f"JSON-LD {i + 1} parses", str(e))
if not online:
return r
# fetch what a preview fetches
for u in icons[:1] + touch[:1]:
st, _, _, _ = fetch(absu(u), limit=1)
r.ok(st == 200, f"icon reachable: {u}", str(st))
declared = list(zip(imgs, widths + [None] * len(imgs), heights + [None] * len(imgs), types + [None] * len(imgs)))
tw = h.first("twitter:image")
if tw and tw not in imgs:
declared.append((tw, None, None, None))
for u, w, hh, ty in declared:
st, final, hd, body = fetch(absu(u))
if not r.ok(st == 200, f"image reachable: {u}", str(st)):
continue
size = image_size(body)
if not r.ok(size is not None, f"image format readable: {u}"):
continue
kind, W, H = size
if w and hh:
r.ok((str(W), str(H)) == (w, hh), f"declared {w}×{hh} = real {W}×{H}", u)
if ty:
r.ok(ty.endswith(kind), f"og:image:type {ty} matches the file ({kind})", level="WARN")
r.ok(W >= 200 and H >= 200, f"{W}×{H} at least 200×200", u)
r.ok(len(body) < 5 << 20, "under 5 MB (X's limit)", f"{len(body) / 1024:.0f} KB")
if W > H:
r.ok(abs(W / H - 1.91) < 0.15, "wide card near 1.91:1 (1200×630)", f"{W / H:.2f}:1", level="WARN")
if card == "summary_large_image" and declared:
st, _, _, body = fetch(absu(tw or declared[0][0]))
s = image_size(body) if st == 200 else None
r.ok(bool(s) and s[1] > s[2], "summary_large_image uses a wide image", level="WARN")
return r
def main():
ap = argparse.ArgumentParser(description=__doc__.split("\n")[0])
ap.add_argument("target", help="a page URL, or a local .html file")
ap.add_argument("--base", help="for a local file: the URL it will be served at (resolves relative image and icon paths)")
ap.add_argument("--offline", action="store_true", help="read the tags only; fetch nothing")
a = ap.parse_args()
if a.target.startswith(("http://", "https://")):
st, final, _, body = fetch(a.target)
if st != 200:
sys.exit(f"{a.target}: HTTP {st}")
if final != a.target:
print(f"note: {a.target} redirects to {final}\n")
text, url = body.decode("utf-8", "replace"), final
else:
text, url = open(a.target, encoding="utf-8").read(), a.base
sys.exit(0 if audit(text, url, online=not a.offline and bool(url)).show() else 1)
if __name__ == "__main__":
main()