El motor se calibró con el corpus inglés y al estrenarlo en zonadeexclusion
daba 6 avisos, 4 de ellos falsos: «que», «los» y «del» pasaban el filtro de
3 caracteres y contaban como identidad de caso. Kenneth Arnold 1947 «colisionaba»
con Roswell 1947 sin compartir más que el año y un artículo determinado.
Dos cambios, y van juntos a propósito:
- Stopwords ES (glue + genéricos del dominio, espejo del bloque inglés).
- _deaccent en el tokenizador: [a-z0-9]+ PARTE en cualquier acento, así que
"Pentágono" era {pent, gono} y "Fenómenos" era {fen, menos} — basura de 3
caracteres que ninguna lista de stopwords puede cubrir, y que además nunca
casaba con el slug de Ghost (que ya viene sin acentos). Sin esto, la mitad
de las stopwords nuevas no llegarían a aplicarse.
Medido sobre los dos corpus antes de tocar nada: ES 6 → 2 avisos, EN idéntico
(mismos 4 pares: PURSUE 3/4 y el par Grusch). Peaje asumido: «los» deja de
identificar a Los Alamos en EN, pero «alamos» sigue ahí y el informe no cambia.
Los 2 avisos ES que quedan son reales: dos artículos publicados sobre Roswell
1947 sin canonical entre ellos, la misma canibalización que ya arreglamos en
el EN con el par Grusch. Queda pendiente de decisión editorial.
347 lines
14 KiB
Python
347 lines
14 KiB
Python
"""
|
|
Reusable SEO rule engine for The Exclusion Zone (EN) — theexclusionzone.com (Ghost).
|
|
|
|
Pure functions, NO I/O. Feed it a Ghost Admin API post dict (with `html` format
|
|
included) and it returns a list of Violation(rule, severity, message, fix).
|
|
|
|
Shared by:
|
|
- seo_audit.py — Tool A, site-wide auditor (this engine, run over all posts)
|
|
- (future) seo_validate.py — Tool C, pre-publish validator (same engine, one draft)
|
|
|
|
Design note: every check is an independent function registered in RULES. To add a
|
|
rule, write a function (post) -> list[Violation] and append it to RULES. The auditor
|
|
and the validator both just call check_post(); they never re-implement a check.
|
|
"""
|
|
import re
|
|
import unicodedata
|
|
from collections import namedtuple
|
|
|
|
SITE_HOST = "theexclusionzone.com"
|
|
|
|
# ---- thresholds (single source of truth, reused by validator) -------------
|
|
META_TITLE_MAX = 60
|
|
META_DESC_MAX = 145
|
|
CUSTOM_EXCERPT_MAX = 300
|
|
MIN_INTERNAL_LINKS = 2
|
|
|
|
# ---- severity weights (used to rank "worst first") ------------------------
|
|
HIGH, MED, LOW, INFO = 3, 2, 1, 0
|
|
SEV_NAME = {HIGH: "HIGH", MED: "MED", LOW: "LOW", INFO: "INFO"}
|
|
|
|
# The Edition-main theme injects BlogPosting JSON-LD globally in default.hbs
|
|
# ({{#is "post"}} ... <script type="application/ld+json"> @type BlogPosting),
|
|
# plus Ghost's own {{ghost_head}}. So JSON-LD is handled site-wide and is NOT a
|
|
# per-post rule. Flip to False only if that theme block is ever removed.
|
|
THEME_HANDLES_JSONLD = True
|
|
|
|
Violation = namedtuple("Violation", "rule severity message fix")
|
|
|
|
|
|
def _s(v):
|
|
return v if isinstance(v, str) else ""
|
|
|
|
|
|
def _empty(v):
|
|
return not (isinstance(v, str) and v.strip())
|
|
|
|
|
|
# First path segments that are NOT article posts on a Ghost site (tags, authors,
|
|
# pagination, static content, etc.). Keeps root-relative link counting from
|
|
# treating /tag/foo or /content/images/... as an internal article link.
|
|
NON_POST_PREFIXES = {
|
|
"tag", "tags", "author", "page", "p", "content", "assets",
|
|
"rss", "ghost", "members", "404", "sitemap",
|
|
}
|
|
|
|
|
|
def _internal_slug(href, self_slug):
|
|
"""Return the article slug an href points to if it's an internal POST link,
|
|
else None. Accepts both absolute (contains SITE_HOST) and root-relative
|
|
("/slug/") forms; rejects protocol-relative ("//host"), anchors, mailto,
|
|
external links, static assets, and known non-post sections."""
|
|
if SITE_HOST in href:
|
|
path = re.sub(r"^https?://[^/]+/", "", href)
|
|
elif href.startswith("/") and not href.startswith("//"):
|
|
path = href[1:]
|
|
else:
|
|
return None
|
|
slug = path.strip("/").split("/")[0]
|
|
if not slug or slug == self_slug:
|
|
return None
|
|
if slug in NON_POST_PREFIXES or "." in slug: # section page or static asset
|
|
return None
|
|
return slug
|
|
|
|
|
|
def internal_links(post):
|
|
"""Distinct internal article slugs linked from the body, excluding self-links.
|
|
|
|
Counts BOTH absolute internal links (href containing SITE_HOST) and
|
|
root-relative links ("/slug/"), so a Ghost-relative link isn't miscounted as
|
|
"too few". Tool A (auditor) and Tool C (validator) share this, staying in sync.
|
|
"""
|
|
html = _s(post.get("html"))
|
|
self_slug = post.get("slug")
|
|
out = set()
|
|
for m in re.finditer(r'href="([^"#]+)"', html):
|
|
slug = _internal_slug(m.group(1), self_slug)
|
|
if slug:
|
|
out.add(slug)
|
|
return out
|
|
|
|
|
|
# --- individual rules -------------------------------------------------------
|
|
|
|
def r_meta_title(p):
|
|
mt = _s(p.get("meta_title"))
|
|
title = _s(p.get("title"))
|
|
if _empty(mt):
|
|
sev = MED if len(title) > META_TITLE_MAX else LOW
|
|
return [Violation("meta_title.missing", sev,
|
|
f"meta_title missing → falls back to title ({len(title)} chars"
|
|
+ (f", which is >{META_TITLE_MAX}!" if len(title) > META_TITLE_MAX else "") + ")",
|
|
"set meta_title")]
|
|
if len(mt) > META_TITLE_MAX:
|
|
return [Violation("meta_title.too_long", MED,
|
|
f"meta_title {len(mt)} chars > {META_TITLE_MAX}", "shorten meta_title")]
|
|
return []
|
|
|
|
|
|
def r_meta_description(p):
|
|
md = _s(p.get("meta_description"))
|
|
if _empty(md):
|
|
return [Violation("meta_description.missing", HIGH,
|
|
"meta_description MISSING (no SERP snippet control)", "write meta_description")]
|
|
if len(md) > META_DESC_MAX:
|
|
return [Violation("meta_description.too_long", MED,
|
|
f"meta_description {len(md)} chars > {META_DESC_MAX} (will be truncated in SERP)",
|
|
"shorten meta_description")]
|
|
return []
|
|
|
|
|
|
def r_custom_excerpt(p):
|
|
ce = _s(p.get("custom_excerpt"))
|
|
if ce and len(ce) > CUSTOM_EXCERPT_MAX:
|
|
return [Violation("custom_excerpt.too_long", LOW,
|
|
f"custom_excerpt {len(ce)} chars > {CUSTOM_EXCERPT_MAX}", "shorten custom_excerpt")]
|
|
return []
|
|
|
|
|
|
def r_social_fields(p):
|
|
"""OG/Twitter empties — the safest auto-fixable category (mirror meta_*)."""
|
|
out = []
|
|
fallbacks = {
|
|
"og_title": "meta_title",
|
|
"og_description": "meta_description",
|
|
"twitter_title": "meta_title",
|
|
"twitter_description": "meta_description",
|
|
}
|
|
for field, src in fallbacks.items():
|
|
if _empty(p.get(field)):
|
|
out.append(Violation(f"{field}.empty", LOW,
|
|
f"{field} empty", f"mirror from {src}"))
|
|
return out
|
|
|
|
|
|
def r_feature_image(p):
|
|
out = []
|
|
if _empty(p.get("feature_image")):
|
|
out.append(Violation("feature_image.missing", MED,
|
|
"feature_image missing (no social/share card image)", "add feature image"))
|
|
else:
|
|
if _empty(p.get("feature_image_alt")):
|
|
out.append(Violation("feature_image_alt.missing", LOW,
|
|
"feature_image_alt missing (a11y + image SEO)", "add feature image alt text"))
|
|
return out
|
|
|
|
|
|
def r_internal_links(p):
|
|
n = len(internal_links(p))
|
|
if n < MIN_INTERNAL_LINKS:
|
|
return [Violation("internal_links.too_few", MED,
|
|
f"{n} internal link(s) < {MIN_INTERNAL_LINKS} (weak interlinking)",
|
|
"add internal links to related articles")]
|
|
return []
|
|
|
|
|
|
def r_title_equals_meta(p):
|
|
mt = _s(p.get("meta_title"))
|
|
title = _s(p.get("title"))
|
|
if mt and mt == title:
|
|
return [Violation("title_eq_meta_title", INFO,
|
|
"meta_title identical to title (often fine; review)", "")]
|
|
return []
|
|
|
|
|
|
def r_jsonld(p):
|
|
if THEME_HANDLES_JSONLD:
|
|
return []
|
|
return [Violation("jsonld.missing", MED, "no BlogPosting JSON-LD", "add JSON-LD")]
|
|
|
|
|
|
# ---- topic collision (corpus-aware; NOT in RULES) ---------------------------
|
|
# Two posts about the same case cannibalize each other in the SERP (2026-07-10:
|
|
# a second Kecksburg post was published while another sat scheduled; a "When
|
|
# Nuclear ... Went/Go Silent" near-twin title was already queued). RULES functions
|
|
# are (post) -> violations; this one also needs the rest of the site, so callers
|
|
# (seo_validate.py) pass the corpus explicitly: published + scheduled posts as
|
|
# dicts with at least {id, title, slug}.
|
|
|
|
TOPIC_STOPWORDS = {
|
|
# english glue
|
|
"the", "a", "an", "of", "and", "in", "at", "on", "to", "that", "what",
|
|
"when", "who", "why", "how", "its", "his", "her", "their", "our", "one",
|
|
"still", "cant", "couldnt", "went", "go", "goes", "most", "from", "with",
|
|
"they", "them", "these", "this", "are", "were", "was", "is", "be", "been",
|
|
"has", "have", "had", "but", "for", "all", "than", "then", "ever", "never",
|
|
# domain-generic (present in half the catalog — carry no case identity)
|
|
"ufo", "ufos", "uap", "uaps", "incident", "incidents", "case", "cases",
|
|
"file", "files", "mystery", "declassified", "declassification", "pentagon",
|
|
"government", "military", "congress", "secret", "program", "investigation",
|
|
"evidence", "witness", "witnesses", "document", "documents", "documented",
|
|
"unexplained", "encounter", "sighting", "sightings", "alien", "aliens",
|
|
"phenomena", "aerial", "unidentified", "extraordinary", "americas",
|
|
"american", "video", "footage",
|
|
# spanish glue (added 2026-07-21 with the ES site — zonadeexclusion.com).
|
|
# Without these, "que"/"los"/"del" counted as case identity: Kenneth Arnold
|
|
# 1947 "collided" with Roswell 1947 on nothing but «que» + the shared year.
|
|
# Cost: "los" no longer identifies Los Alamos on EN — "alamos" still does,
|
|
# and the EN corpus reports the same collisions before and after.
|
|
"los", "las", "una", "unos", "unas", "del", "por", "para", "con", "sin",
|
|
"sus", "que", "cual", "cuales", "quien", "quienes", "donde", "cuando",
|
|
"como", "pero", "porque", "aunque", "sobre", "entre", "hasta", "desde",
|
|
"tras", "ante", "bajo", "durante", "segun", "este", "esta", "esto",
|
|
"estos", "estas", "ese", "esa", "eso", "esos", "esas", "aquel", "aquella",
|
|
"otro", "otra", "otros", "otras", "todo", "toda", "todos", "todas",
|
|
"mismo", "misma", "cada", "algo", "alguien", "nada", "nadie", "mas", "muy",
|
|
"aun", "solo", "tambien", "siempre", "nunca", "jamas", "casi", "menos",
|
|
"fue", "fueron", "era", "eran", "ser", "son", "estan", "estaba",
|
|
"estaban", "haber", "habia", "han", "hay", "hizo", "hacer", "hace",
|
|
"tiene", "tienen", "tenia", "puede", "pueden", "podria", "sigue",
|
|
"siguen", "sabe", "dice", "dicen", "ano", "anos", "dia", "dias", "vez",
|
|
"veces", "despues", "antes", "hoy", "ahora",
|
|
# domain-generic ES — mirror of the English block above
|
|
"ovni", "ovnis", "fenomeno", "fenomenos", "caso", "casos", "incidente",
|
|
"incidentes", "misterio", "misterios", "expediente", "expedientes",
|
|
"archivo", "archivos", "documento", "documentos", "desclasificado",
|
|
"desclasificados", "desclasificacion", "gobierno", "militar", "militares",
|
|
"ejercito", "secreto", "secretos", "investigacion", "testigo", "testigos",
|
|
"avistamiento", "avistamientos", "encuentro", "encuentros",
|
|
"extraterrestre", "extraterrestres", "alienigena", "alienigenas",
|
|
"inexplicable", "inexplicables", "aereo", "aerea", "videos",
|
|
}
|
|
# 0.70 calibrated 2026-07-10: the "When Nuclear Weapons Go / Arsenal Went
|
|
# Silent" near-twin pair scores 0.742 (char-level penalizes weapons/arsenal);
|
|
# the closest legit-distinct pair in the catalog scores 0.65.
|
|
TITLE_HOOK_SIM_MIN = 0.70 # SequenceMatcher on the pre-colon hook
|
|
SLUG_JACCARD_MIN = 0.5 # shared slug-token ratio
|
|
# Years >= this are "news era", not case identity: every contemporary post
|
|
# carries the current year (PURSUE 2026, Grusch 2026...) without being the same
|
|
# story. Case years in the catalog run 1947-2019.
|
|
NEWS_YEAR_MIN = 2020
|
|
|
|
_YEAR_RE = re.compile(r"\b(19|20)\d{2}\b")
|
|
|
|
|
|
def _deaccent(text):
|
|
"""Fold accents to ASCII. Required for Spanish: the [a-z0-9]+ tokenizer
|
|
SPLITS on any accented char, so "Pentágono" became {pent, gono} and
|
|
"Fenómenos" became {fen, menos} — 3-char garbage that no stopword list can
|
|
ever cover, and that never matched the (already accent-free) Ghost slug.
|
|
No-op on EN, whose only non-ASCII are dashes and curly apostrophes."""
|
|
return unicodedata.normalize("NFKD", text).encode("ascii", "ignore").decode()
|
|
|
|
|
|
def _tokens(text):
|
|
return set(re.findall(r"[a-z0-9]+", _deaccent(_s(text)).lower()))
|
|
|
|
|
|
def _case_years(title, slug):
|
|
"""Historical case years (pre news-era) found in title+slug."""
|
|
return {m.group(0) for m in _YEAR_RE.finditer(title + " " + slug)
|
|
if int(m.group(0)) < NEWS_YEAR_MIN}
|
|
|
|
|
|
def _sig_tokens(title, slug):
|
|
"""Case-identity tokens: title+slug minus glue/domain words, years and
|
|
fragments shorter than 3 chars (possessive 's', initials...)."""
|
|
toks = _tokens(title) | _tokens(slug.replace("-", " "))
|
|
return {t for t in toks
|
|
if len(t) >= 3 and t not in TOPIC_STOPWORDS and not _YEAR_RE.fullmatch(t)}
|
|
|
|
|
|
def _hook(title):
|
|
return _s(title).split(":")[0].strip().lower()
|
|
|
|
|
|
def topic_collision(post, corpus):
|
|
"""Compare one candidate post against the site corpus → list[Violation].
|
|
|
|
Fires when the candidate and an existing post look like the same story:
|
|
- share a case year AND a case-identity token (Kecksburg+1965), or
|
|
- their pre-colon title hooks read nearly the same, or
|
|
- their slugs share most of their tokens.
|
|
"""
|
|
from difflib import SequenceMatcher
|
|
|
|
out = []
|
|
c_years = _case_years(_s(post.get("title")), _s(post.get("slug")))
|
|
c_sig = _sig_tokens(post.get("title"), _s(post.get("slug")))
|
|
c_hook = _hook(post.get("title"))
|
|
c_slug_toks = _tokens(_s(post.get("slug")).replace("-", " "))
|
|
|
|
for other in corpus:
|
|
if other.get("id") == post.get("id"):
|
|
continue
|
|
o_title, o_slug = _s(other.get("title")), _s(other.get("slug"))
|
|
o_years = _case_years(o_title, o_slug)
|
|
o_sig = _sig_tokens(o_title, o_slug)
|
|
|
|
reasons = []
|
|
if (c_years & o_years) and (c_sig & o_sig):
|
|
shared = ", ".join(sorted(c_sig & o_sig)[:3] + sorted(c_years & o_years))
|
|
reasons.append((HIGH, f"same case + year ({shared})"))
|
|
hook_sim = SequenceMatcher(None, c_hook, _hook(o_title)).ratio()
|
|
if c_hook and hook_sim >= TITLE_HOOK_SIM_MIN:
|
|
reasons.append((MED, f"title hooks {hook_sim:.0%} similar"))
|
|
o_slug_toks = _tokens(o_slug.replace("-", " "))
|
|
union = c_slug_toks | o_slug_toks
|
|
if union:
|
|
jac = len(c_slug_toks & o_slug_toks) / len(union)
|
|
if jac >= SLUG_JACCARD_MIN:
|
|
reasons.append((MED, f"slugs {jac:.0%} overlapping"))
|
|
|
|
if reasons:
|
|
sev = max(s for s, _ in reasons)
|
|
why = "; ".join(r for _, r in reasons)
|
|
out.append(Violation(
|
|
"topic.collision", sev,
|
|
f"collides with [{other.get('status', '?')}] \"{o_title[:60]}\" — {why}",
|
|
"merge, retitle to a distinct angle, or interlink deliberately"))
|
|
return out
|
|
|
|
|
|
RULES = [
|
|
r_meta_title,
|
|
r_meta_description,
|
|
r_custom_excerpt,
|
|
r_social_fields,
|
|
r_feature_image,
|
|
r_internal_links,
|
|
r_title_equals_meta,
|
|
r_jsonld,
|
|
]
|
|
|
|
|
|
def check_post(post):
|
|
"""Run every rule against one post dict → list[Violation]."""
|
|
out = []
|
|
for rule in RULES:
|
|
out.extend(rule(post))
|
|
return out
|
|
|
|
|
|
def score(violations):
|
|
"""Total severity weight (for ranking posts worst-first); INFO counts 0."""
|
|
return sum(v.severity for v in violations)
|