Files
chemavx-seo-tools/seo_audit.py
T
ChemaVX cc2853850c auditor: auditar también el ES, y dejar de llamar publicado a lo programado
El blog ES nunca se había auditado entero: seo_audit era EN-only aunque
seo_exceptions ya estuviera indexado por sitio desde el principio. Ahora
acepta --site es. Resultado del estreno: 40 de 40 limpios.

Para que eso signifique algo hubo que arreglar SITE_HOST, que era una
constante con el dominio del EN y decide qué href cuenta como enlace
interno. Auditando el ES con el host del EN salían 21 posts marcados por
enlaces internos; con el host correcto, cero. Un informe entero de
hallazgos falsos que nadie habría podido arreglar.

Y `ghst post list --status published` NO filtra: devuelve published Y
scheduled. El encabezado llevaba desde siempre diciendo "42 published
posts" cuando eran 33 publicados y 9 programados. Se pide sin filtro y se
separa en el código, que además deja explícito que los programados también
se auditan — que está bien, pero hay que decirlo.
2026-07-28 07:21:32 +00:00

235 lines
9.9 KiB
Python
Executable File
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
Tool A — site-wide SEO auditor para los dos blogs (Ghost).
READ-ONLY. Fetches every published post via the ghst-en wrapper (Ghost Admin API,
staff token pulled on-demand from the pod — nothing written to disk), runs the
shared rule engine (seo_rules.py), and prints:
1. a summary of aggregate violations,
2. a per-post table sorted worst-first,
3. a ranked "fix this first" list.
Usage:
python3 seo_audit.py # EN (por defecto), en vivo
python3 seo_audit.py --site es # el blog ES
python3 seo_audit.py --json cache.json # audit a saved posts JSON instead of fetching
python3 seo_audit.py --save cache.json # fetch live AND save the raw JSON
Changes NOTHING in Ghost.
"""
import argparse
import json
import os
import subprocess
import sys
import tempfile
from collections import Counter
import seo_rules as R
import seo_exceptions as X
def partition(post, violations, site="en"):
"""Split one post's violations into (actionable, accepted, info).
'accepted' = an actionable-severity violation whose (rule, slug) is a
documented waiver in seo_exceptions — kept OUT of the to-fix count and shown
in its own section. The rule still ran; this is a reporting-side filter only.
"""
slug = post.get("slug")
actionable, accepted, info = [], [], []
for v in violations:
if v.severity == R.INFO:
info.append(v)
elif X.accepted_reason(v.rule, slug, site):
accepted.append(v)
else:
actionable.append(v)
return actionable, accepted, info
SITIOS = {"en": {"cli": "ghst-en", "host": "theexclusionzone.com"},
"es": {"cli": "ghst-es", "host": "zonadeexclusion.com"}}
def fetch_posts(cli="ghst-en"):
"""Pull all published posts (with html) via the site's ghst wrapper.
We redirect ghst's stdout to a temp file rather than capturing the pipe:
ghst's output is ~750 KB and a captured PIPE truncates at the 64 KB pipe
buffer, whereas a file redirect gets the whole document.
"""
fd, path = tempfile.mkstemp(suffix=".json", prefix="seo_audit_")
os.close(fd)
try:
# ⚠️ `--status published` NO filtra: ghst devuelve published Y scheduled
# (medido el 2026-07-28: 33 + 9 = los 42 que el informe llamaba
# "published"). Se pide sin filtro y se separa aquí, que es lo honesto.
cmd = (f"{cli} post list --limit all "
"--formats html,lexical --json > " + path)
r = subprocess.run(cmd, shell=True, stderr=subprocess.PIPE, text=True)
if r.returncode != 0:
sys.exit(f"{cli} fetch failed:\n{r.stderr}")
data = json.load(open(path))
posts = data["posts"] if isinstance(data, dict) and "posts" in data else data
return [x for x in posts if x.get("status") in ("published", "scheduled")]
finally:
os.unlink(path)
def load_posts(path):
data = json.load(open(path))
return data["posts"] if isinstance(data, dict) and "posts" in data else data
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--site", choices=("en", "es"), default="en")
ap.add_argument("--json", help="audit a saved posts JSON instead of fetching live")
ap.add_argument("--save", help="save fetched raw posts JSON to this path")
args = ap.parse_args()
if args.json:
posts = load_posts(args.json)
src = f"file {args.json}"
else:
posts = fetch_posts(SITIOS[args.site]["cli"])
src = f"live ({SITIOS[args.site]['cli']}, Ghost Admin API)"
if args.save:
json.dump({"posts": posts}, open(args.save, "w"), indent=2)
# El motor decide qué href es interno con R.SITE_HOST: sin esto, auditar
# el ES contaría cero enlaces internos en los 31 posts.
R.usar_sitio(args.site)
results = [] # (post, actionable, accepted, info, score)
for p in posts:
v = R.check_post(p)
actionable, accepted, info = partition(p, v, args.site)
# score ranks worst-first on ACTIONABLE severity only — accepted waivers
# and INFO don't push a post up the list.
results.append((p, actionable, accepted, info, R.score(actionable)))
# worst-first: by actionable severity score, then actionable count, then slug
results.sort(key=lambda t: (-t[4], -len(t[1]), t[0]["slug"]))
total_posts = len(posts)
# SUMMARY counts reflect ACTIONABLE violations only (waived items excluded).
posts_with_rule = Counter()
for _, actionable, _, _, _ in results:
for rule in {v.rule for v in actionable}:
posts_with_rule[rule] += 1
clean = sum(1 for _, actionable, _, _, _ in results if not actionable)
accepted_total = sum(len(acc) for _, _, acc, _, _ in results)
# ---- header ----
print("=" * 78)
print(f" SEO AUDIT — {SITIOS[args.site]['host']} source: {src}")
n_pub = sum(1 for p in posts if p.get("status") == "published")
n_prog = total_posts - n_pub
print(f" {n_pub} publicados"
+ (f" + {n_prog} programados" if n_prog else "")
+ f" | {clean} clean (no actionable issues)"
+ (f" | {accepted_total} accepted exception(s) waived" if accepted_total else ""))
print(f" JSON-LD: handled globally by Edition-main theme — not flagged per post")
print("=" * 78)
# ---- summary ----
print("\nSUMMARY — violations by rule (most posts affected first):\n")
LABEL = {
"meta_description.missing": "meta_description MISSING",
"meta_description.too_long": f"meta_description > {R.META_DESC_MAX}",
"meta_title.missing": "meta_title missing (falls back to title)",
"meta_title.too_long": f"meta_title > {R.META_TITLE_MAX}",
"custom_excerpt.too_long": f"custom_excerpt > {R.CUSTOM_EXCERPT_MAX}",
"og_title.empty": "og_title empty",
"og_description.empty": "og_description empty",
"twitter_title.empty": "twitter_title empty",
"twitter_description.empty": "twitter_description empty",
"feature_image.missing": "feature_image missing",
"feature_image_alt.missing": "feature_image_alt missing",
"internal_links.too_few": f"internal links < {R.MIN_INTERNAL_LINKS}",
"title_eq_meta_title": "meta_title == title (info)",
}
# severity per rule (look one up from any occurrence, incl. waived ones)
sev_of = {}
for _, actionable, accepted, info, _ in results:
for v in actionable + accepted + info:
sev_of[v.rule] = v.severity
for rule, _ in sorted(posts_with_rule.items(),
key=lambda kv: (-sev_of.get(kv[0], 0), -kv[1])):
sev = R.SEV_NAME[sev_of.get(rule, 0)]
print(f" [{sev:4}] {posts_with_rule[rule]:2}/{total_posts} posts "
f"{LABEL.get(rule, rule)}")
# ---- per-post table ----
print("\n" + "-" * 78)
print("PER-POST (worst first) — score = summed severity (HIGH 3 / MED 2 / LOW 1)\n")
print(f"{'score':>5} {'slug':45} issues")
print(f"{'-'*5} {'-'*45} {'-'*20}")
for p, actionable, accepted, info, sc in results:
if not actionable:
note = "✓ clean" + (f" ({len(accepted)} accepted exception)" if accepted else "")
print(f"{sc:5} {p['slug'][:45]:45} {note}")
continue
tags = ", ".join(sorted({v.rule for v in actionable}))
print(f"{sc:5} {p['slug'][:45]:45} {tags}")
# ---- detail per non-clean post ----
print("\n" + "-" * 78)
print("DETAIL\n")
for p, actionable, accepted, info, sc in results:
if not actionable and not info:
continue
print(f"• {p['slug']} (score {sc})")
for v in sorted(actionable, key=lambda x: -x.severity):
print(f" [{R.SEV_NAME[v.severity]:4}] {v.message}"
+ (f" → {v.fix}" if v.fix else ""))
for v in info:
print(f" [INFO] {v.message}")
print()
# ---- accepted exceptions (won't fix — documented) ----
if accepted_total:
print("-" * 78)
print("ACCEPTED EXCEPTIONS (won't fix — documented decisions, NOT in the count)\n")
for p, actionable, accepted, info, sc in sorted(results, key=lambda t: t[0]["slug"]):
for v in accepted:
reason = X.accepted_reason(v.rule, p["slug"])
print(f"• {p['slug']}")
print(f" [{R.SEV_NAME[v.severity]:4}] {v.message}")
print(f" waived: {reason}")
print()
# ---- ranked guidance ----
print("-" * 78)
print("FIX FIRST (by impact / effort):\n")
md_missing = posts_with_rule.get("meta_description.missing", 0)
md_long = posts_with_rule.get("meta_description.too_long", 0)
links = posts_with_rule.get("internal_links.too_few", 0)
og = posts_with_rule.get("og_title.empty", 0)
alt = posts_with_rule.get("feature_image_alt.missing", 0)
rank = []
if md_missing:
rank.append(f"meta_description MISSING on {md_missing} post(s) — HIGH impact, "
f"controls the Google snippet. Write unique 140145-char descriptions.")
if md_long:
rank.append(f"meta_description > {R.META_DESC_MAX} on {md_long} post(s) — trim so the "
f"snippet isn't truncated (real ranking/CTR impact).")
if links:
rank.append(f"<{R.MIN_INTERNAL_LINKS} internal links on {links} post(s) — add 23 contextual "
f"links to related articles (helps crawl + topical authority).")
if og:
rank.append(f"Empty OG/Twitter fields on {og} post(s) — SAFEST AUTO-FIX: mirror "
f"meta_title/description into og_* and twitter_*. Highest count, lowest risk.")
if alt:
rank.append(f"Missing feature_image_alt on {alt} post(s) — quick accessibility + image-SEO win.")
for i, line in enumerate(rank, 1):
print(f" {i}. {line}")
print()
if __name__ == "__main__":
main()