Files
researchowl/tests/test_spec_contract.py
T
ChemaVXandClaude Opus 5 9bfa0fdac2 fix(short): la voz se midió con una sola frase, y por eso el bot reescribía Shorts que ya cabían
`NARRATION_CHARS_PER_SECOND` era 14,2, sacado de una única línea de 82
caracteres. Sintetizando de verdad las 28 líneas que el bot ha escrito hasta
hoy — mismo Piper, mismo modelo, mismas banderas deterministas — la voz lee a
18,5 car/s y se calla 0,25 s en cada punto. Contar las frases aparte es lo que
arregla el caso raro: "Witness identities. Sensor details. Locations redacted."
son tres cuartos de segundo de silencio que un modelo de caracteres a secas
regala.

El error del modelo viejo era de cuatro a seis segundos sobre un Short entero,
siempre por arriba, y con eso el aviso de duración saltaba en vídeos que
estaban dentro del objetivo. Contrastado ahora contra los tres MP4 que hay
renderizados: 39,42 / 47,19 / 45,81 s estimados contra 39,57 / 47,53 / 45,40
reales.

Dos cosas más, del mismo tirón:

- Un margen de 1,5 s antes de avisar. La estimación acierta dentro de un
  segundo por línea, así que medio segundo de exceso puede ser del estimador y
  no del spec; la sesión 168 se llevó una generación entera por ochocientas
  milésimas. El objetivo sigue siendo 20-45.
- El consejo va en palabras, no en "recorta narración", y señala el plano que
  más habla. Las tres veces que saltó, el modelo devolvió un spec que seguía
  pasándose: no sabía cuánto.

Los segundos medidos entran en los tests como tabla, no como número redondo.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-12 21:37:17 +00:00

455 lines
19 KiB
Python

"""Validación local del spec contra el esquema publicado por shortsmith.
Los esquemas de abajo son una COPIA REDUCIDA de lo que devuelve
`GET /templates`, sólo para los tests: en producción se piden en vivo. Si
shortsmith cambia el contrato, quien lo nota es `test_shortsmith_live.py`, no
esto.
"""
import copy
import json
from pathlib import Path
import pytest
from src.generator.spec_contract import (
SpecInvalid, describe_templates, editorial_notes, validate_spec,
)
EXAMPLE = Path(__file__).resolve().parents[1] / "src/generator/examples/jal1628.json"
TEMPLATES = {
"radar_sweep": {
"type": "object", "additionalProperties": False,
"required": ["headline"],
"properties": {
"headline": {"type": "string", "minLength": 1},
"subline": {"type": "string", "default": ""},
"contact_bearing_deg": {"type": "number", "minimum": 0,
"exclusiveMaximum": 360, "default": 210.0},
"sweeps": {"type": "number", "exclusiveMinimum": 0, "maximum": 10,
"default": 2.0},
},
},
"scale_bars": {
"type": "object", "additionalProperties": False,
"required": ["headline", "bars"],
"$defs": {"Bar": {
"type": "object", "additionalProperties": False,
"required": ["label", "value"],
"properties": {
"label": {"type": "string", "minLength": 1},
"value": {"type": "number", "exclusiveMinimum": 0},
"unit": {"type": "string", "default": ""},
"color": {"enum": ["ink", "amber", "amber_dark", "muted", "dim", "red"],
"type": "string", "default": "ink"},
"value_label": {"type": "string", "default": ""},
},
}},
"properties": {
"headline": {"type": "string", "minLength": 1},
"bars": {"type": "array", "items": {"$ref": "#/$defs/Bar"},
"minItems": 1, "maxItems": 3},
"quote": {"type": "array", "items": {"type": "string"}, "maxItems": 2},
"attribution": {"type": "string", "default": ""},
},
},
}
def shot(template="radar_sweep", duration=6.0, **props):
base = {"radar_sweep": {"headline": "3 RADARS"},
"scale_bars": {"headline": "ESCALA",
"bars": [{"label": "BOEING 747", "value": 232}]}}[template]
return {"template": template, "duration": duration, "props": {**base, **props}}
def spec_with(*shots, **meta):
return {
"version": 1,
"meta": {"id": "caso", "title": "Un caso", **meta},
"shots": list(shots) or [shot()],
}
def errors_of(spec, templates=None):
with pytest.raises(SpecInvalid) as exc:
validate_spec(spec, templates if templates is not None else TEMPLATES)
return exc.value.errors
# --- lo que pasa ------------------------------------------------------------
def test_a_minimal_valid_spec_passes():
validate_spec(spec_with(shot(duration=25.0)), TEMPLATES)
def test_the_reference_example_passes_against_its_own_templates():
"""El ejemplo de referencia es válido; se comprueba con esquemas laxos para
las plantillas que este fichero no copia (lo estricto lo cubre el test vivo)."""
spec = json.loads(EXAMPLE.read_text())
permissive = {name: {"type": "object"} for name in
{s["template"] for s in spec["shots"]}}
permissive.update(TEMPLATES)
validate_spec(spec, permissive)
# --- rutas de error ---------------------------------------------------------
def test_unknown_prop_is_reported_with_its_full_path():
"""El typo en un nombre de prop es el bug más probable de un spec escrito
por un LLM, y la ruta exacta es lo que se le devuelve para arreglarlo."""
errors = errors_of(spec_with(shot(sweeeps=2)))
assert any(e.startswith("shots.0.radar_sweep.props.sweeeps: campo no permitido")
for e in errors), errors
assert "sweeps" in errors[0], "hay que decirle cuáles SÍ valen"
def test_unknown_template_lists_the_valid_names():
errors = errors_of(spec_with({"template": "radar_swep", "duration": 6.0,
"props": {"headline": "X"}}))
assert errors[0].startswith("shots.0.template:")
assert "radar_sweep" in errors[0] and "scale_bars" in errors[0]
def test_missing_required_prop():
bad = spec_with(shot()); del bad["shots"][0]["props"]["headline"]
assert "shots.0.radar_sweep.props.headline: falta y es obligatorio" in errors_of(bad)
def test_empty_string_where_a_non_empty_one_is_required():
assert any("shots.0.radar_sweep.props.headline" in e
for e in errors_of(spec_with(shot(headline=""))))
def test_numeric_bounds():
errors = errors_of(spec_with(shot(contact_bearing_deg=400)))
assert "shots.0.radar_sweep.props.contact_bearing_deg: 400 debe ser < 360" in errors
def test_list_length_limits_are_enforced():
bars = [{"label": f"B{i}", "value": i + 1} for i in range(4)]
errors = errors_of(spec_with(shot("scale_bars", bars=bars)))
assert "shots.0.scale_bars.props.bars: 4 elementos, el máximo es 3" in errors
def test_colour_must_be_a_palette_name_never_hex():
errors = errors_of(spec_with(shot("scale_bars", bars=[
{"label": "OBJETO", "value": 2000, "color": "#ffbf00"}])))
assert any("color" in e and "amber" in e for e in errors)
def test_nested_paths_survive_lists():
errors = errors_of(spec_with(shot("scale_bars", bars=[
{"label": "BOEING 747", "value": 232},
{"label": "OBJETO", "value": -5}])))
assert "shots.0.scale_bars.props.bars.1.value: -5 debe ser > 0" in errors
def test_every_error_comes_back_at_once():
"""Se devuelven todos: arreglar cinco de una vez sale más barato que cinco vueltas."""
errors = errors_of(spec_with(shot(headline="", sweeeps=1, contact_bearing_deg=999)))
assert len(errors) >= 3
# --- el sobre ---------------------------------------------------------------
def test_meta_id_pattern():
assert any(e.startswith("meta.id:") for e in errors_of(spec_with(id="Caso Roswell")))
def test_resolution_must_be_a_shorts_one():
assert any("no es una resolución admitida" in e
for e in errors_of(spec_with(shot(), width=800, height=600)))
def test_total_duration_ceiling_is_the_contract_not_the_target():
"""45 s es el objetivo editorial; 180 s es el límite duro. Pasarse de 45 no
invalida el spec — eso es una nota, no un error."""
long_spec = spec_with(*[shot(duration=10.0) for _ in range(6)]) # 60 s
validate_spec(long_spec, TEMPLATES)
assert editorial_notes(long_spec)
too_long = spec_with(*[shot(duration=30.0) for _ in range(7)]) # 210 s
assert any("pasa del límite" in e for e in errors_of(too_long))
def test_total_duration_floor():
assert any("no llega al mínimo" in e
for e in errors_of(spec_with(shot(duration=2.0))))
def test_silence_window_cannot_run_past_the_end():
bad = spec_with(shot(duration=25.0))
bad["audio"] = {"preset": "sonar", "silence": [[20.0, 40.0]]}
assert any("se sale de la duración total" in e for e in errors_of(bad))
def test_the_live_palette_widens_what_a_preset_may_be():
"""Con la paleta de GET /audio, un preset nuevo en shortsmith llega aquí
sin tocar este repo — el mismo pacto que las plantillas."""
doc = spec_with(shot(duration=25.0))
doc["audio"] = {"preset": "pulse"}
validate_spec(doc, TEMPLATES, presets=("sonar", "pulse", "static", "none"))
def test_without_the_palette_only_the_baseline_presets_pass():
"""El default es conservador a propósito: nunca acepta lo que un shortsmith
viejo no renderice."""
doc = spec_with(shot(duration=25.0))
doc["audio"] = {"preset": "pulse"}
assert any("audio.preset" in e for e in errors_of(doc))
def test_an_unknown_preset_error_names_the_palette():
doc = spec_with(shot(duration=25.0))
doc["audio"] = {"preset": "vaporwave"}
with pytest.raises(SpecInvalid) as exc:
validate_spec(doc, TEMPLATES, presets=("sonar", "pulse", "none"))
line = next(e for e in exc.value.errors if "audio.preset" in e)
assert "pulse" in line and "vaporwave" in line
def test_extra_root_key_is_rejected():
bad = spec_with(shot(duration=25.0)); bad["narrative_shape"] = "case_file"
assert any(e.startswith("narrative_shape:") for e in errors_of(bad))
def test_editorial_notes_flag_both_ends():
assert "queda corto" in editorial_notes(spec_with(shot(duration=8.0)))[0]
assert "recorta" in editorial_notes(
spec_with(*[shot(duration=10.0) for _ in range(6)]))[0]
assert editorial_notes(spec_with(shot(duration=30.0))) == []
def test_a_spec_that_is_not_even_a_dict():
with pytest.raises(SpecInvalid):
validate_spec([1, 2, 3], TEMPLATES)
# --- descripción para el prompt ---------------------------------------------
def test_describe_templates_is_driven_by_what_the_service_publishes():
text = describe_templates(TEMPLATES)
assert "radar_sweep:" in text and "scale_bars:" in text
assert "headline: string, no vacío, OBLIGATORIO" in text
assert "1-3 elementos" in text # los límites llegan al prompt
assert "ink, amber, amber_dark, muted, dim, red" in text
assert "label: string, no vacío, OBLIGATORIO" in text # despliega los objetos anidados
def test_a_template_nobody_wrote_here_still_gets_described():
"""La prueba de que el contrato no está copiado: una plantilla inventada,
que este repo no conoce, se describe igual."""
text = describe_templates({**TEMPLATES, "holo_scan": {
"type": "object", "required": ["title"],
"properties": {"title": {"type": "string", "minLength": 1},
"depth_m": {"type": "number", "maximum": 999}}}})
assert "holo_scan:" in text
assert "depth_m: number, ≤ 999" in text
def test_validation_accepts_a_template_nobody_wrote_here():
templates = {**TEMPLATES, "holo_scan": {
"type": "object", "additionalProperties": False, "required": ["title"],
"properties": {"title": {"type": "string", "minLength": 1}}}}
validate_spec(spec_with({"template": "holo_scan", "duration": 30.0,
"props": {"title": "X"}}), templates)
# --- narración (fase 4b) ----------------------------------------------------
def test_a_shot_may_carry_narration():
doc = spec_with(shot(duration=25.0))
doc["shots"][0]["narration"] = "Three radars tracked it that night."
validate_spec(doc, TEMPLATES)
def test_an_overlong_narration_is_rejected_with_its_path():
doc = spec_with(shot(duration=25.0))
doc["shots"][0]["narration"] = "x" * 400
assert any("shots.0.narration" in e and "320" in e for e in errors_of(doc))
def test_narration_that_is_not_text_is_rejected():
doc = spec_with(shot(duration=25.0))
doc["shots"][0]["narration"] = ["a", "b"]
assert any("shots.0.narration" in e for e in errors_of(doc))
def test_an_unknown_shot_key_still_names_the_valid_ones():
doc = spec_with(shot(duration=25.0))
doc["shots"][0]["voiceover"] = "nope"
assert any("narration" in e for e in errors_of(doc))
#: Líneas de narración de specs que se renderizaron de verdad, con lo que tarda
#: Piper en decirlas. Medido el 2026-08-12 con el binario, el modelo y las
#: banderas de shortsmith (`en_US-lessac-medium`, length_scale 1.0,
#: --noise_scale 0 --noise_w 0), que son deterministas: estos segundos se
#: reproducen. Se eligieron los extremos del muestreo de 28 líneas — la más
#: rápida, la más lenta y las dos más largas — porque son las que rompen un
#: modelo mal calibrado; la media la aguanta cualquiera.
MEASURED = [
("Eight FBI witness interviews. Five digital renderings. All describe the "
"same shape flying across America for twenty-four years.", 7.809),
("The files are public now, but sections remain blacked out. Witness "
"identities. Sensor details. Locations redacted.", 8.140),
("Three hundred seventy-eight files released. Hundreds of incidents "
"documented. And the government still cannot explain what those shapes "
"were.", 7.681),
("The files came out. The numbers stayed classified.", 3.310),
("Nothing should have been able to hold station beside them up there.", 3.396),
("The Air Force's own investigators called it unexplained.", 2.990),
]
@pytest.mark.parametrize("line,real", MEASURED)
def test_the_estimate_lands_within_a_second_of_the_voice(line, real):
"""La estimación es lo único que separa un aviso útil de una reescritura
inventada, así que se contrasta contra audio medido, no contra sí misma.
El margen es un segundo. Más apretado sería falso — esto estima, no
sintetiza — y más ancho deja de decir nada: el error del modelo anterior
sobre un Short entero era de cuatro a seis segundos, y de ahí salían los
tres intentos que se gastaban en cada generación.
"""
from src.generator.spec_contract import spoken_seconds
assert spoken_seconds(line) == pytest.approx(real, abs=1.0)
def test_a_line_of_short_sentences_is_not_taken_for_fast_prose():
"""Piper calla un cuarto de segundo en cada punto. Cuatro frases cortas son
un segundo de silencio, y contarlas como texto corrido las da por rápidas:
es el caso donde más se equivocaba el modelo de sólo caracteres."""
from src.generator.spec_contract import spoken_seconds
chopped = "The files are public now, but sections remain blacked out. " \
"Witness identities. Sensor details. Locations redacted."
flowing = "The files are public now but sections remain blacked out with " \
"witness identities sensor details and locations redacted"
assert len(chopped) < len(flowing)
assert spoken_seconds(chopped) > spoken_seconds(flowing)
def test_a_decimal_point_is_not_the_end_of_a_sentence():
from src.generator.spec_contract import spoken_seconds
assert spoken_seconds("It climbed to 1.5 miles") == \
pytest.approx(spoken_seconds("It climbed to 155 miles"))
def test_the_estimate_counts_the_voice_not_just_the_declared_seconds():
"""La duración declarada es un suelo: shortsmith estira el shot si la frase
no cabe, y el modelo tiene que enterarse ANTES de pagar el render."""
from src.generator.spec_contract import estimated_duration
doc = spec_with(shot(duration=3.0))
doc["shots"][0]["narration"] = MEASURED[0][0] # 7,81 s de voz medidos
assert estimated_duration(doc) > 8.0
def test_a_shot_with_room_for_its_line_is_estimated_as_declared():
from src.generator.spec_contract import estimated_duration
doc = spec_with(shot(duration=30.0))
doc["shots"][0]["narration"] = "Short line."
assert estimated_duration(doc) == pytest.approx(30.0)
def test_narration_that_overshoots_the_target_is_flagged_as_narration():
"""El consejo tiene que decir QUÉ recortar: con la voz mandando, acortar
duraciones no arregla nada."""
# 3 shots de 8 s = 24 s declarados, dentro del objetivo y sin avisos. Con
# ~21 s de voz cada uno se van a 65 s: sin la estimación, silencio absoluto.
quiet = spec_with(*[shot(duration=8.0) for _ in range(3)])
assert editorial_notes(quiet) == []
doc = copy.deepcopy(quiet)
for s in doc["shots"]:
s["narration"] = "A" * 300
note = editorial_notes(doc)[0]
assert "narración" in note and "estimada" in note
def test_a_second_over_the_target_is_not_worth_a_rewrite():
"""El objetivo sigue siendo 45 s, pero la estimación tiene un segundo de
error por línea: avisar por medio segundo es avisar del estimador. Caso
real — la sesión 168 salió a 45,4 s y se pagó una generación por ello."""
from src.generator.spec_contract import TARGET_GRACE, TARGET_MAX_DURATION
justo = spec_with(shot(duration=TARGET_MAX_DURATION + TARGET_GRACE - 0.1))
pasado = spec_with(shot(duration=TARGET_MAX_DURATION + TARGET_GRACE + 0.1))
assert editorial_notes(justo) == []
assert editorial_notes(pasado)
# Y el consejo se mide contra el objetivo, no contra el margen: se pide
# bajar hasta 45, no hasta 46,5.
assert "sobran 1.6s" in editorial_notes(pasado)[0]
def test_the_grace_works_at_both_ends():
from src.generator.spec_contract import TARGET_GRACE, TARGET_MIN_DURATION
assert editorial_notes(spec_with(shot(duration=TARGET_MIN_DURATION
- TARGET_GRACE + 0.1))) == []
assert editorial_notes(spec_with(shot(duration=TARGET_MIN_DURATION
- TARGET_GRACE - 0.1)))
def test_the_advice_says_how_much_to_cut_and_from_where():
""""Recorta narración" no dice cuánta, y las tres veces que saltó este aviso
el modelo devolvió un spec que seguía pasándose. El exceso va en palabras
porque es lo que el modelo escribe, y señalando el plano que más habla."""
doc = spec_with(shot(duration=4.0), shot(duration=4.0))
doc["shots"][0]["narration"] = "Short line."
doc["shots"][1]["narration"] = " ".join(["word"] * 200)
note = editorial_notes(doc)[0]
assert "palabras de narración" in note
assert "shots.1" in note and "shots.0" not in note
def test_the_advice_for_a_silent_spec_never_mentions_narration():
"""Sin voz, pedir que recorte narración es mandarlo a arreglar algo que no
existe: lo que sobra son duraciones declaradas."""
note = editorial_notes(spec_with(*[shot(duration=10.0) for _ in range(6)]))[0]
assert "narración" not in note and "duraciones declaradas" in note
def test_a_spec_without_narration_keeps_the_old_wording():
note = editorial_notes(spec_with(*[shot(duration=10.0) for _ in range(6)]))[0]
assert "duración total" in note and "estimada" not in note
def test_the_prompt_carries_how_much_text_actually_fits():
"""`x-fits` es el único límite que nada rechaza: si no llega al prompt, el
modelo escribe una cita de 58 caracteres para un hueco de 16."""
templates = {"document_quote": {
"type": "object", "required": ["quote_a"],
"properties": {"quote_a": {"type": "string", "minLength": 1, "x-fits": 16}}}}
described = describe_templates(templates)
assert "CABE ~16 caracteres" in described
def test_a_string_longer_than_it_fits_is_still_valid():
"""Los caracteres son un proxy de los píxeles: rechazar por ancho estimado
tiraría specs que se dibujan perfectamente."""
templates = {"radar_sweep": {
"type": "object", "additionalProperties": False, "required": ["headline"],
"properties": {"headline": {"type": "string", "minLength": 1, "x-fits": 13}}}}
doc = spec_with({"template": "radar_sweep", "duration": 25.0,
"props": {"headline": "UN TITULAR BASTANTE MAS LARGO QUE ESO"}})
validate_spec(doc, templates)