fix(short): la voz se midió con una sola frase, y por eso el bot reescribía Shorts que ya cabían
`NARRATION_CHARS_PER_SECOND` era 14,2, sacado de una única línea de 82 caracteres. Sintetizando de verdad las 28 líneas que el bot ha escrito hasta hoy — mismo Piper, mismo modelo, mismas banderas deterministas — la voz lee a 18,5 car/s y se calla 0,25 s en cada punto. Contar las frases aparte es lo que arregla el caso raro: "Witness identities. Sensor details. Locations redacted." son tres cuartos de segundo de silencio que un modelo de caracteres a secas regala. El error del modelo viejo era de cuatro a seis segundos sobre un Short entero, siempre por arriba, y con eso el aviso de duración saltaba en vídeos que estaban dentro del objetivo. Contrastado ahora contra los tres MP4 que hay renderizados: 39,42 / 47,19 / 45,81 s estimados contra 39,57 / 47,53 / 45,40 reales. Dos cosas más, del mismo tirón: - Un margen de 1,5 s antes de avisar. La estimación acierta dentro de un segundo por línea, así que medio segundo de exceso puede ser del estimador y no del spec; la sesión 168 se llevó una generación entera por ochocientas milésimas. El objetivo sigue siendo 20-45. - El consejo va en palabras, no en "recorta narración", y señala el plano que más habla. Las tres veces que saltó, el modelo devolvió un spec que seguía pasándose: no sabía cuánto. Los segundos medidos entran en los tests como tabla, no como número redondo. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
+107
-2
@@ -283,15 +283,73 @@ def test_an_unknown_shot_key_still_names_the_valid_ones():
|
||||
assert any("narration" in e for e in errors_of(doc))
|
||||
|
||||
|
||||
#: Líneas de narración de specs que se renderizaron de verdad, con lo que tarda
|
||||
#: Piper en decirlas. Medido el 2026-08-12 con el binario, el modelo y las
|
||||
#: banderas de shortsmith (`en_US-lessac-medium`, length_scale 1.0,
|
||||
#: --noise_scale 0 --noise_w 0), que son deterministas: estos segundos se
|
||||
#: reproducen. Se eligieron los extremos del muestreo de 28 líneas — la más
|
||||
#: rápida, la más lenta y las dos más largas — porque son las que rompen un
|
||||
#: modelo mal calibrado; la media la aguanta cualquiera.
|
||||
MEASURED = [
|
||||
("Eight FBI witness interviews. Five digital renderings. All describe the "
|
||||
"same shape flying across America for twenty-four years.", 7.809),
|
||||
("The files are public now, but sections remain blacked out. Witness "
|
||||
"identities. Sensor details. Locations redacted.", 8.140),
|
||||
("Three hundred seventy-eight files released. Hundreds of incidents "
|
||||
"documented. And the government still cannot explain what those shapes "
|
||||
"were.", 7.681),
|
||||
("The files came out. The numbers stayed classified.", 3.310),
|
||||
("Nothing should have been able to hold station beside them up there.", 3.396),
|
||||
("The Air Force's own investigators called it unexplained.", 2.990),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("line,real", MEASURED)
|
||||
def test_the_estimate_lands_within_a_second_of_the_voice(line, real):
|
||||
"""La estimación es lo único que separa un aviso útil de una reescritura
|
||||
inventada, así que se contrasta contra audio medido, no contra sí misma.
|
||||
|
||||
El margen es un segundo. Más apretado sería falso — esto estima, no
|
||||
sintetiza — y más ancho deja de decir nada: el error del modelo anterior
|
||||
sobre un Short entero era de cuatro a seis segundos, y de ahí salían los
|
||||
tres intentos que se gastaban en cada generación.
|
||||
"""
|
||||
from src.generator.spec_contract import spoken_seconds
|
||||
|
||||
assert spoken_seconds(line) == pytest.approx(real, abs=1.0)
|
||||
|
||||
|
||||
def test_a_line_of_short_sentences_is_not_taken_for_fast_prose():
|
||||
"""Piper calla un cuarto de segundo en cada punto. Cuatro frases cortas son
|
||||
un segundo de silencio, y contarlas como texto corrido las da por rápidas:
|
||||
es el caso donde más se equivocaba el modelo de sólo caracteres."""
|
||||
from src.generator.spec_contract import spoken_seconds
|
||||
|
||||
chopped = "The files are public now, but sections remain blacked out. " \
|
||||
"Witness identities. Sensor details. Locations redacted."
|
||||
flowing = "The files are public now but sections remain blacked out with " \
|
||||
"witness identities sensor details and locations redacted"
|
||||
|
||||
assert len(chopped) < len(flowing)
|
||||
assert spoken_seconds(chopped) > spoken_seconds(flowing)
|
||||
|
||||
|
||||
def test_a_decimal_point_is_not_the_end_of_a_sentence():
|
||||
from src.generator.spec_contract import spoken_seconds
|
||||
|
||||
assert spoken_seconds("It climbed to 1.5 miles") == \
|
||||
pytest.approx(spoken_seconds("It climbed to 155 miles"))
|
||||
|
||||
|
||||
def test_the_estimate_counts_the_voice_not_just_the_declared_seconds():
|
||||
"""La duración declarada es un suelo: shortsmith estira el shot si la frase
|
||||
no cabe, y el modelo tiene que enterarse ANTES de pagar el render."""
|
||||
from src.generator.spec_contract import estimated_duration
|
||||
|
||||
doc = spec_with(shot(duration=3.0))
|
||||
doc["shots"][0]["narration"] = "A" * 142 # ~10 s de voz
|
||||
doc["shots"][0]["narration"] = MEASURED[0][0] # 7,81 s de voz medidos
|
||||
|
||||
assert estimated_duration(doc) > 10.0
|
||||
assert estimated_duration(doc) > 8.0
|
||||
|
||||
|
||||
def test_a_shot_with_room_for_its_line_is_estimated_as_declared():
|
||||
@@ -320,6 +378,53 @@ def test_narration_that_overshoots_the_target_is_flagged_as_narration():
|
||||
assert "narración" in note and "estimada" in note
|
||||
|
||||
|
||||
def test_a_second_over_the_target_is_not_worth_a_rewrite():
|
||||
"""El objetivo sigue siendo 45 s, pero la estimación tiene un segundo de
|
||||
error por línea: avisar por medio segundo es avisar del estimador. Caso
|
||||
real — la sesión 168 salió a 45,4 s y se pagó una generación por ello."""
|
||||
from src.generator.spec_contract import TARGET_GRACE, TARGET_MAX_DURATION
|
||||
|
||||
justo = spec_with(shot(duration=TARGET_MAX_DURATION + TARGET_GRACE - 0.1))
|
||||
pasado = spec_with(shot(duration=TARGET_MAX_DURATION + TARGET_GRACE + 0.1))
|
||||
|
||||
assert editorial_notes(justo) == []
|
||||
assert editorial_notes(pasado)
|
||||
# Y el consejo se mide contra el objetivo, no contra el margen: se pide
|
||||
# bajar hasta 45, no hasta 46,5.
|
||||
assert "sobran 1.6s" in editorial_notes(pasado)[0]
|
||||
|
||||
|
||||
def test_the_grace_works_at_both_ends():
|
||||
from src.generator.spec_contract import TARGET_GRACE, TARGET_MIN_DURATION
|
||||
|
||||
assert editorial_notes(spec_with(shot(duration=TARGET_MIN_DURATION
|
||||
- TARGET_GRACE + 0.1))) == []
|
||||
assert editorial_notes(spec_with(shot(duration=TARGET_MIN_DURATION
|
||||
- TARGET_GRACE - 0.1)))
|
||||
|
||||
|
||||
def test_the_advice_says_how_much_to_cut_and_from_where():
|
||||
""""Recorta narración" no dice cuánta, y las tres veces que saltó este aviso
|
||||
el modelo devolvió un spec que seguía pasándose. El exceso va en palabras
|
||||
porque es lo que el modelo escribe, y señalando el plano que más habla."""
|
||||
doc = spec_with(shot(duration=4.0), shot(duration=4.0))
|
||||
doc["shots"][0]["narration"] = "Short line."
|
||||
doc["shots"][1]["narration"] = " ".join(["word"] * 200)
|
||||
|
||||
note = editorial_notes(doc)[0]
|
||||
|
||||
assert "palabras de narración" in note
|
||||
assert "shots.1" in note and "shots.0" not in note
|
||||
|
||||
|
||||
def test_the_advice_for_a_silent_spec_never_mentions_narration():
|
||||
"""Sin voz, pedir que recorte narración es mandarlo a arreglar algo que no
|
||||
existe: lo que sobra son duraciones declaradas."""
|
||||
note = editorial_notes(spec_with(*[shot(duration=10.0) for _ in range(6)]))[0]
|
||||
|
||||
assert "narración" not in note and "duraciones declaradas" in note
|
||||
|
||||
|
||||
def test_a_spec_without_narration_keeps_the_old_wording():
|
||||
note = editorial_notes(spec_with(*[shot(duration=10.0) for _ in range(6)]))[0]
|
||||
assert "duración total" in note and "estimada" not in note
|
||||
|
||||
Reference in New Issue
Block a user