afa67's picture
deploy dry-run demo
257a670 verified
Raw
History Blame
4.34 kB
"""Dynamic Veo clip duration from spoken word count (SPEC_NEW §11.1, §12.1).
The single source of truth for how long a clip should be and how many words a
clip may hold. The prompt validator (`prompts/veo_segment.py`) and the duration
decision share `count_spoken_words`, so the word cap and the chosen bucket can
never disagree (SPEC_NEW §12.2).
Model (SPEC_NEW §11.1):
WORDS_PER_SECOND = 2.3 Swedish UGC speaking pace
TAIL_BUDGET_S = 0.7 required trailing silence per clip (prompt rule 7)
BUCKETS = 4,6,8 integer seconds Veo 3.1 accepts
Caps (owner-tuned): 4s → ≤9 words · 6s → 10–14 · 8s → 15–18 · >18 → split
These constants mirror the .env defaults WORDS_PER_SECOND / TAIL_BUDGET_S.
Re-pacing is a deliberate owner decision (SPEC_NEW §11.1): change both together.
"""
from __future__ import annotations
import re
import secrets
WORDS_PER_SECOND = 2.3
TAIL_BUDGET_S = 0.7
BUCKETS: tuple[int, int, int] = (4, 6, 8)
# Keep letters (incl. åäö/é), digits, underscore and the apostrophe; everything
# else becomes a separator. Matches the donor's robust counter.
_WORD_RE = re.compile(r"[^\w\s']", re.UNICODE)
# Spoken sentence embedded verbatim in a Veo prompt by prompts/veo_segment.py.
# Used by the RAI retry to re-derive a bucket if it ever needs to read the line
# back out of a prompt (preferred path: read segments.spoken_text directly).
_SPOKEN_RE = re.compile(r"exactly these words and nothing else: \"([^\"]*)\"")
def count_spoken_words(text: str) -> int:
"""Robust spoken-word count. Shared by decide_duration and the validator."""
cleaned = _WORD_RE.sub(" ", text.strip())
return len([w for w in cleaned.split() if w])
# Owner-tuned explicit caps: max spoken words per clip length.
# 4s → ≤9 words · 6s → 10–14 · 8s → 15–18 · >18 → split.
# (Veo's real Swedish delivery is a touch faster than the 2.3 words/s model, so more
# words fit per clip — these caps are the tuned values, not the raw formula.)
BUCKET_MAX_WORDS: dict[int, int] = {4: 9, 6: 14, 8: 18}
def max_words(
duration_s: int,
words_per_second: float = WORDS_PER_SECOND,
tail_budget_s: float = TAIL_BUDGET_S,
) -> int:
"""Max spoken words a clip of this length holds (owner-tuned caps; falls back to
the words/sec model for any non-standard duration)."""
return BUCKET_MAX_WORDS.get(duration_s, int((duration_s - tail_budget_s) * words_per_second))
# The hard cap = capacity of the largest bucket.
MAX_SPOKEN_WORDS = max_words(BUCKETS[-1]) # 18
class SegmentTooLong(ValueError):
"""The spoken line cannot fit the largest clip; the scriptwriter must split it."""
def decide_duration(
text: str,
words_per_second: float = WORDS_PER_SECOND,
tail_budget_s: float = TAIL_BUDGET_S,
) -> int:
"""Smallest bucket whose capacity holds the line (SPEC_NEW §11.1).
≤9 → 4s · 10–14 → 6s · 15–18 → 8s · >18 → SegmentTooLong.
"""
n = count_spoken_words(text)
for d in BUCKETS:
if n <= max_words(d, words_per_second, tail_budget_s):
return d
raise SegmentTooLong(
f"{n} words exceeds the {max_words(BUCKETS[-1], words_per_second, tail_budget_s)}-word "
f"cap for an {BUCKETS[-1]}s clip; split this segment in two: {text!r}"
)
def fits_a_bucket(text: str) -> bool:
try:
decide_duration(text)
return True
except SegmentTooLong:
return False
def next_bucket(duration_s: int) -> int | None:
"""The next-larger bucket, or None if already at the max (SPEC_NEW §12.1
clip-clipping escalation: bump a crammed segment up one bucket on regen)."""
for d in BUCKETS:
if d > duration_s:
return d
return None
def extract_spoken_sentence(prompt: str) -> str | None:
"""Pull the verbatim Swedish line back out of a built Veo prompt.
NOTE: this regex matches prompts/veo_segment.py's wording, not the donor's.
Prefer reading segments.spoken_text directly (SPEC_NEW §11.1).
"""
m = _SPOKEN_RE.search(prompt)
return m.group(1) if m else None
def random_seed() -> int:
"""A positive seed that fits a signed Postgres bigint / int32 Veo seed.
31 bits keeps it valid as both a Postgres integer and a Vertex seed.
"""
return secrets.randbits(31)