Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 12 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -396,7 +396,18 @@ Portuguese and `es_419` Latin American Spanish; plain `pt` and `es` stay
unqualified, as does `ar`). The optional `ducking` boolean (default off, free) ducks the
background music/effects bed under the dubbed voice while it speaks; when off
the bed is kept at a constant level. (Every endpoint's `ducking` defaults off,
so this one is no exception.) Source videos may be
so this one is no exception.)

The optional `lipsync` boolean is the one parameter here that defaults **on**:
the speaker's mouth is re-rendered to match the dubbed speech. Pass
`lipsync=False` to leave the picture completely untouched instead — the video
comes back at its original resolution and frame rate rather than re-rendered,
and only the audio is replaced, so the mouths keep moving to the original
language. Reach for it on footage with no on-camera speaker, or when
preserving the exact original picture matters more than matching lip movement.
The background bed is rebuilt either way, so `ducking` behaves the same.

Source videos may be
at most 300 seconds long, and billing is per language: a 3-language call
costs three times as much as one. Dubbing has no free trial allowance — see
[Free trial](#free-trial).
Expand Down
5 changes: 5 additions & 0 deletions sonilo-cli/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -277,6 +277,11 @@ command that produces no media file — nothing is generated:
tr, vi, id` (`pt_br` is Brazilian Portuguese and `es_419` Latin American
Spanish; plain `pt` and `es` stay unqualified, as does `ar`).
- Source videos may be at most 300 seconds long.
- `--no-lipsync` leaves the picture completely untouched. By default the speaker's mouth is
re-rendered to match the dubbed speech; with this flag the video comes back at its original
resolution and frame rate and only the audio is replaced, so the mouths keep moving to the
original language. Use it for footage with no on-camera speaker, or when preserving the exact
original picture matters more than matching lip movement.
- `--output` is a filename template, not a single destination: a dubbing task returns one video
per language, so `--output clip.mp4` writes `clip.es.mp4`, `clip.fr.mp4`, etc.
- Billing is per language, and dubbing has **no free trial runs** — see [Free trial](#free-trial)
Expand Down
13 changes: 13 additions & 0 deletions sonilo-cli/src/sonilo_cli/__main__.py
Original file line number Diff line number Diff line change
Expand Up @@ -635,6 +635,9 @@ def cmd_dubbing(client: Sonilo, args: argparse.Namespace) -> None:
video=args.video,
video_url=args.video_url,
languages=languages,
# Only sent when the flag is present, so the server keeps owning the
# default (lip sync on).
lipsync=False if args.no_lipsync else None,
timeout=args.timeout,
)
if not result.outputs:
Expand Down Expand Up @@ -1010,6 +1013,16 @@ def build_parser() -> argparse.ArgumentParser:
"es_419 Latin American Spanish; plain pt and es stay "
"unqualified, as does ar.",
)
p_dub.add_argument(
"--no-lipsync", dest="no_lipsync", action="store_true",
help="Leave the picture completely untouched. By default the speaker's "
"mouth is re-rendered to match the dubbed speech; with this the "
"video comes back at its original resolution and frame rate and "
"only the audio is replaced, so the mouths keep moving to the "
"original language. Use it for footage with no on-camera speaker, "
"or when preserving the exact original picture matters more than "
"matching lip movement.",
)
p_dub.add_argument(
"--output", default=None,
help="Filename template; one file is written per language with the code "
Expand Down
6 changes: 6 additions & 0 deletions src/sonilo/_requests.py
Original file line number Diff line number Diff line change
Expand Up @@ -174,6 +174,7 @@ def build_dubbing_parts(
video_url: Optional[str],
languages: Optional[List[str]],
ducking: Optional[bool] = None,
lipsync: Optional[bool] = None,
) -> Tuple[Dict[str, str], Optional[Dict[str, tuple]], bool]:
"""Build the multipart parts for POST /v1/dubbing.

Expand Down Expand Up @@ -206,6 +207,11 @@ def build_dubbing_parts(
# when unset so the server default applies.
if ducking is not None:
data["ducking"] = "true" if ducking else "false"
# Default-ON server-side, unlike ducking — this is the one parameter here
# whose useful direction is turning it off. Omitted when unset all the
# same, so the server keeps owning the default.
if lipsync is not None:
data["lipsync"] = "true" if lipsync else "false"

# Now open files (only after data is fully assembled)
files: Optional[Dict[str, tuple]] = None
Expand Down
29 changes: 24 additions & 5 deletions src/sonilo/resources/dubbing.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,7 +25,16 @@ class Dubbing:
`ducking` (default off, free) ducks the background music/effects bed
under the dubbed voice while it speaks; when off the bed is kept at a
constant level. Every endpoint's `ducking` defaults off, so this one is
no exception."""
no exception.

`lipsync` is the one parameter here that defaults **on**: the speaker's
mouth is re-rendered to match the dubbed speech. Pass `lipsync=False` to
leave the picture completely untouched instead — the video comes back at
its original resolution and frame rate rather than re-rendered, and only
the audio is replaced, so the mouths keep moving to the original language.
Reach for it on footage with no on-camera speaker, or when preserving the
exact original picture matters more than matching lip movement. The
background bed is rebuilt either way, so `ducking` behaves the same."""

def __init__(self, client: "Sonilo") -> None:
self._client = client
Expand All @@ -37,8 +46,11 @@ def submit(
video_url: Optional[str] = None,
languages: Optional[List[str]] = None,
ducking: Optional[bool] = None,
lipsync: Optional[bool] = None,
) -> SfxTask:
data, files, opened = build_dubbing_parts(video, video_url, languages, ducking)
data, files, opened = build_dubbing_parts(
video, video_url, languages, ducking, lipsync
)
close_after = files["video"][1] if files is not None and opened else None
return parse_sfx_task(
self._client._post_json(PATH, data=data, files=files, close_after=close_after)
Expand All @@ -51,11 +63,13 @@ def generate(
video_url: Optional[str] = None,
languages: Optional[List[str]] = None,
ducking: Optional[bool] = None,
lipsync: Optional[bool] = None,
poll_interval: float = DEFAULT_POLL_INTERVAL,
timeout: float = DEFAULT_WAIT_TIMEOUT,
) -> DubbingResult:
task = self.submit(
video=video, video_url=video_url, languages=languages, ducking=ducking
video=video, video_url=video_url, languages=languages,
ducking=ducking, lipsync=lipsync,
)
return self._client.tasks.wait(
task.task_id,
Expand All @@ -76,8 +90,11 @@ async def submit(
video_url: Optional[str] = None,
languages: Optional[List[str]] = None,
ducking: Optional[bool] = None,
lipsync: Optional[bool] = None,
) -> SfxTask:
data, files, opened = build_dubbing_parts(video, video_url, languages, ducking)
data, files, opened = build_dubbing_parts(
video, video_url, languages, ducking, lipsync
)
close_after = files["video"][1] if files is not None and opened else None
return parse_sfx_task(
await self._client._post_json(
Expand All @@ -92,11 +109,13 @@ async def generate(
video_url: Optional[str] = None,
languages: Optional[List[str]] = None,
ducking: Optional[bool] = None,
lipsync: Optional[bool] = None,
poll_interval: float = DEFAULT_POLL_INTERVAL,
timeout: float = DEFAULT_WAIT_TIMEOUT,
) -> DubbingResult:
task = await self.submit(
video=video, video_url=video_url, languages=languages, ducking=ducking
video=video, video_url=video_url, languages=languages,
ducking=ducking, lipsync=lipsync,
)
return await self._client.tasks.wait(
task.task_id,
Expand Down
18 changes: 18 additions & 0 deletions tests/test_dubbing.py
Original file line number Diff line number Diff line change
Expand Up @@ -119,6 +119,24 @@ def test_submit_sends_ducking_only_when_set():
assert "ducking" not in bodies[2]


@respx.mock
def test_submit_sends_lipsync_only_when_set():
"""The mirror of ducking, with the default the other way up: absent must
mean lip sync ON, which is what every dubbing call did before the
parameter existed."""
route = respx.post("https://api.sonilo.com/v1/dubbing").mock(
return_value=httpx.Response(202, json=ACK)
)
with Sonilo(api_key="sk-test") as client:
client.dubbing.submit(video_url="https://x/v.mp4", lipsync=False)
client.dubbing.submit(video_url="https://x/v.mp4", lipsync=True)
client.dubbing.submit(video_url="https://x/v.mp4")
bodies = [unquote_plus(c.request.content.decode()) for c in route.calls]
assert "lipsync=false" in bodies[0]
assert "lipsync=true" in bodies[1]
assert "lipsync" not in bodies[2]


@respx.mock
def test_generate_polls_to_a_dubbing_result():
respx.post("https://api.sonilo.com/v1/dubbing").mock(
Expand Down
Loading