Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 8 additions & 5 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -97,11 +97,14 @@ schema version, and config path. **Open Voxtype configuration** launches
writes every setting itself.

For each confirmed dictation, Doubao Say asks Voxtype to write one transcript in
a private per-user runtime directory, waits for the `.done` completion signal,
reads the atomic final text, and deletes both files. It refuses to take over when
Voxtype is already recording or transcribing. Cancel only targets a recording
started by this app. Voxtype 1.0.1 and newer can suppress its own OSD for these
sessions; 1.0.0 remains compatible but may show both overlays.
a private per-user runtime directory. It reads the completed file and removes
it; Voxtype 1.0.x also publishes a `.done` completion record. Voxtype 1.1
streaming sessions are accepted and their final transcript can be polished in
Doubao Say. Voxtype does not expose interim streaming text through its file or
status interface, so Doubao Say's overlay still shows only the final text.
The app refuses to take over an existing recording, and cancellation only
targets one it started. Voxtype 1.0.1 and newer can suppress its own OSD for
these sessions; 1.0.0 remains compatible but may show both overlays.

Voxtype currently returns only the final transcript through this integration.
Its microphone audio and live level meter are not exposed to Doubao Say, so the
Expand Down
10 changes: 6 additions & 4 deletions README.zh-CN.md
Original file line number Diff line number Diff line change
Expand Up @@ -71,10 +71,12 @@ Voxtype,运行 `voxtype setup --download`,启动 daemon,再在首次设置
模型、音频设备、计算后端、Schema 版本和配置文件路径。点击**打开 Voxtype 配置**后,
应用会在可用终端里启动 `voxtype configure`,所有配置仍由 Voxtype 自己校验和写入。

每次听写手势确认后,豆包说会让 Voxtype 把一份转写写入当前用户的私有 runtime 目录,
等待 `.done` 完成信号,读取原子写入的最终文字,然后删除这两个文件。若 Voxtype 已在
录音或转写,豆包说不会接管;取消操作也只针对本应用启动的录音。Voxtype 1.0.1 及更高
版本可以在本次会话中关闭自带 OSD;1.0.0 也能使用,但可能同时显示两套悬浮窗。
每次听写手势确认后,豆包说会让 Voxtype 把转写写入当前用户的私有 runtime 目录,
读取完成后的文字并清理文件。Voxtype 1.0.x 还会写入 `.done` 完成记录。Voxtype 1.1
的流式会话也能交回最终文字,继续使用豆包说的润色功能;Voxtype 的文件和状态接口目前
不提供录音中的部分文字,因此豆包说悬浮窗仍只显示最终结果。若 Voxtype 已在录音或
转写,豆包说不会接管;取消操作也只针对本应用启动的录音。Voxtype 1.0.1 及更高版本
可以在本次会话中关闭自带 OSD;1.0.0 也能使用,但可能同时显示两套悬浮窗。

当前接入只能取得最终转写,Voxtype 没有把麦克风音频和实时音量交给豆包说,因此本地
预录音和实时波形不可用。使用长按说话时,请看到“正在聆听”后再开口,以免句首被截断。
Expand Down
16 changes: 13 additions & 3 deletions src/doubao_input/voxtype/asr_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -135,7 +135,7 @@ def _start_recording(self, session) -> None:
session.runtime,
runner=lambda command: self._runner(command, timeout=2),
)
if state == "recording":
if state in ("recording", "streaming"):
break
if state != "idle":
raise RuntimeError(f"Voxtype entered unexpected state: {state}")
Expand Down Expand Up @@ -183,17 +183,27 @@ def _finish_recording(self, session) -> None:
if not session.owns_recording:
return
try:
result = self._runner([
command = [
session.runtime.executable,
"record",
"stop",
"--wait",
"--timeout",
"30",
], timeout=35)
]
if session.runtime.supports_wait_file:
command.extend(("--wait-file", str(session.transcript_path)))
result = self._runner(command, timeout=35)
if result.returncode == 3:
# Voxtype 1.1 streaming file mode writes the final transcript
# but does not publish the completion sidecar expected by
# `record stop --wait`, which then reports an empty outcome.
text = (_read_transcript(session.transcript_path)
if session.transcript_path.exists() else "")
self._release_recording(session)
_cleanup(session.transcript_path)
if text:
self._emit(session, "on_result", text)
self._emit(session, "on_finish")
return
if result.returncode == 4:
Expand Down
2 changes: 2 additions & 0 deletions src/doubao_input/voxtype/runtime.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@ class VoxtypeRuntime:
executable: str
version: str
supports_no_osd: bool = False
supports_wait_file: bool = False


def _run(command, *, timeout=2):
Expand Down Expand Up @@ -50,6 +51,7 @@ def inspect_runtime(*, which=shutil.which, runner=_run) -> VoxtypeRuntime:
executable=executable,
version=".".join(str(value) for value in version_tuple),
supports_no_osd="--no-osd" in start_help.stdout,
supports_wait_file="--wait-file" in stop_help.stdout,
)


Expand Down
32 changes: 27 additions & 5 deletions tests/unit/test_voxtype_asr_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,25 +11,28 @@


class FakeCommands:
def __init__(self, transcript="local transcript", stop_code=0):
def __init__(self, transcript="local transcript", stop_code=0,
streaming=False):
self.transcript = transcript
self.stop_code = stop_code
self.streaming = streaming
self.commands = []
self.path = None

def __call__(self, command, *, timeout):
self.commands.append(command)
if command[1] == "status":
state = "recording" if self.path else "idle"
state = ("streaming" if self.streaming else "recording") if self.path else "idle"
return SimpleNamespace(returncode=0, stdout=f'{{"alt":"{state}"}}', stderr="")
if command[1:3] == ["record", "start"]:
file_arg = next(value for value in command if value.startswith("--file="))
self.path = Path(file_arg.removeprefix("--file="))
return SimpleNamespace(returncode=0, stdout="", stderr="")
if command[1:3] == ["record", "stop"]:
if self.stop_code == 0:
if self.stop_code == 0 or self.streaming:
self.path.write_text(self.transcript)
Path(f"{self.path}.done").write_text("ok")
if not self.streaming:
Path(f"{self.path}.done").write_text("ok")
return SimpleNamespace(
returncode=self.stop_code, stdout="", stderr="")
return SimpleNamespace(returncode=0, stdout="", stderr="")
Expand All @@ -52,7 +55,8 @@ def client(self, commands, state="idle"):
client = VoxtypeASRClient(
commands,
lambda _runtime, **_kwargs: (
state if state != "idle" or commands.path is None else "recording"
state if state != "idle" or commands.path is None
else ("streaming" if commands.streaming else "recording")
),
)
self.addCleanup(client.disconnect)
Expand Down Expand Up @@ -128,6 +132,24 @@ def test_empty_completion_finishes_without_result(self):
self.assertTrue(finished.wait(2))
result.assert_not_called()

def test_streaming_file_output_delivers_final_without_sidecar(self):
commands = FakeCommands(transcript="streamed final", stop_code=3,
streaming=True)
client = self.client(commands)
opened, finished = threading.Event(), threading.Event()
results, errors = [], []
client.on_open = opened.set
client.on_result = results.append
client.on_finish = finished.set
client.on_error = errors.append
client.connect(VoxtypeRuntime("/usr/bin/voxtype", "1.1.0", True, True))
self.assertTrue(opened.wait(2), errors)
client.finish_sending()
self.assertTrue(finished.wait(2), errors)
self.assertEqual(results, ["streamed final"])
self.assertIn("--wait-file", commands.commands[1])
self.assertFalse(commands.path.exists())

def test_does_not_take_over_an_existing_voxtype_session(self):
commands = FakeCommands()
client = self.client(commands, state="recording")
Expand Down
15 changes: 15 additions & 0 deletions tests/unit/test_voxtype_runtime.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,21 @@ def run(command):
runtime = inspect_runtime(which=lambda _name: "/usr/bin/voxtype", runner=run)
self.assertEqual(runtime.version, "1.0.1")
self.assertTrue(runtime.supports_no_osd)
self.assertFalse(runtime.supports_wait_file)

def test_detects_explicit_wait_file_support(self):
def run(command):
if command[-1] == "--version":
value = "voxtype 1.1.0\n"
elif command[2] == "start":
value = "--file PATH\n--no-osd\n"
else:
value = "--wait\n--timeout SECONDS\n--wait-file FILE\n"
return SimpleNamespace(returncode=0, stdout=value)

runtime = inspect_runtime(which=lambda _name: "/usr/bin/voxtype",
runner=run)
self.assertTrue(runtime.supports_wait_file)

def test_rejects_old_or_incomplete_cli(self):
def old(command):
Expand Down