diff --git a/README.md b/README.md index c54cb59..cc930a6 100644 --- a/README.md +++ b/README.md @@ -97,11 +97,14 @@ schema version, and config path. **Open Voxtype configuration** launches writes every setting itself. For each confirmed dictation, Doubao Say asks Voxtype to write one transcript in -a private per-user runtime directory, waits for the `.done` completion signal, -reads the atomic final text, and deletes both files. It refuses to take over when -Voxtype is already recording or transcribing. Cancel only targets a recording -started by this app. Voxtype 1.0.1 and newer can suppress its own OSD for these -sessions; 1.0.0 remains compatible but may show both overlays. +a private per-user runtime directory. It reads the completed file and removes +it; Voxtype 1.0.x also publishes a `.done` completion record. Voxtype 1.1 +streaming sessions are accepted and their final transcript can be polished in +Doubao Say. Voxtype does not expose interim streaming text through its file or +status interface, so Doubao Say's overlay still shows only the final text. +The app refuses to take over an existing recording, and cancellation only +targets one it started. Voxtype 1.0.1 and newer can suppress its own OSD for +these sessions; 1.0.0 remains compatible but may show both overlays. Voxtype currently returns only the final transcript through this integration. Its microphone audio and live level meter are not exposed to Doubao Say, so the diff --git a/README.zh-CN.md b/README.zh-CN.md index 7d7cdfe..5549b77 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -71,10 +71,12 @@ Voxtype,运行 `voxtype setup --download`,启动 daemon,再在首次设置 模型、音频设备、计算后端、Schema 版本和配置文件路径。点击**打开 Voxtype 配置**后, 应用会在可用终端里启动 `voxtype configure`,所有配置仍由 Voxtype 自己校验和写入。 -每次听写手势确认后,豆包说会让 Voxtype 把一份转写写入当前用户的私有 runtime 目录, -等待 `.done` 完成信号,读取原子写入的最终文字,然后删除这两个文件。若 Voxtype 已在 -录音或转写,豆包说不会接管;取消操作也只针对本应用启动的录音。Voxtype 1.0.1 及更高 -版本可以在本次会话中关闭自带 OSD;1.0.0 也能使用,但可能同时显示两套悬浮窗。 +每次听写手势确认后,豆包说会让 Voxtype 把转写写入当前用户的私有 runtime 目录, +读取完成后的文字并清理文件。Voxtype 1.0.x 还会写入 `.done` 完成记录。Voxtype 1.1 +的流式会话也能交回最终文字,继续使用豆包说的润色功能;Voxtype 的文件和状态接口目前 +不提供录音中的部分文字,因此豆包说悬浮窗仍只显示最终结果。若 Voxtype 已在录音或 +转写,豆包说不会接管;取消操作也只针对本应用启动的录音。Voxtype 1.0.1 及更高版本 +可以在本次会话中关闭自带 OSD;1.0.0 也能使用,但可能同时显示两套悬浮窗。 当前接入只能取得最终转写,Voxtype 没有把麦克风音频和实时音量交给豆包说,因此本地 预录音和实时波形不可用。使用长按说话时,请看到“正在聆听”后再开口,以免句首被截断。 diff --git a/src/doubao_input/voxtype/asr_client.py b/src/doubao_input/voxtype/asr_client.py index fe093ca..f30f375 100644 --- a/src/doubao_input/voxtype/asr_client.py +++ b/src/doubao_input/voxtype/asr_client.py @@ -135,7 +135,7 @@ def _start_recording(self, session) -> None: session.runtime, runner=lambda command: self._runner(command, timeout=2), ) - if state == "recording": + if state in ("recording", "streaming"): break if state != "idle": raise RuntimeError(f"Voxtype entered unexpected state: {state}") @@ -183,17 +183,27 @@ def _finish_recording(self, session) -> None: if not session.owns_recording: return try: - result = self._runner([ + command = [ session.runtime.executable, "record", "stop", "--wait", "--timeout", "30", - ], timeout=35) + ] + if session.runtime.supports_wait_file: + command.extend(("--wait-file", str(session.transcript_path))) + result = self._runner(command, timeout=35) if result.returncode == 3: + # Voxtype 1.1 streaming file mode writes the final transcript + # but does not publish the completion sidecar expected by + # `record stop --wait`, which then reports an empty outcome. + text = (_read_transcript(session.transcript_path) + if session.transcript_path.exists() else "") self._release_recording(session) _cleanup(session.transcript_path) + if text: + self._emit(session, "on_result", text) self._emit(session, "on_finish") return if result.returncode == 4: diff --git a/src/doubao_input/voxtype/runtime.py b/src/doubao_input/voxtype/runtime.py index ae0439a..75e4ba2 100644 --- a/src/doubao_input/voxtype/runtime.py +++ b/src/doubao_input/voxtype/runtime.py @@ -17,6 +17,7 @@ class VoxtypeRuntime: executable: str version: str supports_no_osd: bool = False + supports_wait_file: bool = False def _run(command, *, timeout=2): @@ -50,6 +51,7 @@ def inspect_runtime(*, which=shutil.which, runner=_run) -> VoxtypeRuntime: executable=executable, version=".".join(str(value) for value in version_tuple), supports_no_osd="--no-osd" in start_help.stdout, + supports_wait_file="--wait-file" in stop_help.stdout, ) diff --git a/tests/unit/test_voxtype_asr_client.py b/tests/unit/test_voxtype_asr_client.py index e84fe00..23ce9fa 100644 --- a/tests/unit/test_voxtype_asr_client.py +++ b/tests/unit/test_voxtype_asr_client.py @@ -11,25 +11,28 @@ class FakeCommands: - def __init__(self, transcript="local transcript", stop_code=0): + def __init__(self, transcript="local transcript", stop_code=0, + streaming=False): self.transcript = transcript self.stop_code = stop_code + self.streaming = streaming self.commands = [] self.path = None def __call__(self, command, *, timeout): self.commands.append(command) if command[1] == "status": - state = "recording" if self.path else "idle" + state = ("streaming" if self.streaming else "recording") if self.path else "idle" return SimpleNamespace(returncode=0, stdout=f'{{"alt":"{state}"}}', stderr="") if command[1:3] == ["record", "start"]: file_arg = next(value for value in command if value.startswith("--file=")) self.path = Path(file_arg.removeprefix("--file=")) return SimpleNamespace(returncode=0, stdout="", stderr="") if command[1:3] == ["record", "stop"]: - if self.stop_code == 0: + if self.stop_code == 0 or self.streaming: self.path.write_text(self.transcript) - Path(f"{self.path}.done").write_text("ok") + if not self.streaming: + Path(f"{self.path}.done").write_text("ok") return SimpleNamespace( returncode=self.stop_code, stdout="", stderr="") return SimpleNamespace(returncode=0, stdout="", stderr="") @@ -52,7 +55,8 @@ def client(self, commands, state="idle"): client = VoxtypeASRClient( commands, lambda _runtime, **_kwargs: ( - state if state != "idle" or commands.path is None else "recording" + state if state != "idle" or commands.path is None + else ("streaming" if commands.streaming else "recording") ), ) self.addCleanup(client.disconnect) @@ -128,6 +132,24 @@ def test_empty_completion_finishes_without_result(self): self.assertTrue(finished.wait(2)) result.assert_not_called() + def test_streaming_file_output_delivers_final_without_sidecar(self): + commands = FakeCommands(transcript="streamed final", stop_code=3, + streaming=True) + client = self.client(commands) + opened, finished = threading.Event(), threading.Event() + results, errors = [], [] + client.on_open = opened.set + client.on_result = results.append + client.on_finish = finished.set + client.on_error = errors.append + client.connect(VoxtypeRuntime("/usr/bin/voxtype", "1.1.0", True, True)) + self.assertTrue(opened.wait(2), errors) + client.finish_sending() + self.assertTrue(finished.wait(2), errors) + self.assertEqual(results, ["streamed final"]) + self.assertIn("--wait-file", commands.commands[1]) + self.assertFalse(commands.path.exists()) + def test_does_not_take_over_an_existing_voxtype_session(self): commands = FakeCommands() client = self.client(commands, state="recording") diff --git a/tests/unit/test_voxtype_runtime.py b/tests/unit/test_voxtype_runtime.py index 47c9e4a..d3da072 100644 --- a/tests/unit/test_voxtype_runtime.py +++ b/tests/unit/test_voxtype_runtime.py @@ -18,6 +18,21 @@ def run(command): runtime = inspect_runtime(which=lambda _name: "/usr/bin/voxtype", runner=run) self.assertEqual(runtime.version, "1.0.1") self.assertTrue(runtime.supports_no_osd) + self.assertFalse(runtime.supports_wait_file) + + def test_detects_explicit_wait_file_support(self): + def run(command): + if command[-1] == "--version": + value = "voxtype 1.1.0\n" + elif command[2] == "start": + value = "--file PATH\n--no-osd\n" + else: + value = "--wait\n--timeout SECONDS\n--wait-file FILE\n" + return SimpleNamespace(returncode=0, stdout=value) + + runtime = inspect_runtime(which=lambda _name: "/usr/bin/voxtype", + runner=run) + self.assertTrue(runtime.supports_wait_file) def test_rejects_old_or_incomplete_cli(self): def old(command):