Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .github/workflows/midscene-omarchy-4.0.3.yml
Original file line number Diff line number Diff line change
Expand Up @@ -50,6 +50,7 @@ on:
- omarchy-shard-4
- omarchy-shell
- omarchy-providers
- omarchy-voxtype-settings

permissions:
contents: read
Expand Down
1 change: 1 addition & 0 deletions .github/workflows/midscene-ubuntu-22.04.yml
Original file line number Diff line number Diff line change
Expand Up @@ -41,6 +41,7 @@ on:
- ubuntu-shard-4
- ubuntu-polishing
- ubuntu-providers
- ubuntu-voxtype-settings

permissions:
contents: read
Expand Down
5 changes: 5 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -87,6 +87,11 @@ file-mode completion API. Install Voxtype separately, run
`voxtype setup --download`, start its daemon, then use **Refresh Voxtype status**
in onboarding. The engine, model, language, audio device, and acceleration remain
configured in Voxtype; this integration does not rewrite its configuration.
The Settings window reads Voxtype's versioned schema and extended status to show
the running daemon version, state, engine, model, audio device, compute backend,
schema version, and config path. **Open Voxtype configuration** launches
`voxtype configure` in an available terminal, so Voxtype still validates and
writes every setting itself.

For each confirmed dictation, Doubao Say asks Voxtype to write one transcript in
a private per-user runtime directory, waits for the `.done` completion signal,
Expand Down
3 changes: 3 additions & 0 deletions README.zh-CN.md
Original file line number Diff line number Diff line change
Expand Up @@ -66,6 +66,9 @@ API Key 保存在 `~/.config/doubao-say/deepgram_api_key`(或对应的
Voxtype,运行 `voxtype setup --download`,启动 daemon,再在首次设置中点击
**刷新 Voxtype 状态**。引擎、模型、语言、音频设备和加速方式仍在 Voxtype 中配置;
该接入不会改写 Voxtype 配置。
设置窗口会读取 Voxtype 的版本化 Schema 和扩展状态,显示 daemon 版本、状态、引擎、
模型、音频设备、计算后端、Schema 版本和配置文件路径。点击**打开 Voxtype 配置**后,
应用会在可用终端里启动 `voxtype configure`,所有配置仍由 Voxtype 自己校验和写入。

每次听写手势确认后,豆包说会让 Voxtype 把一份转写写入当前用户的私有 runtime 目录,
等待 `.done` 完成信号,读取原子写入的最终文字,然后删除这两个文件。若 Voxtype 已在
Expand Down
15 changes: 15 additions & 0 deletions src/doubao_input/app.py
Original file line number Diff line number Diff line change
Expand Up @@ -854,11 +854,26 @@ def _show_settings(self):
preview=self._preview_overlay, apply_key=self._apply_trigger_key,
asr_has_key=self._api_key_has_saved, save_asr=self._save_asr_key,
clear_asr=self._clear_asr_key, test_asr=self._test_asr_key,
voxtype_details=self._voxtype_details,
configure_voxtype=self._configure_voxtype,
diagnostic_report=lambda: diagnostic_report(
self.settings, recording=self.app_state.is_recording,
trace=self._diagnostics))
self._settings_window.show()

@staticmethod
def _voxtype_details():
from doubao_input.voxtype.control import inspect_details
return inspect_details()

@staticmethod
def _configure_voxtype():
from doubao_input.voxtype.control import launch_configure
try:
launch_configure()
except RuntimeError as error:
raise ValueError(str(error)) from error

def _preview_overlay(self):
if self._busy():
raise ValueError(tr("Finish the current operation first.", "请先结束当前操作。"))
Expand Down
84 changes: 84 additions & 0 deletions src/doubao_input/ui/settings_window.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,7 @@
"""Grouped preferences, safe hardware-key capture and allowlisted diagnostics."""
from dataclasses import replace
import subprocess
import threading
from gi.repository import GLib, Gtk, Gdk
from doubao_input.ui.trigger_picker import TriggerPicker
from doubao_input.i18n import LANGUAGES, tr
Expand All @@ -22,6 +24,8 @@ def __init__(self, parent, settings, apply, *, capture_key=None,
apply_key=None, asr_has_key=lambda: False,
save_asr=lambda key: None, clear_asr=lambda: None,
test_asr=lambda key, completed: None,
voxtype_details=lambda: None,
configure_voxtype=lambda: None,
diagnostic_report=None):
self.window = Gtk.Window(title=tr("Doubao Say Settings", "豆包说设置"), transient_for=parent, modal=True)
self.window.set_default_size(540, 580)
Expand All @@ -35,6 +39,9 @@ def __init__(self, parent, settings, apply, *, capture_key=None,
self._save_asr = save_asr
self._clear_asr = clear_asr
self._test_asr = test_asr
self._voxtype_details = voxtype_details
self._configure_voxtype = configure_voxtype
self._voxtype_request_id = 0
self._asr_save_source = 0
self._asr_testing = False
self._diagnostic_report = diagnostic_report or (lambda: report(self._settings))
Expand Down Expand Up @@ -141,6 +148,25 @@ def run(*_):
self.asr_details.append(self.asr_status)
self.asr_details.set_visible(selected_provider.uses_api_key)
box.append(self.asr_details)
self.voxtype_details = Gtk.Box(
orientation=Gtk.Orientation.VERTICAL, spacing=8)
self.voxtype_summary = Gtk.Label(
xalign=0, wrap=True, selectable=True)
self.voxtype_details.append(self.voxtype_summary)
voxtype_actions = Gtk.Box(spacing=8, homogeneous=True)
self.voxtype_refresh = Gtk.Button(label=tr(
"Refresh Voxtype status", "刷新 Voxtype 状态"))
self.voxtype_refresh.connect(
"clicked", self._refresh_voxtype_details)
self.voxtype_configure = Gtk.Button(label=tr(
"Open Voxtype configuration", "打开 Voxtype 配置"))
self.voxtype_configure.connect(
"clicked", self._open_voxtype_configuration)
voxtype_actions.append(self.voxtype_refresh)
voxtype_actions.append(self.voxtype_configure)
self.voxtype_details.append(voxtype_actions)
self.voxtype_details.set_visible(selected_provider.is_local)
box.append(self.voxtype_details)

section(tr("Input", "输入"))
self.input_method = Gtk.DropDown.new_from_strings([
Expand Down Expand Up @@ -468,8 +494,66 @@ def _sync_provider_details(self):
self.clear_credentials_button.set_visible(provider.interactive_auth)
self.provider_help.set_text(provider.credential_help)
self.provider_help.set_visible(bool(provider.credential_help))
self.voxtype_details.set_visible(provider.is_local)
if provider.is_local:
self._refresh_voxtype_details()
else:
self._voxtype_request_id += 1
self.voxtype_refresh.set_sensitive(True)
self.privacy_copy.set_text(provider.privacy)

def _refresh_voxtype_details(self, *_):
self._voxtype_request_id += 1
request_id = self._voxtype_request_id
self.voxtype_refresh.set_sensitive(False)
self.voxtype_summary.set_text(tr(
"Reading Voxtype status…", "正在读取 Voxtype 状态…"))

def read_details():
try:
details = self._voxtype_details()
if details is None:
raise RuntimeError("Voxtype details are unavailable")
error = None
except (OSError, RuntimeError, subprocess.SubprocessError) as exc:
details, error = None, str(exc)
GLib.idle_add(self._show_voxtype_details, request_id, details, error)

threading.Thread(target=read_details, name="voxtype-status",
daemon=True).start()

def _show_voxtype_details(self, request_id, details, error):
if request_id != self._voxtype_request_id:
return GLib.SOURCE_REMOVE
self.voxtype_refresh.set_sensitive(True)
if error:
self.voxtype_summary.set_text(tr(
"Voxtype status unavailable: ",
"无法读取 Voxtype 状态:") + error)
return GLib.SOURCE_REMOVE
self.voxtype_summary.set_text("\n".join((
tr("CLI version: ", "CLI 版本:") + details.cli_version,
tr("Daemon version: ", "Daemon 版本:") + details.daemon_version,
tr("State: ", "状态:") + details.state,
tr("Engine: ", "引擎:") + details.engine,
tr("Model: ", "模型:") + details.model,
tr("Audio device: ", "音频设备:") + details.device,
tr("Compute backend: ", "计算后端:") + details.backend,
tr("Config schema: ", "配置 Schema:") + str(details.schema_version),
tr("Config file: ", "配置文件:") + details.config_path,
)))
return GLib.SOURCE_REMOVE

def _open_voxtype_configuration(self, *_):
try:
self._configure_voxtype()
except (OSError, ValueError) as error:
self.voxtype_summary.set_text(str(error))
return
self.voxtype_summary.set_text(tr(
"Voxtype configuration opened in a terminal.",
"已在终端中打开 Voxtype 配置。"))

def _queue_asr_key_save(self, *_):
if self._updating or not self.asr_key.get_text().strip():
return
Expand Down
119 changes: 119 additions & 0 deletions src/doubao_input/voxtype/control.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,119 @@
"""Read Voxtype's stable GUI-facing contracts and open its own configurator."""

from __future__ import annotations

from dataclasses import dataclass
import json
import os
import shlex
import shutil
import subprocess

from doubao_input.voxtype.runtime import VoxtypeRuntime, inspect_runtime


def _run(command, *, timeout=3):
return subprocess.run(
command,
capture_output=True,
text=True,
timeout=timeout,
check=False,
)


@dataclass(frozen=True)
class VoxtypeDetails:
cli_version: str
daemon_version: str
state: str
engine: str
model: str
device: str
backend: str
schema_version: int | str
config_path: str


def inspect_details(*, runtime=None, runner=_run) -> VoxtypeDetails:
runtime = runtime or inspect_runtime(runner=runner)
status = _json_command([
runtime.executable, "status", "--extended", "--format", "json",
], runner)
state = status.get("alt") or status.get("class")
if state == "stopped":
raise RuntimeError("Voxtype daemon is not running")
schema = _json_command([
runtime.executable, "config", "schema", "--json",
], runner)
return VoxtypeDetails(
cli_version=str(schema.get("voxtype_version") or runtime.version),
daemon_version=str(schema.get("daemon_version_label") or "unknown"),
state=str(state or "unknown"),
engine=str(schema.get("engine") or "unknown"),
model=str(status.get("model") or "unknown"),
device=str(status.get("device") or "system default"),
backend=str(status.get("backend") or "unknown"),
schema_version=schema.get("schema_version", "unknown"),
config_path=str(schema.get("config_path") or "unknown"),
)


def _json_command(command, runner):
result = runner(command)
if result.returncode:
message = (result.stderr or result.stdout or "").strip()
raise RuntimeError(message[:500] or "Voxtype command failed")
try:
payload = json.loads(result.stdout)
except (TypeError, ValueError) as error:
raise RuntimeError("Voxtype returned invalid JSON") from error
if not isinstance(payload, dict):
raise RuntimeError("Voxtype returned invalid JSON")
return payload


def launch_configure(*, runtime: VoxtypeRuntime | None = None,
which=shutil.which, popen=subprocess.Popen,
environment=None) -> None:
runtime = runtime or inspect_runtime()
if environment is None:
environment = os.environ
terminal = environment.get("TERMINAL", "").strip()
if terminal:
prefix = shlex.split(terminal)
executable = which(prefix[0]) if prefix else None
if executable:
_launch([
executable, *prefix[1:], "-e",
runtime.executable, "configure",
], popen)
return
candidates = (
("kitty", "-e"),
("alacritty", "-e"),
("foot", "-e"),
("wezterm", "start", "--"),
("gnome-terminal", "--"),
("konsole", "-e"),
("xfce4-terminal", "-e"),
("xterm", "-e"),
)
for candidate in candidates:
executable = which(candidate[0])
if executable:
_launch([
executable, *candidate[1:], runtime.executable, "configure",
], popen)
return
raise RuntimeError("No supported terminal was found for Voxtype configure")


def _launch(command, popen):
popen(
command,
stdin=subprocess.DEVNULL,
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
start_new_session=True,
)
4 changes: 2 additions & 2 deletions tests/contracts/test_pages_report_history.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -124,9 +124,9 @@ test('assigns every product case to exactly one balanced shard', async () => {
caseCount += 1;
}
}
assert.equal(caseCount, 16);
assert.equal(caseCount, 17);
assert.deepEqual(Object.fromEntries(shardCounts), {
'shard-1': 2,
'shard-1': 3,
'shard-2': 3,
'shard-3': 7,
'shard-4': 4,
Expand Down
10 changes: 10 additions & 0 deletions tests/e2e/cases/onboarding-regressions.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -213,6 +213,16 @@ cases:
- aiTap: Finish & check result button in the Voice test content panel
- wait: { duration: 1250 }
- aiAssert: The voice test shows Synthetic Voxtype English transcript, the Listening overlay is gone, and feedback says the synthetic Voxtype CLI completed with transcript files removed and nothing pasted or sent.
- name: Inspect Voxtype configuration in Settings
tags: [shard-1, provider, voxtype-settings]
steps:
- fixture.prepare: { mode: voxtype-settings }
- aiAssert: The Doubao Say Settings window is open with the Voxtype status and action buttons visible.
- aiAssert: The Voxtype status shows CLI version 1.0.1, daemon version 1.0.1, idle state, whisper engine, synthetic-english-model, CI-only synthetic microphone, CPU backend, and config schema 1. The API Key field is hidden.
- aiTap: Refresh Voxtype status button in the Settings window
- aiAssert: The Voxtype status still shows idle and synthetic-english-model, and Open Voxtype configuration is available.
- aiTap: Open Voxtype configuration button in the Settings window
- aiAssert: The Settings window says Voxtype configuration opened in a terminal.
- name: Changing microphone requires another check
tags: [shard-2]
steps:
Expand Down
2 changes: 1 addition & 1 deletion tests/e2e/gtk_fixture.py
Original file line number Diff line number Diff line change
Expand Up @@ -36,7 +36,7 @@ def main():
Gtk.init()
set_language("en")
mode = fixture_mode()
if mode == "voxtype-live":
if mode in {"voxtype-live", "voxtype-settings"}:
fake_cli = Path(__file__).parent / "fakes"
os.environ["PATH"] = str(fake_cli) + os.pathsep + os.environ["PATH"]
cleanup = (
Expand Down
Loading