diff --git a/linux/blitztext/config.py b/linux/blitztext/config.py index bc11962..04164ee 100644 --- a/linux/blitztext/config.py +++ b/linux/blitztext/config.py @@ -17,6 +17,8 @@ class Workflow: hotkey: str mode: str # "transcribe" | "rewrite" prompt: str = "" + # Spoken trigger phrases for voice routing (matched at start/end of speech). + keywords: list[str] = field(default_factory=list) # Optional per-workflow overrides of the [rewrite] defaults. model: str | None = None temperature: float | None = None @@ -44,9 +46,30 @@ class Config: rewrite_model: str = "gpt-4o-mini" temperature: float = 0.3 timeout: int = 45 + # voice-keyword routing + routing_enabled: bool = True + routing_hotkey: str = "++" + routing_default: str = "" # preset name used when no keyword matches; "" = first + routing_threshold: float = 0.82 # workflows workflows: list[Workflow] = field(default_factory=list) + def preset_by_name(self, name: str | None) -> "Workflow | None": + if not name: + return None + return next((w for w in self.workflows if w.name == name), None) + + @property + def default_preset(self) -> "Workflow | None": + return self.preset_by_name(self.routing_default) or (self.workflows[0] if self.workflows else None) + + @property + def all_keywords(self) -> list[str]: + out: list[str] = [] + for w in self.workflows: + out.extend(w.keywords) + return out + @property def api_key(self) -> str | None: return os.environ.get(self.api_key_env) or None @@ -63,6 +86,7 @@ def load(path: Path = CONFIG_PATH) -> Config: g = data.get("general", {}) w = data.get("whisper", {}) r = data.get("rewrite", {}) + rt = data.get("routing", {}) cfg = Config( recorder=g.get("recorder", "auto"), @@ -79,15 +103,20 @@ def load(path: Path = CONFIG_PATH) -> Config: rewrite_model=r.get("model", "gpt-4o-mini"), temperature=float(r.get("temperature", 0.3)), timeout=int(r.get("timeout", 45)), + routing_enabled=bool(rt.get("enabled", True)), + routing_hotkey=rt.get("hotkey", "++"), + routing_default=rt.get("default", ""), + routing_threshold=float(rt.get("threshold", 0.82)), ) for entry in data.get("workflow", []): cfg.workflows.append( Workflow( name=entry["name"], - hotkey=entry["hotkey"], + hotkey=entry.get("hotkey", ""), mode=entry.get("mode", "transcribe"), prompt=entry.get("prompt", ""), + keywords=list(entry.get("keywords", [])), model=entry.get("model"), temperature=entry.get("temperature"), description=entry.get("description", ""), @@ -129,10 +158,18 @@ def save(cfg: Config, path: Path = CONFIG_PATH) -> None: "temperature": cfg.temperature, "timeout": cfg.timeout, }, + "routing": { + "enabled": cfg.routing_enabled, + "hotkey": cfg.routing_hotkey, + "default": cfg.routing_default, + "threshold": cfg.routing_threshold, + }, "workflow": [], } for wf in cfg.workflows: entry: dict = {"name": wf.name, "hotkey": wf.hotkey, "mode": wf.mode} + if wf.keywords: + entry["keywords"] = wf.keywords if wf.prompt: entry["prompt"] = wf.prompt if wf.model: @@ -189,25 +226,35 @@ model = "gpt-4o-mini" # default model for rewrite workflows temperature = 0.3 timeout = 45 +[routing] +# Voice-keyword routing: ONE hotkey to dictate. Say a preset's keyword at the +# START or END of your speech and that preset is applied; otherwise the default +# preset is used. (Per-preset hotkeys below still work as direct shortcuts.) +enabled = true +hotkey = "++" +default = "Transcribe" # preset used when no keyword is recognised +threshold = 0.82 # 0..1 fuzzy-match strictness (higher = stricter) + # ---------------------------------------------------------------------------- -# Workflows. mode = "transcribe" types the raw transcript. mode = "rewrite" -# sends the transcript through the LLM with `prompt` as the system prompt. -# Any workflow may override the [rewrite] defaults with its own: -# model = "gpt-4o" -# temperature = 0.4 +# Workflows / presets. mode = "transcribe" types the raw transcript. mode = +# "rewrite" sends it through the LLM with `prompt` as the system prompt. +# keywords = spoken trigger phrases for voice routing (start or end of speech) +# hotkey = optional direct global hotkey ("" = none; voice routing is primary) +# A workflow may override the [rewrite] defaults with its own model/temperature. # ---------------------------------------------------------------------------- [[workflow]] name = "Transcribe" icon = "⚡" description = "Speak, get plain text." -hotkey = "++" +hotkey = "" mode = "transcribe" [[workflow]] name = "Nicer email" icon = "✉" description = "Rough notes → polished email." +keywords = ["nicer email", "bessere email", "schöne mail"] hotkey = "++e" mode = "rewrite" prompt = '''Du bist ein Schreibassistent fuer E-Mails. Du erhaeltst ein gesprochenes Transkript. @@ -223,6 +270,7 @@ Schreibe daraus eine freundliche, gut formulierte und etwas ausfuehrlichere E-Ma name = "Improve text" icon = "✨" description = "Speak → cleaner writing." +keywords = ["improve text", "verbessere text", "bessere schreibweise"] hotkey = "++i" mode = "rewrite" prompt = '''Du bist ein Lektor und Schreibassistent. Verbessere den folgenden gesprochenen Text: @@ -235,6 +283,7 @@ prompt = '''Du bist ein Lektor und Schreibassistent. Verbessere den folgenden ge name = "Calm down" icon = "☺" description = "Frustrated in → calm out." +keywords = ["calm down", "beruhige das", "entspannte nachricht"] hotkey = "++c" mode = "rewrite" prompt = '''Du erhaeltst ein gesprochenes, frustriertes oder veraergertes Transkript. @@ -246,6 +295,7 @@ Gib NUR die umformulierte Nachricht zurueck, keine Erklaerungen.''' name = "Add emojis" icon = "✿" description = "Text in → emojis out." +keywords = ["add emojis", "mit emojis", "emojis dazu"] hotkey = "++j" mode = "rewrite" prompt = '''Du erhaeltst ein gesprochenes Transkript. Gib den Text moeglichst originalgetreu diff --git a/linux/blitztext/daemon.py b/linux/blitztext/daemon.py index ecb1797..092f48d 100644 --- a/linux/blitztext/daemon.py +++ b/linux/blitztext/daemon.py @@ -1,7 +1,8 @@ """Hotkey-driven engine: toggle recording per workflow, then transcribe + deliver. -Used by both the headless CLI (`run`) and the tkinter GUI. A status callback -lets the GUI reflect each phase; desktop notifications fire regardless. +Used by the headless CLI (`run`) and the GTK GUI. A status callback lets the UI +reflect each phase; desktop notifications fire regardless. Supports voice-keyword +routing: one hotkey records, then the spoken keyword selects the preset. """ from __future__ import annotations @@ -15,6 +16,7 @@ from .notify import notify from .paste import active_window_id, deliver from .recorder import Recording, detect_recorder from .rewrite import RewriteError, rewrite +from .routing import route from .transcribe import Transcriber # status_cb(state, workflow_name, message) @@ -32,6 +34,8 @@ class Daemon: self._target_window: str | None = None self._busy = False self._listener = None + # Synthetic preset used by the voice-routing hotkey. + self._route_workflow = Workflow(name="Voice", hotkey=cfg.routing_hotkey, mode="route") self.recorder_name = detect_recorder(cfg.recorder) self.transcriber: Transcriber | None = None @@ -96,44 +100,66 @@ class Daemon: # -- worker --------------------------------------------------------------- def _process(self, audio_path, workflow: Workflow, window_id) -> None: + label = workflow.name try: - self._emit("busy", workflow.name, "Transcribing…") - self._notify(f"⌛ {workflow.name}", "Transcribing…") - text = self.transcriber.transcribe(audio_path, language=self.cfg.language) + self._emit("busy", label, "Transcribing…") + self._notify(f"⌛ {label}", "Transcribing…") + hotwords = ", ".join(self.cfg.all_keywords) if workflow.mode == "route" else "" + text = self.transcriber.transcribe(audio_path, language=self.cfg.language, hotwords=hotwords) if not text: - self._emit("idle", workflow.name, "No speech detected") + self._emit("idle", label, "No speech detected") self._notify("Nothing heard", "No speech detected.", "low") return - if workflow.mode == "rewrite" and workflow.prompt: - self._emit("busy", workflow.name, "Rewriting…") - self._notify(f"⌛ {workflow.name}", "Rewriting…") + # Voice routing: pick the preset from a spoken keyword, strip it. + if workflow.mode == "route": + res = route(text, self.cfg.workflows, threshold=self.cfg.routing_threshold) + target = self.cfg.preset_by_name(res.preset_name) or self.cfg.default_preset + text = res.text + label = target.name if target else "Transcribe" + via = f"“{res.keyword}”" if res.keyword else "default" + self._emit("busy", label, f"→ {label} ({via})") + self._notify(f"🎙 {label}", f"matched: {via}") + else: + target = workflow + + if target and target.mode == "rewrite" and target.prompt: + if not text: + self._emit("idle", label, "Only a keyword heard") + self._notify("Nothing to do", "Only the keyword was heard.", "low") + return + self._emit("busy", label, "Rewriting…") + self._notify(f"⌛ {label}", "Rewriting…") try: text = rewrite( text, - workflow.prompt, + target.prompt, base_url=self.cfg.base_url, api_key=self.cfg.api_key, - model=workflow.model or self.cfg.rewrite_model, - temperature=workflow.temperature if workflow.temperature is not None else self.cfg.temperature, + model=target.model or self.cfg.rewrite_model, + temperature=target.temperature if target.temperature is not None else self.cfg.temperature, timeout=self.cfg.timeout, ) except RewriteError as exc: - self._emit("error", workflow.name, str(exc)) + self._emit("error", label, str(exc)) self._notify("Rewrite failed", str(exc), "critical") return + if not text: + self._emit("idle", label, "Nothing to type") + return + deliver( text, mode=self.cfg.output, window_id=window_id, type_delay_ms=self.cfg.type_delay_ms, ) - self._emit("done", workflow.name, text) - self._notify(f"✓ {workflow.name}", text[:80] + ("…" if len(text) > 80 else "")) + self._emit("done", label, text) + self._notify(f"✓ {label}", text[:80] + ("…" if len(text) > 80 else "")) except Exception as exc: # noqa: BLE001 - surface any failure - self._emit("error", workflow.name, str(exc)) + self._emit("error", label, str(exc)) self._notify("Error", str(exc), "critical") print(f"[blitztext] error: {exc}", file=sys.stderr) finally: @@ -143,12 +169,21 @@ class Daemon: self._emit("idle", None, "Ready") # -- hotkeys -------------------------------------------------------------- + def _build_mapping(self) -> dict: + """hotkey -> callback, skipping empty hotkeys, plus the routing hotkey.""" + mapping: dict = {} + for wf in self.cfg.workflows: + if wf.hotkey: + mapping[wf.hotkey] = (lambda wf=wf: self.toggle(wf)) + if self.cfg.routing_enabled and self.cfg.routing_hotkey: + mapping[self.cfg.routing_hotkey] = (lambda: self.toggle(self._route_workflow)) + return mapping + def start_hotkeys(self): """Register global hotkeys non-blocking; returns the pynput listener.""" from pynput import keyboard - mapping = {wf.hotkey: (lambda wf=wf: self.toggle(wf)) for wf in self.cfg.workflows} - self._listener = keyboard.GlobalHotKeys(mapping) + self._listener = keyboard.GlobalHotKeys(self._build_mapping()) self._listener.start() return self._listener @@ -160,12 +195,12 @@ class Daemon: # -- headless run loop ---------------------------------------------------- def run(self) -> None: self.prepare() - lines = "\n".join(f" {wf.hotkey} → {wf.name}" for wf in self.cfg.workflows) - print(f"[blitztext] ready. Recorder: {self.recorder_name}. Hotkeys:\n{lines}", file=sys.stderr) + lines = [f" {self.cfg.routing_hotkey} → Voice routing (speak a keyword)"] if self.cfg.routing_enabled else [] + lines += [f" {wf.hotkey} → {wf.name}" for wf in self.cfg.workflows if wf.hotkey] + print(f"[blitztext] ready. Recorder: {self.recorder_name}. Hotkeys:\n" + "\n".join(lines), file=sys.stderr) self._notify("Blitztext ready", "Focus a text field and press a hotkey.") from pynput import keyboard - mapping = {wf.hotkey: (lambda wf=wf: self.toggle(wf)) for wf in self.cfg.workflows} - with keyboard.GlobalHotKeys(mapping) as listener: + with keyboard.GlobalHotKeys(self._build_mapping()) as listener: listener.join() diff --git a/linux/blitztext/routing.py b/linux/blitztext/routing.py new file mode 100644 index 0000000..12ee8e4 --- /dev/null +++ b/linux/blitztext/routing.py @@ -0,0 +1,128 @@ +"""Voice-keyword routing. + +Given a transcript, find a trigger keyword at the start or end that selects a +preset, strip it, and return the cleaned text. ASR-tolerant: normalizes text and +fuzzy-matches so "nicer e-mail." still routes to the "nicer email" preset. + +Only the leading/trailing word-windows are scanned, so a keyword spoken inside +the body doesn't misfire. Returns the best match across all presets. +""" + +from __future__ import annotations + +import re +import unicodedata +from dataclasses import dataclass +from difflib import SequenceMatcher + +# How similar a window has to be to a keyword phrase to count as a match. +DEFAULT_THRESHOLD = 0.82 +# A keyword phrase of N words is matched against the first/last N words (+ a +# little slack) of the transcript. +EDGE_SLACK = 1 + + +@dataclass +class RouteResult: + preset_name: str | None # None -> no keyword matched (use default) + text: str # transcript with the keyword stripped + keyword: str | None # the configured keyword that matched + position: str | None # "start" | "end" + score: float + + +def _strip_accents(s: str) -> str: + return "".join(c for c in unicodedata.normalize("NFKD", s) if not unicodedata.combining(c)) + + +def normalize(text: str) -> list[str]: + """Lowercase, drop accents/punctuation, return word tokens.""" + text = _strip_accents(text.lower()) + text = re.sub(r"[^\w\s]", " ", text, flags=re.UNICODE) + return text.split() + + +def _similar(a: str, b: str) -> float: + return SequenceMatcher(None, a, b).ratio() + + +def _match_window(tokens: list[str], kw_tokens: list[str], threshold: float): + """Try to match kw_tokens against the start and end of tokens. + + Tries span lengths of n-1..n+1 words at each edge so ASR token drift + (e.g. "e-mail" -> "e mail") still matches. Returns (position, score, + span_len) for the best edge match, or None. + """ + n = len(kw_tokens) + if not n or not tokens: + return None + kw = " ".join(kw_tokens) + + best = None + for span in range(max(1, n - EDGE_SLACK), n + EDGE_SLACK + 1): + if span > len(tokens): + continue + for pos, window in (("start", tokens[:span]), ("end", tokens[-span:])): + score = _similar(" ".join(window), kw) + if score >= threshold and (best is None or score > best[1]): + best = (pos, score, span) + return best + + +def route( + transcript: str, + presets, + *, + threshold: float = DEFAULT_THRESHOLD, +) -> RouteResult: + """Pick the preset whose keyword best matches an edge of the transcript. + + `presets` is any iterable of objects with `.name` and `.keywords` (list[str]). + """ + tokens = normalize(transcript) + if not tokens: + return RouteResult(None, transcript.strip(), None, None, 0.0) + + best = None # (score, span_len, position, preset_name, keyword) + for preset in presets: + for keyword in getattr(preset, "keywords", None) or []: + kw_tokens = normalize(keyword) + m = _match_window(tokens, kw_tokens, threshold) + if m is None: + continue + position, score, span = m + # Prefer higher score, then longer (more specific) keyword phrase. + key = (score, span) + if best is None or key > (best[0], best[1]): + best = (score, span, position, preset.name, keyword) + + if best is None: + return RouteResult(None, transcript.strip(), None, None, 0.0) + + score, span, position, name, keyword = best + cleaned = _strip_span(transcript, span, position) + return RouteResult(name, cleaned, keyword, position, score) + + +def _strip_span(transcript: str, span_words: int, position: str) -> str: + """Remove the matched keyword from the given edge of the original transcript. + + `span_words` counts *normalized* tokens; a raw word like "e-mail." may + normalize to two tokens, so we consume raw words until their cumulative + normalized-token count reaches the span. Original casing/spacing is kept. + """ + words = transcript.split() + seq = words if position == "start" else list(reversed(words)) + + consumed = 0 + ntok = 0 + for w in seq: + ntok += len(normalize(w)) + consumed += 1 + if ntok >= span_words: + break + + remaining = words[consumed:] if position == "start" else words[: len(words) - consumed] + out = " ".join(remaining) + # Trim leftover separators left where the keyword was removed. + return out.strip(" \t\r\n,.;:!?-–—\"'").strip() diff --git a/linux/blitztext/transcribe.py b/linux/blitztext/transcribe.py index 770b424..7a4073d 100644 --- a/linux/blitztext/transcribe.py +++ b/linux/blitztext/transcribe.py @@ -49,11 +49,15 @@ class Transcriber: print(f"[blitztext] {dev} unavailable ({exc}); trying next device", file=sys.stderr) raise RuntimeError(f"Failed to load Whisper model '{model}': {last_err}") - def transcribe(self, audio_path: Path, language: str = "") -> str: - segments, _info = self._model.transcribe( - str(audio_path), - language=language or None, - beam_size=self.beam_size, - vad_filter=True, - ) + def transcribe(self, audio_path: Path, language: str = "", hotwords: str = "") -> str: + kwargs = dict(language=language or None, beam_size=self.beam_size, vad_filter=True) + if hotwords: + # Bias recognition toward the routing keywords so they transcribe + # reliably. Older faster-whisper builds lack `hotwords`; fall back. + kwargs["hotwords"] = hotwords + try: + segments, _info = self._model.transcribe(str(audio_path), **kwargs) + except TypeError: + kwargs.pop("hotwords", None) + segments, _info = self._model.transcribe(str(audio_path), **kwargs) return " ".join(seg.text.strip() for seg in segments).strip()