irongit

refactor: Move whisper dictate script into current repo

huncholanehuncholaneauthored
parent 0d8dea6commit 1ca94e7930fc878df24dc671ae4ba2ae82b43e47Browse files

4 files changed, +183 -2

+2-2UserConfigs/UserKeybinds.conf
@@ -136,5 +136,5 @@ bind = $mainMod CTRL SHIFT, 0, movetoworkspace, 20
136136 bind = $mainMod, i, exec, foot -a floating-term sh -c 'hyprctl clients | nvim' # Show the current hyprland client info
137137
138138 # Whisper voice dictation (push-to-talk: hold Super+` to record, release to type)
139-bind = $mainMod, grave, exec, ~/.local/bin/whisper-dictate start
140-bindr = $mainMod, grave, exec, ~/.local/bin/whisper-dictate stop
139+bind = $mainMod, grave, exec, $UserScripts/whisper/whisper-dictate start
140+bindr = $mainMod, grave, exec, $UserScripts/whisper/whisper-dictate stop
+79-0UserScripts/whisper/whisper-dictate
@@ -0,0 +1,79 @@
1+#!/usr/bin/env bash
2+# Push-to-talk dictation: `start` begins recording, `stop` ends + transcribes + pastes.
3+
4+set -euo pipefail
5+
6+CACHE=~/.cache/whisper-dictate
7+PID_FILE="$CACHE/recorder.pid"
8+WAV_FILE="$CACHE/audio.wav"
9+TARGET_FILE="$CACHE/target.json"
10+PYTHON=~/.venv/bin/python
11+MODEL="${WHISPER_MODEL:-small.en}"
12+DEVICE="${WHISPER_DEVICE:-cuda}"
13+
14+mkdir -p "$CACHE"
15+
16+CUDA_LIB_PATHS=""
17+for sub in nvidia/cublas/lib nvidia/cudnn/lib; do
18+ for dir in ~/.venv/lib/python*/site-packages/$sub; do
19+ [[ -d "$dir" ]] && CUDA_LIB_PATHS+="$dir:"
20+ done
21+done
22+export LD_LIBRARY_PATH="${CUDA_LIB_PATHS}${LD_LIBRARY_PATH:-}"
23+
24+notify() {
25+ command -v notify-send >/dev/null && notify-send -t 1500 "Whisper" "$1" || true
26+}
27+
28+start() {
29+ if [[ -f "$PID_FILE" ]] && kill -0 "$(cat "$PID_FILE")" 2>/dev/null; then
30+ kill -INT "$(cat "$PID_FILE")" 2>/dev/null || true
31+ rm -f "$PID_FILE"
32+ fi
33+ rm -f "$WAV_FILE"
34+ hyprctl activewindow -j > "$TARGET_FILE" 2>/dev/null || true
35+ notify "Recording..."
36+ "$PYTHON" "$HOME/.dotfiles/hypr/UserScripts/whisper/whisper-record.py" "$WAV_FILE" </dev/null >/dev/null 2>&1 &
37+ echo "$!" > "$PID_FILE"
38+}
39+
40+stop() {
41+ [[ -f "$PID_FILE" ]] || exit 0
42+ PID="$(cat "$PID_FILE")"
43+ rm -f "$PID_FILE"
44+ notify "Transcribing..."
45+ kill -INT "$PID" 2>/dev/null || true
46+ for _ in {1..30}; do
47+ kill -0 "$PID" 2>/dev/null || break
48+ sleep 0.1
49+ done
50+ if [[ ! -s "$WAV_FILE" ]]; then
51+ notify "No audio recorded"
52+ exit 0
53+ fi
54+ TEXT="$("$PYTHON" "$HOME/.dotfiles/hypr/UserScripts/whisper/whisper-transcribe.py" "$WAV_FILE" "$MODEL" "$DEVICE" 2>/tmp/whisper-error.log || true)"
55+ if [[ -z "$TEXT" ]]; then
56+ notify "No speech detected (see /tmp/whisper-error.log)"
57+ exit 0
58+ fi
59+ ADDR="$(grep -oP '"address":\s*"\K[^"]+' "$TARGET_FILE" 2>/dev/null || true)"
60+ CLASS="$(grep -oP '"class":\s*"\K[^"]+' "$TARGET_FILE" 2>/dev/null || true)"
61+ if [[ -n "$ADDR" ]]; then
62+ hyprctl dispatch focuswindow "address:$ADDR" >/dev/null 2>&1 || true
63+ sleep 0.05
64+ fi
65+ printf "%s" "$TEXT" | wl-copy
66+ case "$CLASS" in
67+ foot|kitty|Alacritty|alacritty|WezTerm|org.wezfurlong.wezterm|com.mitchellh.ghostty|ghostty|XTerm|st-*)
68+ wtype -M ctrl -M shift -k v -m shift -m ctrl ;;
69+ *)
70+ wtype -M ctrl -k v -m ctrl ;;
71+ esac
72+ notify "Pasted ${#TEXT} chars"
73+}
74+
75+case "${1:-}" in
76+ start) start ;;
77+ stop) stop ;;
78+ *) echo "usage: $0 {start|stop}" >&2; exit 1 ;;
79+esac
+46-0UserScripts/whisper/whisper-record.py
@@ -0,0 +1,46 @@
1+#!/usr/bin/env python3
2+"""Record audio from default input until SIGINT, save as 16kHz mono WAV."""
3+
4+import signal
5+import struct
6+import sys
7+import wave
8+
9+import numpy as np
10+import sounddevice as sd
11+
12+SAMPLE_RATE = 16000
13+CHANNELS = 1
14+DTYPE = "int16"
15+
16+if len(sys.argv) < 2:
17+ print("usage: whisper-record.py <out.wav>", file=sys.stderr)
18+ sys.exit(1)
19+
20+out_path = sys.argv[1]
21+chunks: list[np.ndarray] = []
22+stop = False
23+
24+
25+def on_signal(_sig, _frame):
26+ global stop
27+ stop = True
28+
29+
30+signal.signal(signal.SIGINT, on_signal)
31+signal.signal(signal.SIGTERM, on_signal)
32+
33+with sd.InputStream(
34+ samplerate=SAMPLE_RATE, channels=CHANNELS, dtype=DTYPE
35+) as stream:
36+ while not stop:
37+ chunk, _ = stream.read(1600) # 100ms blocks
38+ chunks.append(chunk.copy())
39+
40+if chunks:
41+ audio = np.concatenate(chunks).flatten().astype(np.int16)
42+ with wave.open(out_path, "wb") as w:
43+ w.setnchannels(CHANNELS)
44+ w.setsampwidth(2)
45+ w.setframerate(SAMPLE_RATE)
46+ w.writeframes(audio.tobytes())
+56-0UserScripts/whisper/whisper-transcribe.py
@@ -0,0 +1,56 @@
1+#!/usr/bin/env python3
2+"""Transcribe a WAV with faster-whisper. Print plain text to stdout."""
3+
4+import glob
5+import os
6+import site
7+import sys
8+
9+
10+def _add_cuda_libs():
11+ """Locate pip-installed cuBLAS / cuDNN .so dirs and prepend to LD_LIBRARY_PATH."""
12+ candidates = []
13+ for base in site.getsitepackages() + [site.getusersitepackages()]:
14+ for sub in ("nvidia/cublas/lib", "nvidia/cudnn/lib"):
15+ path = os.path.join(base, sub)
16+ if os.path.isdir(path):
17+ candidates.append(path)
18+ # Also check virtualenv site-packages explicitly
19+ venv = os.environ.get("VIRTUAL_ENV") or os.path.dirname(
20+ os.path.dirname(sys.executable)
21+ )
22+ for sub in ("nvidia/cublas/lib", "nvidia/cudnn/lib"):
23+ for hit in glob.glob(
24+ os.path.join(venv, "lib", "python*", "site-packages", sub)
25+ ):
26+ candidates.append(hit)
27+ candidates = list(dict.fromkeys(candidates)) # dedupe
28+ if candidates:
29+ os.environ["LD_LIBRARY_PATH"] = ":".join(
30+ candidates + [os.environ.get("LD_LIBRARY_PATH", "")]
31+ ).rstrip(":")
32+
33+
34+_add_cuda_libs()
35+
36+from faster_whisper import WhisperModel # noqa: E402
37+
38+if len(sys.argv) < 2:
39+ print("usage: whisper-transcribe.py <in.wav> [model] [device]", file=sys.stderr)
40+ sys.exit(1)
41+
42+wav = sys.argv[1]
43+model_name = sys.argv[2] if len(sys.argv) > 2 else "small.en"
44+device = sys.argv[3] if len(sys.argv) > 3 else "cuda"
45+
46+compute_type = "float16" if device == "cuda" else "int8"
47+
48+try:
49+ model = WhisperModel(model_name, device=device, compute_type=compute_type)
50+except Exception as e:
51+ print(f"[whisper-transcribe] CUDA init failed: {e}; falling back to CPU", file=sys.stderr)
52+ model = WhisperModel(model_name, device="cpu", compute_type="int8")
53+
54+segments, _info = model.transcribe(wav, language="en", beam_size=5, vad_filter=True)
55+text = " ".join(seg.text.strip() for seg in segments).strip()
56+print(text)