-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathconfig.example.yaml
More file actions
213 lines (202 loc) · 11.6 KB
/
Copy pathconfig.example.yaml
File metadata and controls
213 lines (202 loc) · 11.6 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
# LocalFlow configuration - reference copy.
#
# On first run LocalFlow copies this file to config.yaml next to the localflow package.
# Edit config.yaml (not this file) and restart; changes made from the tray/dot menu apply
# immediately and are saved back. Missing keys are filled in from the built-in defaults,
# so you can delete anything you do not care about.
# ---------------------------------------------------------------- hotkeys
# Key names: ctrl win alt shift space esc tab enter a-z 0-9 f1-f12
hotkeys:
ptt: [ctrl, win] # hold to talk; release pastes the text
handsfree: [ctrl, win, space] # toggle continuous dictation
double_tap_toggle: true # two quick PTT taps also start hands-free
double_tap_ms: 400
tap_max_ms: 250 # a hold shorter than this counts as a tap, not speech
cancel: [esc] # discard the current recording
repaste: [shift, alt, z] # paste the last result again
polish: [ctrl, win, alt] # hold: like PTT but cleaned with the "high" prompt
# ---------------------------------------------------------------- microphone / capture
audio:
device: '' # substring of the input device name; '' = system default
sample_rate: 16000
preroll_ms: 400 # audio kept from *before* you pressed the key
min_duration_ms: 400 # shorter clips are dropped
rms_threshold: 0.0005 # only digital silence is dropped here
normalize: true # peak-normalise each clip so whispering still transcribes
normalize_dbfs: -3.0
max_seconds: 1200 # hands-free safety cap (20 min)
# How hands-free (double-tap PTT / Ctrl+Win+Space / click the dot) behaves:
# whole - default. Records everything and does nothing until you stop, then transcribes the
# whole speech in one pass and cleans it at cleanup.handsfree_level. Nothing is
# pasted while you talk, so nothing can be cut off mid-sentence.
# chunked - the old behaviour: a pause in your speech closes a chunk, which is transcribed and
# pasted while you keep talking. Live text, at the cost of accuracy across a pause.
handsfree_mode: whole
# The three keys below do nothing in whole mode; they only shape chunked mode.
handsfree_silence_ms: 700 # chunked only: how long a pause must be to close a chunk
handsfree_vad_threshold: 0.002 # chunked only: loudness counted as speech. A normal speaking
# voice is around 0.003 RMS, so keep this well under it or your
# speech is read as silence. Raise it only in a noisy room.
handsfree_max_chunk_s: 30 # chunked only: close a chunk anyway after this long without a pause
# ---------------------------------------------------------------- speech recognition
asr:
engine: parakeet # parakeet (fast, English) | whisper (multilingual fallback)
parakeet_model: nemo-parakeet-tdt-0.6b-v2
parakeet_quantization: null # null = full precision (about 3.0 GB of VRAM, fastest)
# int8 = Low VRAM mode: about 0.6 GB, roughly a third of a second
# slower per dictation, same words. Also in the dot menu.
whisper_model: turbo # large-v3-turbo; use "small" on CPU-only machines
whisper_compute_type: float16 # use int8 on CPU
language: en
allow_cpu_fallback: false # true = run on the CPU (slower) instead of failing without CUDA
models_dir: models # model cache, relative to the project folder
gpu_mem_limit_mb: 3072 # cap the VRAM the ASR session may use (keeps games happy)
# ---------------------------------------------------------------- text cleanup
cleanup:
# LLM level: none | light | medium | high. The rules below always run (~1 ms) and work
# with no LLM installed at all.
# none - rules only, lowest latency
# light - fillers, punctuation, self-corrections; no rephrasing
# medium - plus grammar and light restructuring (recommended)
# high - rewrite for clarity and brevity
level: medium
# Level used for a whole hands-free speech (same enhanced pass as the polish hotkey). "high"
# handles spoken corrections best but will also tidy away throwaway asides and repeated content;
# use "medium" to keep your wording closer to what you said, or "none" for a raw transcript.
handsfree_level: high
rules: true # false = completely raw ASR output
spoken_punctuation: true # "comma", "period", "new line", "at sign", ...
remove_fillers: true
fillers: [um, umm, uh, uhm, uhh, erm, er, hmm, hm, mhm, ah]
filler_phrases: [] # e.g. ["you know", "sort of"]
collapse_repeats: true # "the the" -> "the"
backtrack: true # spoken self-corrections drop the previous clause
backtrack_strong:
- scratch that
- strike that
- no no wait
- no wait
- wait no
- no no
- oops
- I meant to say
- I mean to say
- I meant
- let me rephrase
- correction
backtrack_weak: [i mean, actually] # only when they begin a comma-clause
press_enter_phrases: [press enter, hit enter] # the ONLY way Enter is ever pressed
auto_lists: true # 3+ spoken items after a lead-in become a bulleted list
list_leadins:
- I need to get
- I need to buy
- I need to pick up
- I need to do
- we need to get
- we need to buy
- the list is
- the items are
- the following
- here is the list
- here's the list
- things to do
- things to get
- things to buy
- shopping list
- to do list
- to-do list
- the steps are
- the options are
# Fix words the recogniser gets wrong (whole word, case-insensitive).
dictionary:
local flow: LocalFlow
# Say the trigger phrase, get the expansion typed. Handy for addresses, signatures, boilerplate.
snippets:
my signature: "Best regards,\n<your name>"
# Different cleanup level per app, matched on a foreground window-title substring.
per_app: {}
# per_app:
# Slack: medium
# Visual Studio Code: none
# ---------------------------------------------------------------- optional local LLM (Ollama)
# Entirely optional. With Ollama absent or stopped, LocalFlow logs it once and uses the rules text.
llm:
host: http://127.0.0.1:11434
model: gemma3:4b # ollama pull gemma3:4b
polish_model: gemma3:4b # keep one resident model; a 12B next to Parakeet spills out of VRAM
skip_when_cold: true # before each cleanup call, ask Ollama whether the model is in
# VRAM (a few ms). If it is not, do not wait for it: paste the
# rules-cleaned text now, warm the model in the background, and
# the next dictation gets it. Cold, one call takes 7-19 s;
# warm, 150-600 ms. Set false to wait for the cold load instead.
timeout_ms: 2500 # base timeout; on timeout the rules-cleaned text is pasted.
# Sized for a warm model (measured 150-600 ms): waiting 6-10 s
# and then discarding the answer helps nobody.
timeout_per_word_ms: 80 # added per input word ...
timeout_max_ms: 20000 # ... up to this cap
polish_timeout_ms: 20000
handsfree_timeout_ms: 60000 # timeout per segment of a whole hands-free speech. Generous on
# purpose: a long speech is worth waiting for, and a segment that
# does time out falls back to the rules cleanup on its own.
num_ctx: 4096 # model context window. The largest request LocalFlow can ever send
# is one full segment at level high: 835 prompt tokens plus at
# most 1,040 of output = 1,875, so 4096 leaves over 2x headroom
# and keeps gemma3:4b at 3,832 MB instead of 4,060 MB. Only raise
# it if you raise segment_words below.
segment_words: 400 # a long text is cleaned in sentence-aligned segments of about this
# many words, then joined. Lower it if the model is short on
# context; a segment that fails only affects itself.
keep_alive: 1800 # seconds the model stays in VRAM after a call (-1 = forever)
temperature: 0
autostart_ollama: true # if a health probe finds Ollama down and nothing named ollama* is
# running, start it (Ollama's own Startup shortcut does not
# always fire). Local hosts only, at most 3 tries per session and
# 5 minutes apart. Set to false to never launch anything.
# ---------------------------------------------------------------- how text is inserted
inject:
method: auto # auto | paste | scancode | type
# auto = scancode in fullscreen apps / type_apps, else paste
# paste = clipboard + Ctrl+V (fastest, works nearly everywhere)
# scancode = per-key SendInput for games that ignore Ctrl+V
# type = legacy per-character typing
force_scancode: false # tray toggle "Force keystroke typing"
scancode_delay_ms: 12 # inter-key delay in scancode mode (+/- 4 ms jitter)
scancode_newlines: false # false = newlines become spaces (chat boxes are single-line)
scancode_max_chars: 500 # never type more than this into a game
restore_clipboard_ms: 150 # raise it if a slow app reads the clipboard late
enter_delay_ms: 150 # raise it if Enter lands before the text
trailing_space: false
handsfree_trailing_space: true
# Window-title substrings / exe-name prefixes that ALWAYS get keystroke typing instead of a
# paste. Empty by default: fullscreen apps are detected automatically, so you only need this
# for a windowed app that ignores Ctrl+V. Example:
# type_apps: [FiveM]
type_apps: []
# ---------------------------------------------------------------- on-screen dot + tray
ui:
overlay: true # show the Flow Dot
sounds: true
pill_x: null # saved automatically when you drag the dot
pill_y: null
hide_when_fullscreen: false # hide the dot while a fullscreen app is in front (fixes some stutter)
# ---------------------------------------------------------------- GPU sharing
gpu:
auto_pause_fullscreen: false # unload the models while a fullscreen app (a game) is in front
auto_resume_delay_s: 5
# ---------------------------------------------------------------- local history
# One JSON line per utterance, stored next to the app. Nothing is ever uploaded.
history:
enabled: true
path: history.jsonl
max_entries: 5000
# ---------------------------------------------------------------- phone access (docs/API.md)
# A small HTTP server on 127.0.0.1 only; the Android app reaches it over your Tailscale network
# via `tailscale serve` (no firewall change, WireGuard-encrypted). Off by default.
server:
enabled: false # tray: "Enable phone access"
port: 8770
token: '' # generated on first enable and saved here; tray: "Phone setup"
bind: 127.0.0.1 # keep it loopback: tailscale serve is the only way in
tailscale_serve: true # map http://<pc>.<tailnet>.ts.net -> 127.0.0.1:<port> at startup
# Verbose logging (also: run.ps1 -Debug). Logs the full pasted text and LLM rejects.
debug: false