Cap buffered audio and keep transcripts out of INFO logs

Wyoming has no authentication, so buffered audio is attacker-controlled.
self.audio grew until AudioStop with no bound: measured at ~11 MB/s over
loopback, one connection exhausts 32 GB in under an hour, and a stuck
satellite that never sends AudioStop does the same by accident. Cap it at
--max-audio-seconds (default 120), dropping the excess with a single warning
while still transcribing what was captured. Verified: a client streaming
10.8 GB now moves server RSS by 213 MB rather than 10.8 GB.

Transcripts were logged at INFO. Log files are long-lived and world-readable
under /tmp on macOS, so every voice command sat in plaintext readable by any
local account. INFO now records duration, latency and character count; the
text moved behind --debug.

Both are covered by mutation-checked tests, and the README gains a Security
section covering the unauthenticated trust boundary, the 0.0.0.0 bind that
also exposes VPN interfaces, and running the daemon as a non-admin user.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
2026-07-29 03:44:25 +01:00
co-authored by Claude Opus 5
parent 9bf887e780
commit 5a6fc62106
4 changed files with 170 additions and 9 deletions
+29 -3
View File
@@ -12,6 +12,13 @@ from .engine import SAMPLE_RATE
_LOGGER = logging.getLogger(__name__)
# Wyoming has no authentication, so any client that can reach the port can
# stream audio. Without a cap, self.audio grows until AudioStop -- measured at
# ~11 MB/s over loopback, which exhausts 32 GB in under an hour from a single
# connection. A stuck satellite that never sends AudioStop does the same thing
# by accident. No real voice command approaches this bound.
DEFAULT_MAX_AUDIO_SECONDS = 120
class ParakeetEventHandler(AsyncEventHandler):
def __init__(self, wyoming_info: Info, cli_args, engine, *args, **kwargs):
@@ -20,13 +27,27 @@ class ParakeetEventHandler(AsyncEventHandler):
self.wyoming_info_event = wyoming_info.event()
self.engine = engine
self.audio = bytes()
self.truncated = False
self.converter = AudioChunkConverter(rate=SAMPLE_RATE, width=2, channels=1)
max_seconds = getattr(cli_args, "max_audio_seconds", DEFAULT_MAX_AUDIO_SECONDS)
self.max_bytes = int(max_seconds * SAMPLE_RATE * 2)
def _append(self, chunk: bytes) -> None:
room = self.max_bytes - len(self.audio)
if room > 0:
self.audio += chunk[:room]
if len(self.audio) >= self.max_bytes and not self.truncated:
self.truncated = True
_LOGGER.warning(
"Audio exceeded %.0fs; ignoring the rest of this utterance",
self.max_bytes / (SAMPLE_RATE * 2),
)
async def handle_event(self, event: Event) -> bool:
if AudioChunk.is_type(event.type):
if not self.audio:
_LOGGER.debug("Receiving audio")
self.audio += self.converter.convert(AudioChunk.from_event(event)).audio
self._append(self.converter.convert(AudioChunk.from_event(event)).audio)
return True
if AudioStop.is_type(event.type):
@@ -34,12 +55,16 @@ class ParakeetEventHandler(AsyncEventHandler):
started = time.monotonic()
try:
text = await self.engine.transcribe(self.audio)
# The transcript is everything the user said, and the log file
# is long-lived and readable by other local accounts. Keep the
# operational signal at INFO and the content behind --debug.
_LOGGER.info(
"%.2fs audio -> %.0fms :: %r",
"%.2fs audio -> %.0fms, %d chars",
duration,
(time.monotonic() - started) * 1000,
text,
len(text),
)
_LOGGER.debug("Transcript: %r", text)
except Exception:
# wyoming's run loop has no except clause, so letting this
# propagate closes the connection without ever sending a
@@ -49,6 +74,7 @@ class ParakeetEventHandler(AsyncEventHandler):
text = ""
finally:
self.audio = bytes()
self.truncated = False
await self.write_event(Transcript(text=text).event())
return False