* fix(engines): make SubprocessBackend.generate() path-aware on the GPU pool generate()'s slot handling had two bugs: 1. On-pool self-deadlock: /v1/audio/speech and /generate (and audiobook, dub, batch) dispatch generate() via run_on_gpu_pool_guarded, already on a pool worker, so the inner slot submit queued behind the very job running it on a 1-worker (MPS) pool and timed out before the sidecar spawned. Every subprocess engine surfaced the in-process 300s-abandon instead of synthesizing. 2. Off-pool no hold: the off-pool slot was a bare no-op that released the worker before _spawn(), so off-pool callers (engine self-test, diagnostics) could synthesize concurrently with a pool job and over-subscribe the GPU. Make the slot block path-aware: on-pool callers skip (the outer run_on_gpu_pool_guarded already holds _running for the whole sidecar exchange); off-pool callers hold a real slot for the whole synthesis via an _occupy task that blocks the worker until _held is set in the finally. Single release point in the finally. Regression tests: generate dispatched on a pool worker (on-pool skip) and a concurrent pool job blocked during an off-pool generate (off-pool hold). Both verified fail-before / pass-after. Supersedes #1296 (on-pool-skip-only). Closes #1295, #1297. * Address review: couple on-pool skip to the pool prefix; fix comment /simplify + /code-review flagged that the on-pool skip keyed on the literal "gpu-pool" string, decoupled from _build_gpu_pool's thread_name_prefix. A rename would silently re-introduce the exact self-deadlock this PR fixes (and the tests can't catch it, since they hardcode the prefix). Centralise the prefix in _GPU_POOL_THREAD_PREFIX + a running_on_gpu_pool() helper, used by _build_gpu_pool, the skip in generate(), and _heal_tts_placement. Also fix the comment: the Settings engine self-test rejects subprocess-isolated engines with a 400, so the only real off-pool caller is the diagnose.py deep-synth probe. * fix(engines): bind slot_future before the off-pool branch CodeQL py/uninitialized-local-variable (error, blocking CI). `_held is not None` does imply slot_future was assigned, so the current code is correct — but the two are only coupled by convention, which the analyser cannot see and a third exit path would quietly break. Binds it to None up front and guards the cancel. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> * test(engines): make the slot-hold regression deterministic and leak-free CodeRabbit, valid on both counts. The test used sleep(0.8)/sleep(0.5) as synchronization — the tests/** contract forbids it, and on a slow runner the marker could be enqueued before the generator had reserved anything, so the assertion passed for the wrong reason. It now waits on an event signalled when the slot task actually starts, and asserts "did not run" via a result() timeout rather than a bare sleep. Cleanup moved into finally: an assertion failure used to leak the sidecar process and the pool thread into the rest of the session. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> --------- Co-authored-by: debpalash <4178343+debpalash@users.noreply.github.com> Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
117 lines
3.7 KiB
Python
117 lines
3.7 KiB
Python
"""Regression: SubprocessBackend.generate() must not self-deadlock when called
|
|
from a gpu-pool worker.
|
|
|
|
/v1/audio/speech and /generate dispatch backend.generate() via
|
|
run_on_gpu_pool_guarded, i.e. already ON a gpu-pool worker. generate() used to
|
|
unconditionally submit a no-op to the same pool to "acquire a slot"; on a
|
|
1-worker pool (MPS) that submit queued behind the very job running it and
|
|
result(timeout=10) raised before the sidecar spawned. This test reproduces that
|
|
dispatch shape (generate on a pool worker) against a stub sidecar.
|
|
"""
|
|
import base64
|
|
import json
|
|
import math
|
|
import array
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from services.subprocess_backend import SubprocessBackend
|
|
|
|
# Model-free stub sidecar: speaks the length-prefixed-JSON protocol and returns
|
|
# a 1s sine wave for any synthesize.
|
|
STUB_SIDECAR = r'''
|
|
import sys, json, struct, math, array, base64
|
|
|
|
def _send(o):
|
|
b = json.dumps(o, separators=(",", ":")).encode()
|
|
sys.stdout.buffer.write(struct.pack("!I", len(b)) + b)
|
|
sys.stdout.buffer.flush()
|
|
|
|
def _recv():
|
|
h = sys.stdin.buffer.read(4)
|
|
if len(h) < 4:
|
|
return None
|
|
(n,) = struct.unpack("!I", h)
|
|
body = bytearray()
|
|
while len(body) < n:
|
|
c = sys.stdin.buffer.read(n - len(body))
|
|
if not c:
|
|
return None
|
|
body.extend(c)
|
|
return json.loads(bytes(body).decode())
|
|
|
|
_send({"op": "ready", "engine": "stub", "sample_rate": 24000})
|
|
while True:
|
|
m = _recv()
|
|
if m is None:
|
|
sys.exit(0)
|
|
op = m.get("op")
|
|
if op == "ping":
|
|
_send({"op": "pong", "vram_mb": 0.0})
|
|
elif op == "shutdown":
|
|
sys.exit(0)
|
|
elif op == "synthesize":
|
|
sr = 24000
|
|
pcm = array.array("h", (int(32767 * math.sin(2 * math.pi * 440 * i / sr)) for i in range(sr)))
|
|
_send({"op": "audio", "audio_pcm_b64": base64.b64encode(pcm.tobytes()).decode(),
|
|
"sample_rate": sr, "n_samples": sr})
|
|
else:
|
|
_send({"op": "error", "stage": "dispatch", "message": "unknown op %r" % op})
|
|
'''
|
|
|
|
|
|
class _StubSubprocessBackend(SubprocessBackend):
|
|
"""Minimal concrete SubprocessBackend pointing at the stub sidecar."""
|
|
id = "stub-subprocess"
|
|
display_name = "stub"
|
|
gpu_compat = ("cuda", "mps", "cpu")
|
|
|
|
@classmethod
|
|
def is_available(cls):
|
|
return True, "ok"
|
|
|
|
@classmethod
|
|
def venv_python(cls):
|
|
return Path(sys.executable)
|
|
|
|
@classmethod
|
|
def sidecar_script(cls):
|
|
raise NotImplementedError # patched per-test
|
|
|
|
@property
|
|
def sample_rate(self):
|
|
return 24000
|
|
|
|
@property
|
|
def supported_languages(self):
|
|
return ["multi"]
|
|
|
|
|
|
def test_generate_on_pool_worker_does_not_deadlock(tmp_path, monkeypatch):
|
|
# Mirror /v1/audio/speech + /generate: dispatch generate() ON a gpu-pool
|
|
# worker. Force a 1-worker pool so the self-deadlock reproduces
|
|
# deterministically regardless of host (the ambient pool may have >1
|
|
# worker): pre-fix, generate()'s inner slot-submit queued behind this very
|
|
# job and slot_future.result(timeout=10) raised at ~10s, before the sidecar
|
|
# spawned.
|
|
stub = tmp_path / "stub_sidecar.py"
|
|
stub.write_text(STUB_SIDECAR)
|
|
monkeypatch.setattr(_StubSubprocessBackend, "sidecar_script",
|
|
classmethod(lambda cls: stub))
|
|
|
|
from concurrent.futures import ThreadPoolExecutor
|
|
import services.model_manager as mm
|
|
pool = ThreadPoolExecutor(max_workers=1, thread_name_prefix="gpu-pool")
|
|
monkeypatch.setattr(mm, "_get_gpu_pool", lambda: pool)
|
|
|
|
b = _StubSubprocessBackend()
|
|
try:
|
|
fut = pool.submit(lambda: b.generate("on-pool"))
|
|
tensor = fut.result(timeout=30) # pre-fix: raised ~10s slot timeout
|
|
assert tensor.shape[1] == 24000
|
|
finally:
|
|
b.shutdown()
|
|
pool.shutdown(wait=False)
|