Skip to content

Commit acb5416

Browse files
authored
Merge branch 'main' into feature/rocm-windows-support
2 parents bdf476f + ed2eec5 commit acb5416

16 files changed

Lines changed: 210 additions & 35 deletions

‎.bumpversion.cfg‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,5 @@
11
[bumpversion]
2-
current_version = 0.4.4
2+
current_version = 0.4.5
33
commit = True
44
tag = True
55
tag_name = v{new_version}

‎CHANGELOG.md‎

Lines changed: 11 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -7,6 +7,15 @@
77

88
## [Unreleased]
99

10+
## [0.4.5] - 2026-04-22
11+
12+
Second hotfix for the "offline mode is enabled" crash on model load. 0.4.4 reverted the inference-path offline guards but kept the same trap on the load path, so users who updated to 0.4.4 kept hitting the exact error the release was supposed to fix ([#526](https://github.com/jamiepine/voicebox/issues/526)). This release removes the load-path guards and patches the transformers tokenizer load to be robust to HuggingFace metadata failures at the source, so the class of bug can't recur.
13+
14+
### Reliability
15+
16+
- **Load no longer fails with "offline mode is enabled"** ([#530](https://github.com/jamiepine/voicebox/pull/530), fixes [#526](https://github.com/jamiepine/voicebox/issues/526)). transformers 4.57.x added an unconditional `huggingface_hub.model_info()` call inside `AutoTokenizer.from_pretrained` (via `_patch_mistral_regex`) that runs for every non-local repo load, regardless of cache state or whether the target model is actually a Mistral variant. The load-time `HF_HUB_OFFLINE` guard from 0.4.2 turned that into a hard crash for cached online users the moment 0.4.4 removed the inference-path guard that had been masking the problem. Fix wraps `_patch_mistral_regex` so any exception from the HF metadata check is caught and the tokenizer is returned unchanged — matching the success-path behavior for non-Mistral repos. The wrapper installs at `backend.backends` import time so it covers Qwen Base, Qwen CustomVoice, TADA, and every other transformers-backed engine on Windows, Linux, and CUDA alike. The load-time `force_offline_if_cached` guards were removed — with the wrapper in place they provide zero value and only risk re-introducing the same failure mode.
17+
- **No more 30s pause when generating without a network.** The HuggingFace metadata timeout called out as a known caveat in 0.4.4 is covered by the same patch; offline users no longer wait for the check to time out before load completes.
18+
1019
## [0.4.4] - 2026-04-21
1120

1221
Hotfix for a regression in 0.4.3 where generation and transcription could fail outright with "offline mode is enabled" even when the user was online.
@@ -648,7 +657,8 @@ The first public release of Voicebox — an open-source voice synthesis studio p
648657

649658
Tauri v2, React, TypeScript, Tailwind CSS, FastAPI, Qwen3-TTS, Whisper, SQLite
650659

651-
[Unreleased]: https://github.com/jamiepine/voicebox/compare/v0.4.4...HEAD
660+
[Unreleased]: https://github.com/jamiepine/voicebox/compare/v0.4.5...HEAD
661+
[0.4.5]: https://github.com/jamiepine/voicebox/compare/v0.4.4...v0.4.5
652662
[0.4.4]: https://github.com/jamiepine/voicebox/compare/v0.4.3...v0.4.4
653663
[0.4.3]: https://github.com/jamiepine/voicebox/compare/v0.4.2...v0.4.3
654664
[0.4.2]: https://github.com/jamiepine/voicebox/compare/v0.4.1...v0.4.2

‎app/package.json‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,6 @@
11
{
22
"name": "@voicebox/app",
3-
"version": "0.4.4",
3+
"version": "0.4.5",
44
"private": true,
55
"type": "module",
66
"scripts": {

‎backend/__init__.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,3 +1,3 @@
11
# Backend package
22

3-
__version__ = "0.4.4"
3+
__version__ = "0.4.5"

‎backend/backends/__init__.py‎

Lines changed: 7 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -5,6 +5,13 @@
55
and a model config registry that eliminates per-engine dispatch maps.
66
"""
77

8+
# Install HF compatibility patches before any backend imports transformers /
9+
# huggingface_hub. The module runs ``patch_transformers_mistral_regex`` at
10+
# import time, which wraps transformers' tokenizer load against the
11+
# unconditional HuggingFace metadata call that otherwise raises on
12+
# HF_HUB_OFFLINE=1 and on network failures.
13+
from ..utils import hf_offline_patch # noqa: F401
14+
815
import threading
916
from dataclasses import dataclass, field
1017
from typing import Protocol, Optional, Tuple, List

‎backend/backends/mlx_backend.py‎

Lines changed: 2 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -20,7 +20,6 @@
2020
from . import TTSBackend, STTBackend, LANGUAGE_CODE_TO_NAME, WHISPER_HF_REPOS
2121
from .base import is_model_cached, combine_voice_prompts as _combine_voice_prompts, model_load_progress
2222
from ..utils.cache import get_cache_key, get_cached_voice_prompt, cache_voice_prompt
23-
from ..utils.hf_offline_patch import force_offline_if_cached
2423

2524

2625
class MLXTTSBackend:
@@ -99,8 +98,7 @@ def _load_model_sync(self, model_size: str):
9998

10099
logger.info("Loading MLX TTS model %s...", model_size)
101100

102-
with force_offline_if_cached(is_cached, model_name):
103-
self.model = load(model_path)
101+
self.model = load(model_path)
104102

105103
self._current_model_size = model_size
106104
self.model_size = model_size
@@ -311,8 +309,7 @@ def _load_model_sync(self, model_size: str):
311309
model_name = WHISPER_HF_REPOS.get(model_size, f"openai/whisper-{model_size}")
312310
logger.info("Loading MLX Whisper model %s...", model_size)
313311

314-
with force_offline_if_cached(is_cached, progress_model_name):
315-
self.model = load(model_name)
312+
self.model = load(model_name)
316313

317314
self.model_size = model_size
318315
logger.info("MLX Whisper model %s loaded successfully", model_size)

‎backend/backends/pytorch_backend.py‎

Lines changed: 16 additions & 19 deletions
Original file line numberDiff line numberDiff line change
@@ -21,7 +21,6 @@
2121
)
2222
from ..utils.cache import get_cache_key, get_cached_voice_prompt, cache_voice_prompt
2323
from ..utils.audio import load_audio
24-
from ..utils.hf_offline_patch import force_offline_if_cached
2524

2625

2726
class PyTorchTTSBackend:
@@ -106,21 +105,20 @@ def _load_model_sync(self, model_size: str):
106105
from huggingface_hub import constants as hf_constants
107106
tts_cache_dir = hf_constants.HF_HUB_CACHE
108107

109-
with force_offline_if_cached(is_cached, model_name):
110-
if self.device == "cpu":
111-
self.model = Qwen3TTSModel.from_pretrained(
112-
model_path,
113-
cache_dir=tts_cache_dir,
114-
torch_dtype=torch.float32,
115-
low_cpu_mem_usage=False,
116-
)
117-
else:
118-
self.model = Qwen3TTSModel.from_pretrained(
119-
model_path,
120-
cache_dir=tts_cache_dir,
121-
device_map=self.device,
122-
torch_dtype=torch.bfloat16,
123-
)
108+
if self.device == "cpu":
109+
self.model = Qwen3TTSModel.from_pretrained(
110+
model_path,
111+
cache_dir=tts_cache_dir,
112+
torch_dtype=torch.float32,
113+
low_cpu_mem_usage=False,
114+
)
115+
else:
116+
self.model = Qwen3TTSModel.from_pretrained(
117+
model_path,
118+
cache_dir=tts_cache_dir,
119+
device_map=self.device,
120+
torch_dtype=torch.bfloat16,
121+
)
124122

125123
self._current_model_size = model_size
126124
self.model_size = model_size
@@ -297,9 +295,8 @@ def _load_model_sync(self, model_size: str):
297295
model_name = WHISPER_HF_REPOS.get(model_size, f"openai/whisper-{model_size}")
298296
logger.info("Loading Whisper model %s on %s...", model_size, self.device)
299297

300-
with force_offline_if_cached(is_cached, progress_model_name):
301-
self.processor = WhisperProcessor.from_pretrained(model_name)
302-
self.model = WhisperForConditionalGeneration.from_pretrained(model_name)
298+
self.processor = WhisperProcessor.from_pretrained(model_name)
299+
self.model = WhisperForConditionalGeneration.from_pretrained(model_name)
303300

304301
self.model.to(self.device)
305302
self.model_size = model_size

‎backend/backends/qwen_custom_voice_backend.py‎

Lines changed: 0 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -28,7 +28,6 @@
2828
combine_voice_prompts as _combine_voice_prompts,
2929
model_load_progress,
3030
)
31-
from ..utils.hf_offline_patch import force_offline_if_cached
3231

3332
logger = logging.getLogger(__name__)
3433

Lines changed: 113 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,113 @@
1+
"""
2+
Unit tests for ``patch_transformers_mistral_regex``.
3+
4+
Verifies that our wrapper around
5+
``transformers.PreTrainedTokenizerBase._patch_mistral_regex`` catches
6+
exceptions from the unconditional ``huggingface_hub.model_info()`` lookup
7+
and returns the tokenizer unchanged — matching the success-path behavior
8+
for non-Mistral repos (transformers 4.57.3, ``tokenization_utils_base.py:2503``).
9+
10+
NOTE: These tests mutate ``transformers.PreTrainedTokenizerBase`` globally;
11+
run serially, not under ``pytest-xdist`` with per-worker process isolation.
12+
"""
13+
14+
import sys
15+
from pathlib import Path
16+
17+
import pytest
18+
19+
sys.path.insert(0, str(Path(__file__).parent.parent))
20+
21+
from huggingface_hub.errors import OfflineModeIsEnabled # noqa: E402
22+
from transformers.tokenization_utils_base import PreTrainedTokenizerBase # noqa: E402
23+
24+
import utils.hf_offline_patch as hf_offline_patch # noqa: E402
25+
26+
27+
@pytest.fixture(autouse=True)
28+
def restore_mistral_regex():
29+
"""Snapshot the current ``_patch_mistral_regex`` and restore after each test."""
30+
saved = PreTrainedTokenizerBase.__dict__.get("_patch_mistral_regex")
31+
saved_flag = hf_offline_patch._mistral_regex_patched
32+
try:
33+
yield
34+
finally:
35+
if saved is not None:
36+
PreTrainedTokenizerBase._patch_mistral_regex = saved
37+
hf_offline_patch._mistral_regex_patched = saved_flag
38+
39+
40+
def _apply_patch():
41+
hf_offline_patch._mistral_regex_patched = False
42+
hf_offline_patch.patch_transformers_mistral_regex()
43+
44+
45+
def test_suppresses_offline_mode_is_enabled(monkeypatch):
46+
_apply_patch()
47+
48+
import huggingface_hub
49+
50+
def raise_offline(*_args, **_kwargs):
51+
raise OfflineModeIsEnabled("offline")
52+
53+
monkeypatch.setattr(huggingface_hub, "model_info", raise_offline)
54+
55+
sentinel = object()
56+
result = PreTrainedTokenizerBase._patch_mistral_regex(
57+
sentinel, "Qwen/Qwen3-TTS-12Hz-1.7B-Base"
58+
)
59+
assert result is sentinel
60+
61+
62+
def test_suppresses_connection_errors(monkeypatch):
63+
_apply_patch()
64+
65+
import huggingface_hub
66+
67+
def raise_connection(*_args, **_kwargs):
68+
raise ConnectionError("network unreachable")
69+
70+
monkeypatch.setattr(huggingface_hub, "model_info", raise_connection)
71+
72+
sentinel = object()
73+
result = PreTrainedTokenizerBase._patch_mistral_regex(
74+
sentinel, "Qwen/Qwen3-TTS-12Hz-1.7B-Base"
75+
)
76+
assert result is sentinel
77+
78+
79+
def test_passthrough_on_success(monkeypatch):
80+
"""When model_info returns non-Mistral tags the original falls through and returns the tokenizer unchanged."""
81+
_apply_patch()
82+
83+
import huggingface_hub
84+
85+
class FakeInfo:
86+
tags = ["model-type:qwen", "language:en"]
87+
88+
monkeypatch.setattr(huggingface_hub, "model_info", lambda *_a, **_kw: FakeInfo())
89+
90+
sentinel = object()
91+
result = PreTrainedTokenizerBase._patch_mistral_regex(
92+
sentinel, "Qwen/Qwen3-TTS-12Hz-1.7B-Base"
93+
)
94+
assert result is sentinel
95+
96+
97+
def test_idempotent():
98+
_apply_patch()
99+
first = PreTrainedTokenizerBase._patch_mistral_regex
100+
hf_offline_patch.patch_transformers_mistral_regex()
101+
second = PreTrainedTokenizerBase._patch_mistral_regex
102+
assert first.__func__ is second.__func__
103+
104+
105+
def test_missing_method_is_noop(monkeypatch):
106+
monkeypatch.delattr(PreTrainedTokenizerBase, "_patch_mistral_regex", raising=False)
107+
hf_offline_patch._mistral_regex_patched = False
108+
hf_offline_patch.patch_transformers_mistral_regex()
109+
assert hf_offline_patch._mistral_regex_patched is False
110+
111+
112+
if __name__ == "__main__":
113+
pytest.main([__file__, "-v"])

‎backend/utils/hf_offline_patch.py‎

Lines changed: 52 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -142,6 +142,57 @@ def force_offline_if_cached(is_cached: bool, model_label: str = ""):
142142
_saved_transformers_const = None
143143

144144

145+
_mistral_regex_patched = False
146+
147+
148+
def patch_transformers_mistral_regex():
149+
"""Make transformers' tokenizer load robust to HuggingFace metadata failures.
150+
151+
transformers 4.57.x added ``PreTrainedTokenizerBase._patch_mistral_regex``
152+
which unconditionally calls ``huggingface_hub.model_info(repo_id)`` during
153+
every non-local tokenizer load to check whether the model is a Mistral
154+
variant. That call raises on ``HF_HUB_OFFLINE=1`` and on plain network
155+
failures, killing unrelated loads (Qwen TTS, TADA, etc.).
156+
157+
Voicebox never loads Mistral models, so the rewrite the function would
158+
apply is a no-op for us anyway. Wrap the method so any exception from the
159+
metadata lookup returns the tokenizer unchanged — matching the success-path
160+
behavior for non-Mistral repos (transformers 4.57.3,
161+
``tokenization_utils_base.py:2503``).
162+
"""
163+
global _mistral_regex_patched
164+
if _mistral_regex_patched:
165+
return
166+
167+
try:
168+
from transformers.tokenization_utils_base import PreTrainedTokenizerBase
169+
except ImportError:
170+
logger.debug("transformers not available, skipping mistral-regex patch")
171+
return
172+
173+
original = getattr(PreTrainedTokenizerBase, "_patch_mistral_regex", None)
174+
if original is None:
175+
logger.debug(
176+
"transformers has no _patch_mistral_regex attribute, skipping patch",
177+
)
178+
return
179+
180+
def safe_patch_mistral_regex(cls, tokenizer, pretrained_model_name_or_path, *args, **kwargs):
181+
try:
182+
return original(tokenizer, pretrained_model_name_or_path, *args, **kwargs)
183+
except Exception as exc:
184+
logger.debug(
185+
"[mistral-regex-patch] suppressed %s for %r, returning tokenizer as-is",
186+
type(exc).__name__,
187+
pretrained_model_name_or_path,
188+
)
189+
return tokenizer
190+
191+
PreTrainedTokenizerBase._patch_mistral_regex = classmethod(safe_patch_mistral_regex)
192+
_mistral_regex_patched = True
193+
logger.debug("installed _patch_mistral_regex wrapper")
194+
195+
145196
def patch_huggingface_hub_offline():
146197
"""Monkey-patch huggingface_hub to force offline mode."""
147198
try:
@@ -215,4 +266,5 @@ def ensure_original_qwen_config_cached():
215266

216267
if os.environ.get("VOICEBOX_OFFLINE_PATCH", "1") != "0":
217268
patch_huggingface_hub_offline()
269+
patch_transformers_mistral_regex()
218270
ensure_original_qwen_config_cached()

0 commit comments

Comments
 (0)