michael-chan-000 commited on 2 days ago

Commit

79529ed

verified ·

1 Parent(s): b04ca6a

Upload model

Browse files

Files changed (21) hide show

.gitattributes +1 -0
README.md +14 -0
__pycache__/miner.cpython-312.pyc +0 -0
added_tokens.json +35 -0
chute_config.yml +23 -0
config.json +163 -0
demo.py +98 -0
generation_config.json +12 -0
merges.txt +0 -0
miner.py +364 -0
model.safetensors +3 -0
preprocessor_config.json +6 -0
special_tokens_map.json +44 -0
speech_tokenizer/config.json +94 -0
speech_tokenizer/configuration.json +1 -0
speech_tokenizer/model.safetensors +3 -0
speech_tokenizer/preprocessor_config.json +10 -0
tokenizer.json +3 -0
tokenizer_config.json +318 -0
vocab.json +0 -0
vocence_config.yaml +16 -0

.gitattributes CHANGED Viewed

@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+tokenizer.json filter=lfs diff=lfs merge=lfs -text

README.md ADDED Viewed

	@@ -0,0 +1,14 @@

+---
+license: cc-by-nc-sa-4.0
+base_model: magma90909/vocence_miner_v8
+pipeline_tag: text-to-speech
+library_name: transformers
+language:
+  - en
+tags:
+  - tts
+  - prompttts
+  - qwen3-tts
+  - voice-design
+  - vocence
+---

__pycache__/miner.cpython-312.pyc ADDED Viewed

Binary file (17.8 kB). View file

added_tokens.json ADDED Viewed

	@@ -0,0 +1,35 @@

+{
+  "</think>": 151668,
+  "</tool_call>": 151658,
+  "</tool_response>": 151666,
+  "<think>": 151667,
+  "<tool_call>": 151657,
+  "<tool_response>": 151665,
+  "<tts_pad>": 151671,
+  "<tts_text_bos>": 151672,
+  "<tts_text_bos_single>": 151674,
+  "<tts_text_eod>": 151673,
+  "<|audio_end|>": 151670,
+  "<|audio_pad|>": 151675,
+  "<|audio_start|>": 151669,
+  "<|box_end|>": 151649,
+  "<|box_start|>": 151648,
+  "<|endoftext|>": 151643,
+  "<|file_sep|>": 151664,
+  "<|fim_middle|>": 151660,
+  "<|fim_pad|>": 151662,
+  "<|fim_prefix|>": 151659,
+  "<|fim_suffix|>": 151661,
+  "<|im_end|>": 151645,
+  "<|im_start|>": 151644,
+  "<|image_pad|>": 151655,
+  "<|object_ref_end|>": 151647,
+  "<|object_ref_start|>": 151646,
+  "<|quad_end|>": 151651,
+  "<|quad_start|>": 151650,
+  "<|repo_name|>": 151663,
+  "<|video_pad|>": 151656,
+  "<|vision_end|>": 151653,
+  "<|vision_pad|>": 151654,
+  "<|vision_start|>": 151652
+}

chute_config.yml ADDED Viewed

	@@ -0,0 +1,23 @@

+# Image + node + Chute for Vocence deploy. Required in the HF repo at build time.
+Image:
+  from_base: parachutes/python:3.12
+  run_command:
+    - pip install torch torchaudio transformers accelerate huggingface_hub pyyaml soundfile librosa
+    - pip install -U qwen-tts
+  set_workdir: /app
+NodeSelector:
+  gpu_count: 1
+  min_vram_gb_per_gpu: 24
+  include: ["pro_6000"]
+  exclude: []
+Chute:
+  tagline: Vocence TTS — Qwen3 PromptTTS (weights in repo)
+  readme: Qwen3 12Hz TTS snapshot + miner.py for Vocence
+  shutdown_after_seconds: 86400
+  concurrency: 1
+  max_instances: 1
+  scaling_threshold: 0.5
+  tee: true

config.json ADDED Viewed

	@@ -0,0 +1,163 @@

+{
+  "architectures": [
+    "Qwen3TTSForConditionalGeneration"
+  ],
+  "assistant_token_id": 77091,
+  "im_end_token_id": 151645,
+  "im_start_token_id": 151644,
+  "tts_bos_token_id": 151672,
+  "tts_eos_token_id": 151673,
+  "tts_pad_token_id": 151671,
+  "model_type": "qwen3_tts",
+  "tokenizer_type": "qwen3_tts_tokenizer_12hz",
+  "tts_model_size": "1b7",
+  "tts_model_type": "voice_design",
+  "talker_config": {
+    "attention_bias": false,
+    "attention_dropout": 0,
+    "code_predictor_config": {
+      "_name_or_path": "",
+      "add_cross_attention": false,
+      "architectures": null,
+      "attention_bias": false,
+      "attention_dropout": 0,
+      "bad_words_ids": null,
+      "begin_suppress_tokens": null,
+      "bos_token_id": null,
+      "chunk_size_feed_forward": 0,
+      "cross_attention_hidden_size": null,
+      "decoder_start_token_id": null,
+      "diversity_penalty": 0.0,
+      "do_sample": false,
+      "early_stopping": false,
+      "encoder_no_repeat_ngram_size": 0,
+      "eos_token_id": null,
+      "exponential_decay_length_penalty": null,
+      "finetuning_task": null,
+      "forced_bos_token_id": null,
+      "forced_eos_token_id": null,
+      "head_dim": 128,
+      "hidden_act": "silu",
+      "hidden_size": 1024,
+      "id2label": {
+        "0": "LABEL_0",
+        "1": "LABEL_1"
+      },
+      "initializer_range": 0.02,
+      "intermediate_size": 3072,
+      "is_decoder": false,
+      "is_encoder_decoder": false,
+      "label2id": {
+        "LABEL_0": 0,
+        "LABEL_1": 1
+      },
+      "layer_types": [
+        "full_attention",
+        "full_attention",
+        "full_attention",
+        "full_attention",
+        "full_attention"
+      ],
+      "length_penalty": 1.0,
+      "max_length": 20,
+      "max_position_embeddings": 65536,
+      "max_window_layers": 28,
+      "min_length": 0,
+      "model_type": "qwen3_tts_talker_code_predictor",
+      "no_repeat_ngram_size": 0,
+      "num_attention_heads": 16,
+      "num_beam_groups": 1,
+      "num_beams": 1,
+      "num_code_groups": 16,
+      "num_hidden_layers": 5,
+      "num_key_value_heads": 8,
+      "num_return_sequences": 1,
+      "output_attentions": false,
+      "output_hidden_states": false,
+      "output_scores": false,
+      "pad_token_id": null,
+      "prefix": null,
+      "problem_type": null,
+      "pruned_heads": {},
+      "remove_invalid_values": false,
+      "repetition_penalty": 1.0,
+      "return_dict": true,
+      "return_dict_in_generate": false,
+      "rms_norm_eps": 1e-06,
+      "rope_scaling": null,
+      "rope_theta": 1000000,
+      "sep_token_id": null,
+      "sliding_window": null,
+      "suppress_tokens": null,
+      "task_specific_params": null,
+      "temperature": 1.0,
+      "tf_legacy_loss": false,
+      "tie_encoder_decoder": false,
+      "tie_word_embeddings": false,
+      "tokenizer_class": null,
+      "top_k": 50,
+      "top_p": 1.0,
+      "dtype": null,
+      "torchscript": false,
+      "typical_p": 1.0,
+      "use_bfloat16": false,
+      "use_cache": true,
+      "use_sliding_window": false,
+      "vocab_size": 2048
+    },
+    "codec_bos_id": 2149,
+    "codec_eos_token_id": 2150,
+    "codec_think_id": 2154,
+    "codec_language_id": {
+        "chinese": 2055,
+        "english": 2050,
+        "german": 2053,
+        "italian": 2070,
+        "portuguese": 2071,
+        "spanish": 2054,
+        "japanese": 2058,
+        "korean": 2064,
+        "french": 2061,
+        "russian": 2069
+    },
+    "codec_nothink_id": 2155,
+    "codec_pad_id": 2148,
+    "codec_think_bos_id": 2156,
+    "codec_think_eos_id": 2157,
+    "spk_id": {
+    },
+    "spk_is_dialect": {
+    },
+    "head_dim": 128,
+    "hidden_act": "silu",
+    "hidden_size": 2048,
+    "initializer_range": 0.02,
+    "intermediate_size": 6144,
+    "max_position_embeddings": 32768,
+    "model_type": "qwen3_tts_talker",
+    "num_attention_heads": 16,
+    "num_code_groups": 16,
+    "num_hidden_layers": 28,
+    "num_key_value_heads": 8,
+    "position_id_per_seconds": 13,
+    "rms_norm_eps": 1e-06,
+    "rope_scaling": {
+      "interleaved": true,
+      "mrope_section": [
+        24,
+        20,
+        20
+      ],
+      "rope_type": "default",
+      "type": "default"
+    },
+    "rope_theta": 1000000,
+    "sliding_window": null,
+    "text_hidden_size": 2048,
+    "text_vocab_size": 151936,
+    "use_cache": true,
+    "use_sliding_window": false,
+    "vocab_size": 3072
+  },
+  "transformers_version": "4.57.3"
+}

demo.py ADDED Viewed

	@@ -0,0 +1,98 @@

+"""demo.py — quick smoke test for vocence_miner_v1.
+Reads the merged checkpoint either from a local path or from the Hugging Face Hub,
+then generates a small set of preset clips that exercise the prompt-following range.
+    pip install qwen-tts transformers torch soundfile
+    python demo.py                                       # uses the current directory
+    python demo.py --source magma90909/vocence_miner_v8  # pull from HF
+"""
+from __future__ import annotations
+import argparse
+import dataclasses
+import sys
+from pathlib import Path
+import soundfile as sf
+import torch
+from qwen_tts import Qwen3TTSModel
+@dataclasses.dataclass(frozen=True)
+class Sample:
+    slug: str
+    say: str
+    voice: str
+SAMPLES: tuple[Sample, ...] = (
+    Sample(
+        slug="warm_male_storyteller",
+        say="Long ago, in a kingdom by the sea, a young girl made a remarkable discovery.",
+        voice="An older male narrator reads a bedtime story slowly, with warmth.",
+    ),
+    Sample(
+        slug="whisper_female",
+        say="Don't say a word. Just listen carefully.",
+        voice="A young woman whispers, conspiratorial, low energy, very quiet.",
+    ),
+    Sample(
+        slug="projecting_announcer",
+        say="And he scores in the final second of the match!",
+        voice="A high-pitched announcer projects an exciting headline at a fast pace.",
+    ),
+)
+SAMPLER = dict(
+    temperature=0.85,
+    top_k=50,
+    top_p=0.95,
+    repetition_penalty=1.05,
+    max_new_tokens=600,
+    do_sample=True,
+)
+def parse_args(argv: list[str] | None = None) -> argparse.Namespace:
+    p = argparse.ArgumentParser(description=__doc__.split("\n", 1)[0])
+    p.add_argument("--source", default=".", help="HF repo id or local checkpoint dir")
+    p.add_argument("--out", default="./demo_out", help="output dir for wav files")
+    p.add_argument("--precision", default="bfloat16", choices=("bfloat16", "float16", "float32"))
+    p.add_argument("--device", default="cuda:0" if torch.cuda.is_available() else "cpu")
+    return p.parse_args(argv)
+def load(source: str, device: str, precision: str) -> Qwen3TTSModel:
+    dtype = {"bfloat16": torch.bfloat16, "float16": torch.float16, "float32": torch.float32}[precision]
+    print(f"[demo] loading {source!r} -> {device} ({precision})", flush=True)
+    return Qwen3TTSModel.from_pretrained(source, device_map=device, dtype=dtype)
+def synth_one(model: Qwen3TTSModel, sample: Sample, out_dir: Path) -> Path:
+    wavs, sr = model.generate_voice_design(
+        text=sample.say,
+        instruct=sample.voice,
+        language="english",
+        **SAMPLER,
+    )
+    target = out_dir / f"{sample.slug}.wav"
+    sf.write(target, wavs[0], sr)
+    duration = len(wavs[0]) / sr
+    print(f"  -> {target.name}  ({duration:.2f}s @ {sr} Hz)")
+    return target
+def run(args: argparse.Namespace) -> int:
+    out_dir = Path(args.out)
+    out_dir.mkdir(parents=True, exist_ok=True)
+    model = load(args.source, args.device, args.precision)
+    for sample in SAMPLES:
+        synth_one(model, sample, out_dir)
+    print(f"[demo] {len(SAMPLES)} clips written to {out_dir}/", flush=True)
+    return 0
+if __name__ == "__main__":
+    sys.exit(run(parse_args()))

generation_config.json ADDED Viewed

	@@ -0,0 +1,12 @@

+{
+  "do_sample": true,
+  "repetition_penalty": 1.05,
+  "temperature": 0.9,
+  "top_p": 1.0,
+  "top_k": 50,
+  "subtalker_dosample": true,
+  "subtalker_temperature": 0.9,
+  "subtalker_top_p": 1.0,
+  "subtalker_top_k": 50,
+  "max_new_tokens": 8192
+}

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

miner.py ADDED Viewed

	@@ -0,0 +1,364 @@

+"""Vocence engine for the merged Qwen3-TTS VoiceDesign checkpoint.
+The Vocence Chutes wrapper instantiates ``Miner`` with the on-disk path of the HF
+snapshot and then drives it through the contract:
+    Miner(path_hf_repo: Path)
+    warmup() -> None
+    generate_wav(instruction: str, text: str) -> tuple[np.ndarray, int]
+All weights, the audio codec, and the tokenizer ship together in the snapshot —
+nothing is fetched at runtime.
+"""
+from __future__ import annotations
+import dataclasses
+import re
+import threading
+from pathlib import Path
+from typing import Any
+import numpy as np
+_REPO_REQUIRED_FILE = "config.json"
+_RUNTIME_CONFIG_FILE = "vocence_config.yaml"
+# --------------------------------------------------------------------------- #
+# Instruction rewrite (tag -> natural-language preamble)                      #
+# --------------------------------------------------------------------------- #
+#
+# Validators may send instructions in the legacy pipe-tag form, e.g.
+# ``| gender: male | pitch: mid | accent: uk |``. The base voice_design
+# checkpoint was conditioned on natural-language descriptions, so we paraphrase
+# the tags into a short imperative preamble and *prepend* it to whatever the
+# caller sent. Free-form prompts (no ``| key: value |`` pairs) pass through
+# unchanged because ``_parse_instruction`` returns ``{}`` for them.
+# One ``| key: value |`` pair. Value runs until the next ``|`` or end-of-string;
+# the lookahead keeps the trailing ``|`` available for the next iteration.
+_INSTRUCTION_TAG_RE = re.compile(
+    r"\|\s*([A-Za-z_]+)\s*:\s*([^|]+?)\s*(?=\||$)"
+)
+_GENDER_PHRASE = {
+    "male": "male", "female": "female", "neutral": "gender-neutral",
+}
+_PITCH_PHRASE = {
+    "low": "deep low-pitched voice", "mid": "medium natural pitch", "high": "high-pitched voice",
+}
+_SPEED_PHRASE = {
+    "slow": "slow deliberate pace", "normal": "natural conversational pace", "fast": "brisk fast pace",
+}
+_AGE_PHRASE = {
+    "child": "child", "young_adult": "young adult", "adult": "adult", "senior": "elderly senior",
+}
+_EMOTION_PHRASE = {
+    "neutral": "neutral composed delivery",
+    "happy": "cheerful happy upbeat warm",
+    "sad": "sorrowful sad subdued downcast",
+    "angry": "firm angry forceful assertive tense",
+    "calm": "calm relaxed measured peaceful unhurried",
+    "excited": "excited enthusiastic energetic lively",
+    "serious": "serious grave deliberate weighty",
+    "fearful": "nervous fearful hesitant trembling",
+}
+_TONE_PHRASE = {
+    "warm": "warm", "cold": "cold detached", "friendly": "friendly",
+    "formal": "formal", "casual": "casual", "authoritative": "authoritative commanding",
+}
+_ACCENT_PHRASE = {
+    "us": "standard American English accent with rhotic r sounds",
+    "uk": "standard British English accent with non-rhotic received pronunciation",
+    "au": "Australian English accent",
+    "in": "Indian English accent",
+    "neutral": "neutral international English accent",
+    "other": "non-native English accent",
+}
+def _parse_instruction(instruction: str) -> dict[str, str]:
+    """Parse a pipe-tag instruction (``| key: value | ...``) into a flat dict.
+    Keys are lowercased; values are lowercased and stripped. Returns ``{}``
+    for free-form natural-language prompts (no tag pairs found), which
+    signals ``_enhance_instruction`` to pass them through unchanged. Unknown
+    or out-of-vocabulary values quietly drop out at preamble-build time
+    because the phrase tables only contain mappings we trust to be in the
+    base model's training distribution.
+    """
+    if not instruction or "|" not in instruction:
+        return {}
+    out: dict[str, str] = {}
+    for m in _INSTRUCTION_TAG_RE.finditer(instruction):
+        key = m.group(1).strip().lower()
+        val = m.group(2).strip().lower()
+        if key and val:
+            out[key] = val
+    return out
+def _build_natural_preamble(parsed: dict[str, str]) -> str:
+    gender = _GENDER_PHRASE.get(parsed.get("gender", ""), "")
+    age = _AGE_PHRASE.get(parsed.get("age_group", ""), "")
+    pitch = _PITCH_PHRASE.get(parsed.get("pitch", ""), "")
+    speed = _SPEED_PHRASE.get(parsed.get("speed", ""), "")
+    emotion = _EMOTION_PHRASE.get(parsed.get("emotion", ""), "")
+    tone = _TONE_PHRASE.get(parsed.get("tone", ""), "")
+    accent = _ACCENT_PHRASE.get(parsed.get("accent", ""), "")
+    parts: list[str] = []
+    # Gender-first to avoid timbre drift on emotion-heavy prompts
+    identity = " ".join(p for p in [gender, age] if p)
+    if identity:
+        parts.append(f"a {identity} voice")
+    if emotion:
+        parts.append(emotion)
+    if accent:
+        parts.append(f"speaking with a {accent}")
+    if pitch:
+        parts.append(pitch)
+    if speed:
+        parts.append(speed)
+    if tone:
+        parts.append(f"{tone} tone")
+    if not parts:
+        return ""
+    preamble = "Speak as " + ", ".join(parts) + "."
+    return preamble + " Use natural human prosody with realistic breath placement and varied intonation."
+def _enhance_instruction(instruction: str) -> str:
+    """Prepend a natural-language preamble derived from any pipe tags.
+    Pass-through when the input has no parseable tags or none of them map
+    to a known phrase (so the preamble would be empty). Always keeps the
+    original instruction at the end so the caller's free-form instructions
+    still influence the model.
+    """
+    parsed = _parse_instruction(instruction)
+    if not parsed:
+        return instruction
+    preamble = _build_natural_preamble(parsed)
+    if not preamble:
+        return instruction
+    return f"{preamble} {instruction}"
+# --------------------------------------------------------------------------- #
+# Text normalization                                                          #
+# --------------------------------------------------------------------------- #
+_NUM_WORDS = {
+    "0": "zero", "1": "one", "2": "two", "3": "three", "4": "four",
+    "5": "five", "6": "six", "7": "seven", "8": "eight", "9": "nine",
+    "10": "ten", "11": "eleven", "12": "twelve", "13": "thirteen",
+    "14": "fourteen", "15": "fifteen", "16": "sixteen", "17": "seventeen",
+    "18": "eighteen", "19": "nineteen", "20": "twenty", "30": "thirty",
+    "40": "forty", "50": "fifty", "60": "sixty", "70": "seventy",
+    "80": "eighty", "90": "ninety", "100": "one hundred",
+}
+_ABBREV = {
+    "Mr.": "Mister", "Mrs.": "Missus", "Dr.": "Doctor", "St.": "Saint",
+    "etc.": "et cetera", "vs.": "versus", "approx.": "approximately",
+    "dept.": "department", "govt.": "government", "mgr.": "manager",
+}
+# Pre-compiled at module load so we don't recompile on every call.
+_DOLLAR_RE = re.compile(r"\$(\d+)")
+_POUND_RE = re.compile(r"£(\d+)")
+_EURO_RE = re.compile(r"€(\d+)")
+_SMALL_INT_RE = re.compile(r"\b(\d{1,2})\b")
+_CONJ_RE = re.compile(
+    r"(?<!\,)\s+(but|however|although|though|yet)\s+",
+    flags=re.IGNORECASE,
+)
+def _normalize_text_for_tts(text: str) -> str:
+    """Rewrite a transcript so the talker emits cleaner, more prosodic speech.
+    Concretely: expand a small list of common abbreviations, turn currency-
+    prefixed integers into spelled-out phrases (``$5`` -> ``five dollars``),
+    spell out 1-2 digit standalone integers, and insert a comma before
+    coordinating conjunctions in long sentences so the model hears a beat
+    where humans naturally take one. Larger numbers, decimals, and unknown
+    abbreviations pass through unchanged.
+    """
+    # Expand known abbreviations
+    for abbr, expansion in _ABBREV.items():
+        text = text.replace(abbr, expansion)
+    # Expand $N / £N / €N → "N dollars/pounds/euros"
+    text = _DOLLAR_RE.sub(
+        lambda m: f"{_NUM_WORDS.get(m.group(1), m.group(1))} dollars", text
+    )
+    text = _POUND_RE.sub(
+        lambda m: f"{_NUM_WORDS.get(m.group(1), m.group(1))} pounds", text
+    )
+    text = _EURO_RE.sub(
+        lambda m: f"{_NUM_WORDS.get(m.group(1), m.group(1))} euros", text
+    )
+    # Expand standalone small integers (not part of larger numbers)
+    text = _SMALL_INT_RE.sub(
+        lambda m: _NUM_WORDS.get(m.group(1), m.group(1)),
+        text,
+    )
+    # Add comma pause before coordinating conjunctions in long sentences
+    text = _CONJ_RE.sub(r", \1 ", text)
+    return text.strip()
+@dataclasses.dataclass
+class _RuntimeOpts:
+    """Subset of vocence_config.yaml that the engine actually consumes."""
+    language: str = "English"
+    sample_rate: int = 24000
+    max_instruction_chars: int = 600
+    max_text_chars: int = 2000
+    device_pref: str = "cuda"
+    dtype_pref: str = "bfloat16"
+    flash_attention_2: bool = False
+    @classmethod
+    def from_repo(cls, repo: Path) -> "_RuntimeOpts":
+        cfg_path = repo / _RUNTIME_CONFIG_FILE
+        if not cfg_path.is_file():
+            return cls()
+        from yaml import safe_load
+        with cfg_path.open("r", encoding="utf-8") as fh:
+            data = safe_load(fh) or {}
+        runtime = data.get("runtime") or {}
+        generation = data.get("generation") or {}
+        limits = data.get("limits") or {}
+        return cls(
+            language=str(limits.get("default_language") or runtime.get("default_language") or "English"),
+            sample_rate=int(generation.get("sample_rate", 24000)),
+            max_instruction_chars=int(limits.get("max_instruction_chars", 600)),
+            max_text_chars=int(limits.get("max_text_chars", 2000)),
+            device_pref=str(runtime.get("device_preference", "cuda")).lower(),
+            dtype_pref=str(runtime.get("dtype", "bfloat16")).lower(),
+            flash_attention_2=bool(runtime.get("use_flash_attention_2", False)),
+        )
+class Miner:
+    """Loads merged Qwen3-TTS weights from the snapshot and serves the Vocence API."""
+    WARMUP_BUDGET_S = 180.0
+    def __init__(self, path_hf_repo: Path) -> None:
+        self.repo = Path(path_hf_repo).resolve()
+        if not (self.repo / _REPO_REQUIRED_FILE).is_file():
+            raise FileNotFoundError(
+                f"Snapshot incomplete: {self.repo / _REPO_REQUIRED_FILE} not found"
+            )
+        self.opts = _RuntimeOpts.from_repo(self.repo)
+        self.model = self._build_model()
+    def __repr__(self) -> str:
+        return f"<Miner repo={self.repo.name} language={self.opts.language!r}>"
+    # ------------------------------------------------------------------ #
+    # Vocence contract                                                    #
+    # ------------------------------------------------------------------ #
+    def warmup(self) -> None:
+        outcome: dict[str, Any] = {"ok": False, "err": None}
+        def _heat() -> None:
+            try:
+                self.generate_wav(instruction="Calm neutral delivery.", text="Warmup.")
+                outcome["ok"] = True
+            except Exception as exc:  # noqa: BLE001 — surface to host
+                outcome["err"] = repr(exc)
+        worker = threading.Thread(target=_heat, daemon=True)
+        worker.start()
+        worker.join(timeout=self.WARMUP_BUDGET_S)
+        if not outcome["ok"]:
+            raise RuntimeError(f"Miner warmup did not complete: {outcome['err'] or 'timeout'}")
+    def generate_wav(self, instruction: str, text: str) -> tuple[np.ndarray, int]:
+        # Cap raw inputs first so an oversized payload never reaches the
+        # rewriter (which would just throw away the surplus anyway).
+        prompt = self._truncate(instruction, self.opts.max_instruction_chars)
+        body = self._truncate(text, self.opts.max_text_chars)
+        # Tag-form instructions get a natural-language preamble prepended;
+        # already-natural instructions pass through untouched.
+        prompt = _enhance_instruction(prompt)
+        # Spell out numbers/currency, expand a few abbreviations, and add
+        # a beat before coordinating conjunctions in long sentences.
+        body = _normalize_text_for_tts(body)
+        # The preamble + abbreviation/number expansion can lengthen the
+        # strings; re-clip to the same limits so we honour the contract
+        # advertised in vocence_config.yaml's ``limits`` block.
+        prompt = self._truncate(prompt, self.opts.max_instruction_chars)
+        body = self._truncate(body, self.opts.max_text_chars)
+        wavs, sample_rate = self.model.generate_voice_design(
+            text=body,
+            instruct=prompt,
+            language=self.opts.language,
+        )
+        if not wavs or wavs[0] is None:
+            raise ValueError("Qwen3-TTS returned no audio")
+        wave = self._coerce_mono_float32(wavs[0])
+        return wave, int(sample_rate)
+    # ------------------------------------------------------------------ #
+    # Internal                                                            #
+    # ------------------------------------------------------------------ #
+    @staticmethod
+    def _truncate(value: str, limit: int) -> str:
+        return value[:limit] if limit and limit > 0 else value
+    @staticmethod
+    def _coerce_mono_float32(arr: Any) -> np.ndarray:
+        wave = np.asarray(arr, dtype=np.float32)
+        if wave.ndim > 1:
+            wave = wave.mean(axis=1)
+        return wave
+    def _build_model(self):
+        import torch
+        from qwen_tts import Qwen3TTSModel
+        cuda_available = bool(torch.cuda.is_available())
+        device_map = "cuda:0" if (self.opts.device_pref == "cuda" and cuda_available) else "cpu"
+        torch_dtype = (
+            torch.bfloat16
+            if (self.opts.dtype_pref == "bfloat16" and cuda_available)
+            else torch.float32
+        )
+        attempt_order = ("flash_attention_2", "sdpa") if self.opts.flash_attention_2 else ("sdpa",)
+        last_error: BaseException | None = None
+        for attn in attempt_order:
+            try:
+                model = Qwen3TTSModel.from_pretrained(
+                    pretrained_model_name_or_path=str(self.repo),
+                    device_map=device_map,
+                    dtype=torch_dtype,
+                    attn_implementation=attn,
+                )
+                print(
+                    f"[Miner] Qwen3-TTS ready on {device_map} "
+                    f"(dtype={self.opts.dtype_pref}, attn={attn})"
+                )
+                return model
+            except Exception as exc:  # noqa: BLE001 — try next attn variant
+                last_error = exc
+        raise RuntimeError(f"Qwen3-TTS failed to load: {last_error!r}")

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:2151faf8114f41ac60301872e5e1b50020a865fde1e2d15862231aff7a7c04be
+size 3833402644

preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "padding_side": "left",
+  "padding_value": 0.0,
+  "processor_class": "Qwen3TTSProcessor",
+  "return_attention_mask": true
+}

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,44 @@

+{
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>",
+    "<|audio_start|>",
+    "<|audio_end|>",
+    "<tts_pad>",
+    "<tts_text_bos>",
+    "<tts_text_bos_single>",
+    "<|audio_pad|>"
+  ],
+  "audio_bos_token": "<|audio_start|>",
+  "audio_eos_token": "<|audio_end|>",
+  "audio_token": "<|audio_pad|>",
+  "eos_token": {
+    "content": "<|im_end|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "image_token": "<|image_pad|>",
+  "pad_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "video_token": "<|video_pad|>",
+  "vision_bos_token": "<|vision_start|>",
+  "vision_eos_token": "<|vision_end|>"
+}

speech_tokenizer/config.json ADDED Viewed

	@@ -0,0 +1,94 @@

+{
+  "architectures": [
+    "Qwen3TTSTokenizerV2Model"
+  ],
+  "model_type": "qwen3_tts_tokenizer_12hz",
+  "encoder_valid_num_quantizers": 16,
+  "input_sample_rate": 24000,
+  "output_sample_rate": 24000,
+  "decode_upsample_rate": 1920,
+  "encode_downsample_rate": 1920,
+  "decoder_config": {
+    "attention_bias": false,
+    "attention_dropout": 0.0,
+    "latent_dim": 1024,
+    "codebook_dim": 512,
+    "codebook_size": 2048,
+    "decoder_dim": 1536,
+    "hidden_act": "silu",
+    "hidden_size": 512,
+    "intermediate_size": 1024,
+    "layer_scale_initial_scale": 0.01,
+    "max_position_embeddings": 8000,
+    "head_dim": 64,
+    "num_attention_heads": 16,
+    "num_hidden_layers": 8,
+    "num_key_value_heads": 16,
+    "num_quantizers": 16,
+    "num_semantic_quantizers": 1,
+    "rms_norm_eps": 1e-05,
+    "rope_theta": 10000,
+    "semantic_codebook_size": 4096,
+    "sliding_window": 72,
+    "upsample_rates": [
+      8,
+      5,
+      4,
+      3
+    ],
+    "upsampling_ratios": [
+      2,
+      2
+    ],
+    "vector_quantization_hidden_dimension": 512
+  },
+  "encoder_config": {
+    "_frame_rate": 12.5,
+    "attention_bias": false,
+    "attention_dropout": 0.0,
+    "audio_channels": 1,
+    "codebook_dim": 256,
+    "codebook_size": 2048,
+    "compress": 2,
+    "dilation_growth_rate": 2,
+    "dtype": "float32",
+    "head_dim": 64,
+    "hidden_act": "gelu",
+    "hidden_size": 512,
+    "initializer_range": 0.02,
+    "intermediate_size": 2048,
+    "kernel_size": 7,
+    "last_kernel_size": 3,
+    "layer_scale_initial_scale": 0.01,
+    "max_position_embeddings": 8000,
+    "norm_eps": 1e-05,
+    "normalize": false,
+    "num_attention_heads": 8,
+    "num_filters": 64,
+    "num_hidden_layers": 8,
+    "num_key_value_heads": 8,
+    "num_quantizers": 32,
+    "num_residual_layers": 1,
+    "num_semantic_quantizers": 1,
+    "pad_mode": "constant",
+    "residual_kernel_size": 3,
+    "rope_theta": 10000.0,
+    "sampling_rate": 24000,
+    "sliding_window": 250,
+    "transformers_version": "4.57.0.dev0",
+    "trim_right_ratio": 1.0,
+    "upsample_groups": 512,
+    "upsampling_ratios": [
+      8,
+      6,
+      5,
+      4
+    ],
+    "use_cache": false,
+    "use_causal_conv": true,
+    "use_conv_shortcut": false,
+    "use_streaming": false,
+    "vector_quantization_hidden_dimension": 256
+  },
+  "transformers_version": "4.57.3"
+}

speech_tokenizer/configuration.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"framework": "pytorch", "task": "feature-extraction", "allow_remote": true}

speech_tokenizer/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:836b7b357f5ea43e889936a3709af68dfe3751881acefe4ecf0dbd30ba571258
+size 682293092

speech_tokenizer/preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,10 @@

+{
+  "chunk_length_s": null,
+  "feature_extractor_type": "EncodecFeatureExtractor",
+  "feature_size": 1,
+  "overlap": null,
+  "padding_side": "right",
+  "padding_value": 0.0,
+  "return_attention_mask": true,
+  "sampling_rate": 24000
+}

tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:09267689b8362020b9763b65dd5be7e086b31e28d72e02837a9e781de9a91bc7
+size 11423986

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,318 @@

+{
+  "add_bos_token": false,
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "151643": {
+      "content": "<|endoftext|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151644": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151645": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151646": {
+      "content": "<|object_ref_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151647": {
+      "content": "<|object_ref_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151648": {
+      "content": "<|box_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151649": {
+      "content": "<|box_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151650": {
+      "content": "<|quad_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151651": {
+      "content": "<|quad_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151652": {
+      "content": "<|vision_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151653": {
+      "content": "<|vision_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151654": {
+      "content": "<|vision_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151655": {
+      "content": "<|image_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151656": {
+      "content": "<|video_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151657": {
+      "content": "<tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151658": {
+      "content": "</tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151659": {
+      "content": "<|fim_prefix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151660": {
+      "content": "<|fim_middle|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151661": {
+      "content": "<|fim_suffix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151662": {
+      "content": "<|fim_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151663": {
+      "content": "<|repo_name|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151664": {
+      "content": "<|file_sep|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151665": {
+      "content": "<tool_response>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151666": {
+      "content": "</tool_response>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151667": {
+      "content": "<think>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151668": {
+      "content": "</think>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151669": {
+      "content": "<|audio_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151670": {
+      "content": "<|audio_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151671": {
+      "content": "<tts_pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151672": {
+      "content": "<tts_text_bos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151673": {
+      "content": "<tts_text_eod>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151674": {
+      "content": "<tts_text_bos_single>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151675": {
+      "content": "<|audio_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>",
+    "<|audio_start|>",
+    "<|audio_end|>",
+    "<tts_pad>",
+    "<tts_text_bos>",
+    "<tts_text_bos_single>",
+    "<|audio_pad|>"
+  ],
+  "audio_bos_token": "<|audio_start|>",
+  "audio_eos_token": "<|audio_end|>",
+  "audio_token": "<|audio_pad|>",
+  "bos_token": null,
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|im_end|>",
+  "errors": "replace",
+  "extra_special_tokens": {
+    "audio_bos_token": "<|audio_start|>",
+    "audio_eos_token": "<|audio_end|>",
+    "audio_token": "<|audio_pad|>",
+    "image_token": "<|image_pad|>",
+    "video_token": "<|video_pad|>",
+    "vision_bos_token": "<|vision_start|>",
+    "vision_eos_token": "<|vision_end|>"
+  },
+  "fix_mistral_regex": true,
+  "image_token": "<|image_pad|>",
+  "model_max_length": 131072,
+  "pad_token": "<|endoftext|>",
+  "processor_class": "Qwen3TTSProcessor",
+  "split_special_tokens": false,
+  "tokenizer_class": "Qwen2Tokenizer",
+  "unk_token": null,
+  "video_token": "<|video_pad|>",
+  "vision_bos_token": "<|vision_start|>",
+  "vision_eos_token": "<|vision_end|>"
+}

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff

vocence_config.yaml ADDED Viewed

	@@ -0,0 +1,16 @@

+# Miner + /health metadata. Weights live in this HF repo (no runtime model_id).
+runtime:
+  adapter: "qwen3_tts_repo_snapshot"
+  device_preference: "cuda"
+  dtype: "bfloat16"
+  default_language: "English"
+  use_flash_attention_2: false
+generation:
+  sample_rate: 24000
+  max_seconds: 30
+limits:
+  max_text_chars: 2000
+  max_instruction_chars: 600
+  default_language: "English"