michael-chan-000 commited on Apr 29

Commit

aedb5d2

verified ·

1 Parent(s): 3c13b21

Upload model

Browse files

Files changed (19) hide show

.gitattributes +1 -0
__pycache__/miner.cpython-312.pyc +0 -0
added_tokens.json +35 -0
chute_config.yml +23 -0
config.json +304 -0
generation_config.json +12 -0
merges.txt +0 -0
miner.py +240 -0
model.safetensors +3 -0
preprocessor_config.json +6 -0
special_tokens_map.json +44 -0
speech_tokenizer/config.json +94 -0
speech_tokenizer/configuration.json +1 -0
speech_tokenizer/model.safetensors +3 -0
speech_tokenizer/preprocessor_config.json +10 -0
tokenizer.json +3 -0
tokenizer_config.json +316 -0
vocab.json +0 -0
vocence_config.yaml +16 -0

.gitattributes CHANGED Viewed

@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+tokenizer.json filter=lfs diff=lfs merge=lfs -text

__pycache__/miner.cpython-312.pyc ADDED Viewed

Binary file (11.1 kB). View file

added_tokens.json ADDED Viewed

	@@ -0,0 +1,35 @@

+{
+  "</think>": 151668,
+  "</tool_call>": 151658,
+  "</tool_response>": 151666,
+  "<think>": 151667,
+  "<tool_call>": 151657,
+  "<tool_response>": 151665,
+  "<tts_pad>": 151671,
+  "<tts_text_bos>": 151672,
+  "<tts_text_bos_single>": 151674,
+  "<tts_text_eod>": 151673,
+  "<|audio_end|>": 151670,
+  "<|audio_pad|>": 151675,
+  "<|audio_start|>": 151669,
+  "<|box_end|>": 151649,
+  "<|box_start|>": 151648,
+  "<|endoftext|>": 151643,
+  "<|file_sep|>": 151664,
+  "<|fim_middle|>": 151660,
+  "<|fim_pad|>": 151662,
+  "<|fim_prefix|>": 151659,
+  "<|fim_suffix|>": 151661,
+  "<|im_end|>": 151645,
+  "<|im_start|>": 151644,
+  "<|image_pad|>": 151655,
+  "<|object_ref_end|>": 151647,
+  "<|object_ref_start|>": 151646,
+  "<|quad_end|>": 151651,
+  "<|quad_start|>": 151650,
+  "<|repo_name|>": 151663,
+  "<|video_pad|>": 151656,
+  "<|vision_end|>": 151653,
+  "<|vision_pad|>": 151654,
+  "<|vision_start|>": 151652
+}

chute_config.yml ADDED Viewed

	@@ -0,0 +1,23 @@

+# Image + node + Chute for Vocence deploy. Required in the HF repo at build time.
+Image:
+  from_base: parachutes/python:3.12
+  run_command:
+    - pip install torch torchaudio transformers accelerate huggingface_hub pyyaml soundfile librosa
+    - pip install -U qwen-tts
+  set_workdir: /app
+NodeSelector:
+  gpu_count: 1
+  min_vram_gb_per_gpu: 24
+  include: ["pro_6000"]
+  exclude: []
+Chute:
+  tagline: Vocence TTS — Qwen3 PromptTTS (weights in repo)
+  readme: Qwen3 12Hz TTS snapshot + miner.py for Vocence
+  shutdown_after_seconds: 86400
+  concurrency: 1
+  max_instances: 1
+  scaling_threshold: 0.5
+  tee: true

config.json ADDED Viewed

	@@ -0,0 +1,304 @@

+{
+  "_name_or_path": "/home/base-model",
+  "add_cross_attention": false,
+  "architectures": [
+    "Qwen3TTSForConditionalGeneration"
+  ],
+  "assistant_token_id": 77091,
+  "bad_words_ids": null,
+  "begin_suppress_tokens": null,
+  "bos_token_id": null,
+  "chunk_size_feed_forward": 0,
+  "cross_attention_hidden_size": null,
+  "decoder_start_token_id": null,
+  "diversity_penalty": 0.0,
+  "do_sample": false,
+  "early_stopping": false,
+  "encoder_no_repeat_ngram_size": 0,
+  "eos_token_id": null,
+  "exponential_decay_length_penalty": null,
+  "finetuning_task": null,
+  "forced_bos_token_id": null,
+  "forced_eos_token_id": null,
+  "id2label": {
+    "0": "LABEL_0",
+    "1": "LABEL_1"
+  },
+  "im_end_token_id": 151645,
+  "im_start_token_id": 151644,
+  "is_decoder": false,
+  "is_encoder_decoder": false,
+  "label2id": {
+    "LABEL_0": 0,
+    "LABEL_1": 1
+  },
+  "length_penalty": 1.0,
+  "max_length": 20,
+  "min_length": 0,
+  "model_type": "qwen3_tts",
+  "no_repeat_ngram_size": 0,
+  "num_beam_groups": 1,
+  "num_beams": 1,
+  "num_return_sequences": 1,
+  "output_attentions": false,
+  "output_hidden_states": false,
+  "output_scores": false,
+  "pad_token_id": null,
+  "prefix": null,
+  "problem_type": null,
+  "pruned_heads": {},
+  "remove_invalid_values": false,
+  "repetition_penalty": 1.0,
+  "return_dict": true,
+  "return_dict_in_generate": false,
+  "sep_token_id": null,
+  "speaker_encoder_config": {
+    "enc_attention_channels": 128,
+    "enc_channels": [
+      512,
+      512,
+      512,
+      512,
+      1536
+    ],
+    "enc_dilations": [
+      1,
+      2,
+      3,
+      4,
+      1
+    ],
+    "enc_dim": 1024,
+    "enc_kernel_sizes": [
+      5,
+      3,
+      3,
+      3,
+      1
+    ],
+    "enc_res2net_scale": 8,
+    "enc_se_channels": 128,
+    "mel_dim": 128,
+    "sample_rate": 24000
+  },
+  "suppress_tokens": null,
+  "talker_config": {
+    "_name_or_path": "",
+    "add_cross_attention": false,
+    "architectures": null,
+    "attention_bias": false,
+    "attention_dropout": 0,
+    "bad_words_ids": null,
+    "begin_suppress_tokens": null,
+    "bos_token_id": null,
+    "chunk_size_feed_forward": 0,
+    "code_predictor_config": {
+      "_name_or_path": "",
+      "add_cross_attention": false,
+      "architectures": null,
+      "attention_bias": false,
+      "attention_dropout": 0,
+      "bad_words_ids": null,
+      "begin_suppress_tokens": null,
+      "bos_token_id": null,
+      "chunk_size_feed_forward": 0,
+      "cross_attention_hidden_size": null,
+      "decoder_start_token_id": null,
+      "diversity_penalty": 0.0,
+      "do_sample": false,
+      "early_stopping": false,
+      "encoder_no_repeat_ngram_size": 0,
+      "eos_token_id": null,
+      "exponential_decay_length_penalty": null,
+      "finetuning_task": null,
+      "forced_bos_token_id": null,
+      "forced_eos_token_id": null,
+      "head_dim": 128,
+      "hidden_act": "silu",
+      "hidden_size": 1024,
+      "id2label": {
+        "0": "LABEL_0",
+        "1": "LABEL_1"
+      },
+      "initializer_range": 0.02,
+      "intermediate_size": 3072,
+      "is_decoder": false,
+      "is_encoder_decoder": false,
+      "label2id": {
+        "LABEL_0": 0,
+        "LABEL_1": 1
+      },
+      "layer_types": [
+        "full_attention",
+        "full_attention",
+        "full_attention",
+        "full_attention",
+        "full_attention"
+      ],
+      "length_penalty": 1.0,
+      "max_length": 20,
+      "max_position_embeddings": 65536,
+      "max_window_layers": 28,
+      "min_length": 0,
+      "no_repeat_ngram_size": 0,
+      "num_attention_heads": 16,
+      "num_beam_groups": 1,
+      "num_beams": 1,
+      "num_code_groups": 16,
+      "num_hidden_layers": 5,
+      "num_key_value_heads": 8,
+      "num_return_sequences": 1,
+      "output_attentions": false,
+      "output_hidden_states": false,
+      "output_scores": false,
+      "pad_token_id": 2148,
+      "prefix": null,
+      "problem_type": null,
+      "pruned_heads": {},
+      "remove_invalid_values": false,
+      "repetition_penalty": 1.0,
+      "return_dict": true,
+      "return_dict_in_generate": false,
+      "rms_norm_eps": 1e-06,
+      "rope_scaling": null,
+      "rope_theta": 1000000,
+      "sep_token_id": null,
+      "sliding_window": null,
+      "suppress_tokens": null,
+      "task_specific_params": null,
+      "temperature": 1.0,
+      "tf_legacy_loss": false,
+      "tie_encoder_decoder": false,
+      "tie_word_embeddings": false,
+      "tokenizer_class": null,
+      "top_k": 50,
+      "top_p": 1.0,
+      "torchscript": false,
+      "typical_p": 1.0,
+      "use_bfloat16": false,
+      "use_cache": true,
+      "use_sliding_window": false,
+      "vocab_size": 2048
+    },
+    "codec_bos_id": 2149,
+    "codec_eos_token_id": 2150,
+    "codec_language_id": {
+      "chinese": 2055,
+      "english": 2050,
+      "french": 2061,
+      "german": 2053,
+      "italian": 2070,
+      "japanese": 2058,
+      "korean": 2064,
+      "portuguese": 2071,
+      "russian": 2069,
+      "spanish": 2054
+    },
+    "codec_nothink_id": 2155,
+    "codec_pad_id": 2148,
+    "codec_think_bos_id": 2156,
+    "codec_think_eos_id": 2157,
+    "codec_think_id": 2154,
+    "cross_attention_hidden_size": null,
+    "decoder_start_token_id": null,
+    "diversity_penalty": 0.0,
+    "do_sample": false,
+    "early_stopping": false,
+    "encoder_no_repeat_ngram_size": 0,
+    "eos_token_id": null,
+    "exponential_decay_length_penalty": null,
+    "finetuning_task": null,
+    "forced_bos_token_id": null,
+    "forced_eos_token_id": null,
+    "head_dim": 128,
+    "hidden_act": "silu",
+    "hidden_size": 2048,
+    "id2label": {
+      "0": "LABEL_0",
+      "1": "LABEL_1"
+    },
+    "initializer_range": 0.02,
+    "intermediate_size": 6144,
+    "is_decoder": false,
+    "is_encoder_decoder": false,
+    "label2id": {
+      "LABEL_0": 0,
+      "LABEL_1": 1
+    },
+    "length_penalty": 1.0,
+    "max_length": 20,
+    "max_position_embeddings": 32768,
+    "min_length": 0,
+    "no_repeat_ngram_size": 0,
+    "num_attention_heads": 16,
+    "num_beam_groups": 1,
+    "num_beams": 1,
+    "num_code_groups": 16,
+    "num_hidden_layers": 28,
+    "num_key_value_heads": 8,
+    "num_return_sequences": 1,
+    "output_attentions": false,
+    "output_hidden_states": false,
+    "output_scores": false,
+    "pad_token_id": 151671,
+    "position_id_per_seconds": 13,
+    "prefix": null,
+    "problem_type": null,
+    "pruned_heads": {},
+    "remove_invalid_values": false,
+    "repetition_penalty": 1.0,
+    "return_dict": true,
+    "return_dict_in_generate": false,
+    "rms_norm_eps": 1e-06,
+    "rope_scaling": {
+      "interleaved": true,
+      "mrope_section": [
+        24,
+        20,
+        20
+      ],
+      "rope_type": "default",
+      "type": "default"
+    },
+    "rope_theta": 1000000,
+    "sep_token_id": null,
+    "sliding_window": null,
+    "spk_id": {},
+    "spk_is_dialect": {},
+    "suppress_tokens": null,
+    "task_specific_params": null,
+    "temperature": 1.0,
+    "text_hidden_size": 2048,
+    "text_vocab_size": 151936,
+    "tf_legacy_loss": false,
+    "tie_encoder_decoder": false,
+    "tie_word_embeddings": false,
+    "tokenizer_class": null,
+    "top_k": 50,
+    "top_p": 1.0,
+    "torchscript": false,
+    "typical_p": 1.0,
+    "use_bfloat16": false,
+    "use_cache": true,
+    "use_sliding_window": false,
+    "vocab_size": 3072
+  },
+  "task_specific_params": null,
+  "temperature": 1.0,
+  "tf_legacy_loss": false,
+  "tie_encoder_decoder": false,
+  "tie_word_embeddings": true,
+  "tokenizer_class": null,
+  "tokenizer_type": "qwen3_tts_tokenizer_12hz",
+  "top_k": 50,
+  "top_p": 1.0,
+  "torchscript": false,
+  "transformers_version": "4.57.3",
+  "tts_bos_token_id": 151672,
+  "tts_eos_token_id": 151673,
+  "tts_model_size": "1b7",
+  "tts_model_type": "voice_design",
+  "tts_pad_token_id": 151671,
+  "typical_p": 1.0,
+  "use_bfloat16": false
+}

generation_config.json ADDED Viewed

	@@ -0,0 +1,12 @@

+{
+  "do_sample": true,
+  "repetition_penalty": 1.05,
+  "temperature": 0.9,
+  "top_p": 1.0,
+  "top_k": 50,
+  "subtalker_dosample": true,
+  "subtalker_temperature": 0.9,
+  "subtalker_top_p": 1.0,
+  "subtalker_top_k": 50,
+  "max_new_tokens": 8192
+}

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

miner.py ADDED Viewed

	@@ -0,0 +1,240 @@

+"""
+Vocence TTS engine: Qwen3 12Hz checkpoint in the HF repo snapshot.
+The chute snapshot is the only weight source: nothing is pulled from an external
+model id at inference time. Optional vocence_config.yaml tweaks device, dtype,
+attention, and language defaults.
+This file is self-contained: transformers + qwen-tts compatibility (config
+sanitization, check_model_inputs shim) is inlined so Chute does not need a
+separate helper module.
+Model load: Miner.__init__ -> _instantiate_qwen() -> Qwen3TTSModel.from_pretrained(repo_path).
+Contract (Vocence):
+  Miner(path_hf_repo: Path)
+  warmup() -> None
+  generate_wav(instruction: str, text: str) -> tuple[np.ndarray, int]
+"""
+from __future__ import annotations
+import json
+import threading
+from pathlib import Path
+from typing import Any, Mapping
+import numpy as np
+import transformers.utils.generic as _g
+# --- Inlined from qwen3_tts_load_utils: must run before ``from qwen_tts`` (via _instantiate_qwen) ---
+_apply_shim_done = False
+def _apply_check_model_inputs_shim() -> None:
+    """qwen-tts @check_model_inputs() vs current transformers check_model_inputs factory API."""
+    global _apply_shim_done
+    if _apply_shim_done or getattr(_g, "_qwen_tts_check_model_inputs_shim_applied", False):
+        return
+    _orig = _g.check_model_inputs
+    def check_model_inputs(*args, **kwargs):
+        if not args and not kwargs:
+            return _orig()
+        if len(args) == 1 and callable(args[0]) and not kwargs:
+            return _orig()(args[0])
+        return _orig(*args, **kwargs)
+    _g.check_model_inputs = check_model_inputs
+    _g._qwen_tts_check_model_inputs_shim_applied = True
+    _apply_shim_done = True
+_UNWANTED_DTYPE_KEYS = frozenset({"dtype", "torch_dtype"})
+_NESTED_METADATA_KEYS = frozenset({"model_type"})
+def _strip_config_for_qwen3_load(obj: object, depth: int = 0) -> int:
+    n = 0
+    if isinstance(obj, dict):
+        for k in _UNWANTED_DTYPE_KEYS:
+            if k in obj:
+                del obj[k]
+                n += 1
+        if depth > 0:
+            for k in _NESTED_METADATA_KEYS:
+                if k in obj:
+                    del obj[k]
+                    n += 1
+        for v in obj.values():
+            n += _strip_config_for_qwen3_load(v, depth + 1)
+    elif isinstance(obj, list):
+        for v in obj:
+            n += _strip_config_for_qwen3_load(v, depth)
+    return n
+def sanitize_qwen3_tts_config_json(repo_or_config: Path) -> int:
+    """In-place fix for merged config.json (nested dtype / model_type keys Qwen3 sub-configs reject)."""
+    path = Path(repo_or_config)
+    if path.is_dir():
+        path = path / "config.json"
+    if not path.is_file():
+        return 0
+    with path.open("r", encoding="utf-8") as f:
+        data = json.load(f)
+    n = _strip_config_for_qwen3_load(data, 0)
+    if n:
+        with path.open("w", encoding="utf-8") as f:
+            json.dump(data, f, indent=2, ensure_ascii=False)
+    return n
+_apply_check_model_inputs_shim()
+# --- end inlined helpers ---
+_CONFIG_NAME = "config.json"
+_VOCENCE_YAML = "vocence_config.yaml"
+def _merge_vocence_yaml(repo: Path) -> dict[str, Any]:
+    path = repo / _VOCENCE_YAML
+    if not path.is_file():
+        return {}
+    from yaml import safe_load
+    with path.open("r", encoding="utf-8") as fh:
+        data = safe_load(fh)
+    return data if isinstance(data, Mapping) else {}
+def _ensure_repo_checkpoint(repo: Path) -> Path:
+    repo = repo.resolve()
+    marker = repo / _CONFIG_NAME
+    if not marker.is_file():
+        raise FileNotFoundError(
+            f"Model snapshot incomplete: {marker} missing. "
+            "Host the full Qwen3-TTS weights (checkpoint + tokenizers) in this repository."
+        )
+    return repo
+def _resolve_compute_device(prefer_cuda: bool) -> str:
+    import torch
+    if prefer_cuda and torch.cuda.is_available():
+        return "cuda:0"
+    return "cpu"
+def _resolve_torch_dtype(torch, prefer_bf16: bool):
+    if prefer_bf16 and torch.cuda.is_available():
+        return torch.bfloat16
+    return torch.float32
+def _instantiate_qwen(checkpoint_dir: str, device_map: str, torch_dtype, use_flash2: bool):
+    """Load Qwen3TTSModel from a local directory (same call shape as example_inference / qwen-tts)."""
+    from qwen_tts import Qwen3TTSModel
+    # API: from_pretrained(path, device_map=..., dtype=..., attn_implementation=...)
+    # Do not use kwargs-only form with ``pretrained_model_name_or_path=``; use path as first arg.
+    kwargs = dict(device_map=device_map, dtype=torch_dtype, attn_implementation="sdpa")
+    if use_flash2:
+        try:
+            return Qwen3TTSModel.from_pretrained(
+                checkpoint_dir,
+                device_map=device_map,
+                dtype=torch_dtype,
+                attn_implementation="flash_attention_2",
+            )
+        except Exception:
+            # flash-attn missing or kernel mismatch — fall back to SDPA
+            pass
+    return Qwen3TTSModel.from_pretrained(checkpoint_dir, **kwargs)
+def _to_mono_f32(segment: np.ndarray) -> np.ndarray:
+    x = np.asarray(segment, dtype=np.float32)
+    if x.ndim > 1:
+        x = x.mean(axis=1)
+    return x
+class Miner:
+    """
+    Loads the checkpoint from the Hugging Face repo directory Chutes downloaded.
+    Synthesis uses natural-language instruction + text (qwen-tts API).
+    """
+    def __init__(self, path_hf_repo: Path) -> None:
+        self._root = _ensure_repo_checkpoint(Path(path_hf_repo))
+        self._cfg = _merge_vocence_yaml(self._root)
+        rt = self._cfg.get("runtime") or {}
+        gen = self._cfg.get("generation") or {}
+        lim = self._cfg.get("limits") or {}
+        self._language = str(lim.get("default_language") or rt.get("default_language", "English"))
+        self._output_sr = int(gen.get("sample_rate", 24000))
+        self._cap_instruction = int(lim.get("max_instruction_chars", 600))
+        self._cap_text = int(lim.get("max_text_chars", 2000))
+        prefer_cuda = str(rt.get("device_preference", "cuda")).lower() == "cuda"
+        want_bf16 = str(rt.get("dtype", "bfloat16")).lower() == "bfloat16"
+        flash = bool(rt.get("use_flash_attention_2", False))
+        import torch
+        device_map = _resolve_compute_device(prefer_cuda)
+        torch_dtype = _resolve_torch_dtype(torch, want_bf16)
+        ckpt = str(self._root)
+        # Merged HF saves may put dtype / model_type into nested sub-configs; Qwen3 rejects them.
+        sanitize_qwen3_tts_config_json(self._root)
+        self._tts = _instantiate_qwen(ckpt, device_map, torch_dtype, flash)
+        # Qwen3TTSModel is a thin wrapper, not nn.Module — no .eval()
+        print("Qwen3-TTS checkpoint ready (loaded from repo snapshot).")
+    def __repr__(self) -> str:
+        return "Miner(qwen3-tts-local, local_snapshot=True)"
+    def warmup(self) -> None:
+        """Force one cheap synthesis on a background thread (startup SLAs)."""
+        status: dict[str, object] = {"done": False, "error": None}
+        def _once() -> None:
+            try:
+                self.generate_wav(
+                    instruction="Clear, neutral delivery.",
+                    text="Warmup.",
+                )
+                status["done"] = True
+            except Exception as exc:  # noqa: BLE001 — surface to host
+                status["error"] = str(exc)
+        worker = threading.Thread(target=_once, daemon=True)
+        worker.start()
+        worker.join(timeout=180.0)
+        if not status["done"]:
+            raise RuntimeError(status["error"] or "warmup exceeded 180s")
+    def generate_wav(self, instruction: str, text: str) -> tuple[np.ndarray, int]:
+        if self._cap_instruction > 0:
+            instruction = instruction[: self._cap_instruction]
+        if self._cap_text > 0:
+            text = text[: self._cap_text]
+        # Upstream qwen-tts method name (instruct + text -> waveform).
+        waves, sr = self._tts.generate_voice_design(
+            text=text,
+            language=self._language,
+            instruct=instruction,
+        )
+        if not waves:
+            raise ValueError("TTS generation returned no audio")
+        first = waves[0]
+        if first is None:
+            raise ValueError("TTS generation returned empty channel")
+        return _to_mono_f32(first), int(sr)

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e52ff5b60e47c4d5f32184a840f89eb684ff7562afc5b6bdab80cff01ccbbcca
+size 3833402552

preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "padding_side": "left",
+  "padding_value": 0.0,
+  "processor_class": "Qwen3TTSProcessor",
+  "return_attention_mask": true
+}

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,44 @@

+{
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>",
+    "<|audio_start|>",
+    "<|audio_end|>",
+    "<tts_pad>",
+    "<tts_text_bos>",
+    "<tts_text_bos_single>",
+    "<|audio_pad|>"
+  ],
+  "audio_bos_token": "<|audio_start|>",
+  "audio_eos_token": "<|audio_end|>",
+  "audio_token": "<|audio_pad|>",
+  "eos_token": {
+    "content": "<|im_end|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "image_token": "<|image_pad|>",
+  "pad_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "video_token": "<|video_pad|>",
+  "vision_bos_token": "<|vision_start|>",
+  "vision_eos_token": "<|vision_end|>"
+}

speech_tokenizer/config.json ADDED Viewed

	@@ -0,0 +1,94 @@

+{
+  "architectures": [
+    "Qwen3TTSTokenizerV2Model"
+  ],
+  "model_type": "qwen3_tts_tokenizer_12hz",
+  "encoder_valid_num_quantizers": 16,
+  "input_sample_rate": 24000,
+  "output_sample_rate": 24000,
+  "decode_upsample_rate": 1920,
+  "encode_downsample_rate": 1920,
+  "decoder_config": {
+    "attention_bias": false,
+    "attention_dropout": 0.0,
+    "latent_dim": 1024,
+    "codebook_dim": 512,
+    "codebook_size": 2048,
+    "decoder_dim": 1536,
+    "hidden_act": "silu",
+    "hidden_size": 512,
+    "intermediate_size": 1024,
+    "layer_scale_initial_scale": 0.01,
+    "max_position_embeddings": 8000,
+    "head_dim": 64,
+    "num_attention_heads": 16,
+    "num_hidden_layers": 8,
+    "num_key_value_heads": 16,
+    "num_quantizers": 16,
+    "num_semantic_quantizers": 1,
+    "rms_norm_eps": 1e-05,
+    "rope_theta": 10000,
+    "semantic_codebook_size": 4096,
+    "sliding_window": 72,
+    "upsample_rates": [
+      8,
+      5,
+      4,
+      3
+    ],
+    "upsampling_ratios": [
+      2,
+      2
+    ],
+    "vector_quantization_hidden_dimension": 512
+  },
+  "encoder_config": {
+    "_frame_rate": 12.5,
+    "attention_bias": false,
+    "attention_dropout": 0.0,
+    "audio_channels": 1,
+    "codebook_dim": 256,
+    "codebook_size": 2048,
+    "compress": 2,
+    "dilation_growth_rate": 2,
+    "dtype": "float32",
+    "head_dim": 64,
+    "hidden_act": "gelu",
+    "hidden_size": 512,
+    "initializer_range": 0.02,
+    "intermediate_size": 2048,
+    "kernel_size": 7,
+    "last_kernel_size": 3,
+    "layer_scale_initial_scale": 0.01,
+    "max_position_embeddings": 8000,
+    "norm_eps": 1e-05,
+    "normalize": false,
+    "num_attention_heads": 8,
+    "num_filters": 64,
+    "num_hidden_layers": 8,
+    "num_key_value_heads": 8,
+    "num_quantizers": 32,
+    "num_residual_layers": 1,
+    "num_semantic_quantizers": 1,
+    "pad_mode": "constant",
+    "residual_kernel_size": 3,
+    "rope_theta": 10000.0,
+    "sampling_rate": 24000,
+    "sliding_window": 250,
+    "transformers_version": "4.57.0.dev0",
+    "trim_right_ratio": 1.0,
+    "upsample_groups": 512,
+    "upsampling_ratios": [
+      8,
+      6,
+      5,
+      4
+    ],
+    "use_cache": false,
+    "use_causal_conv": true,
+    "use_conv_shortcut": false,
+    "use_streaming": false,
+    "vector_quantization_hidden_dimension": 256
+  },
+  "transformers_version": "4.57.3"
+}

speech_tokenizer/configuration.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"framework": "pytorch", "task": "feature-extraction", "allow_remote": true}

speech_tokenizer/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:836b7b357f5ea43e889936a3709af68dfe3751881acefe4ecf0dbd30ba571258
+size 682293092

speech_tokenizer/preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,10 @@

+{
+  "chunk_length_s": null,
+  "feature_extractor_type": "EncodecFeatureExtractor",
+  "feature_size": 1,
+  "overlap": null,
+  "padding_side": "right",
+  "padding_value": 0.0,
+  "return_attention_mask": true,
+  "sampling_rate": 24000
+}

tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:21cbba93fc2b3723c68db44ba5a8ff1a22a3f7bed34a5adbd5a165e17fff551c
+size 11424108

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,316 @@

+{
+  "add_bos_token": false,
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "151643": {
+      "content": "<|endoftext|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151644": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151645": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151646": {
+      "content": "<|object_ref_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151647": {
+      "content": "<|object_ref_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151648": {
+      "content": "<|box_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151649": {
+      "content": "<|box_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151650": {
+      "content": "<|quad_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151651": {
+      "content": "<|quad_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151652": {
+      "content": "<|vision_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151653": {
+      "content": "<|vision_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151654": {
+      "content": "<|vision_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151655": {
+      "content": "<|image_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151656": {
+      "content": "<|video_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151657": {
+      "content": "<tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151658": {
+      "content": "</tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151659": {
+      "content": "<|fim_prefix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151660": {
+      "content": "<|fim_middle|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151661": {
+      "content": "<|fim_suffix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151662": {
+      "content": "<|fim_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151663": {
+      "content": "<|repo_name|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151664": {
+      "content": "<|file_sep|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151665": {
+      "content": "<tool_response>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151666": {
+      "content": "</tool_response>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151667": {
+      "content": "<think>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151668": {
+      "content": "</think>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151669": {
+      "content": "<|audio_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151670": {
+      "content": "<|audio_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151671": {
+      "content": "<tts_pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151672": {
+      "content": "<tts_text_bos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151673": {
+      "content": "<tts_text_eod>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151674": {
+      "content": "<tts_text_bos_single>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151675": {
+      "content": "<|audio_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>",
+    "<|audio_start|>",
+    "<|audio_end|>",
+    "<tts_pad>",
+    "<tts_text_bos>",
+    "<tts_text_bos_single>",
+    "<|audio_pad|>"
+  ],
+  "extra_special_tokens": {
+    "image_token": "<|image_pad|>",
+    "audio_token": "<|audio_pad|>",
+    "video_token": "<|video_pad|>",
+    "vision_bos_token": "<|vision_start|>",
+    "vision_eos_token": "<|vision_end|>",
+    "audio_bos_token": "<|audio_start|>",
+    "audio_eos_token": "<|audio_end|>"
+  },
+  "bos_token": null,
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|im_end|>",
+  "errors": "replace",
+  "model_max_length": 131072,
+  "pad_token": "<|endoftext|>",
+  "split_special_tokens": false,
+  "tokenizer_class": "Qwen2Tokenizer",
+  "unk_token": null,
+  "image_token": "<|image_pad|>",
+  "audio_token": "<|audio_pad|>",
+  "video_token": "<|video_pad|>",
+  "vision_bos_token": "<|vision_start|>",
+  "vision_eos_token": "<|vision_end|>",
+  "audio_bos_token": "<|audio_start|>",
+  "audio_eos_token": "<|audio_end|>"
+}

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff

vocence_config.yaml ADDED Viewed

	@@ -0,0 +1,16 @@

+# Miner + /health metadata. Weights live in this HF repo (no runtime model_id).
+runtime:
+  adapter: "qwen3_tts_repo_snapshot"
+  device_preference: "cuda"
+  dtype: "bfloat16"
+  default_language: "English"
+  use_flash_attention_2: false
+generation:
+  sample_rate: 24000
+  max_seconds: 30
+limits:
+  max_text_chars: 2000
+  max_instruction_chars: 600
+  default_language: "English"