MCPcopy Create free account
hub / github.com/OpenMOSS/MOSS-TTS / __init__

Method __init__

moss_tts_delay/llama_cpp/backbone.py:123–161  ·  view source on GitHub ↗
(
        self,
        model_path: str | Path,
        n_ctx: int = 4096,
        n_batch: int = 512,
        n_threads: int = 4,
        n_gpu_layers: int = -1,
        type_k: str = "f16",
        type_v: str = "f16",
        flash_attn: str | bool = "auto",
    )

Source from the content-addressed store, hash-verified

121 """
122
123 def __init__(
124 self,
125 model_path: str | Path,
126 n_ctx: int = 4096,
127 n_batch: int = 512,
128 n_threads: int = 4,
129 n_gpu_layers: int = -1,
130 type_k: str = "f16",
131 type_v: str = "f16",
132 flash_attn: str | bool = "auto",
133 ):
134 lib_path = _find_bridge_lib()
135 log.info("Loading bridge from %s", lib_path)
136 self._lib = _load_bridge(lib_path)
137
138 ggml_type_k = _resolve_ggml_type(type_k)
139 ggml_type_v = _resolve_ggml_type(type_v)
140 fa_type = _resolve_flash_attn(flash_attn)
141
142 model_path = str(Path(model_path).resolve())
143 log.info(
144 "Loading GGUF model: %s (type_k=%s, type_v=%s, flash_attn=%s)",
145 model_path, type_k, type_v, flash_attn,
146 )
147 self._handle = self._lib.bridge_create(
148 model_path.encode("utf-8"), n_ctx, n_batch, n_threads, n_gpu_layers,
149 ggml_type_k, ggml_type_v, fa_type,
150 )
151 if not self._handle:
152 raise RuntimeError(f"Failed to load model from {model_path}")
153
154 self.n_embd = self._lib.bridge_n_embd(self._handle)
155 self.n_vocab = self._lib.bridge_n_vocab(self._handle)
156 self.n_batch = n_batch
157 self.n_ctx = n_ctx
158 log.info(
159 "LlamaCppBackbone ready: n_embd=%d, n_vocab=%d, n_ctx=%d, n_batch=%d",
160 self.n_embd, self.n_vocab, n_ctx, n_batch,
161 )
162
163 def decode_single(self, embd: np.ndarray, pos: int, output: bool = True) -> None:
164 """Feed a single embedding vector at the given position."""

Callers

nothing calls this directly

Calls 5

_find_bridge_libFunction · 0.85
_load_bridgeFunction · 0.85
_resolve_ggml_typeFunction · 0.85
_resolve_flash_attnFunction · 0.85
encodeMethod · 0.45

Tested by

no test coverage detected