(
self,
*,
model: LlamaModel,
params: llama_cpp.llama_context_params,
verbose: bool = True,
)
| 249 | NOTE: For stability it's recommended you use the Llama class instead.""" |
| 250 | |
| 251 | def __init__( |
| 252 | self, |
| 253 | *, |
| 254 | model: LlamaModel, |
| 255 | params: llama_cpp.llama_context_params, |
| 256 | verbose: bool = True, |
| 257 | ): |
| 258 | self.model = model |
| 259 | self.params = params |
| 260 | self.verbose = verbose |
| 261 | self._exit_stack = ExitStack() |
| 262 | |
| 263 | ctx = llama_cpp.llama_init_from_model(self.model.model, self.params) |
| 264 | |
| 265 | if ctx is None: |
| 266 | raise ValueError("Failed to create llama_context") |
| 267 | |
| 268 | self.ctx = ctx |
| 269 | self.memory = llama_cpp.llama_get_memory(self.ctx) |
| 270 | self.sampler = None # LlamaContext doesn't manage samplers directly, but some cleanup code expects this attribute |
| 271 | |
| 272 | def free_ctx(): |
| 273 | if self.ctx is None: |
| 274 | return |
| 275 | llama_cpp.llama_free(self.ctx) |
| 276 | self.ctx = None |
| 277 | |
| 278 | self._exit_stack.callback(free_ctx) |
| 279 | |
| 280 | def close(self): |
| 281 | self._exit_stack.close() |
nothing calls this directly
no outgoing calls
no test coverage detected