MCPcopy Create free account
hub / github.com/Tiiny-AI/PowerInfer / llama_set_state_data

Function llama_set_state_data

llama.cpp:10370–10498  ·  view source on GitHub ↗

Sets the state reading from the specified source address

Source from the content-addressed store, hash-verified

10368
10369// Sets the state reading from the specified source address
10370size_t llama_set_state_data(struct llama_context * ctx, uint8_t * src) {
10371 uint8_t * inp = src;
10372
10373 // set rng
10374 {
10375 size_t rng_size;
10376 char rng_buf[LLAMA_MAX_RNG_STATE];
10377
10378 memcpy(&rng_size, inp, sizeof(rng_size)); inp += sizeof(rng_size);
10379 memcpy(&rng_buf[0], inp, LLAMA_MAX_RNG_STATE); inp += LLAMA_MAX_RNG_STATE;
10380
10381 std::stringstream rng_ss;
10382 rng_ss.str(std::string(&rng_buf[0], rng_size));
10383 rng_ss >> ctx->rng;
10384
10385 GGML_ASSERT(!rng_ss.fail());
10386 }
10387
10388 // set logits
10389 {
10390 size_t logits_cap;
10391 size_t logits_size;
10392
10393 memcpy(&logits_cap, inp, sizeof(logits_cap)); inp += sizeof(logits_cap);
10394 memcpy(&logits_size, inp, sizeof(logits_size)); inp += sizeof(logits_size);
10395
10396 GGML_ASSERT(ctx->logits.capacity() == logits_cap);
10397
10398 if (logits_size) {
10399 ctx->logits.resize(logits_size);
10400 memcpy(ctx->logits.data(), inp, logits_size * sizeof(float));
10401 }
10402
10403 inp += logits_cap * sizeof(float);
10404 }
10405
10406 // set embeddings
10407 {
10408 size_t embedding_size;
10409
10410 memcpy(&embedding_size, inp, sizeof(embedding_size)); inp += sizeof(embedding_size);
10411
10412 GGML_ASSERT(ctx->embedding.capacity() == embedding_size);
10413
10414 if (embedding_size) {
10415 memcpy(ctx->embedding.data(), inp, embedding_size * sizeof(float));
10416 inp += embedding_size * sizeof(float);
10417 }
10418 }
10419
10420 // set kv cache
10421 {
10422 const auto & kv_self = ctx->kv_self;
10423 const auto & hparams = ctx->model.hparams;
10424 const auto & cparams = ctx->cparams;
10425
10426 const int n_layer = hparams.n_layer;
10427 const int n_embd = hparams.n_embd_gqa();

Callers 2

mainFunction · 0.50

Calls 15

ggml_element_sizeFunction · 0.70
ggml_initFunction · 0.70
ggml_tensor_overheadFunction · 0.70
ggml_graph_overheadFunction · 0.70
ggml_new_graphFunction · 0.70
ggml_new_tensor_3dFunction · 0.70
ggml_nbytesFunction · 0.70
ggml_view_3dFunction · 0.70
ggml_cpyFunction · 0.70
ggml_freeFunction · 0.70

Tested by

no test coverage detected