/
vshmidt
/
llama.cpp
Обзор
Документация
Войти
/
vshmidt
/
llama.cpp
Код
Запросы
0
Задачи
Вики
Пакеты
0
Релизы
0
Аналитика
Безопасность
master
src/llama-cparams.h
47 строк
1 KB
Georgi Gerganov
llama : enable chunked fused GDN path (#20340)
11 мар 2026, 23:46
Не верифицирован
11 мар 2026, 23:46
d28961d
Код
Авторство
О чём код?
#pragma once #include "llama.h" #include <cstdint> #define LLAMA_MAX_SEQ 256 struct llama_cparams { uint32_t n_ctx; // context size used during inference uint32_t n_ctx_seq; // context for a single sequence uint32_t n_batch; uint32_t n_ubatch; uint32_t n_seq_max; int32_t n_threads; // number of threads to use for generation int32_t n_threads_batch; // number of threads to use for batch processing float rope_freq_base; float rope_freq_scale; uint32_t n_ctx_orig_yarn; // These hyperparameters are not exposed in GGUF, because all // existing YaRN models use the same values for them. float yarn_ext_factor; float yarn_attn_factor; float yarn_beta_fast; float yarn_beta_slow; bool embeddings; bool causal_attn; bool offload_kqv; bool flash_attn; bool auto_fa; bool fused_gdn_ar; // use fused gated delta net (autoregressive) bool fused_gdn_ch; // use fused gated delta net (chunked) bool auto_fgdn; bool no_perf; bool warmup; bool op_offload; bool kv_unified; bool pipeline_parallel; enum llama_pooling_type pooling_type; ggml_backend_sched_eval_callback cb_eval; void * cb_eval_user_data; };