mirror of
https://github.com/LostRuins/koboldcpp.git
synced 2026-08-01 04:03:48 +00:00
server: correct accepted tokens when need draft token replay (#26320)
python-type-check.yml #426 -Commit
000547513f
pushed by
vrr
fit : count nextn (MTP) blocks in n_gpu_layers so front layers stay on GPU (#26177)
python-type-check.yml #425 -Commit
0324696b8e
pushed by
vrr
fit : count nextn (MTP) blocks in n_gpu_layers so front layers stay on GPU (#26177)
pre-tokenizer-hashes.yml #424 -Commit
0324696b8e
pushed by
vrr
opencl: cache compiled cl_program binaries on disk (#26050)
update-ops-docs.yml #423 -Commit
ed7adbfefd
pushed by
vrr
opencl: cache compiled cl_program binaries on disk (#26050)
python-type-check.yml #422 -Commit
ed7adbfefd
pushed by
vrr
opencl: cache compiled cl_program binaries on disk (#26050)
python-check-requirements.yml #421 -Commit
ed7adbfefd
pushed by
vrr
opencl: cache compiled cl_program binaries on disk (#26050)
pre-tokenizer-hashes.yml #420 -Commit
ed7adbfefd
pushed by
vrr
opencl: load and use `kernel_gemm_moe_q6_k_f32_ns` from bin kernel lib (#25797)
python-type-check.yml #419 -Commit
86a9c79f86
pushed by
vrr
DeepseekV4: reduce graph splits (#25702)
update-ops-docs.yml #418 -Commit
33a75f41c3
pushed by
vrr
DeepseekV4: reduce graph splits (#25702)
python-type-check.yml #417 -Commit
33a75f41c3
pushed by
vrr
DeepseekV4: reduce graph splits (#25702)
python-check-requirements.yml #416 -Commit
33a75f41c3
pushed by
vrr
DeepseekV4: reduce graph splits (#25702)
pre-tokenizer-hashes.yml #415 -Commit
33a75f41c3
pushed by
vrr
tests: Harmonize header use (#25616)
python-type-check.yml #414 -Commit
f4253ef965
pushed by
vrr
server: accept null sampling params (#25538)
update-ops-docs.yml #413 -Commit
4f37f51972
pushed by
vrr
server: accept null sampling params (#25538)
python-type-check.yml #412 -Commit
4f37f51972
pushed by
vrr
server-stream: follow-up on SSE Replay Buffer (#23226) (#25047)
update-ops-docs.yml #411 -Commit
bbebeec4a8
pushed by
vrr
server-stream: follow-up on SSE Replay Buffer (#23226) (#25047)
python-type-check.yml #410 -Commit
bbebeec4a8
pushed by
vrr
CUDA: extend K-type validation to V-types for flash attention (#24403)
python-type-check.yml #409 -Commit
cb295bf596
pushed by
vrr
llama : add guard for K/V rotation input when buffer is unallocated (#25215)
python-type-check.yml #408 -Commit
a4107133a6
pushed by
vrr
opencl: allow loading precompiled binary kernels from library (#23042)
python-type-check.yml #407 -Commit
4fc4ec5541
pushed by
vrr
common : dedup preset and cached model entries in /v1/models (#25131)
python-type-check.yml #406 -Commit
6f4f53f2b7
pushed by
vrr
common : dedup preset and cached model entries in /v1/models (#25131)
pre-tokenizer-hashes.yml #405 -Commit
6f4f53f2b7
pushed by
vrr
common : remove unused regex-partial (#25118)
python-type-check.yml #404 -Commit
277a105dc8
pushed by
vrr
app : allow --version, --licenses & --help (#25054)
python-type-check.yml #403 -Commit
050ee92d04
pushed by
vrr
vulkan: allow reducing the graph submission batches to avoid timeouts (#24872)
python-type-check.yml #402 -Commit
51eae8cfca
pushed by
vrr
server: fix edit_file crash on append at end of file (line_start -1) (#24893)
python-type-check.yml #401 -Commit
d0f9d2e5ac
pushed by
vrr
server: fix edit_file crash on append at end of file (line_start -1) (#24893)
pre-tokenizer-hashes.yml #400 -Commit
d0f9d2e5ac
pushed by
vrr
ggml-webgpu: add adapter toggles for F16 on Vulkan + NVIDIA
python-type-check.yml #399 -Commit
f449e05537
pushed by
vrr
[SYCL] rename GGML_SYCL_SUPPORT_LEVEL_ZERO (#24719)
update-ops-docs.yml #398 -Commit
9724f664e8
pushed by
vrr
[SYCL] rename GGML_SYCL_SUPPORT_LEVEL_ZERO (#24719)
python-type-check.yml #397 -Commit
9724f664e8
pushed by
vrr