Merge branch 'upstream' into concedo_experimental

# Conflicts: # .github/workflows/docker.yml # Makefile # examples/CMakeLists.txt # ggml/CMakeLists.txt # ggml/src/CMakeLists.txt # ggml/src/ggml-sycl/common.hpp # ggml/src/ggml-sycl/convert.cpp # ggml/src/ggml-sycl/convert.hpp # ggml/src/ggml-sycl/ggml-sycl.cpp # scripts/sync-ggml.last
2025-09-11 01:24:36 +00:00 · 2025-05-08 23:07:33 +08:00 · 2025-05-08 23:07:33 +08:00 · b6220669f4
commit b6220669f4
parent 7c5d47f688 8733e0cf6e
14 changed files with 410 additions and 117 deletions
--- a/src/llama-model.cpp
+++ b/src/llama-model.cpp
@ -3605,7 +3605,11 @@ bool llama_model::load_tensors(llama_model_loader & ml) {

                    // output
                    output_norm   = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), {n_embd}, 0);
-                    output        = create_tensor(tn(LLM_TENSOR_OUTPUT,      "weight"), {n_embd, n_vocab}, 0);
+                    output        = create_tensor(tn(LLM_TENSOR_OUTPUT,      "weight"), {n_embd, n_vocab}, TENSOR_NOT_REQUIRED);
+                    // if output is NULL, init from the input tok embed
+                    if (output == NULL) {
+                        output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, TENSOR_DUPLICATED);
+                    }

                    for (int i = 0; i < n_layer; ++i) {
                        auto & layer = layers[i];
@ -4891,7 +4895,7 @@ struct llm_build_deci : public llm_graph_context {
            }

            // FFN-free layer of Llama-3_1-Nemotron-Ultra-253B
-            if (n_head == 0 && n_ff == 0) {
+            if (n_ff == 0) {
                continue;
            }