janhq
diff --git a/‎.clang-format‎
Lines changed: 7 additions & 0 deletions b/‎.clang-format‎
Lines changed: 7 additions & 0 deletions
diff --git a/‎.github/workflows/build.yml‎
Lines changed: 20 additions & 9 deletions b/‎.github/workflows/build.yml‎
Lines changed: 20 additions & 9 deletions
diff --git a/‎CMakeLists.txt‎
Lines changed: 7 additions & 0 deletions b/‎CMakeLists.txt‎
Lines changed: 7 additions & 0 deletions
diff --git a/‎common/arg.cpp‎
Lines changed: 5 additions & 5 deletions b/‎common/arg.cpp‎
Lines changed: 5 additions & 5 deletions
diff --git a/‎common/common.h‎
Lines changed: 14 additions & 0 deletions b/‎common/common.h‎
Lines changed: 14 additions & 0 deletions
diff --git a/‎convert_hf_to_gguf.py‎
Lines changed: 73 additions & 0 deletions b/‎convert_hf_to_gguf.py‎
Lines changed: 73 additions & 0 deletions
diff --git a/‎convert_hf_to_gguf_update.py‎
Lines changed: 1 addition & 0 deletions b/‎convert_hf_to_gguf_update.py‎
Lines changed: 1 addition & 0 deletions
diff --git a/‎examples/diffusion/diffusion-cli.cpp‎
Lines changed: 17 additions & 7 deletions b/‎examples/diffusion/diffusion-cli.cpp‎
Lines changed: 17 additions & 7 deletions
diff --git a/‎ggml/src/ggml-cpu/ops.cpp‎
Lines changed: 0 additions & 1 deletion b/‎ggml/src/ggml-cpu/ops.cpp‎
Lines changed: 0 additions & 1 deletion
diff --git a/‎ggml/src/ggml-cuda/tsembd.cu‎
Lines changed: 3 additions & 3 deletions b/‎ggml/src/ggml-cuda/tsembd.cu‎
Lines changed: 3 additions & 3 deletions
@@ -22,6 +22,13 @@ AllowShortIfStatementsOnASingleLine: Never
 AllowShortLambdasOnASingleLine: Inline
 AllowShortLoopsOnASingleLine: false
 AlwaysBreakBeforeMultilineStrings: true
+# Treat CUDA keywords/attributes as "attribute macros" and avoid breaking lines inside them
+AttributeMacros:
+  - __host__
+  - __device__
+  - __global__
+  - __forceinline__
+  - __launch_bounds__
 BinPackArguments: true
 BinPackParameters: false # OnePerLine
 BitFieldColonSpacing: Both
 
@@ -56,7 +56,7 @@ env:
 
 jobs:
   macOS-latest-cmake-arm64:
-    runs-on: macos-14
+    runs-on: macos-latest
 
     steps:
       - name: Clone
@@ -97,7 +97,7 @@ jobs:
           ctest -L 'main|curl' --verbose --timeout 900
 
   macOS-latest-cmake-x64:
-    runs-on: macos-13
+    runs-on: macos-latest
 
     steps:
       - name: Clone
@@ -138,7 +138,7 @@ jobs:
           ctest -L main --verbose --timeout 900
 
   macOS-latest-cmake-arm64-webgpu:
-    runs-on: macos-14
+    runs-on: macos-latest
 
     steps:
       - name: Clone
@@ -711,6 +711,7 @@ jobs:
 
   macOS-latest-swift:
     runs-on: macos-latest
+    needs: ios-xcode-build
 
     strategy:
       matrix:
@@ -727,6 +728,12 @@ jobs:
           key: macOS-latest-swift
           evict-old-files: 1d
 
+      - name: Download xcframework artifact
+        uses: actions/download-artifact@v4
+        with:
+          name: llama-xcframework
+          path: build-apple/llama.xcframework/
+
       - name: Dependencies
         id: depends
         continue-on-error: true
@@ -748,11 +755,6 @@ jobs:
             -DCMAKE_OSX_ARCHITECTURES="arm64;x86_64"
           cmake --build build --config Release -j $(sysctl -n hw.logicalcpu)
 
-      - name: xcodebuild for swift package
-        id: xcodebuild
-        run: |
-          ./build-xcframework.sh
-
   windows-msys2:
     runs-on: windows-2025
 
@@ -1170,8 +1172,17 @@ jobs:
         run: |
           ./build-xcframework.sh
 
+      - name: Upload xcframework artifact
+        uses: actions/upload-artifact@v4
+        with:
+          name: llama-xcframework
+          path: build-apple/llama.xcframework/
+          retention-days: 1
+
       - name: Build Xcode project
-        run: xcodebuild -project examples/llama.swiftui/llama.swiftui.xcodeproj -scheme llama.swiftui -sdk iphoneos CODE_SIGNING_REQUIRED=NO CODE_SIGN_IDENTITY= -destination 'generic/platform=iOS' FRAMEWORK_FOLDER_PATH=./build-ios build
+        run: |
+          xcodebuild -downloadPlatform iOS
+          xcodebuild -project examples/llama.swiftui/llama.swiftui.xcodeproj -scheme llama.swiftui -sdk iphoneos CODE_SIGNING_REQUIRED=NO CODE_SIGN_IDENTITY= -destination 'generic/platform=iOS' FRAMEWORK_FOLDER_PATH=./build-ios build
 
   android-build:
     runs-on: ubuntu-latest
 
@@ -58,6 +58,12 @@ if (MSVC)
     add_compile_options("$<$<COMPILE_LANGUAGE:CXX>:/bigobj>")
 endif()
 
+if (CMAKE_SYSTEM_NAME STREQUAL "iOS")
+    set(LLAMA_TOOLS_INSTALL_DEFAULT OFF)
+else()
+    set(LLAMA_TOOLS_INSTALL_DEFAULT ${LLAMA_STANDALONE})
+endif()
+
 #
 # option list
 #
@@ -82,6 +88,7 @@ option(LLAMA_BUILD_TESTS    "llama: build tests"          ${LLAMA_STANDALONE})
 option(LLAMA_BUILD_TOOLS    "llama: build tools"          ${LLAMA_STANDALONE})
 option(LLAMA_BUILD_EXAMPLES "llama: build examples"       ${LLAMA_STANDALONE})
 option(LLAMA_BUILD_SERVER   "llama: build server example" ${LLAMA_STANDALONE})
+option(LLAMA_TOOLS_INSTALL  "llama: install tools"        ${LLAMA_TOOLS_INSTALL_DEFAULT})
 
 # 3rd party libs
 option(LLAMA_CURL       "llama: use libcurl to download model from an URL" ON)
 
@@ -1704,7 +1704,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
         [](common_params & params, const std::string & value) {
             params.system_prompt = value;
         }
-    ).set_examples({LLAMA_EXAMPLE_MAIN}));
+    ).set_examples({LLAMA_EXAMPLE_MAIN, LLAMA_EXAMPLE_DIFFUSION}));
     add_opt(common_arg(
         {"--no-perf"},
         string_format("disable internal libllama performance timings (default: %s)", params.no_perf ? "true" : "false"),
@@ -2548,7 +2548,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
         {"--cpu-moe", "-cmoe"},
         "keep all Mixture of Experts (MoE) weights in the CPU",
         [](common_params & params) {
-            params.tensor_buft_overrides.push_back({"\\.ffn_(up|down|gate)_exps", ggml_backend_cpu_buffer_type()});
+            params.tensor_buft_overrides.push_back(llm_ffn_exps_cpu_override());
         }
     ).set_env("LLAMA_ARG_CPU_MOE"));
     add_opt(common_arg(
@@ -2561,7 +2561,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
             for (int i = 0; i < value; ++i) {
                 // keep strings alive and avoid leaking memory by storing them in a static vector
                 static std::list<std::string> buft_overrides;
-                buft_overrides.push_back(string_format("blk\\.%d\\.ffn_(up|down|gate)_exps", i));
+                buft_overrides.push_back(llm_ffn_exps_block_regex(i));
                 params.tensor_buft_overrides.push_back({buft_overrides.back().c_str(), ggml_backend_cpu_buffer_type()});
             }
         }
@@ -2570,7 +2570,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
         {"--cpu-moe-draft", "-cmoed"},
         "keep all Mixture of Experts (MoE) weights in the CPU for the draft model",
         [](common_params & params) {
-            params.speculative.tensor_buft_overrides.push_back({"\\.ffn_(up|down|gate)_exps", ggml_backend_cpu_buffer_type()});
+            params.speculative.tensor_buft_overrides.push_back(llm_ffn_exps_cpu_override());
         }
     ).set_examples({LLAMA_EXAMPLE_SPECULATIVE, LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_CPU_MOE_DRAFT"));
     add_opt(common_arg(
@@ -2582,7 +2582,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
             }
             for (int i = 0; i < value; ++i) {
                 static std::list<std::string> buft_overrides_draft;
-                buft_overrides_draft.push_back(string_format("blk\\.%d\\.ffn_(up|down|gate)_exps", i));
+                buft_overrides_draft.push_back(llm_ffn_exps_block_regex(i));
                 params.speculative.tensor_buft_overrides.push_back({buft_overrides_draft.back().c_str(), ggml_backend_cpu_buffer_type()});
             }
         }
 
@@ -734,6 +734,20 @@ const char * const LLM_KV_SPLIT_TENSORS_COUNT = "split.tensors.count";
 
 }
 
+//
+// MoE utils
+//
+
+const char * const LLM_FFN_EXPS_REGEX = "\\.ffn_(up|down|gate)_exps";
+
+static std::string llm_ffn_exps_block_regex(int idx) {
+    return string_format("blk\\.%d%s", idx, LLM_FFN_EXPS_REGEX);
+}
+
+static llama_model_tensor_buft_override llm_ffn_exps_cpu_override() {
+    return { LLM_FFN_EXPS_REGEX, ggml_backend_cpu_buffer_type() };
+}
+
 //
 // training utils
 //
 
@@ -888,6 +888,9 @@ def get_vocab_base_pre(self, tokenizer) -> str:
         if chkhsh == "a1e163ecab2e718a4c829d1148b6e86824ec36163bb71941c3dca9cd5ac25756":
             # ref: https://huggingface.co/JetBrains/Mellum-4b-base
             res = "mellum"
+        if chkhsh == "9b1be57e70d20d9501b2b3186e792d81181ae36ada3903c26f9fea418cf87206":
+            # ref: https://huggingface.co/inclusionAI/LLaDA-MoE-7B-A1B-Base
+            res = "llada-moe"
 
         if res is None:
             logger.warning("\n")
@@ -8239,6 +8242,76 @@ def prepare_tensors(self):
                 raise ValueError(f"Unprocessed experts: {experts}")
 
 
+@ModelBase.register("LLaDAMoEModel", "LLaDAMoEModelLM")
+class LLaDAMoEModel(TextModel):
+    model_arch = gguf.MODEL_ARCH.LLADA_MOE
+
+    def set_gguf_parameters(self):
+        super().set_gguf_parameters()
+        if (n_experts := self.hparams.get("num_experts")) is not None:
+            self.gguf_writer.add_expert_count(n_experts)
+
+        if (expert_intermediate_size := self.hparams.get("expert_intermediate_size")) is not None:
+            self.gguf_writer.add_expert_feed_forward_length(expert_intermediate_size)
+
+        # number of experts used per token (top-k)
+        if (n_experts_used := self.hparams.get("num_experts_per_tok")) is not None:
+            self.gguf_writer.add_expert_used_count(n_experts_used)
+
+        self.gguf_writer.add_mask_token_id(156895)
+        self.gguf_writer.add_causal_attention(False)
+        self.gguf_writer.add_diffusion_shift_logits(False)
+
+    _experts: list[dict[str, Tensor]] | None = None
+
+    # Copied from: Qwen2MoeModel
+    def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
+        # process the experts separately
+        if name.find("experts") != -1:
+            n_experts = self.hparams["num_experts"]
+            assert bid is not None
+
+            if self._experts is None:
+                self._experts = [{} for _ in range(self.block_count)]
+
+            self._experts[bid][name] = data_torch
+
+            if len(self._experts[bid]) >= n_experts * 3:
+                tensors: list[tuple[str, Tensor]] = []
+
+                # merge the experts into a single 3d tensor
+                for w_name in ["down_proj", "gate_proj", "up_proj"]:
+                    datas: list[Tensor] = []
+
+                    for xid in range(n_experts):
+                        ename = f"model.layers.{bid}.mlp.experts.{xid}.{w_name}.weight"
+                        datas.append(self._experts[bid][ename])
+                        del self._experts[bid][ename]
+
+                    data_torch = torch.stack(datas, dim=0)
+
+                    merged_name = f"model.layers.{bid}.mlp.experts.{w_name}.weight"
+
+                    new_name = self.map_tensor_name(merged_name)
+
+                    tensors.append((new_name, data_torch))
+                return tensors
+            else:
+                return []
+
+        return [(self.map_tensor_name(name), data_torch)]
+
+    # Copied from: Qwen2MoeModel
+    def prepare_tensors(self):
+        super().prepare_tensors()
+
+        if self._experts is not None:
+            # flatten `list[dict[str, Tensor]]` into `list[str]`
+            experts = [k for d in self._experts for k in d.keys()]
+            if len(experts) > 0:
+                raise ValueError(f"Unprocessed experts: {experts}")
+
+
 @ModelBase.register("HunYuanDenseV1ForCausalLM")
 class HunYuanModel(TextModel):
     model_arch = gguf.MODEL_ARCH.HUNYUAN_DENSE
 
@@ -139,6 +139,7 @@ class TOKENIZER_TYPE(IntEnum):
     {"name": "lfm2",             "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/LiquidAI/LFM2-Tokenizer"},
     {"name": "exaone4",          "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/LGAI-EXAONE/EXAONE-4.0-32B", },
     {"name": "mellum",           "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/JetBrains/Mellum-4b-base", },
+    {"name": "llada-moe",        "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/inclusionAI/LLaDA-MoE-7B-A1B-Base", },
 ]
 
 # some models are known to be broken upstream, so we will skip them as exceptions
 
@@ -510,19 +510,27 @@ static void diffusion_generate(llama_context *          ctx,
     n_generated = params.max_length;
 }
 
-static std::string format_input_text(const std::string & prompt, bool use_chat_template, llama_model * model) {
+static std::string format_input_text(const std::string & prompt, const std::string & system_prompt, bool use_chat_template, llama_model * model) {
     if (!use_chat_template) {
         return prompt;
     }
 
     auto chat_templates = common_chat_templates_init(model, "");
-
     common_chat_templates_inputs inputs;
-    common_chat_msg              user_msg;
-    user_msg.role                = "user";
-    user_msg.content             = prompt;
-    inputs.add_generation_prompt = true;
+    common_chat_msg system_msg;
+
+    if (!system_prompt.empty()) {
+        system_msg.role = "system";
+        system_msg.content = system_prompt;
+        inputs.messages.push_back(system_msg);
+    }
+
+    common_chat_msg user_msg;
+    user_msg.role = "user";
+    user_msg.content = prompt;
+
     inputs.messages.push_back(user_msg);
+    inputs.add_generation_prompt = true;
 
     auto result = common_chat_templates_apply(chat_templates.get(), inputs);
 
@@ -579,7 +587,8 @@ int main(int argc, char ** argv) {
     llama_set_n_threads(ctx, params.cpuparams.n_threads, params.cpuparams_batch.n_threads);
 
     const llama_vocab * vocab            = llama_model_get_vocab(model);
-    std::string         formatted_prompt = format_input_text(params.prompt, params.enable_chat_template, model);
+
+    std::string         formatted_prompt = format_input_text(params.prompt, params.system_prompt, params.enable_chat_template, model);
 
     std::vector<llama_token> input_tokens = common_tokenize(vocab,
                                                             formatted_prompt,
@@ -596,6 +605,7 @@ int main(int argc, char ** argv) {
     }
 
     llama_token mask_token_id = llama_vocab_mask(vocab);
+
     GGML_ASSERT(mask_token_id != LLAMA_TOKEN_NULL);
 
     bool visual_mode = params.diffusion.visual_mode;
 
@@ -8599,7 +8599,6 @@ static void ggml_compute_forward_timestep_embedding_f32(
         }
         if (dim % 2 != 0 && ith == 0) {
             embed_data[2 * half] = 0.f;
-            embed_data[dim] = 0.f;
         }
     }
 }
 
@@ -7,11 +7,11 @@ static __global__ void timestep_embedding_f32(const float * timesteps, float * d
     int j = threadIdx.x + blockIdx.x * blockDim.x;
     float * embed_data = (float *)((char *)dst +  i*nb1);
 
-    if (dim % 2 != 0 && j == ((dim + 1) / 2)) {
-        embed_data[dim] = 0.f;
+    int half = dim / 2;
+    if (dim % 2 != 0 && j == half) {
+        embed_data[2 * half] = 0.f;
     }
 
-    int half = dim / 2;
     if (j >= half) {
         return;
     }
Original file line number	Diff line number	Diff line change
`@@ -1704,7 +1704,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex`
`1704`	`1704`	`[](common_params & params, const std::string & value) {`
`1705`	`1705`	`params.system_prompt = value;`
`1706`	`1706`	`}`
`1707`		`- ).set_examples({LLAMA_EXAMPLE_MAIN}));`
	`1707`	`+ ).set_examples({LLAMA_EXAMPLE_MAIN, LLAMA_EXAMPLE_DIFFUSION}));`
`1708`	`1708`	`add_opt(common_arg(`
`1709`	`1709`	`{"--no-perf"},`
`1710`	`1710`	`string_format("disable internal libllama performance timings (default: %s)", params.no_perf ? "true" : "false"),`
`@@ -2548,7 +2548,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex`
`2548`	`2548`	`{"--cpu-moe", "-cmoe"},`
`2549`	`2549`	`"keep all Mixture of Experts (MoE) weights in the CPU",`
`2550`	`2550`	`[](common_params & params) {`
`2551`		`- params.tensor_buft_overrides.push_back({"\\.ffn_(up\|down\|gate)_exps", ggml_backend_cpu_buffer_type()});`
	`2551`	`+ params.tensor_buft_overrides.push_back(llm_ffn_exps_cpu_override());`
`2552`	`2552`	`}`
`2553`	`2553`	`).set_env("LLAMA_ARG_CPU_MOE"));`
`2554`	`2554`	`add_opt(common_arg(`
`@@ -2561,7 +2561,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex`
`2561`	`2561`	`for (int i = 0; i < value; ++i) {`
`2562`	`2562`	`// keep strings alive and avoid leaking memory by storing them in a static vector`
`2563`	`2563`	`static std::list<std::string> buft_overrides;`
`2564`		`- buft_overrides.push_back(string_format("blk\\.%d\\.ffn_(up\|down\|gate)_exps", i));`
	`2564`	`+ buft_overrides.push_back(llm_ffn_exps_block_regex(i));`
`2565`	`2565`	`params.tensor_buft_overrides.push_back({buft_overrides.back().c_str(), ggml_backend_cpu_buffer_type()});`
`2566`	`2566`	`}`
`2567`	`2567`	`}`
`@@ -2570,7 +2570,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex`
`2570`	`2570`	`{"--cpu-moe-draft", "-cmoed"},`
`2571`	`2571`	`"keep all Mixture of Experts (MoE) weights in the CPU for the draft model",`
`2572`	`2572`	`[](common_params & params) {`
`2573`		`- params.speculative.tensor_buft_overrides.push_back({"\\.ffn_(up\|down\|gate)_exps", ggml_backend_cpu_buffer_type()});`
	`2573`	`+ params.speculative.tensor_buft_overrides.push_back(llm_ffn_exps_cpu_override());`
`2574`	`2574`	`}`
`2575`	`2575`	`).set_examples({LLAMA_EXAMPLE_SPECULATIVE, LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_CPU_MOE_DRAFT"));`
`2576`	`2576`	`add_opt(common_arg(`
`@@ -2582,7 +2582,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex`
`2582`	`2582`	`}`
`2583`	`2583`	`for (int i = 0; i < value; ++i) {`
`2584`	`2584`	`static std::list<std::string> buft_overrides_draft;`
`2585`		`- buft_overrides_draft.push_back(string_format("blk\\.%d\\.ffn_(up\|down\|gate)_exps", i));`
	`2585`	`+ buft_overrides_draft.push_back(llm_ffn_exps_block_regex(i));`
`2586`	`2586`	`params.speculative.tensor_buft_overrides.push_back({buft_overrides_draft.back().c_str(), ggml_backend_cpu_buffer_type()});`
`2587`	`2587`	`}`
`2588`	`2588`	`}`
Original file line number	Diff line number	Diff line change
`@@ -139,6 +139,7 @@ class TOKENIZER_TYPE(IntEnum):`
`139`	`139`	`{"name": "lfm2", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/LiquidAI/LFM2-Tokenizer"},`
`140`	`140`	`{"name": "exaone4", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/LGAI-EXAONE/EXAONE-4.0-32B", },`
`141`	`141`	`{"name": "mellum", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/JetBrains/Mellum-4b-base", },`
	`142`	`+ {"name": "llada-moe", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/inclusionAI/LLaDA-MoE-7B-A1B-Base", },`
`142`	`143`	`]`
`143`	`144`
`144`	`145`	`# some models are known to be broken upstream, so we will skip them as exceptions`
Original file line number	Diff line number	Diff line change
`@@ -8599,7 +8599,6 @@ static void ggml_compute_forward_timestep_embedding_f32(`
`8599`	`8599`	`}`
`8600`	`8600`	`if (dim % 2 != 0 && ith == 0) {`
`8601`	`8601`	`embed_data[2 * half] = 0.f;`
`8602`		`- embed_data[dim] = 0.f;`
`8603`	`8602`	`}`
`8604`	`8603`	`}`
`8605`	`8604`	`}`
Original file line number	Diff line number	Diff line change
`@@ -7,11 +7,11 @@ static __global__ void timestep_embedding_f32(const float * timesteps, float * d`
`7`	`7`	`int j = threadIdx.x + blockIdx.x * blockDim.x;`
`8`	`8`	`float * embed_data = (float )((char )dst + i*nb1);`
`9`	`9`
`10`		`- if (dim % 2 != 0 && j == ((dim + 1) / 2)) {`
`11`		`- embed_data[dim] = 0.f;`
	`10`	`+ int half = dim / 2;`
	`11`	`+ if (dim % 2 != 0 && j == half) {`
	`12`	`+ embed_data[2 * half] = 0.f;`
`12`	`13`	`}`
`13`	`14`
`14`		`- int half = dim / 2;`
`15`	`15`	`if (j >= half) {`
`16`	`16`	`return;`
`17`	`17`	`}`