diff --git a/python/freetoken/models/glm4_moe/moe.py b/python/freetoken/models/glm4_moe/moe.py index 8e595ee7c7..648eb9c32a 100644 --- a/python/freetoken/models/glm4_moe/moe.py +++ b/python/freetoken/models/glm4_moe/moe.py @@ -32,10 +32,8 @@ def __init__(self, config: ModelConfig, layer_id: int, *, prefix: str = ""): self.topk_group = config.topk_group self.gate = LinearReplicated(config.hidden_size, config.num_experts, has_bias=False) - # DeepSeek-style selection bias; a registered buffer in HF (kept fp32 there). We - # store it in the model's bf16 dtype and upcast at use, which is exact enough for - # the argmax-style top-k selection. - self.e_score_correction_bias = torch.empty(config.num_experts) + # Keep selection bias in fp32: rounding can change the selected experts. + self.e_score_correction_bias = torch.empty(config.num_experts, dtype=torch.float32) # The offload cache indexes experts by *MoE* layer (global layer minus # first_k_dense_replace), matching how the loader packs the expert banks. The diff --git a/python/freetoken/models/glm4_moe/weight.py b/python/freetoken/models/glm4_moe/weight.py index 38b17a0536..aa6538dcf9 100644 --- a/python/freetoken/models/glm4_moe/weight.py +++ b/python/freetoken/models/glm4_moe/weight.py @@ -168,11 +168,11 @@ def _iter_resident_weights(reader, config, primary) -> Iterator[tuple[str, torch for proj in ("gate_proj", "up_proj", "down_proj"): yield from _iter_nvfp4_resident(reader, f"{m}.{proj}", f"{m}.{proj}") else: - # router (bf16 gate + fp32 selection bias -> bf16) and shared expert. + # router (bf16 gate + fp32 selection bias) and shared expert. yield f"{m}.gate.weight", reader.get(f"{m}.gate.weight") yield ( f"{m}.e_score_correction_bias", - reader.get(f"{m}.gate.e_score_correction_bias").to(torch.bfloat16), + reader.get(f"{m}.gate.e_score_correction_bias").to(torch.float32), ) s = f"{m}.shared_experts" for proj in ("gate_proj", "up_proj", "down_proj"):