Skip to content

Commit cf61db7

Browse files
hans00claude
andcommitted
perf(bluemagpie): drop per-eval weight dequant + fuse QKV/gate-up + native rope
Phase-4 profiling (new CODEC_OP_PROFILE per-node profiler in graph_exec.cpp, ggml_backend_sched_set_eval_callback based, zero-cost when unset) showed the CFM step spent 31% of compute in CPY nodes: codec_graph_weight bakes an F16->F32 dequant of every weight into the graph, re-run every eval, while the matmuls themselves were 1%. - tensor_utils: new codec_graph_weight_mat — F16/BF16 matmul weights pass through to ggml_mul_mat natively (no dequant CPY); quantized types still cast to F32 for parity (native quantized mul_mat drops Q8 audio corr to 0.9981; CODEC_MAT_NATIVE_QUANT=1 opts into the lossy fast path). codec_graph_weight itself unchanged, other models unaffected; only BlueMagpie call sites switched. - Converter writes fused .attn_qkv.w / .gate_up.w for the three MiniCPM stacks (7 -> 5 matmuls per block); runtime splits with views and keeps unfused-name fallback. _emit_lm now honors --quantization (F16 default restores a genuine F16 lm gguf). - RoPE via native ggml_rope_ext (NEOX, freq_factors=rope_short_factor, attn_factor baked at convert time); ~1540 fewer nodes; baked cos/sin fallback kept. Scale folded into soft_max_ext. - flash_attn_ext and in-graph set_rows KV write evaluated and deliberately skipped: attention is <8% of compute and the KV host round-trip <3 ms; measurements in the phase report. At official defaults (cfg=2.8, ts=9): CFM-only CPU-8 F16 305 -> 154 ms/step, Vulkan F16 116 ms/step (new best; beats Q8 on both speed and precision on UMA), full step_generate 173 ms/step warm (~realtime for 160 ms audio/step). Parity: F16 patch corr 0.999997 / audio 0.999977; Q8 unchanged; decode + moss cross-model smokes PASS; 72-step runs bit-identical. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
1 parent 94dd98d commit cf61db7

6 files changed

Lines changed: 388 additions & 73 deletions

File tree

‎scripts/converters/bluemagpie.py‎

Lines changed: 34 additions & 14 deletions
Original file line numberDiff line numberDiff line change
@@ -290,19 +290,22 @@ def lm_t(name: str) -> np.ndarray:
290290
raise KeyError(f"missing LM tensor: {name}")
291291
return np.asarray(arr)
292292

293-
# LM weight-streaming acceleration: 2D matmul weights under lm.* are
294-
# quantized to Q8_0 (halves the FP16 bytes streamed per CFM step). The
295-
# gguf Q8_0 helper quantizes along ne[0] (the innermost dim = the matmul
296-
# contraction dim), so rows must be a multiple of the 32-elem block.
297-
# Anything else (norms, biases, 1D vectors, non-.w tensors, or rows not
298-
# /32) falls back to F16 — matching the rest of the LM section's default.
299-
# This is independent of the global --quantization flag so the AudioVAE
300-
# section keeps its existing F16 behaviour untouched.
293+
# LM matmul weight dtype follows the --quantization flag:
294+
# - Q8_0 → 2D lm.* matmul weights are Q8_0 (halves the bytes streamed
295+
# per CFM step; the gguf Q8_0 helper quantizes along ne[0] =
296+
# the matmul contraction dim, so rows must be a 32-multiple).
297+
# - else → F16 (the genuine "F16 GGUF"), which the runtime feeds to
298+
# ggml_mul_mat natively (no per-eval dequant CPY, see
299+
# codec_graph_weight_mat). On CPU the F16 path is markedly
300+
# faster AND higher-parity than cast-from-Q8, so the two
301+
# GGUFs are meant to be genuinely different here.
302+
# Anything not eligible (norms, biases, 1D vectors, rows not /32) is F16.
301303
_Q8_BLK = 32
304+
_lm_use_q8 = (self.quantization == "Q8_0")
302305

303306
def _add_lm_weight(name: str, arr: np.ndarray) -> None:
304307
arr = np.asarray(arr)
305-
if arr.ndim == 2 and (int(arr.shape[-1]) % _Q8_BLK) == 0:
308+
if _lm_use_q8 and arr.ndim == 2 and (int(arr.shape[-1]) % _Q8_BLK) == 0:
306309
self._add_tensor(writer, name, arr, "Q8_0")
307310
else:
308311
self._add_tensor(writer, name, arr, "F16")
@@ -318,12 +321,19 @@ def add_norm(prefix: str, out: str) -> None:
318321
def add_minicpm_stack(src: str, out: str, n_layers: int) -> None:
319322
for i in range(n_layers):
320323
s, o = f"{src}.layers.{i}", f"{out}.layers.{i}"
321-
add_lin(f"{s}.self_attn.q_proj", o + ".attn_q")
322-
add_lin(f"{s}.self_attn.k_proj", o + ".attn_k")
323-
add_lin(f"{s}.self_attn.v_proj", o + ".attn_v")
324+
# Fuse Q|K|V and gate|up into single matmul weights (concat along
325+
# the output dim = PyTorch axis 0 of the (out,in) weight). The
326+
# contraction dim ne[0]=in is unchanged, so Q8_0 block alignment
327+
# (in % 32) is preserved. The runtime splits the outputs with
328+
# views (see codec_bm_minicpm_block_htb).
329+
q = lm_t(f"{s}.self_attn.q_proj.weight")
330+
k = lm_t(f"{s}.self_attn.k_proj.weight")
331+
v = lm_t(f"{s}.self_attn.v_proj.weight")
332+
_add_lm_weight(o + ".attn_qkv.w", np.concatenate([q, k, v], axis=0))
324333
add_lin(f"{s}.self_attn.o_proj", o + ".attn_o")
325-
add_lin(f"{s}.mlp.gate_proj", o + ".gate")
326-
add_lin(f"{s}.mlp.up_proj", o + ".up")
334+
g = lm_t(f"{s}.mlp.gate_proj.weight")
335+
up = lm_t(f"{s}.mlp.up_proj.weight")
336+
_add_lm_weight(o + ".gate_up.w", np.concatenate([g, up], axis=0))
327337
add_lin(f"{s}.mlp.down_proj", o + ".down")
328338
add_norm(f"{s}.input_layernorm", o + ".ln1")
329339
add_norm(f"{s}.post_attention_layernorm", o + ".ln2")
@@ -384,6 +394,16 @@ def add_minicpm_stack(src: str, out: str, n_layers: int) -> None:
384394
self._add_tensor(writer, "lm.rope.cos", (np.cos(emb) * scaling).astype(np.float32), "F32")
385395
self._add_tensor(writer, "lm.rope.sin", (np.sin(emb) * scaling).astype(np.float32), "F32")
386396

397+
# Native-rope path: ggml_rope_ext(mode=NEOX, freq_base=theta,
398+
# freq_factors=short, attn_factor=scaling, ext_factor=0, freq_scale=1)
399+
# reproduces the LongRoPE short_factor branch above in a single node per
400+
# q/k (vs the ~8-node baked cos/sin rotate_half). short_factor is stored
401+
# F32 len head_dim/2; theta + attn_factor go to KV. The cos/sin table is
402+
# retained so older runtimes / a fallback path still work.
403+
self._add_tensor(writer, "lm.rope.short_factor", short.astype(np.float32), "F32")
404+
writer.add_float32("codec.lm.rope_theta", float(cfg["rope_theta"]))
405+
writer.add_float32("codec.lm.rope_attn_factor", float(scaling))
406+
387407
# LM metadata
388408
writer.add_bool("codec.lm.has_adaptor", True)
389409
writer.add_string("codec.lm.kind", "continuous_latent_cfm")

‎src/lm/bluemagpie_cfm.cpp‎

Lines changed: 57 additions & 27 deletions
Original file line numberDiff line numberDiff line change
@@ -66,6 +66,10 @@ ggml_tensor * lin(ggml_context * c, ggml_tensor * w, ggml_tensor * x, ggml_tenso
6666
return b ? ggml_add(c, y, codec_graph_cast_f32(c, b)) : y;
6767
}
6868

69+
// Matmul-weight accessor: no dequant CPY for F16/BF16 (see
70+
// codec_graph_weight_mat). Use for every tensor that lands as the LHS of
71+
// ggml_mul_mat in this per-step graph.
72+
6973
// One RALM layer, incremental KV (1 new token). Causal, no rope, no qk_norm.
7074
// Attends over a fixed bucket cache[0..B) plus the 1 new token, where B is a
7175
// 64-multiple >= kv_pos+1 so the graph shape is kv_pos-independent within a
@@ -79,13 +83,24 @@ ggml_tensor * bm_ralm_kv_step(ggml_context * ctx, ggml_tensor * x_ht, const std:
7983
ggml_tensor * k_cache_l, ggml_tensor * v_cache_l, int32_t bucket, ggml_tensor * mask,
8084
ggml_tensor ** kk_out, ggml_tensor ** vv_out) {
8185

82-
auto W = [&](const char * s) { return codec_graph_weight(ctx, model, prefix + s); };
86+
auto W = [&](const char * s) { return codec_graph_weight(ctx, model, prefix + s); };
87+
auto WM = [&](const char * s) { return codec_graph_weight_mat(ctx, model, prefix + s); };
8388
const int32_t q_dim = n_heads * head_dim;
8489

90+
const int32_t kv_dim = n_kv * head_dim;
8591
ggml_tensor * h = codec_op_rms_norm_ct(ctx, x_ht, eps, W(".ln1.w"));
86-
ggml_tensor * q = ggml_reshape_3d(ctx, ggml_mul_mat(ctx, W(".attn_q.w"), h), head_dim, n_heads, 1);
87-
ggml_tensor * kk = ggml_reshape_3d(ctx, ggml_mul_mat(ctx, W(".attn_k.w"), h), head_dim, n_kv, 1);
88-
ggml_tensor * vv = ggml_reshape_3d(ctx, ggml_mul_mat(ctx, W(".attn_v.w"), h), head_dim, n_kv, 1);
92+
ggml_tensor * q, * kk, * vv;
93+
ggml_tensor * w_qkv = codec_graph_weight_mat(ctx, model, prefix + ".attn_qkv.w");
94+
if (w_qkv != nullptr) {
95+
ggml_tensor * qkv = ggml_mul_mat(ctx, w_qkv, h); // (q_dim+2*kv_dim, 1)
96+
q = ggml_reshape_3d(ctx, ggml_cont(ctx, ggml_view_1d(ctx, qkv, q_dim, 0)), head_dim, n_heads, 1);
97+
kk = ggml_reshape_3d(ctx, ggml_cont(ctx, ggml_view_1d(ctx, qkv, kv_dim, (size_t) q_dim * qkv->nb[0])), head_dim, n_kv, 1);
98+
vv = ggml_reshape_3d(ctx, ggml_cont(ctx, ggml_view_1d(ctx, qkv, kv_dim, (size_t) (q_dim + kv_dim) * qkv->nb[0])), head_dim, n_kv, 1);
99+
} else {
100+
q = ggml_reshape_3d(ctx, ggml_mul_mat(ctx, WM(".attn_q.w"), h), head_dim, n_heads, 1);
101+
kk = ggml_reshape_3d(ctx, ggml_mul_mat(ctx, WM(".attn_k.w"), h), head_dim, n_kv, 1);
102+
vv = ggml_reshape_3d(ctx, ggml_mul_mat(ctx, WM(".attn_v.w"), h), head_dim, n_kv, 1);
103+
}
89104
*kk_out = kk;
90105
*vv_out = vv;
91106

@@ -107,12 +122,22 @@ ggml_tensor * bm_ralm_kv_step(ggml_context * ctx, ggml_tensor * x_ht, const std:
107122
ggml_tensor * v_p = ggml_cont(ctx, ggml_permute(ctx, v_all, 1, 2, 0, 3));
108123
ggml_tensor * attn = ggml_cont(ctx, ggml_permute(ctx, ggml_mul_mat(ctx, v_p, scores), 0, 2, 1, 3));
109124
attn = ggml_reshape_2d(ctx, attn, q_dim, 1);
110-
x_ht = ggml_add(ctx, x_ht, ggml_mul_mat(ctx, W(".attn_o.w"), attn));
125+
x_ht = ggml_add(ctx, x_ht, ggml_mul_mat(ctx, WM(".attn_o.w"), attn));
111126

112127
h = codec_op_rms_norm_ct(ctx, x_ht, eps, W(".ln2.w"));
113-
ggml_tensor * g = ggml_silu(ctx, ggml_mul_mat(ctx, W(".gate.w"), h));
114-
ggml_tensor * u = ggml_mul_mat(ctx, W(".up.w"), h);
115-
return ggml_add(ctx, x_ht, ggml_mul_mat(ctx, W(".down.w"), ggml_mul(ctx, g, u)));
128+
ggml_tensor * g, * u;
129+
ggml_tensor * w_gu = codec_graph_weight_mat(ctx, model, prefix + ".gate_up.w");
130+
if (w_gu != nullptr) {
131+
ggml_tensor * gu = ggml_mul_mat(ctx, w_gu, h); // (2*ffn, 1)
132+
const int64_t ffn = gu->ne[0] / 2;
133+
g = ggml_view_1d(ctx, gu, ffn, 0);
134+
u = ggml_view_1d(ctx, gu, ffn, (size_t) ffn * gu->nb[0]);
135+
} else {
136+
g = ggml_mul_mat(ctx, WM(".gate.w"), h);
137+
u = ggml_mul_mat(ctx, WM(".up.w"), h);
138+
}
139+
g = ggml_silu(ctx, g);
140+
return ggml_add(ctx, x_ht, ggml_mul_mat(ctx, WM(".down.w"), ggml_mul(ctx, g, u)));
116141
}
117142

118143
// Build the whole per-step graph.
@@ -138,8 +163,13 @@ bool build_step(ggml_context * ctx, void * ud, ggml_tensor ** out) {
138163
cfm_build * p = static_cast<cfm_build *>(ud);
139164
const cfm_impl & I = p->imp;
140165
const codec_model * model = p->model;
141-
auto W = [&](const char * s) { return codec_graph_weight(ctx, model, s); };
166+
auto W = [&](const char * s) { return codec_graph_weight(ctx, model, s); };
167+
auto WM = [&](const char * s) { return codec_graph_weight_mat(ctx, model, s); };
142168
const int32_t P = I.patch_size, D = I.latent_dim;
169+
// lin() wrapper that fetches the weight via the no-dequant matmul accessor.
170+
auto linW = [&](const char * wn, ggml_tensor * x, const char * bn) {
171+
return lin(ctx, WM(wn), x, bn ? W(bn) : nullptr);
172+
};
143173

144174
ggml_tensor * h_in = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, I.h_barbet, 1); ggml_set_name(h_in, "bm.cfm.h_in");
145175
ggml_tensor * pfb_lm = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, I.h_vox, 1); ggml_set_name(pfb_lm, "bm.cfm.pfb_lm");
@@ -150,21 +180,21 @@ bool build_step(ggml_context * ctx, void * ud, ggml_tensor ** out) {
150180

151181
// ---- tslm_adapter(h_in) → h_vox ----
152182
ggml_tensor * a = codec_op_rms_norm_ct(ctx, h_in, I.eps, W("lm.tslm_adapter.norm.w"));
153-
a = lin(ctx, W("lm.tslm_adapter.proj.w"), a, W("lm.tslm_adapter.proj.b")); // (h_vox,1)
183+
a = linW("lm.tslm_adapter.proj.w", a, "lm.tslm_adapter.proj.b"); // (h_vox,1)
154184
{ // residual SwiGLU block
155185
ggml_tensor * bn = codec_op_rms_norm_ct(ctx, a, I.eps, W("lm.tslm_adapter.blk0.ln.w"));
156-
ggml_tensor * g = ggml_silu(ctx, lin(ctx, W("lm.tslm_adapter.blk0.gate.w"), bn, nullptr));
157-
ggml_tensor * u = lin(ctx, W("lm.tslm_adapter.blk0.up.w"), bn, nullptr);
158-
a = ggml_add(ctx, a, lin(ctx, W("lm.tslm_adapter.blk0.down.w"), ggml_mul(ctx, g, u), nullptr));
186+
ggml_tensor * g = ggml_silu(ctx, linW("lm.tslm_adapter.blk0.gate.w", bn, nullptr));
187+
ggml_tensor * u = linW("lm.tslm_adapter.blk0.up.w", bn, nullptr);
188+
a = ggml_add(ctx, a, linW("lm.tslm_adapter.blk0.down.w", ggml_mul(ctx, g, u), nullptr));
159189
}
160190
// ---- FSQ: round(tanh(in_proj(a))*s)/s then out_proj ----
161-
ggml_tensor * q = ggml_tanh(ctx, lin(ctx, W("lm.fsq.in_proj.w"), a, W("lm.fsq.in_proj.b")));
191+
ggml_tensor * q = ggml_tanh(ctx, linW("lm.fsq.in_proj.w", a, "lm.fsq.in_proj.b"));
162192
q = ggml_scale(ctx, ggml_round(ctx, ggml_scale(ctx, q, (float) I.fsq_scale)), 1.0f / (float) I.fsq_scale);
163-
ggml_tensor * lm_hidden = lin(ctx, W("lm.fsq.out_proj.w"), q, W("lm.fsq.out_proj.b")); // (h_vox,1)
193+
ggml_tensor * lm_hidden = linW("lm.fsq.out_proj.w", q, "lm.fsq.out_proj.b"); // (h_vox,1)
164194

165195
// ---- RALM input = fusion_concat_proj([lm_hidden ; prev_feedback_lm]) ----
166196
ggml_tensor * fus = ggml_concat(ctx, lm_hidden, pfb_lm, 0); // (2*h_vox,1)
167-
ggml_tensor * ralm_new = lin(ctx, W("lm.proj.fusion_concat.w"), fus, W("lm.proj.fusion_concat.b")); // (h_vox,1)
197+
ggml_tensor * ralm_new = linW("lm.proj.fusion_concat.w", fus, "lm.proj.fusion_concat.b"); // (h_vox,1)
168198

169199
// ---- RALM: one incremental step over the persistent KV cache ----
170200
// Shared additive attention mask for all RALM layers. scores has ne0 =
@@ -190,16 +220,16 @@ bool build_step(ggml_context * ctx, void * ud, ggml_tensor ** out) {
190220
ggml_tensor * residual_hidden = codec_op_rms_norm_ct(ctx, rh, I.eps, W("lm.ralm.norm.w")); // (h_vox,1)
191221

192222
// ---- mu = [lm_to_dit(lm_hidden) ; res_to_dit(residual_hidden)] → (h_dit, n_mu) ----
193-
ggml_tensor * mu1 = lin(ctx, W("lm.proj.lm_to_dit.w"), lm_hidden, W("lm.proj.lm_to_dit.b"));
194-
ggml_tensor * mu2 = lin(ctx, W("lm.proj.res_to_dit.w"), residual_hidden, W("lm.proj.res_to_dit.b"));
223+
ggml_tensor * mu1 = linW("lm.proj.lm_to_dit.w", lm_hidden, "lm.proj.lm_to_dit.b");
224+
ggml_tensor * mu2 = linW("lm.proj.res_to_dit.w", residual_hidden, "lm.proj.res_to_dit.b");
195225
ggml_tensor * mu = ggml_concat(ctx, mu1, mu2, 1); // (h_dit, 2)
196226

197227
// ---- CFM Euler solver (unrolled, cfg_zero_star) ----
198-
ggml_tensor * cond_h = lin(ctx, W("lm.locdit.cond_proj.w"), cond, W("lm.locdit.cond_proj.b")); // (h_dit,P)
228+
ggml_tensor * cond_h = linW("lm.locdit.cond_proj.w", cond, "lm.locdit.cond_proj.b"); // (h_dit,P)
199229
ggml_tensor * mu_zero = ggml_scale(ctx, mu, 0.0f);
200230
auto time_mlp = [&](const char * pfx, ggml_tensor * s) {
201-
ggml_tensor * h = ggml_silu(ctx, lin(ctx, W((std::string(pfx) + ".l1.w").c_str()), s, W((std::string(pfx) + ".l1.b").c_str())));
202-
return lin(ctx, W((std::string(pfx) + ".l2.w").c_str()), h, W((std::string(pfx) + ".l2.b").c_str()));
231+
ggml_tensor * h = ggml_silu(ctx, lin(ctx, WM((std::string(pfx) + ".l1.w").c_str()), s, W((std::string(pfx) + ".l1.b").c_str())));
232+
return lin(ctx, WM((std::string(pfx) + ".l2.w").c_str()), h, W((std::string(pfx) + ".l2.b").c_str()));
203233
};
204234
ggml_tensor * dt_emb = time_mlp("lm.locdit.dtime_mlp", dtsin);
205235
const int64_t T = (int64_t) I.n_mu + 1 + 2 * P;
@@ -209,7 +239,7 @@ bool build_step(ggml_context * ctx, void * ud, ggml_tensor ** out) {
209239
const bool cfg_one = (p->cfg_value == 1.0f);
210240
ggml_tensor * x = z;
211241
for (int32_t s = 0; s < p->n_real; ++s) {
212-
ggml_tensor * x_h = lin(ctx, W("lm.locdit.in_proj.w"), x, W("lm.locdit.in_proj.b"));
242+
ggml_tensor * x_h = linW("lm.locdit.in_proj.w", x, "lm.locdit.in_proj.b");
213243
ggml_tensor * tsin_s = ggml_cont(ctx, ggml_view_2d(ctx, tsin, I.h_dit, 1, tsin->nb[1], (size_t) s * tsin->nb[1]));
214244
ggml_tensor * t_h = ggml_add(ctx, time_mlp("lm.locdit.time_mlp", tsin_s), dt_emb);
215245
ggml_tensor * dphi;
@@ -234,13 +264,13 @@ bool build_step(ggml_context * ctx, void * ud, ggml_tensor ** out) {
234264
ggml_set_output(x);
235265

236266
// ---- stop head ----
237-
ggml_tensor * sp = ggml_silu(ctx, lin(ctx, W("lm.stop.proj.w"), lm_hidden, W("lm.stop.proj.b")));
238-
ggml_tensor * stop_logit = lin(ctx, W("lm.stop.head.w"), sp, nullptr); // (2,1)
267+
ggml_tensor * sp = ggml_silu(ctx, linW("lm.stop.proj.w", lm_hidden, "lm.stop.proj.b"));
268+
ggml_tensor * stop_logit = linW("lm.stop.head.w", sp, nullptr); // (2,1)
239269
ggml_set_name(stop_logit, "bm.cfm.stop");
240270
ggml_set_output(stop_logit);
241271

242272
// ---- LocEnc(patch) → feedback ----
243-
ggml_tensor * le = lin(ctx, W("lm.locenc.in_proj.w"), x, W("lm.locenc.in_proj.b")); // (h_enc, P)
273+
ggml_tensor * le = linW("lm.locenc.in_proj.w", x, "lm.locenc.in_proj.b"); // (h_enc, P)
244274
ggml_tensor * sptok = ggml_reshape_2d(ctx, codec_graph_cast_f32(ctx, W("lm.locenc.special_token")), I.h_enc, 1);
245275
le = ggml_concat(ctx, sptok, le, 1); // (h_enc, P+1)
246276
const int64_t Te = le->ne[1];
@@ -252,10 +282,10 @@ bool build_step(ggml_context * ctx, void * ud, ggml_tensor ** out) {
252282
}
253283
le = codec_op_rms_norm_ct(ctx, le, I.eps, W("lm.locenc.norm.w"));
254284
ggml_tensor * cls = ggml_cont(ctx, ggml_view_1d(ctx, le, I.h_enc, 0)); // (h_enc,)
255-
ggml_tensor * fb_tslm = lin(ctx, W("lm.proj.enc_to_tslm.w"), cls, W("lm.proj.enc_to_tslm.b")); // (h_barbet,)
285+
ggml_tensor * fb_tslm = linW("lm.proj.enc_to_tslm.w", cls, "lm.proj.enc_to_tslm.b"); // (h_barbet,)
256286
ggml_set_name(fb_tslm, "bm.cfm.fb_tslm");
257287
ggml_set_output(fb_tslm);
258-
ggml_tensor * fb_lm = lin(ctx, W("lm.proj.enc_to_lm.w"), cls, W("lm.proj.enc_to_lm.b")); // (h_vox,)
288+
ggml_tensor * fb_lm = linW("lm.proj.enc_to_lm.w", cls, "lm.proj.enc_to_lm.b"); // (h_vox,)
259289
ggml_set_name(fb_lm, "bm.cfm.fb_lm");
260290
ggml_set_output(fb_lm);
261291

0 commit comments

Comments
 (0)