diff --git a/c/sample.h b/c/sample.h index 414c6210b..806810b8a 100644 --- a/c/sample.h +++ b/c/sample.h @@ -116,14 +116,21 @@ static void stops_arm_tok(const Cfg *c, int tok_eos, Tok *T){ int nsp = 0; if (T) for (int id = 0; id < T->n_ids && g_nstop < 64; id++) if (T->id_special[id] && !is_stop(id)) { g_stop[g_nstop++] = id; nsp++; } - /* #401: in serve mode keep ONLY <|endoftext|>. Role markers <|user|>/<|observation|> - * (config stops + tokenizer special set) are boundaries the Python server owns; as - * hard stops they cut generation the moment the model opens a block, - * because int4 argmax noise picks a stop-token ID over the correct '<' token. */ + /* #401: in serve mode discard tokenizer-only special tokens, but preserve + * EVERY EOS explicitly declared by config.json/generation_config.json. + * GLM-5.2 declares <|endoftext|>, <|user|>, and <|observation|> as EOS. + * Keeping only tok_eos leaks <|user|> into CLI/API output and generation + * continues until NGEN. Tokenizer-only specials remain filtered so int4 + * argmax noise cannot turn unrelated control tokens into hard stops. */ if (getenv("SERVE") && tok_eos >= 0) { int kept = 0; - for (int i = 0; i < g_nstop; i++) if (g_stop[i] == tok_eos) g_stop[kept++] = g_stop[i]; - if (kept < g_nstop) fprintf(stderr, "[stop] serve mode: filtered %d non-EOS stop tokens (tool-call safety, #401)\n", g_nstop - kept); + for (int i = 0; i < g_nstop; i++) { + int declared = g_stop[i] == tok_eos; + for (int j = 0; !declared && j < c->n_stop; j++) + if (g_stop[i] == c->stop_ids[j]) declared = 1; + if (declared) g_stop[kept++] = g_stop[i]; + } + if (kept < g_nstop) fprintf(stderr, "[stop] serve mode: filtered %d tokenizer-only special tokens (tool-call safety, #401)\n", g_nstop - kept); g_nstop = kept; nsp = 0; } fprintf(stderr, "[stop] %d stop tokens:", g_nstop); diff --git a/c/tests/test_stops.c b/c/tests/test_stops.c index 5e7e9d2ab..2227a3857 100644 --- a/c/tests/test_stops.c +++ b/c/tests/test_stops.c @@ -128,6 +128,31 @@ int main(void){ fail|=expect("T=NULL: no tokenizer sweep",101,0); if(!fail) printf(" T=NULL -> config stops only (validation path untouched) ok\n"); } + /* 6. SERVE filters tokenizer-only specials for tool safety, but MUST keep + * every model-declared EOS. This catches the v1.1.1 role-token leak: + * the old code kept only tok_eos=100 and discarded 101/102. */ + { Cfg c; memset(&c,0,sizeof c); + write_cfg(dir,"config.json","[100,101,102]"); + rm_file(dir,"generation_config.json"); + load_cfg(&c,dir); +#ifdef _WIN32 + _putenv_s("SERVE","1"); +#else + setenv("SERVE","1",1); +#endif + stops_arm_tok(&c,100,&T); + fail|=expect("SERVE: endoftext",100,1); + fail|=expect("SERVE: declared <|user|> EOS",101,1); + fail|=expect("SERVE: declared <|observation|> EOS",102,1); + fail|=expect("SERVE: tokenizer-only <|assistant|>",103,0); + fail|=expect("SERVE: tokenizer-only ",104,0); +#ifdef _WIN32 + _putenv_s("SERVE",""); +#else + unsetenv("SERVE"); +#endif + if(!fail) printf(" SERVE -> all declared EOS kept; tokenizer-only specials filtered ok\n"); } + rm_file(dir,"config.json"); rm_file(dir,"generation_config.json"); rm_file(dir,"tokenizer.json"); rmdir(dir); if(fail){ printf("test_stops: FAIL\n"); return 1; }