Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
19 changes: 13 additions & 6 deletions c/sample.h
Original file line number Diff line number Diff line change
Expand Up @@ -116,14 +116,21 @@ static void stops_arm_tok(const Cfg *c, int tok_eos, Tok *T){
int nsp = 0;
if (T) for (int id = 0; id < T->n_ids && g_nstop < 64; id++)
if (T->id_special[id] && !is_stop(id)) { g_stop[g_nstop++] = id; nsp++; }
/* #401: in serve mode keep ONLY <|endoftext|>. Role markers <|user|>/<|observation|>
* (config stops + tokenizer special set) are boundaries the Python server owns; as
* hard stops they cut generation the moment the model opens a <tool_call> block,
* because int4 argmax noise picks a stop-token ID over the correct '<' token. */
/* #401: in serve mode discard tokenizer-only special tokens, but preserve
* EVERY EOS explicitly declared by config.json/generation_config.json.
* GLM-5.2 declares <|endoftext|>, <|user|>, and <|observation|> as EOS.
* Keeping only tok_eos leaks <|user|> into CLI/API output and generation
* continues until NGEN. Tokenizer-only specials remain filtered so int4
* argmax noise cannot turn unrelated control tokens into hard stops. */
if (getenv("SERVE") && tok_eos >= 0) {
int kept = 0;
for (int i = 0; i < g_nstop; i++) if (g_stop[i] == tok_eos) g_stop[kept++] = g_stop[i];
if (kept < g_nstop) fprintf(stderr, "[stop] serve mode: filtered %d non-EOS stop tokens (tool-call safety, #401)\n", g_nstop - kept);
for (int i = 0; i < g_nstop; i++) {
int declared = g_stop[i] == tok_eos;
for (int j = 0; !declared && j < c->n_stop; j++)
if (g_stop[i] == c->stop_ids[j]) declared = 1;
if (declared) g_stop[kept++] = g_stop[i];
}
if (kept < g_nstop) fprintf(stderr, "[stop] serve mode: filtered %d tokenizer-only special tokens (tool-call safety, #401)\n", g_nstop - kept);
g_nstop = kept; nsp = 0;
}
fprintf(stderr, "[stop] %d stop tokens:", g_nstop);
Expand Down
25 changes: 25 additions & 0 deletions c/tests/test_stops.c
Original file line number Diff line number Diff line change
Expand Up @@ -128,6 +128,31 @@ int main(void){
fail|=expect("T=NULL: no tokenizer sweep",101,0);
if(!fail) printf(" T=NULL -> config stops only (validation path untouched) ok\n"); }

/* 6. SERVE filters tokenizer-only specials for tool safety, but MUST keep
* every model-declared EOS. This catches the v1.1.1 role-token leak:
* the old code kept only tok_eos=100 and discarded 101/102. */
{ Cfg c; memset(&c,0,sizeof c);
write_cfg(dir,"config.json","[100,101,102]");
rm_file(dir,"generation_config.json");
load_cfg(&c,dir);
#ifdef _WIN32
_putenv_s("SERVE","1");
#else
setenv("SERVE","1",1);
#endif
stops_arm_tok(&c,100,&T);
fail|=expect("SERVE: endoftext",100,1);
fail|=expect("SERVE: declared <|user|> EOS",101,1);
fail|=expect("SERVE: declared <|observation|> EOS",102,1);
fail|=expect("SERVE: tokenizer-only <|assistant|>",103,0);
fail|=expect("SERVE: tokenizer-only <sop>",104,0);
#ifdef _WIN32
_putenv_s("SERVE","");
#else
unsetenv("SERVE");
#endif
if(!fail) printf(" SERVE -> all declared EOS kept; tokenizer-only specials filtered ok\n"); }

rm_file(dir,"config.json"); rm_file(dir,"generation_config.json"); rm_file(dir,"tokenizer.json");
rmdir(dir);
if(fail){ printf("test_stops: FAIL\n"); return 1; }
Expand Down
Loading