fix: skip leading EOS tokens in streaming + non-streaming

- Skip <|im_end|> generated as first token (prevents empty responses)
- Clean <think> tags as plain text in generated output
This commit is contained in:
Asep Haryana
2026-07-26 16:09:04 +07:00
parent 495b9ed126
commit 59c77108a5
2 changed files with 43 additions and 0 deletions
+10
View File
@@ -219,6 +219,16 @@ impl LlamaEngine {
let mut current = inner.sample(sampler);
// Skip leading EOS tokens (like <|im_end|> as first token)
while output.is_empty() && self.model.is_eog_token(current) {
let pos = input_tokens.len() as i32 + output.len() as i32;
if let Err(e) = inner.decode(current, pos) {
tracing::info!(" Decode error: {e}");
break;
}
current = inner.sample(sampler);
}
for _ in 0..max_tokens {
if self.model.is_eog_token(current) {
break;