Skip to content

Commit 8065390

Browse files
committed
Fixed: The way tokens were counted with LlamaCpp wasn't accurate in chat mode.
Fixed: Chat processing would taken exponentially more time as you go past 32K max tokens in LlamaCpp+Chat mode. This is now fixed and processing is now snappy.
1 parent 85626e4 commit 8065390

5 files changed

Lines changed: 290 additions & 10 deletions

File tree

‎API/LlamaCppClient.cs‎

Lines changed: 78 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -167,6 +167,22 @@ public TokenCountResponse GetTokenCountSync(MessageListQuery body)
167167
return Task.Run(() => GetTokenCountAsync(body)).ConfigureAwait(false).GetAwaiter().GetResult();
168168
}
169169

170+
/// <summary>
171+
/// Applies the model's chat template to a list of chat messages via POST /apply-template, returning
172+
/// the fully-formatted prompt string exactly as it would be built for /v1/chat/completions (names,
173+
/// roles, template scaffolding included). This does not run inference.
174+
/// </summary>
175+
public async Task<ApplyTemplateResponse> ApplyTemplateAsync(ApplyTemplateQuery body, CancellationToken cancellationToken = default)
176+
{
177+
return await SendRequestAsync<ApplyTemplateResponse>(_httpClient!, HttpMethod.Post, "/apply-template", body, cancellationToken: cancellationToken).ConfigureAwait(false);
178+
}
179+
180+
public ApplyTemplateResponse ApplyTemplateSync(ApplyTemplateQuery body)
181+
{
182+
// Using a new task and ConfigureAwait(false) to avoid deadlocks
183+
return Task.Run(() => ApplyTemplateAsync(body)).ConfigureAwait(false).GetAwaiter().GetResult();
184+
}
185+
170186
public async Task<LlamaServerState> GetServerStateAsync(CancellationToken cancellationToken = default)
171187
{
172188
return await SendRequestAsync<LlamaServerState>(_httpClient!, HttpMethod.Get, "/props", cancellationToken: cancellationToken).ConfigureAwait(false);
@@ -388,6 +404,68 @@ public class MessageListQuery
388404
public List<MessageQuery> messages { get; set; } = [];
389405
}
390406

407+
/// <summary>
408+
/// Request body for POST /apply-template. The messages must be in the same shape used by
409+
/// /v1/chat/completions so the server applies the identical chat template used at generation time.
410+
/// This is a plain, Newtonsoft-serializable structure on purpose: the OpenAI.Chat.* types use
411+
/// System.Text.Json (JsonNode) internals that Newtonsoft cannot serialize.
412+
/// </summary>
413+
public class ApplyTemplateQuery
414+
{
415+
[JsonProperty("messages")]
416+
public List<ApplyTemplateMessage> messages { get; set; } = [];
417+
}
418+
419+
/// <summary>
420+
/// A single chat message for /apply-template, mirroring the /v1/chat/completions message schema.
421+
/// Null members are dropped so the payload matches what the OpenAI-compatible endpoint expects.
422+
/// </summary>
423+
public class ApplyTemplateMessage
424+
{
425+
[JsonProperty("role")]
426+
public string role { get; set; } = "user";
427+
428+
[JsonProperty("content", NullValueHandling = NullValueHandling.Ignore)]
429+
public string? content { get; set; }
430+
431+
[JsonProperty("tool_calls", NullValueHandling = NullValueHandling.Ignore)]
432+
public List<ApplyTemplateToolCall>? tool_calls { get; set; }
433+
434+
[JsonProperty("tool_call_id", NullValueHandling = NullValueHandling.Ignore)]
435+
public string? tool_call_id { get; set; }
436+
}
437+
438+
public class ApplyTemplateToolCall
439+
{
440+
[JsonProperty("id")]
441+
public string id { get; set; } = string.Empty;
442+
443+
[JsonProperty("type")]
444+
public string type { get; set; } = "function";
445+
446+
[JsonProperty("function")]
447+
public ApplyTemplateFunction function { get; set; } = new();
448+
}
449+
450+
public class ApplyTemplateFunction
451+
{
452+
[JsonProperty("name")]
453+
public string name { get; set; } = string.Empty;
454+
455+
// Per the OpenAI spec, function arguments are a JSON *string*, not a nested object.
456+
[JsonProperty("arguments")]
457+
public string arguments { get; set; } = string.Empty;
458+
}
459+
460+
/// <summary>
461+
/// Response body for POST /apply-template. Contains the fully-formatted prompt string.
462+
/// </summary>
463+
public class ApplyTemplateResponse
464+
{
465+
[JsonProperty("prompt")]
466+
public string prompt { get; set; } = string.Empty;
467+
}
468+
391469
public class MessageQuery(string role, string content)
392470
{
393471
public string role { get; set; } = role;

‎Adapters/LlamaCppAdapter.cs‎

Lines changed: 106 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -242,6 +242,47 @@ public void Dispose()
242242

243243
public int CountMessageTokens(List<SingleMessage> messages)
244244
{
245+
// In chat mode the authoritative count must match what /v1/chat/completions actually builds:
246+
// names, macro expansion and the model's chat-template scaffolding all affect the real token
247+
// count. Counting the raw message text (as the legacy /v1/messages/count_tokens path did)
248+
// systematically under-counts, and the gap grows with the number of messages. To get an exact
249+
// count we render the messages through the model's chat template via /apply-template (using the
250+
// same content generation produces: name prefixes, macros and tool-call structure), then
251+
// tokenize that string with BOS.
252+
if (CompletionType == CompletionType.Chat)
253+
{
254+
try
255+
{
256+
var chatMessages = new List<ApplyTemplateMessage>();
257+
foreach (var message in messages)
258+
{
259+
var built = BuildTemplateMessage(message);
260+
if (built is not null)
261+
chatMessages.Add(built);
262+
}
263+
if (chatMessages.Count == 0)
264+
return 0;
265+
266+
var templated = _client.ApplyTemplateSync(new ApplyTemplateQuery { messages = chatMessages });
267+
if (!string.IsNullOrEmpty(templated?.prompt))
268+
{
269+
var tokens = _client.TokenizeSync(new TokenRequest
270+
{
271+
content = templated.prompt,
272+
add_special = true,
273+
parse_special = true
274+
});
275+
return tokens.GetTokenCount();
276+
}
277+
LLMEngine.Logger?.LogWarning("[LlamaCpp] /apply-template returned an empty prompt; falling back to legacy token count.");
278+
}
279+
catch (Exception ex)
280+
{
281+
// Older servers may not expose /apply-template. Fall back to the legacy path below.
282+
LLMEngine.Logger?.LogWarning(ex, "[LlamaCpp] /apply-template unavailable; falling back to legacy token count.");
283+
}
284+
}
285+
245286
var request = new MessageListQuery();
246287
foreach (var message in messages)
247288
{
@@ -268,6 +309,71 @@ public int CountMessageTokens(List<SingleMessage> messages)
268309
return token.input_tokens;
269310
}
270311

312+
/// <summary>
313+
/// Builds a Newtonsoft-serializable /apply-template message from a <see cref="SingleMessage"/>,
314+
/// mirroring the three cases in <see cref="SingleMessage.ToChatCompletion"/> (tool result,
315+
/// assistant tool-call-only, and normal text). Tool-call arguments are kept as the raw JSON
316+
/// *string* (never a System.Text.Json JsonNode) so serialization cannot self-reference.
317+
/// Returns null for roles that are not sent to the template.
318+
/// </summary>
319+
private static ApplyTemplateMessage? BuildTemplateMessage(SingleMessage message)
320+
{
321+
// Tool result messages
322+
if (message.Role == AuthorRole.Tool && message.ToolCalls.Count > 0)
323+
{
324+
return new ApplyTemplateMessage
325+
{
326+
role = "tool",
327+
content = message.Message,
328+
tool_call_id = message.ToolCalls[0].CallId
329+
};
330+
}
331+
332+
// Assistant tool-call-only messages (no text content)
333+
if (message.Role == AuthorRole.Assistant && message.ToolCalls.Count > 0 && string.IsNullOrEmpty(message.Message))
334+
{
335+
var calls = new List<ApplyTemplateToolCall>();
336+
foreach (var record in message.ToolCalls)
337+
{
338+
calls.Add(new ApplyTemplateToolCall
339+
{
340+
id = record.CallId,
341+
type = "function",
342+
function = new ApplyTemplateFunction
343+
{
344+
name = record.FunctionName,
345+
arguments = record.ArgumentsJson
346+
}
347+
});
348+
}
349+
return new ApplyTemplateMessage
350+
{
351+
role = "assistant",
352+
// OpenAI schema requires content (may be null) alongside tool_calls.
353+
content = string.IsNullOrEmpty(message.Message) ? null : message.Message,
354+
tool_calls = calls
355+
};
356+
}
357+
358+
// Normal messages (System / User / Assistant text)
359+
var role = message.Role switch
360+
{
361+
AuthorRole.User => "user",
362+
AuthorRole.Assistant => "assistant",
363+
AuthorRole.System => "system",
364+
AuthorRole.Tool => "tool",
365+
_ => null
366+
};
367+
if (role is null)
368+
return null;
369+
370+
return new ApplyTemplateMessage
371+
{
372+
role = role,
373+
content = message.ToChatContentText()
374+
};
375+
}
376+
271377
private async Task GenerateChatCompletionStreaming(object parameters)
272378
{
273379
if (parameters is not ChatRequest input)

‎Chatlog/SingleMessage.cs‎

Lines changed: 27 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -74,6 +74,33 @@ internal string ToolCallToString()
7474
return tmsg.ToString();
7575
}
7676

77+
/// <summary>
78+
/// Returns the macro-expanded text content of this message as it would appear in the "content"
79+
/// field of a /v1/chat/completions message (including any "Name: " prefix), without image or
80+
/// tool-call structure. Used for accurate chat-template token counting.
81+
/// </summary>
82+
internal string ToChatContentText()
83+
{
84+
var realprompt = Message;
85+
var addname = LLMEngine.NamesInPromptOverride ?? LLMEngine.Settings.AddNamesToPrompt;
86+
87+
if (Bot is GroupPersonaBase)
88+
addname = true;
89+
90+
if (Role != AuthorRole.Assistant && Role != AuthorRole.User)
91+
addname = false;
92+
93+
if (addname || Bot != LLMEngine.Bot || User != LLMEngine.User)
94+
{
95+
if (Role == AuthorRole.Assistant)
96+
realprompt = string.Format("{0}: {1}", Bot.Name, Message);
97+
else if (Role == AuthorRole.User)
98+
realprompt = string.Format("{0}: {1}", User.Name, Message);
99+
}
100+
101+
return Bot.ReplaceMacros(realprompt, User);
102+
}
103+
77104
internal Message ToChatCompletion()
78105
{
79106
// Tool result messages: skip all name/image logic

‎LLM/LLMEngine.cs‎

Lines changed: 4 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1212,7 +1212,10 @@ public static async Task<object> GenerateFullPrompt(SingleMessage message, strin
12121212

12131213
// get the full, formated chat history complemented by the data inserts
12141214
var addinserts = string.IsNullOrEmpty(Instruct.ThinkingStart) || !Settings.RAGMoveToThinkBlock;
1215-
var historytokens = (int)((Client?.CompletionType == CompletionType.Chat) ? availtokens + 2048 : availtokens);
1215+
var cushion = Client?.CompletionType == CompletionType.Chat ? MaxContextLength / 10 : 0;
1216+
if (cushion > 10240)
1217+
cushion = 10240;
1218+
var historytokens = availtokens + cushion;
12161219
History.AddHistoryToPrompt(Settings.SessionHandling, historytokens, addinserts ? dataInserts : null);
12171220
if (!string.IsNullOrEmpty(message.Message) || message.Role != AuthorRole.User)
12181221
{

‎PromptBuilders/ChatPromptBuilder.cs‎

Lines changed: 75 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -183,15 +183,7 @@ public object PromptToQuery(AuthorRole responserole = AuthorRole.Assistant, doub
183183

184184
var total = GetTokenUsage(workingprompt);
185185
var max = LLMEngine.MaxContextLength - (responseoverride == -1 ? LLMEngine.Settings.MaxReplyLength : responseoverride) - 15;
186-
while (total > max && workingprompt.Count > 1)
187-
{
188-
if (total > max + 2048)
189-
{
190-
workingprompt.RemoveAt(1);
191-
}
192-
workingprompt.RemoveAt(1);
193-
total = GetTokenUsage(workingprompt);
194-
}
186+
TrimToFit(workingprompt, ref total, max);
195187

196188
var finalprompt = new List<Message>(workingprompt.ConvertAll(m => m.ToChatCompletion()));
197189
var cleanimages = !LLMEngine.SupportsVision || LLMEngine.Settings.MaxImageCount > 0;
@@ -304,6 +296,80 @@ public object PromptToQuery(AuthorRole responserole = AuthorRole.Assistant, doub
304296

305297
}
306298

299+
/// <summary>
300+
/// Trims the oldest messages (preserving index 0, the system prompt) until the accurate token
301+
/// usage of <paramref name="workingprompt"/> fits within <paramref name="max"/>.
302+
/// </summary>
303+
/// <remarks>
304+
/// The naive approach re-counts the full prompt after every single removal, which is extremely
305+
/// slow on backends whose token-counting is an HTTP round-trip (e.g. llama.cpp). Instead we use a
306+
/// cheap local per-message estimate to decide how many messages to drop in one batch, then perform
307+
/// a single accurate re-count to verify. The batch is intentionally conservative (under-drops) so
308+
/// that we never trim more context than necessary; if the estimate was too optimistic we simply
309+
/// loop again, which converges in one or two accurate counts instead of N.
310+
/// </remarks>
311+
private void TrimToFit(List<SingleMessage> workingprompt, ref int total, int max)
312+
{
313+
// Fast path: nothing to do.
314+
if (total <= max || workingprompt.Count <= 1)
315+
return;
316+
317+
// Guard against pathological loops (e.g. a single message that is itself larger than max).
318+
var safety = 0;
319+
while (total > max && workingprompt.Count > 1)
320+
{
321+
var excess = total - max;
322+
323+
// Walk from the oldest removable message (index 1) forward, accumulating cheap local
324+
// estimates until we've covered the excess. We deliberately stop as soon as the estimate
325+
// meets the excess (conservative / under-drop) rather than padding it, because any
326+
// shortfall is caught by the accurate re-count below.
327+
var removeCount = 0;
328+
var estRemoved = 0;
329+
for (int i = 1; i < workingprompt.Count && estRemoved < excess; i++)
330+
{
331+
estRemoved += EstimateLocalTokens(workingprompt[i]);
332+
removeCount++;
333+
}
334+
335+
// Always remove at least one message so we make progress even if the local estimate is 0.
336+
if (removeCount == 0)
337+
removeCount = 1;
338+
339+
// Never remove the final (newest) message; keep at least the system prompt + one message.
340+
var maxRemovable = workingprompt.Count - 2;
341+
if (maxRemovable < 1)
342+
maxRemovable = 1;
343+
if (removeCount > maxRemovable)
344+
removeCount = maxRemovable;
345+
346+
workingprompt.RemoveRange(1, removeCount);
347+
348+
// One accurate count to verify the batch. If we under-dropped, the outer loop runs again.
349+
total = GetTokenUsage(workingprompt);
350+
351+
if (++safety > workingprompt.Count + 4)
352+
break;
353+
}
354+
}
355+
356+
/// <summary>
357+
/// Cheap, local (no backend call) per-message token estimate used only to decide how many
358+
/// messages to drop in a single trim batch. This is intentionally an over-estimate per message so
359+
/// the batch tends to be conservative; the authoritative count is always the full-prompt count.
360+
/// </summary>
361+
private static int EstimateLocalTokens(SingleMessage message)
362+
{
363+
var text = message.ToTextCompletion();
364+
var total = TokenTools.CountTokens(text);
365+
366+
if (message.Role == AuthorRole.Assistant && message.ToolCalls.Count > 0)
367+
total += TokenTools.CountTokens(message.ToolCallToString());
368+
369+
// Structural per-message overhead (role headers/delimiters) the plain text count misses.
370+
return total + 8;
371+
}
372+
307373
public void Clear()
308374
{
309375
LastQuery = null;

0 commit comments

Comments
 (0)