Report what a reply spent reading its prompt, and pin the clock right

`UsageDelta` gains `prefillMs`, llama-server's own `timings.prompt_ms`, so the
footer under a finished reply is "read 9.5s · 50.3 tok/s · 3:00 PM". Prefill is
the half of a turn that was invisible and is often the larger: measured on the
0.6B here, 1m 4s for the first turn after a model loads against 22ms for the
next, whose prompt the server still had cached.

The clock moves to the end of the line. Everything in front of it is a
provider's own measurement, so a session on another provider has fewer of them
or none, and a reader who has learned where the time is should not have to find
it again because the model changed. The costs grow leftwards into the space
instead, and a test asserts every shape of the line ends with the same thing.

Verified on the emulator against a real llama session: three replies reading
"read 1m 4s · 193 tok/s · 3:54 PM", "read 25ms · 308 tok/s · 3:54 PM" and
"read 22ms · 194 tok/s · 3:54 PM", with the clock in one column.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
iris-aiandClaude Opus 5 committed 2026-09-19 15:57:56 -04:00
1 parent b660905098
commit 369b8f7e52
14 files changed
+135 -42

No files matched your search

@@ -14,6 +14,7 @@ import kotlin.test.assertTrue
* generation speed has no figure -- neither may borrow one.
*/
class ThinkingTest {
private val utc = ZoneId.of("UTC")
private var seq = 0L
private fun fold(items: List<TranscriptItem>, event: SessionEvent, ts: Double = 1.0) =
@@ -77,29 +78,41 @@ class ThinkingTest {
}
@Test
fun `a reply carries when it was sent and what it was generated at`() {
fun `a reply carries when it was sent and what it cost to produce`() {
val items =
fold(emptyList(), SessionEvent.AssistantText("Done."), ts = 1_788_609_600.0).let {
fold(it, SessionEvent.UsageDelta(42, 100, 18.37))
fold(it, SessionEvent.UsageDelta(42, 100, 18.37, 9_489))
}
val reply = items.filterIsInstance<TranscriptItem.AssistantMsg>().single()
assertEquals(1_788_609_600.0, reply.ts)
assertEquals(18.37, reply.tokensPerSecond)
assertEquals(9_489, reply.prefillMs)
val footer = replyFooterText(reply.ts, reply.tokensPerSecond, ZoneId.of("UTC"))
val footer = replyFooterText(reply.ts, reply.tokensPerSecond, reply.prefillMs, utc)
// The clock reading rather than the whole string: the platform's own short-time format
// differs by JDK and locale, which is the point of asking it for one.
assertTrue(footer!!.contains("12:00"), footer)
assertTrue(footer.endsWith("18.4 tok/s"), footer)
assertTrue(footer!!.startsWith("read 9.5s · 18.4 tok/s · "), footer)
assertTrue(footer.contains("12:00"), footer)
}
@Test
fun `a provider that measures no speed gets a footer of the time alone`() {
val footer = replyFooterText(1_788_609_600.0, null, ZoneId.of("UTC"))
assertTrue(footer!!.contains("12:00"), footer)
assertTrue(!footer.contains("tok/s"), footer)
// And a reply with neither has no line at all rather than an empty one.
assertNull(replyFooterText(0.0, null, ZoneId.of("UTC")))
fun `the clock stays at the end however much the provider measured`() {
// What a provider that measures nothing leaves: the time, and nothing in front of it.
val bare = replyFooterText(1_788_609_600.0, null, null, utc)
assertTrue(bare!!.contains("12:00"), bare)
assertTrue(!bare.contains("tok/s") && !bare.contains("read"), bare)
// Every shape ends with the same thing, which is the whole point of the order: the clock
// does not move because the session is on a provider that measures more or less.
val shapes =
listOf(
bare,
replyFooterText(1_788_609_600.0, 18.37, null, utc)!!,
replyFooterText(1_788_609_600.0, null, 9_489, utc)!!,
replyFooterText(1_788_609_600.0, 18.37, 9_489, utc)!!,
)
assertEquals(1, shapes.map { it.substringAfterLast("· ") }.distinct().size, "$shapes")
// A reply with nothing to say has no line at all rather than an empty one.
assertNull(replyFooterText(0.0, null, null, utc))
}
@Test
@@ -130,7 +143,7 @@ class ThinkingTest {
SessionEvent.AssistantText("Reading it."),
SessionEvent.ToolStart("t1", "Read", "{}"),
SessionEvent.ToolEnd("t1", "done"),
SessionEvent.UsageDelta(42, 100, 18.0),
SessionEvent.UsageDelta(42, 100, 18.0, 500),
)
assertNull(items.filterIsInstance<TranscriptItem.AssistantMsg>().single().tokensPerSecond)
}