Compare commits
91
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
03c6be80a3 | ||
|
|
4afc453faa | ||
|
|
1aab61bf26 | ||
|
|
dc01f88d75 | ||
|
|
c589a75fa0 | ||
|
|
4b62cc642e | ||
|
|
80c2eadec9 | ||
|
|
0b587629e6 | ||
|
|
3163256d2c | ||
|
|
6102e0d4d9 | ||
|
|
f0da383e28 | ||
|
|
2d3695a1d3 | ||
|
|
f06ee259b4 | ||
|
|
560a74caf8 | ||
|
|
fd7e17523d | ||
|
|
c7682297fa | ||
|
|
27511302f2 | ||
|
|
184a6c5b33 | ||
|
|
5b2ca039f1 | ||
|
|
a8d24553d5 | ||
|
|
3b80a88f3b | ||
|
|
d8e6bc6e9b | ||
|
|
b887a96765 | ||
|
|
2aaa3733c3 | ||
|
|
46246ea511 | ||
|
|
a27fbdb029 | ||
|
|
c07d544aeb | ||
|
|
46d3a6fd41 | ||
|
|
5655fa8093 | ||
|
|
b3b1d47dd6 | ||
|
|
50fe4828a2 | ||
|
|
800da46188 | ||
|
|
00767eed4d | ||
|
|
683db4908a | ||
|
|
8d23a20792 | ||
|
|
d01c105037 | ||
|
|
88631f5e8b | ||
|
|
e6924298bc | ||
|
|
68b48cfd14 | ||
|
|
e6c884a0cd | ||
|
|
a6cb9a9082 | ||
|
|
0be6a571c4 | ||
|
|
bfe93c4188 | ||
|
|
5b7dc0e4e2 | ||
|
|
621f08d725 | ||
|
|
e49d0e606f | ||
|
|
e2a1fadbec | ||
|
|
0e4629361b | ||
|
|
1e7b1cddb7 | ||
|
|
470f8e5019 | ||
|
|
cf10b17c5b | ||
|
|
7ae53ad797 | ||
|
|
d17040b601 | ||
|
|
bf5087a598 | ||
|
|
9fa09b0af1 | ||
|
|
aa3d11471f | ||
|
|
78aff64844 | ||
|
|
9a33cb5384 | ||
|
|
45ced405f3 | ||
|
|
62199aa3a7 | ||
|
|
b133d85943 | ||
|
|
ba6817fee5 | ||
|
|
73ee63bc1b | ||
|
|
8f0aec449a | ||
|
|
6d5fd64bb0 | ||
|
|
e5880c33f4 | ||
|
|
a853eb5a4d | ||
|
|
eff5c8b0c0 | ||
|
|
7b63330aaa | ||
|
|
22d5c6585a | ||
|
|
b063fbd7f9 | ||
|
|
3f25e7ebca | ||
|
|
0af4c88d08 | ||
|
|
ceabd00805 | ||
|
|
32a5256a0d | ||
|
|
4cfe0ef6e6 | ||
|
|
c9b273ff16 | ||
|
|
8adda94a7a | ||
|
|
3a9208f38b | ||
|
|
e898370bf4 | ||
|
|
6e0bd06e4d | ||
|
|
03da47e550 | ||
|
|
a2cd119985 | ||
|
|
19c36e37f2 | ||
|
|
6bdec6e785 | ||
|
|
e2873df92e | ||
|
|
4821a02bd3 | ||
|
|
1ff662c7c3 | ||
|
|
e4f0935f98 | ||
|
|
3c0214ece8 | ||
|
|
127b25e60a |
No files matched your search
@@ -0,0 +1,6 @@
|
||||
# xtask convention (https://github.com/matklad/cargo-xtask), without folding
|
||||
# every crate in this repo into one workspace -- they are deliberately
|
||||
# independent (see run-tests.sh, which cds into each). `cargo xtask apk`
|
||||
# from the repo root runs xtask/src/main.rs directly.
|
||||
[alias]
|
||||
xtask = "run --quiet --manifest-path xtask/Cargo.toml --"
|
||||
@@ -0,0 +1,244 @@
|
||||
---
|
||||
name: ai-app-rigs
|
||||
description: ai-app's test rigs, harness scripts and reference measurements - ui-sandbox.sh, debug-transcript.sh, transcript-bench.sh, stream-bench.sh, trace-draw.sh, the /usage fixture vocabulary, the fake CLI, the rule that no UI-driving script may tap a coordinate, how to test llama.cpp and ssh on this machine, how importing behaves, and the scroll/stream/explorer numbers not worth re-measuring. Read before running or writing a benchmark, driving the app's UI from a script, exercising the session lifecycle, testing a llama or remote session, or touching the import screen.
|
||||
---
|
||||
|
||||
# ai-app: rigs, harnesses and measurements
|
||||
|
||||
Moved out of `AGENTS.md` on 2026-09-04 so it is read when it is relevant
|
||||
rather than sent with every request in this repo -- it was 12 KB of the 35 KB
|
||||
that file cost on every one. Unchanged in the move, and still the only copy.
|
||||
|
||||
## The rigs
|
||||
|
||||
Each exists because something was invisible without it.
|
||||
|
||||
- **`app/ui-sandbox.sh`** — a second `ai-server` with its own `$HOME`, config
|
||||
and data directory, holding eight invented Claude Code transcripts and a
|
||||
`claude` that is two lines of shell. **That isolation is the point**: the
|
||||
import screen lists whatever is in `~/.claude/projects`, which in this VM is
|
||||
real agent transcripts, so exercising *delete* against the ordinary server
|
||||
deletes somebody's conversation and exercising *import* starts a real
|
||||
`--resume` on the owner's account.
|
||||
Its port and root derive from the checkout's name, so two checkouts'
|
||||
sandboxes cannot reach each other, and its token is generated once into
|
||||
`~/.config/ai-app/sandbox-token` and carried across restarts along with any
|
||||
the enrolment flow appended — so the emulator app is enrolled **once** (the
|
||||
start banner prints the command) and stays enrolled. It shares the real TLS
|
||||
certificates, because the installed APK pins that CA.
|
||||
Driving verbs, so none of this is re-derived per session:
|
||||
`./ui-sandbox.sh spawn [title]` (an echo session, prints its id),
|
||||
`./ui-sandbox.sh send SID text|@file`, and
|
||||
`./ui-sandbox.sh api /path [curl args]`.
|
||||
`./ui-sandbox.sh keep` restarts the server without wiping the sessions and
|
||||
enrolment already there — for when the fixture under test was expensive to
|
||||
build; plain `start` wipes them, which is right for the list-screen
|
||||
fixtures and wrong for that.
|
||||
It passes `--delay` by default, and `AI_SANDBOX_BIG_MB` puts one large
|
||||
transcript among the small ones while `AI_SANDBOX_SPAWN_DELAY` makes the
|
||||
fake CLI slow to start. Both exist because operations that finish in
|
||||
milliseconds have states on the way that nothing can observe, and an
|
||||
unobservable state is one where broken and working look identical.
|
||||
It also builds a fixture tree at the sandbox home's `~/files` for the
|
||||
explorer, holding the states otherwise only reachable by finding a real
|
||||
machine in one: an empty directory, a name with a tab and one with an
|
||||
apostrophe, a binary file, one over `FILE_LIMIT`, one `chmod 000`, a
|
||||
symlink to a directory and a broken one, a source file per language, and
|
||||
the three sizes the limits were measured against (`edit-32k.rs`,
|
||||
`edit-128k.rs`, `big-source.rs`). Point a session at it with
|
||||
`./ui-sandbox.sh api /sessions/<id>/cwd -X POST -H 'content-type: application/json' -d '{"cwd":"~/files"}'`.
|
||||
The explorer's 409 is produced by editing the file on the machine
|
||||
(`printf … > file`) between pressing the pencil and pressing save.
|
||||
- **`app/debug-transcript.sh`** — a real conversation on the emulator. The
|
||||
echo driver is the right rig for most things and the wrong one for anything
|
||||
whose cost scales with what was actually written: a real reply is longer,
|
||||
is real markdown, and carries tool calls whose input and output are
|
||||
kilobytes. Two faults were invisible until a real transcript was loaded — a
|
||||
page of history landing mid-fling threw the reader back to the newest end,
|
||||
and parsing one real reply took 51ms against 4.6ms for a synthetic one.
|
||||
`-b` takes the biggest conversation on the machine rather than the newest,
|
||||
which is what a scrolling test wants; `--stop` takes it down.
|
||||
It copies the transcript into `/tmp` and gives the server a `HOME` of its
|
||||
own, so the import can only see the copy — importing spawns `claude
|
||||
--resume`, and against the real file that is a second CLI writing to a
|
||||
conversation somebody may still be in. **A transcript never goes in this
|
||||
repository**: they hold whatever was said, read and written in that
|
||||
session, and `~/repos` is shared with the host besides.
|
||||
- **`/usage` in an echo session puts up an invented meter**, which is how the
|
||||
rate-limit screens' states are reached without spending quota: `/usage 42`,
|
||||
`/usage 95 20` (minutes left), `/usage 42 never` (the between-blocks window
|
||||
with no reset time), `/usage 42 unreadable`, `/usage notloggedin`,
|
||||
`/usage unreachable`, `/usage failed`, `/usage off`. The vocabulary is
|
||||
`usage::Fixture`'s, since those are its states. With none set an echo
|
||||
session meters nothing, which is the ordinary case and draws no bar.
|
||||
- **A fake CLI exercises the process lifecycle without a token.** Point a
|
||||
`claude_cli` provider's `command` at a two-line script — `#!/bin/sh` and
|
||||
`cat > /dev/null` — and it behaves the way the lifecycle code cares about:
|
||||
it holds the fifo open, records a real pid, writes nothing, and dies on a
|
||||
signal. So adopt, stop, restart and start are all drivable without a real
|
||||
`--resume` and without spending a turn on somebody's account. Reach for
|
||||
this when what is under test is *whether a process is running*, and for
|
||||
`debug-transcript.sh` when it is *what the transcript draws*.
|
||||
- **`app/transcript-bench.sh`** is the standard scroll measurement: it opens
|
||||
the first session (or `-k` keeps the current screen), scrolls a fixed
|
||||
gesture loop, and prints the app's render report — the same one the in-app
|
||||
copy button produces, whose `on screen:` line names what the viewport was
|
||||
holding. Compare two runs with the same gestures; the emulator's absolute
|
||||
frame times transfer nothing, the report's accounting does. Run it either
|
||||
side of any change under `Markdown*.kt`, `Transcript*.kt` or
|
||||
`SessionScreen.kt`'s list, and put the report in the commit. The numbers
|
||||
that move first are the worst `record: one block`, the reparse mean while
|
||||
streaming, and the draw phase's accounting line.
|
||||
- **`app/stream-bench.sh [-k] FILE`** is that measurement for a reply still
|
||||
arriving. It taps "Jump to latest" so the list is pinned to the newest end,
|
||||
resets the report, sends FILE, waits for the transcript to stop growing,
|
||||
and prints. Both of those are corrections to a first version that measured
|
||||
nothing: a transcript parked further back never redraws while a reply
|
||||
streams into it, and a session is idle at *both* ends of a turn, so polling
|
||||
for idle answers before the turn has started.
|
||||
- **`app/trace-draw.sh`** names what a scrolling frame spends inside the
|
||||
framework, from `atrace` text output with no trace processor needed. It is
|
||||
how the cost of a layout node per link was attributed to the framework
|
||||
rather than guessed at.
|
||||
|
||||
### Driving the UI
|
||||
|
||||
**No script that drives this app's UI presses a coordinate.** Every control
|
||||
is found by the name it already carries for assistive technology —
|
||||
`ui-trace record --do "tap 'Session settings'"` — which resolves the label
|
||||
against the screen at the moment of the gesture and fails the whole run when
|
||||
it is not there. `app/bench-lib.sh` is what the bench scripts share for it. A
|
||||
coordinate is a position measured once by hand, and anything that moves the
|
||||
control makes the tap land on whatever now sits there — the bench then
|
||||
reports a number that was never measured, which reads exactly like a result.
|
||||
Both bench scripts pressed the render report at `tap 723 205` until that
|
||||
button moved into the session settings dialog on 2026-09-03. The check that
|
||||
none has crept back:
|
||||
|
||||
grep -n "tap [0-9]" app/*.sh
|
||||
|
||||
Swipes are still coordinates, deliberately: a gesture across a scrolling area
|
||||
is a distance rather than a control.
|
||||
|
||||
**Two traps in the emulator bench loop**, each of which cost a run.
|
||||
`adb shell pm clear` removes the enrolment and the notification permission
|
||||
along with the saved anchors, so the next run measures a permission dialog —
|
||||
re-enrol with the command `ui-sandbox.sh` prints, and
|
||||
`pm grant … POST_NOTIFICATIONS`. And a saved scroll anchor is per session id,
|
||||
so the only way two builds start a scroll from the same place is a *fresh
|
||||
session for each*.
|
||||
|
||||
**The emulator is `~/repos/emulator-tools`' business, not this repo's.**
|
||||
`emu up` creates and boots the AVD named after this checkout — whatever `emu
|
||||
name` prints, never a name typed out here, since this file is the same in
|
||||
every clone. `run-android.sh` is that plus a build and an install. The `adb`
|
||||
on `PATH` after sourcing `android-env.sh` is that repo's wrapper, which fills
|
||||
in `-s` from the same rule. Gradle does not go through it, so a Gradle init
|
||||
script from `emulator-tools` runs `emu check` before `installDebug`,
|
||||
`uninstallDebug` and `connectedAndroidTest` and fails rather than fanning out
|
||||
to every attached device; when it refuses, say which device you mean at the
|
||||
moment you use it — `ANDROID_SERIAL=$(emu serial) ./gradlew …`.
|
||||
|
||||
### Testing llama.cpp and ssh here
|
||||
|
||||
**Both are set up here as of 2026-09-04** and need nothing typed. The
|
||||
prebuilt CPU llama.cpp lives outside the repo at `~/.local/opt/llama.cpp`
|
||||
(the 15 MB `ubuntu-x64` release asset) and is symlinked as
|
||||
`/usr/local/bin/llama-server`, which is what makes **discovery find it over
|
||||
ssh**: `~/.local/bin` is not on the PATH a non-interactive ssh session gets.
|
||||
It resolves its own libraries through `$ORIGIN`, so no `LD_LIBRARY_PATH` is
|
||||
needed. One model is downloaded — `unsloth/Qwen3-0.6B-GGUF/Qwen3-0.6B-Q8_0.gguf`,
|
||||
639 MB under `~/.local/share/ai-app/models` — and answers at usable speed on
|
||||
this VM's 8 cores. **Do not test with a 2-bit quant**: the
|
||||
IQ2_XXS of that model produces fluent nonsense, which reads exactly like a
|
||||
broken driver — `llama-cli` produces the same from the file directly, which
|
||||
is how to tell the two apart in a hurry.
|
||||
|
||||
There is no second machine, so **ssh this VM to itself**. That is set up
|
||||
too: the key is `~/.config/ai-app/ssh-self` (its public half is in
|
||||
`~/.ssh/authorized_keys`, labelled removable), and the real config carries a
|
||||
setup called **"this vm over ssh"** — `bob@127.0.0.1` with that
|
||||
`identityFile` plus
|
||||
`options: ["StrictHostKeyChecking=no", "UserKnownHostsFile=/tmp/ai-app-known-hosts"]`
|
||||
so it touches nothing real — offering `claude-cli` and `llama-cpp`. It is the
|
||||
whole rig for "does a remote llama session work", since the far machine is
|
||||
this one and the model file is the same file. For a throwaway setup of your
|
||||
own, point a provider's `command` at something harmless like `/bin/echo`
|
||||
rather than at `claude`: the transport is what is under test, the process
|
||||
exiting immediately is the signal, and it costs no tokens. The remote login
|
||||
shell here is **fish**; the
|
||||
remote script and `ssh.rs`'s POSIX quoting happen to mean the same thing in
|
||||
both, but that is luck rather than design, and a shell that is neither is the
|
||||
thing to suspect first if a remote spawn ever mangles an argument.
|
||||
|
||||
## Importing
|
||||
|
||||
The import list reports each session's **size as well as its line count**,
|
||||
because the two disagree in the way that matters: these transcripts embed
|
||||
screenshots as base64, so one line can be a megabyte. On this machine a 69 MB
|
||||
session has 3,427 lines and a 44 MB one has 6,792 — nothing about a line
|
||||
count tells you what continuing a session will cost. Shown, not warned about;
|
||||
importing a large session is a choice somebody is entitled to make.
|
||||
|
||||
**Never import a Claude Code session that is open in a terminal.** The app
|
||||
refuses it — see PLAN.md for the incident that made that a refusal rather
|
||||
than a warning.
|
||||
|
||||
**One Claude Code session id can name two files, and the listing offers it
|
||||
once.** Resuming from a different working directory makes the CLI write a
|
||||
second transcript with the same id under that directory's project folder — an
|
||||
ordinary state of a machine, not corruption. Everything downstream addresses
|
||||
a session by id, and the phone keyed its list on it, so two rows sharing one
|
||||
**closed the app** on a Compose duplicate-key throw. `parse_listing` keeps
|
||||
the copy with the most lines, because the other is usually a few-hundred-byte
|
||||
stub and is often the *newer* of the two, so recency is the wrong key.
|
||||
Deleting removes every copy rather than the first, or the row came back after
|
||||
a delete that reported success. The phone's half is `uniqueItems`, which
|
||||
every list keyed on a server-chosen id goes through: a repeat there must
|
||||
never be able to close the app, whatever produced it.
|
||||
|
||||
**Deleting a session offers to take the machine's own transcript with it** —
|
||||
`DELETE /sessions/{id}?deleteForeign=true`, behind a switch in the
|
||||
confirmation, and only where the driver keeps a record of its own
|
||||
(`keepsOwnTranscript`, which today means Claude Code). Off by default,
|
||||
because leaving that copy is what makes an ordinary delete recoverable — and
|
||||
the dialog's paragraph is rewritten when it is on rather than appended to,
|
||||
since the sentence promising the conversation "should still be there to
|
||||
import again" is exactly the one the switch makes false. The server deletes
|
||||
the machine's copy *first*, so a machine it cannot reach leaves the session
|
||||
where it was instead of half-deleted.
|
||||
|
||||
## Measurements worth not re-taking
|
||||
|
||||
- **What the transcript screen costs to scroll.** Taken 2026-08-30 on the GPU
|
||||
emulator against a real imported transcript with the server at
|
||||
`--delay 120`. Settled and flinging fast, both into fresh history and back
|
||||
through rows already drawn: **5.2–5.9% janky frames, 99th percentile
|
||||
29–32ms, 0–2 slow UI-thread frames.** The stock Settings app on the same
|
||||
device is 3.3% and 38ms, so this is at the platform floor. The number that
|
||||
is *not* at the floor is the first few seconds after opening a session,
|
||||
where every row on the way is being composed for the first time; that is
|
||||
inherent to a lazy list and it is why a measurement taken before the screen
|
||||
settles reads three times worse. **Settle first, then reset `gfxinfo`.**
|
||||
- **The reset path is not reachable by reopening a session.** Measured
|
||||
2026-09-04 against a session streaming at 20 events a second: reopening one
|
||||
with an anchor 1,800 events back connects **87–119 events behind**, well
|
||||
under `CATCH_UP_LIMIT`'s 200, because the restore is two requests — the
|
||||
opening page, then one span covering the whole distance. To exercise the
|
||||
reset at all you have to lower `CATCH_UP_LIMIT` in a throwaway build; at 5
|
||||
the app takes the reset on a live connection, clears, refills and carries
|
||||
on without reconnecting.
|
||||
- **The session screen's stream survives backgrounding here** — 20 seconds at
|
||||
the launcher while 415 events were produced brought no reconnect at all,
|
||||
which is not what the comment above that loop expects, and is most likely
|
||||
this emulator being headless rather than the phone's behaviour.
|
||||
- **Reopening a cached session costs one request for one event** (the probe),
|
||||
and scrolling the whole conversation back costs nothing more; a cold open
|
||||
of the same 500-event session is two pages, 100 events. Measured
|
||||
2026-09-04 on the emulator against the sandbox.
|
||||
- **Reading is cheap and editing is not.** The viewer handles a 1 MiB,
|
||||
28,000-line file because it draws one row per line; the editor is one
|
||||
`BasicTextField`, which costs two seconds a frame at 128 kB and stops the
|
||||
app at 1 MiB, so `EDIT_LIMIT` caps it at 32 kB with the reason said on
|
||||
screen. If you make the editor faster, that number is what to move.
|
||||
EXPLORER.md's "What the measurements said" has the rest.
|
||||
@@ -55,4 +55,21 @@ components: [
|
||||
// the terminal the QR would be printed on.
|
||||
enroll: "server/enroll-link.sh",
|
||||
),
|
||||
// E5 (RUST.md): app/shellApp packaged by the xtask instead of Gradle
|
||||
// (cargo ndk -> javac -> d8 -> aapt2 -> zipalign -> apksigner), signed
|
||||
// with the same release key as "app" above so the two can install
|
||||
// over each other -- a separate component, not a mode of "app" above,
|
||||
// because it is a different applicationId (com.example.aiapp.shell)
|
||||
// built by a different tool from different sources. No `cwd`: it
|
||||
// defaults to this checkout's root, which both the `cargo xtask`
|
||||
// alias (`.cargo/config.toml`, resolved relative to the working
|
||||
// directory cargo is run from) and `cargo xtask apk`'s own publishing
|
||||
// step (`xtask/build/outputs/apk/<mode>/*.apk`, matching discover.rs's
|
||||
// `*/build/outputs/apk/*/*.apk` pattern -- see apk.rs's module doc)
|
||||
// both need.
|
||||
Apk(
|
||||
name: "shell",
|
||||
modes: ["release", "debug"],
|
||||
build: "cargo xtask apk",
|
||||
),
|
||||
],
|
||||
+13
@@ -1,6 +1,7 @@
|
||||
.gradle/
|
||||
build/
|
||||
app/androidApp/build/
|
||||
app/shellApp/build/
|
||||
local.properties
|
||||
.kotlin/
|
||||
*.iml
|
||||
@@ -9,6 +10,11 @@ local.properties
|
||||
server/target/
|
||||
event-model/target/
|
||||
client-core/target/
|
||||
android-shell/target/
|
||||
|
||||
# E3's native library, built by cargo-ndk straight into the Gradle module
|
||||
# (RUST.md) -- an artifact, like server/target/ above, not source.
|
||||
app/shellApp/src/main/jniLibs/
|
||||
|
||||
# Server logs from a development run (ai-server.log by convention,
|
||||
# wg-test.log from ./test-wg-tunnel.sh).
|
||||
@@ -27,3 +33,10 @@ sessions/
|
||||
# iris, the in-house UI library, is vendored at iris/ and built by cargo.
|
||||
iris/target/
|
||||
iris/android-app/target/
|
||||
|
||||
# E5's packaging xtask (RUST.md). `build/` above already covers
|
||||
# xtask/build/outputs/apk (the published APK, see apk.rs's module doc).
|
||||
# The repo root has no Cargo workspace, so this is xtask's own
|
||||
# intermediate working files (target/xtask/apk/...), not a shared one.
|
||||
xtask/target/
|
||||
/target/
|
||||
@@ -5,11 +5,14 @@ replacing the Claude app for daily use. Rust/Axum backend on the desktop,
|
||||
Kotlin/Compose Android app, WireGuard + pinned self-signed TLS + bearer token
|
||||
between them.
|
||||
|
||||
**`PLAN.md` is the design source of truth** — every decision with its date,
|
||||
its rationale, and what was rejected. Read it before changing anything
|
||||
**`docs/PLAN.md` is the design source of truth** — every decision with its
|
||||
date, its rationale, and what was rejected. Read it before changing anything
|
||||
structural, and update it in place when a decision changes rather than
|
||||
letting this file and the plan become two versions of the truth. This file is
|
||||
the working notes layer: layout, commands, rigs, and things that have bitten.
|
||||
The design and working documents live under `docs/` — everything except this
|
||||
file and `CLAUDE.md`, which stay at the root because that is where Claude
|
||||
Code and other agent harnesses look for them.
|
||||
|
||||
The central design point, worth not undoing by accident: **a session is a
|
||||
child process, translated into one common event model.** A new session type
|
||||
@@ -22,7 +25,7 @@ Mirrors `../dev-updater` deliberately: same stack (axum 0.8 +
|
||||
axum-server/rustls, tokio, clap; Kotlin 2.4.x + Compose Multiplatform, single
|
||||
`:androidApp` module), same cert scheme, same registry pattern. Read
|
||||
dev-updater's `README.md` and `AGENTS.md` before diverging from them.
|
||||
Module-by-module intent is in PLAN.md's "Backend layout".
|
||||
Module-by-module intent is in `docs/PLAN.md`'s "Backend layout".
|
||||
|
||||
- `server/` — the Rust backend (`ai-server`). `routes.rs`'s module doc
|
||||
comment is the HTTP table and the surface's source of truth.
|
||||
@@ -40,16 +43,23 @@ Module-by-module intent is in PLAN.md's "Backend layout".
|
||||
projects version-locked to the commit this repo pins. What deliberately did
|
||||
**not** move is the API surface and the config *schema*: routes, drivers,
|
||||
sessions and setups are what makes this project itself.
|
||||
- `EXPLORER.md` — the file explorer's design (`server/src/files.rs` and
|
||||
`FilesScreen.kt` / `FileViewer.kt` / `FileEditor.kt`).
|
||||
- `TRANSCRIPT_CACHE.md` — the phone's copy of what it has been sent. Read it
|
||||
before touching `TranscriptCache.kt`, `TranscriptSource.kt`, or the opening
|
||||
and stream effects in `SessionScreen.kt`.
|
||||
- `TODO.md` — the working list.
|
||||
- `RUST.md` — the plan for moving the app to Rust (on the `rustify`
|
||||
branch of the `ai-app-2` clone): what has to be reproduced, the
|
||||
framework decision, and the ordered experiments with their pass
|
||||
conditions. Read it before touching anything under that branch.
|
||||
- `docs/` — every design and working document except this file and
|
||||
`CLAUDE.md`:
|
||||
- `docs/EXPLORER.md` — the file explorer's design (`server/src/files.rs`
|
||||
and `FilesScreen.kt` / `FileViewer.kt` / `FileEditor.kt`).
|
||||
- `docs/TRANSCRIPT_CACHE.md` — the phone's copy of what it has been sent.
|
||||
Read it before touching `TranscriptCache.kt`, `TranscriptSource.kt`, or
|
||||
the opening and stream effects in `SessionScreen.kt`.
|
||||
- `docs/TODO.md` — the working list.
|
||||
- `docs/RUST.md` — the plan for moving the app to Rust (on the `rustify`
|
||||
branch of the `ai-app-2` clone): what has to be reproduced, the
|
||||
framework decision, and the ordered experiments with their pass
|
||||
conditions. Read it before touching anything under that branch.
|
||||
- `docs/IRIS.md`, `docs/IRIS_TODO.md`, `docs/DECISIONS.md`,
|
||||
`docs/LAYOUT.md`, `docs/TEXTURES.md`, `docs/CLIENT_CORE.md` — iris's
|
||||
own public API log, working list, decisions log, layout/render design,
|
||||
and texture-atlas design, and the client-core crate's design,
|
||||
respectively.
|
||||
- `.dev-updater.ron` — what Dev Updater builds here: the server (run as
|
||||
`service: Managed(…)`, supervised by Dev Updater's own implementation
|
||||
rather than a script kept here) and the APK, in parallel. It points at
|
||||
@@ -86,7 +96,11 @@ two icon buttons the same width without either being given one — and why
|
||||
:androidApp:compileDebugKotlin :androidApp:lintDebug
|
||||
:androidApp:testDebugUnitTest`. The unit tests are JVM-only and cover the
|
||||
syntax highlighter, the ANSI parser and the transcript cache — the app's
|
||||
pure logic with no Android in it.
|
||||
pure logic with no Android in it. Touching anything under `BenchFixture.kt`,
|
||||
`BenchNetwork.kt`, `BenchRun.kt` or the `bench` build type also needs
|
||||
`:androidApp:compileBenchKotlin :androidApp:lintBench` — a second build
|
||||
type compiles separately and lint has caught real bugs debug alone never
|
||||
would (see "Android Lint" below).
|
||||
- **Android Lint is not optional and is not run by a build.** It found a
|
||||
crash that had been shipping (`java.time` on a minSdk-24 app with
|
||||
desugaring off) and later a permission check that silently dropped every
|
||||
@@ -146,6 +160,27 @@ two icon buttons the same width without either being given one — and why
|
||||
|
||||
Each exists because something was invisible without it.
|
||||
|
||||
- **The `bench` build type and `app/bench-fixture/`** exist for P0 (RUST.md
|
||||
and DECISIONS.md's 2026-09-05 entries), the phone benchmark gate Iris
|
||||
asked for before porting continues: a deterministic, checked-in synthetic
|
||||
transcript (`app/bench-fixture/generate.py`, never a real one) that both
|
||||
this app and iris open with no server, so a frame-time comparison
|
||||
measures the renderer rather than the data. `./build-apk.sh bench` builds
|
||||
it — own application id (`com.example.aiapp.bench`) and label ("AI
|
||||
Sessions bench") so it installs beside a real enrollment rather than
|
||||
replacing it. Opening it goes straight to a session screen holding the
|
||||
fixture (no enrollment, no permission prompts) with a "Run benchmark"
|
||||
control beside "Copy" in session settings: it drives the same scroll loop
|
||||
and streaming phase `transcript-bench.sh`/`stream-bench.sh` drive over
|
||||
`ui-trace`, but in-process, since a real phone has no usable system
|
||||
tracing and no agent can drive one (this-machine-android's skill).
|
||||
`BenchFixture.kt`/`BenchNetwork.kt` fake the backend by installing a
|
||||
`URLStreamHandlerFactory` that answers `TranscriptSource`/`EventStream`'s
|
||||
requests from an in-memory copy of the fixture instead of opening a
|
||||
socket — so the fold, the paging and `uniqueItems` under test are the
|
||||
screen's real ones, never a shortcut built just for this. The report
|
||||
gains a `bench:` section (process CPU time, peak RSS, battery current) on
|
||||
every build, empty except when `BenchRun.kt` filled it in.
|
||||
- **`app/ui-sandbox.sh`** — a second `ai-server` with its own `$HOME`, config
|
||||
and data directory, holding eight invented Claude Code transcripts and a
|
||||
`claude` that is two lines of shell. **That isolation is the point**: the
|
||||
@@ -226,6 +261,13 @@ Each exists because something was invisible without it.
|
||||
framework, from `atrace` text output with no trace processor needed. It is
|
||||
how the cost of a layout node per link was attributed to the framework
|
||||
rather than guessed at.
|
||||
- **`iris/android-app/build-apk.sh [debug|release] [--abi ...] [--features
|
||||
...]`** builds iris-android-app's cdylib (`cargo ndk`) and its APK
|
||||
(Gradle) in one step and verifies the result (`aapt2`/`apksigner`), and
|
||||
**`iris/android-app/run-bench.sh [--apk PATH]`** installs it on this
|
||||
checkout's own emulator, taps "Run benchmark" by label, and prints the
|
||||
report -- written so the P0 build/install/tap/read-report cycle stops
|
||||
being retyped by hand each time (docs/RUST.md's P0 box).
|
||||
|
||||
### Driving the UI
|
||||
|
||||
@@ -312,7 +354,7 @@ means here:
|
||||
## Sessions outlive the backend
|
||||
|
||||
Since 2026-08-29 a session's process is deliberately left running when
|
||||
`ai-server` stops, and adopted again when it starts. PLAN.md has the design;
|
||||
`ai-server` stops, and adopted again when it starts. docs/PLAN.md has the design;
|
||||
day to day:
|
||||
|
||||
- **Stopping the server no longer stops the sessions.** After `pkill
|
||||
@@ -345,7 +387,7 @@ count tells you what continuing a session will cost. Shown, not warned about;
|
||||
importing a large session is a choice somebody is entitled to make.
|
||||
|
||||
**Never import a Claude Code session that is open in a terminal.** The app
|
||||
refuses it — see PLAN.md for the incident that made that a refusal rather
|
||||
refuses it — see docs/PLAN.md for the incident that made that a refusal rather
|
||||
than a warning.
|
||||
|
||||
**One Claude Code session id can name two files, and the listing offers it
|
||||
@@ -521,4 +563,4 @@ belongs in `~/.claude/TOOLCHAIN.md` or `~/.claude/MACHINE.md` instead.
|
||||
`BasicTextField`, which costs two seconds a frame at 128 kB and stops the
|
||||
app at 1 MiB, so `EDIT_LIMIT` caps it at 32 kB with the reason said on
|
||||
screen. If you make the editor faster, that number is what to move.
|
||||
EXPLORER.md's "What the measurements said" has the rest.
|
||||
docs/EXPLORER.md's "What the measurements said" has the rest.
|
||||
@@ -0,0 +1,44 @@
|
||||
# Decisions awaiting review
|
||||
|
||||
Choices made while working autonomously, for Bryan to keep or change. Each
|
||||
says what was picked and why; the detail is in the design doc it names.
|
||||
Delete an entry once it has been looked at.
|
||||
|
||||
## Subagent views (2026-09-05, `SUBAGENTS.md`)
|
||||
|
||||
Made on my own judgement, limited blast radius:
|
||||
|
||||
1. **A subagent is a transcript, not a session.** It has no process,
|
||||
controls or settings; it is addressed as `/sessions/{id}/subagents/{sub}`
|
||||
and stored under the session's directory, so deleting the session takes
|
||||
it. Alternative rejected: registering it as a session of its own, which
|
||||
would give it a card in the main list and a driver that can do nothing.
|
||||
2. **Read-only view is the session screen minus its controls**, rather than
|
||||
a second, simpler transcript screen. Keeps paging, caching, selection
|
||||
and rendering in one place. Cost: a `readOnly` mode threaded through
|
||||
`SessionScreen`.
|
||||
3. **The list only carries a count.** Each session row says how many
|
||||
subagents it has; their titles and statuses are fetched when the card is
|
||||
expanded. Keeps `GET /sessions` from reading every subagent transcript.
|
||||
Consequence: an expanded card's statuses refresh with the list, not live.
|
||||
4. **Expanded/collapsed is remembered per session on the phone**, not on
|
||||
the server. Collapsed by default, per the transcript convention that new
|
||||
things arrive collapsed.
|
||||
5. **Subagents of imported sessions are not shown.** The import path still
|
||||
skips `isSidechain` records; the CLI's own `subagents/agent-*.jsonl` files
|
||||
are not read. Only subagents run while this backend was watching exist.
|
||||
6. **Echo grows `/subagent [n]`** as the test rig, so nothing here needs a
|
||||
paid turn to exercise.
|
||||
|
||||
Deferred, because they reach further than this feature:
|
||||
|
||||
- **Live status on the list.** Whether the session list should follow a
|
||||
stream at all (it refreshes on demand today) decides whether subagent
|
||||
status can ever be live there. Not changed.
|
||||
- **Nested subagents.** A subagent's own Task calls are shown as tool calls
|
||||
in its transcript and are not given transcripts of their own. Supporting
|
||||
that is the same mechanism one level down, but the UI would need nested
|
||||
expanders.
|
||||
|
||||
- **The subagent status row says "context unknown".** Nothing measures a
|
||||
subagent's context; the row could leave it out rather than admit it.
|
||||
@@ -1,110 +0,0 @@
|
||||
# iris: notable public API changes
|
||||
|
||||
For Iris to read on her own time. Each entry is a change to iris's public
|
||||
surface that a widget author or app author would notice: a trait method
|
||||
added, removed or re-shaped; a type that callers construct differently; a
|
||||
capability that moved. Small and trivial changes do not go here.
|
||||
|
||||
An entry gives the date, what changed, why, and a short before/after where
|
||||
it helps judge the change without the session that made it. Newest first.
|
||||
|
||||
## 2026-09-05: a second backend (android-view), and what moved to make room for it
|
||||
|
||||
RUST.md's I2. Three changes a widget or app author would notice, all in
|
||||
service of the same thing: `default` (winit) and the new `android`
|
||||
(android-view) backends sharing what does not depend on windowing.
|
||||
|
||||
- **`Selector`/`Selectable`'s bound changed from `Rsc::State:
|
||||
HasDefaultUiState` to `Rsc::State: FocusHost`** (new trait, `attr.rs`).
|
||||
`HasDefaultUiState` still exists and still works — `default/attr.rs` now
|
||||
implements `FocusHost` for anything that has it — so a winit app's
|
||||
existing code is unaffected. An Android app implements `FocusHost` via
|
||||
`HasAndroidUiState` instead. Affects only an app that referenced
|
||||
`HasDefaultUiState` directly at a `Selectable`/`Selector` call site
|
||||
rather than through `.attr::<Selectable>(())`, which nothing in-tree
|
||||
does.
|
||||
- **`Tasks::init` takes `Arc<dyn RequestRedraw>` instead of
|
||||
`Arc<winit::window::Window>`.** `RequestRedraw` (`task.rs`) is one method,
|
||||
`fn request_redraw(&self)`; `winit::window::Window` implements it
|
||||
(`default/render.rs`), so `Tasks::init(window)` at a call site is
|
||||
unchanged by inference. Only matters if something constructed a `Tasks`
|
||||
directly rather than through `DefaultRsc`/`AndroidRsc`.
|
||||
- **`TextEdit::apply_event`/`TextInputResult` are `#[cfg(not(target_os =
|
||||
"android"))]`** — they take a `winit::event::KeyEvent`, which does not
|
||||
exist on Android; `android/input.rs` drives the same primitives
|
||||
(`backspace`/`delete`/`motion`/`insert`, all still unconditional) from
|
||||
`ndk::event::Keycode` directly instead. New unconditional getters on the
|
||||
way: `TextEdit::text()`/`selection_range()`/`caret()`, and
|
||||
`TextEditCtx::delete_byte_range`/`set_cursor_byte` — the primitives
|
||||
`android/ime.rs`'s `InputConnection` bridge needed and that were not
|
||||
previously exposed publicly.
|
||||
|
||||
## 2026-09-04: `Widget::draw` reports the size it used; `desired_width`/`desired_height` are gone
|
||||
|
||||
A widget used to implement three methods (`draw`, `desired_width`,
|
||||
`desired_height`); it now implements one, `fn draw(&mut self, painter: &mut
|
||||
Painter) -> Size`, which draws into `painter.region()` and returns how much
|
||||
of it was used. Why: the two extra methods routinely re-simulated what
|
||||
`draw` was about to do anyway (`Span::desired_ortho` copied its own draw
|
||||
loop to get cross-axis sizing right) — one visit per widget per frame
|
||||
instead of up to three. A container that needs a child's size before
|
||||
placing it (alignment, centering) draws the child once at a provisional
|
||||
region, reads the returned `Size`, and calls the new `Painter::reposition`
|
||||
to move it into its final spot — an O(1) offset write, not a second draw. A
|
||||
widget whose drawn output never depends on the size it's given (a
|
||||
fixed-size `Rect`, a decoded `Image`) overrides the new `fn
|
||||
is_size_independent(&self) -> bool { false }` to `true`, which skips
|
||||
redrawing it when only its offered region changes shape.
|
||||
|
||||
```rust
|
||||
// before
|
||||
fn draw(&mut self, painter: &mut Painter) { /* ... */ }
|
||||
fn desired_width(&mut self, ctx: &mut SizeCtx) -> Len { /* ... */ }
|
||||
fn desired_height(&mut self, ctx: &mut SizeCtx) -> Len { /* ... */ }
|
||||
|
||||
// after
|
||||
fn draw(&mut self, painter: &mut Painter) -> Size { /* ... */ }
|
||||
```
|
||||
|
||||
`SizeCtx` and `Cache` are gone with it — see `LAYOUT.md` for the full
|
||||
design, the move-offset mechanism this shipped alongside, and the file
|
||||
list.
|
||||
|
||||
## 2026-09-04: texture pipeline rebuilt off the binding array
|
||||
|
||||
`Textures`/`TextureHandle`, `GlyphPrimitive`, and `UiRenderNode::new` all
|
||||
changed shape. Why: the old pipeline bound every texture ever drawn in one
|
||||
`binding_array<texture_2d<f32>>` and asked every device, unconditionally,
|
||||
for `VK_EXT_descriptor_indexing` — a real share of Android GPUs lack it,
|
||||
and it failed outright on the Android emulator's software Vulkan. See
|
||||
TEXTURES.md's "Recommended shape" and "Implemented, 2026-09-04".
|
||||
|
||||
- **`UiRenderNode::new` drops its `limits: UiLimits` parameter, and
|
||||
`UiLimits` is gone.** Before: `UiRenderNode::new(&device, &queue,
|
||||
&config, UiLimits::default())`. After: `UiRenderNode::new(&device,
|
||||
&queue, &config)`. Nothing replaces it — there are no more
|
||||
binding-array limits to size.
|
||||
- **`src/default/render.rs`'s device request asks for no features and no
|
||||
binding-array limits.** Before: `required_features:
|
||||
Features::TEXTURE_BINDING_ARRAY | Features::PARTIALLY_BOUND_BINDING_ARRAY
|
||||
| Features::SAMPLED_TEXTURE_AND_STORAGE_BUFFER_ARRAY_NON_UNIFORM_INDEXING`
|
||||
plus two `max_binding_array_*` limits. After: `Features::empty()` (the
|
||||
`DeviceDescriptor` default) and only `max_buffer_size` set, which was
|
||||
never about the binding array.
|
||||
- **`TextureHandle` has no `primitive()` method any more**; a caller
|
||||
outside `iris` shouldn't have been calling it (it fed the old renderer's
|
||||
internals), but if something did: use `image_index()` for a standalone
|
||||
image's bind-group index. There is no equivalent for a page — a page has
|
||||
no bind group of its own now, see below.
|
||||
- **`GlyphPrimitive` has no public constructor from a struct literal.**
|
||||
Before: `GlyphPrimitive { uv_min, uv_max, view_idx, sampler_idx, color,
|
||||
flags }`. After: `GlyphPrimitive::new(uv_min, uv_max, layer, color,
|
||||
flags)` — one `layer` (the shared atlas array's layer) instead of a
|
||||
`view_idx`/`sampler_idx` pair, since a page is now a layer of one array
|
||||
texture rather than its own bound texture.
|
||||
- **A widget author drawing images is unaffected**: `Painter::texture`/
|
||||
`texture_at`/`texture_within` and `Textures::add` keep their signatures.
|
||||
What changed underneath is that each standalone image now gets its own
|
||||
`wgpu::BindGroup` and draw call instead of a slot in the shared array —
|
||||
invisible from the widget API, visible only in `UiRenderNode`'s internals
|
||||
and in `iris`'s device requirements.
|
||||
-170
@@ -1,170 +0,0 @@
|
||||
# iris: known problems and things still to build
|
||||
|
||||
Iris's own list for the library, recorded 2026-09-04 in her words where it
|
||||
matters, so the agents working through RUST.md pick these up in a sensible
|
||||
order rather than rediscovering them. Each item says where it sits in the
|
||||
order and what "done" looks like. Tick and date them in place.
|
||||
|
||||
## Fix
|
||||
|
||||
- [x] **Input does not fall through by input type (2026-09-04).**
|
||||
`SensorUi::run_sensors` (`src/default/sense.rs`) used to set "consumed,
|
||||
stop checking lower layers" from mere hover — a widget registered for
|
||||
nothing but `click()` blocked a `Scroll` meant for whatever was behind
|
||||
it, since "the cursor is over this widget" and "this widget handled the
|
||||
event" were the same check. Fixed by judging consumption per input
|
||||
kind: with no button transition and no scroll happening this frame
|
||||
("momentary" activity), the topmost hovered widget still wins, same as
|
||||
before; when something momentary *is* happening, only a widget whose
|
||||
registered senses actually include a matching non-hover one (checked
|
||||
via a new `TypeEventManager::registered`, which lists what a widget
|
||||
registered without running anything) consumes it, so a widget with only
|
||||
`Hovering`/click handlers can no longer block a scroll from reaching a
|
||||
list underneath. `iris/src/sense_tests.rs` builds a button-over-a-list
|
||||
`Stack` with a plain `HasEvents` impl (no GPU or window) and checks both
|
||||
directions: a scroll over the button reaches the list, and a real click
|
||||
still reaches the button — confirmed to fail on the pre-fix code and
|
||||
pass after.
|
||||
|
||||
- [ ] **Appending one image to an already-loaded list rebuilds every other
|
||||
image's bind group (2026-09-05).** Found by the benchmark below, not
|
||||
designed against: `GpuTextures::update` (`core/src/render/texture.rs`)
|
||||
triggers `rebuild_image_bind_groups` — a loop over *every live
|
||||
standalone image*, rebuilding its `BindGroup` — whenever the shared
|
||||
`masks` or `move_offsets` GPU buffer is resized (`masks_resized ||
|
||||
moves_resized` in `UiRenderNode::update`, `core/src/render/mod.rs`), and
|
||||
a widget getting its *first* move-offset slot (LAYOUT.md section 2 —
|
||||
every widget gets one on first draw) can be exactly what grows that
|
||||
buffer. So one new message with one new image, appended to a transcript
|
||||
that already has N images loaded, does not cost O(1): it costs one
|
||||
`create_image` for the new image plus one `make_image_bind_group` per
|
||||
*existing* image, because the new widget's own move slot pushed the
|
||||
arena past its capacity. Measured directly in
|
||||
`iris/examples/bench_images.rs`: appending a 1,001st image to 1,000
|
||||
already-settled ones reports **1,001** bind-group creates for that one
|
||||
frame, not 1 (`./run-bench.sh images`, frame 5 in the transcript below).
|
||||
This is the same class of cost LAYOUT.md's move chain exists to avoid
|
||||
elsewhere in the codebase, just not yet closed off here — the fix is
|
||||
presumably to size `masks`/`move_offsets` with headroom (the array
|
||||
texture already grows by doubling, `grow_array`, for the same reason) so
|
||||
an ordinary append does not cross a capacity boundary, or to stop tying
|
||||
the *image* bind group's contents to a buffer that changes on every new
|
||||
widget in the whole tree, image or not. Not designed further here per
|
||||
the "do not redesign, record it" instruction this benchmark was built
|
||||
under.
|
||||
- [ ] **Bind-group creation takes two frames to reach the steady state, not
|
||||
one (2026-09-05).** Same benchmark: loading 1,000 images cold reports
|
||||
1,000 creates on frame 1 (expected — this is `create_image`, one per
|
||||
new image) *and again* 1,000 on frame 2, with nothing between the two
|
||||
frames marked dirty, before settling to 0 from frame 3. The second
|
||||
frame's 1,000 is `rebuild_image_bind_groups` again, for the same
|
||||
masks/move-offsets buffer-growth reason as the item above — the arena
|
||||
apparently does not finish growing to its steady size within the first
|
||||
frame the tree is drawn. Not chased further; recorded so whoever fixes
|
||||
the item above checks whether the fix also closes this one, since they
|
||||
look like the same root cause measured two different ways.
|
||||
|
||||
## Build
|
||||
|
||||
- [x] **Benchmarks**, not unit tests, run on demand (2026-09-05; a
|
||||
`benches/` or a script under `iris/`, never in `cargo test`). The
|
||||
scenario that matters most is a **message list** — chat apps and this
|
||||
app's transcript alike — stressed with many messages and many images.
|
||||
One case in particular: **resizing an input box** (typing enough text to
|
||||
grow it) that pushes a long list of messages above it must stay very
|
||||
fast and recalculate almost nothing — a move of everything above, not a
|
||||
re-layout. That is exactly the O(1) move chain in LAYOUT.md; the
|
||||
benchmark is what proves it. Done when the numbers are in this file with
|
||||
the command, and the input-box case reports draws re-run, not just frame
|
||||
time.
|
||||
|
||||
**Built as two rigs**, chosen per scenario by whether a real `wgpu`
|
||||
device is needed (`UiRenderState`/`Widgets` touch no GPU or window, so
|
||||
most of this runs as an ordinary binary — the same property
|
||||
`layout_tests.rs` relies on):
|
||||
|
||||
- `iris/benches/message_list.rs` — a plain `Instant`-timed binary
|
||||
(`[[bench]] harness = false` in `iris/Cargo.toml`), not criterion: see
|
||||
the file's own header for why (short version — every scenario here
|
||||
reduces to a *count* `UiRenderState::take_counters` already produces,
|
||||
which criterion's statistical machinery adds nothing to and which a
|
||||
new dependency is not worth pulling in for). Covers (a) first-frame
|
||||
cost of a message list of N wrapped-text rows (one in 20 also carrying
|
||||
a small in-memory image) for N = 100/1,000/10,000; (b) per-frame cost
|
||||
of scrolling that list, 200 ticks; (c) the input-box case — a
|
||||
fixed-height field at the bottom of the screen growing by a line 40
|
||||
times, with the message list above it filling the rest of the screen.
|
||||
Run: `cd iris && cargo bench --bench message_list` (always release —
|
||||
`cargo bench` builds the `bench` profile, which is optimized).
|
||||
- `iris/examples/bench_images.rs` — needs a real device, so it runs
|
||||
through `iris/run-headless.sh bench_images`, printing
|
||||
`UiRenderNode::take_image_bind_group_creates()` (a new counter, added
|
||||
in `core/src/render/texture.rs` and `core/src/render/mod.rs`,
|
||||
mirroring `UiRenderState::take_counters`) each frame. Covers (d): 1,000
|
||||
image rows, checked both cold (does bind-group creation reach zero
|
||||
once loaded) and after appending one more image once settled (does
|
||||
*that* stay cheap) — the second question is what actually matters for
|
||||
a live transcript and is what turned up the two Fix items above.
|
||||
- `iris/run-bench.sh [list|images]` runs either or both and is what to
|
||||
run before/after touching `Scroll`, `Span`, `Sized`, the move-offset
|
||||
chain, or `GpuTextures`.
|
||||
|
||||
**Numbers (2026-09-05, release, `cargo bench`/`run-headless.sh`, this
|
||||
VM: AMD Ryzen 7 3800X, 8 cores, rustc 1.98.0 nightly-2026-09-03):**
|
||||
|
||||
cd iris && cargo bench --bench message_list
|
||||
(a) first frame, N=100: 30.30ms draws=227 rewrites=15 moves=0
|
||||
(a) first frame, N=1000: 186.04ms draws=2252 rewrites=150 moves=0
|
||||
(a) first frame, N=10000:1770.36ms draws=22502 rewrites=1500 moves=0
|
||||
(b) scroll, N=100/1000/10000, 200 ticks each:
|
||||
draws=200 rewrites=0 moves=200 (identical at every N)
|
||||
per-tick average: 0.0002ms (identical at every N)
|
||||
(c) input grows 40 lines, N=100/1000/10000 rows above it:
|
||||
draws=320 rewrites=40 moves=160 (identical at every N)
|
||||
per-line average: 0.0012-0.0013ms (identical at every N)
|
||||
|
||||
cd iris && ./run-bench.sh images
|
||||
frame=1 bind_group_creates=1000 (cold load)
|
||||
frame=2 bind_group_creates=1000 (see Fix item above)
|
||||
frame=3 bind_group_creates=0
|
||||
frame=4 bind_group_creates=0
|
||||
(append one image here)
|
||||
frame=5 bind_group_creates=1001 (see Fix item above)
|
||||
frame=6 bind_group_creates=0
|
||||
|
||||
**Reading it**: (a) is real, necessary work — shaping and laying out N
|
||||
never-before-seen text rows — and scales with N as it must, ~10x cost
|
||||
per 10x N. (b) and (c) are the pass conditions that matter: both are
|
||||
**exactly flat across N = 100 to 10,000**, confirming LAYOUT.md's O(1)
|
||||
move chain holds for both scrolling and for a growing input box pushing
|
||||
the message list — draws/moves per tick or per line do not grow with
|
||||
list size, and the per-operation cost (a fraction of a microsecond) is
|
||||
nowhere near a frame budget. (d)'s cold-load and steady-state halves
|
||||
behave as designed; its *append* half did not, which is the two Fix
|
||||
items above.
|
||||
- [ ] **Masks defined relative to each other.** Wanted: mask A multiplies
|
||||
by something *and also* applies mask B — a mask can reference a parent
|
||||
mask, the way the move chain references a parent offset. Today masks
|
||||
are independent regions. Design it beside the move chain (same shape:
|
||||
a parent index and a bounded walk in the shader); do it when a real
|
||||
widget needs it, not before.
|
||||
- [ ] **Positions as a single float per scroll.** Iris raised, and half
|
||||
rejected, letting a scroll update one float rather than positions:
|
||||
input handling cares about most elements in a list, so absolute
|
||||
positions must be computed on the CPU anyway. LAYOUT.md's design
|
||||
already lands here (GPU walks the chain, CPU resolves on demand for
|
||||
hit tests). Keep the CPU resolution lazy and per query; do not
|
||||
materialise every row's absolute position per frame.
|
||||
- [ ] **Animations, last.** Cosmetic, so after everything above. Must be
|
||||
**modular — a piece of the library rather than a core part forced into
|
||||
everything, the same way input is**. Whatever the mechanism, a widget
|
||||
that does not animate must pay nothing and import nothing for it.
|
||||
|
||||
## Reconsider
|
||||
|
||||
- [ ] **`WidgetView`.** Iris is unsure of it: what she wants is an easy way
|
||||
to compose a widget from others (a button is the main case). With
|
||||
sizing folded into `draw`, composing may be easy enough that `View` is
|
||||
redundant. Decide after the layout change lands, by writing a button
|
||||
both ways and keeping the one that is shorter to explain; delete the
|
||||
other rather than keeping two ways.
|
||||
+141
@@ -0,0 +1,141 @@
|
||||
# Subagents
|
||||
|
||||
A session's subagents -- the helpers a Claude Code session starts through its
|
||||
Task tool -- each get a transcript of their own, listed under the session's
|
||||
card and readable in the same transcript view the session has. Designed
|
||||
2026-09-05; the decisions Bryan has not yet reviewed are in `DECISIONS.md`.
|
||||
|
||||
## What a subagent is here
|
||||
|
||||
**A subagent is a second transcript owned by a session, in the same event
|
||||
model, with no process and no controls.** It is not a session: it cannot be
|
||||
messaged, stopped or started, and it has no setup, model or usage of its
|
||||
own. Everything it shares with a session -- the transcript file format, the
|
||||
paging routes, the SSE stream, the phone's cache and rendering -- is reused
|
||||
by addressing, not by copying.
|
||||
|
||||
The CLI reports a subagent's messages on the parent's own stream-json
|
||||
output, each carrying `parent_tool_use_id` = the id of the Task `tool_use`
|
||||
that started it. Before this the translator dropped those lines
|
||||
(`subagent_events_are_not_duplicated_into_the_transcript`); now it routes
|
||||
them to that subagent's own translator and transcript. The parent's
|
||||
transcript still shows only the Task call itself.
|
||||
|
||||
## Storage
|
||||
|
||||
Under the session directory:
|
||||
|
||||
```
|
||||
<session>/subagents/<tool_use_id>/meta.json {title, created}
|
||||
<session>/subagents/<tool_use_id>/transcript.jsonl same SeqEvent lines as the session's
|
||||
```
|
||||
|
||||
The id is the Task tool_use id (`toolu_…`), which is unique, stable across a
|
||||
backend restart, and already the key everything on the parent side uses.
|
||||
Only ids matching `[A-Za-z0-9_-]+` are ever created or looked up, since the
|
||||
id becomes a path.
|
||||
|
||||
The transcript's sequence numbers are its own, starting at 1. `Transcript`,
|
||||
`read_window`, `catch_up` and `read_after` work on it unchanged.
|
||||
|
||||
Its path out: deleting the session deletes its directory, subagents included.
|
||||
There is no separate delete.
|
||||
|
||||
## Lifecycle, as events in the subagent's transcript
|
||||
|
||||
1. Created on the first child line for an unseen parent id (or, when the
|
||||
parent Task call was seen, at that call). First lines written:
|
||||
`Status Running`, then `UserMessage { text: <the Task's prompt> }` when
|
||||
the prompt is known -- it genuinely is the subagent's first user turn.
|
||||
2. Every child line is translated by that subagent's own `Translator`
|
||||
(one per subagent: tool ids are unique but streaming deltas are by
|
||||
content-block index, and parallel subagents interleave).
|
||||
3. **The parent's `tool_result` never finishes a subagent.** The Task tool
|
||||
runs in the background by default: the `tool_result` -- "Async agent
|
||||
launched..." -- arrives the moment it *starts*, while the subagent goes
|
||||
on working for however long its own turn takes, sometimes minutes. What
|
||||
ends it is its own turn ending: the raw API's `message_delta` on its
|
||||
stream carrying `stop_reason: "end_turn"` (a `stop_reason` of `tool_use`
|
||||
is the model about to call one, not an end), or a `result` line for its
|
||||
own turn if a future CLI version ever sends one. Either maps to
|
||||
`Status Exited`; the subagent's vocabulary has no `Idle`, so the
|
||||
equivalent event `dispatch` produces for an ordinary session is dropped
|
||||
rather than written. A shipped version of this finished on the
|
||||
`tool_result` instead, which read a running background agent as
|
||||
"finished" with its transcript truncated at the moment it launched.
|
||||
4. **A child line for a subagent that already finished reopens it**
|
||||
(`Status Running`) rather than being dropped: a background Task can be
|
||||
sent another message long after its first turn ended, and that is
|
||||
exactly what a further line for it means. Same transcript, same child
|
||||
`Translator`, just picking back up.
|
||||
5. When the parent session's process exits (`Status Exited` on the
|
||||
session), every subagent still `Running` gets `Status Exited` too: its
|
||||
process was the parent's.
|
||||
|
||||
A subagent that was mid-flight when the backend restarted keeps working:
|
||||
the registry reopens the existing transcript on the next child line, and
|
||||
the file continues its sequence -- the same reopening #4 describes, whether
|
||||
what closed it was a restart or its own `end_turn`. If its turn ended while
|
||||
the backend was down nothing recorded that until the next line arrives, so
|
||||
its last status stays `Running`, which the list reports as **unknown**
|
||||
rather than as running (see the wire shape) until then.
|
||||
|
||||
Title: the Task call's `description` input, then ` (<subagent_type>)` when
|
||||
one is given; falling back to the tool's name when the child arrives before
|
||||
(or without) the parent call being seen.
|
||||
|
||||
## Server layout
|
||||
|
||||
- `session/subagent.rs` -- the registry: `Subagents` (per session, in
|
||||
`Shared`), `Subagent` (its `Transcript` behind a mutex plus a
|
||||
`broadcast::Sender<SeqEvent>`), `record(id, event)`, `start(id, title,
|
||||
prompt)`, `finish(id)`, `reopen(id)`, `finish_all()`, `list()` from disk. Drivers get an
|
||||
`Arc<Subagents>` beside their `EventSink`; llama ignores it.
|
||||
- `session/claude/translate.rs` -- routes child lines by parent id, holds
|
||||
one child `Translator` per subagent, remembers pending Task calls'
|
||||
description/prompt/subagent_type.
|
||||
- `session/echo.rs` -- `/subagent [n]`: the test rig. Starts *n* (default 1)
|
||||
subagents at once, each named "helper k". Each writes the prompt as its
|
||||
user message, streams a few words of text, runs one `Bash` tool call, then
|
||||
finishes about three seconds after starting, and the parent's Task calls
|
||||
end when their subagent does. Three seconds so the running state can be
|
||||
seen on the phone.
|
||||
- `routes.rs` -- three routes, in the doc table.
|
||||
|
||||
## Wire shape
|
||||
|
||||
```
|
||||
GET /sessions/{id} SessionInfo gains `subagents: N` (count, 0 when none)
|
||||
GET /sessions same field on each row
|
||||
GET /sessions/{id}/subagents [{id, title, status, created, lastActivity}], oldest first
|
||||
GET /sessions/{id}/subagents/{sub}/transcript exactly the session transcript's query and answer
|
||||
GET /sessions/{id}/subagents/{sub}/events?after=N exactly the session events stream
|
||||
```
|
||||
|
||||
`status` is the transcript's last `Status` event, serialised like a session's
|
||||
(`running`, `exited`), except that a subagent whose session is not itself
|
||||
running cannot be running: the list answers `unknown` for that one. The
|
||||
phone words these as *running*, *finished* and *unknown* on the subcard.
|
||||
|
||||
The count on `SessionInfo` is a directory listing, so the list stays cheap.
|
||||
The per-subagent status is only read when the list route is asked for.
|
||||
|
||||
## Phone
|
||||
|
||||
- `SessionSummary.subagents: Int`. A card with a non-zero count ends in an
|
||||
expander row -- a full-width `Chevron(Pointing.Down)` row that flips to
|
||||
`Pointing.Up` -- collapsed by default. Expanding fetches
|
||||
`/sessions/{id}/subagents` and draws one `OutlinedCard` per subagent,
|
||||
indented inside the session card, the way dev-updater draws a project's
|
||||
components: title, then the status word and a relative time. The
|
||||
expansion state is per session id and survives a refresh of the list.
|
||||
- Tapping a subcard opens `Screen.Subagent`, which is `SessionScreen` in
|
||||
**read-only** form: the same transcript, paging, cache, selection,
|
||||
images and status row, with the composer, the process button, the model
|
||||
picker, the files button, the settings cog and the usage bar left out.
|
||||
The header shows the subagent's title with the session's title beneath
|
||||
it. Back returns to the list.
|
||||
- Addressing: `fetchTranscript`, `EventStream`, `TranscriptSource` and the
|
||||
cache take a transcript address rather than a session id --
|
||||
`sessions/{id}` or `sessions/{id}/subagents/{sub}` -- so the cache nests a
|
||||
subagent's copy under its session's and the same code serves both.
|
||||
Generated
+1081
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,36 @@
|
||||
[package]
|
||||
name = "android-shell"
|
||||
version = "0.1.0"
|
||||
edition = "2024"
|
||||
|
||||
# The JNI bridge behind E3's two Java stub classes (`MainActivity`,
|
||||
# `NotificationService` -- see RUST.md's "How much Java is unavoidable" for
|
||||
# why those two classes cannot be anything but Java/Kotlin, registered from
|
||||
# the manifest by name). Everything they would otherwise have done in
|
||||
# Kotlin -- the SSE follow loop, deciding where a notification is shown,
|
||||
# picking a session for a share -- is here instead, built on `client-core`
|
||||
# so the networking and parsing are not duplicated a third time next to the
|
||||
# server and the Kotlin app.
|
||||
#
|
||||
# `cdylib` for `System.loadLibrary`; `lib` too so `cargo test`/`clippy` run
|
||||
# on a normal host target without an Android NDK toolchain, the same
|
||||
# posture `client-core` and `server` already have.
|
||||
|
||||
[lib]
|
||||
name = "android_shell"
|
||||
crate-type = ["cdylib", "lib"]
|
||||
|
||||
[dependencies]
|
||||
client-core = { path = "../client-core" }
|
||||
jni = "0.22"
|
||||
log = "0.4"
|
||||
|
||||
# `LogErrorAndDefault` (the `native_method!` error policy this crate uses
|
||||
# throughout, see lib.rs) logs through the `log` facade, which is a no-op
|
||||
# without a backend installed -- so without this, every recoverable error
|
||||
# at a native entry point would be silently dropped rather than reaching
|
||||
# logcat. Android-only: nothing else here needs it, and it does not build
|
||||
# off-device (see `notify::ensure_logger`'s call site, the only place this
|
||||
# is used).
|
||||
[target.'cfg(target_os = "android")'.dependencies]
|
||||
android_logger = "0.15"
|
||||
@@ -0,0 +1,152 @@
|
||||
//! Thin wrappers around the five `Env` calls this crate makes constantly
|
||||
//! (a class name, a method name and a signature, all as plain `&str`).
|
||||
//!
|
||||
//! `jni` 0.22 wants a class or method *name* as `AsRef<JNIStr>` (its own
|
||||
//! modified-UTF-8 type; `JNIString::new` is the runtime conversion, used
|
||||
//! here uniformly rather than switching to the compile-time `jni_str!`
|
||||
//! literal macro call by call -- these are a handful of short, one-off
|
||||
//! lookups, not a hot loop, so the difference is not worth two code paths
|
||||
//! for the same thing) and a *signature* as a parsed `MethodSignature`/
|
||||
//! `FieldSignature`, which is why those go through
|
||||
//! `RuntimeMethodSignature`/`RuntimeFieldSignature::from_str` instead: the
|
||||
//! parsed form is what lets these calls skip re-validating the signature
|
||||
//! against the arguments on every call, which is the whole reason `jni`
|
||||
//! moved to it.
|
||||
//!
|
||||
//! **The classloader gotcha, found by testing (2026-09-05).** A class
|
||||
//! lookup by name (`find_class`, `new_object`, `call_static_method`,
|
||||
//! `get_static_field` -- anything that resolves a *class*, as opposed to
|
||||
//! `call_method` on an object it already has, which needs no such lookup)
|
||||
//! defaults to `FindClass`'s ordinary search when it cannot find the
|
||||
//! calling thread a classloader through `Thread.getContextClassLoader()`.
|
||||
//! That default is fine on a thread the JVM itself started -- an
|
||||
//! `onCreate`/`onStartCommand` callback -- but every one of these calls
|
||||
//! from `android-shell`'s own background thread (the notification
|
||||
//! follow-loop, the share upload) is running on a thread *Rust* spawned
|
||||
//! and attached with `JavaVM::attach_current_thread`, which the platform
|
||||
//! never gave an app classloader. Framework classes
|
||||
//! (`android.app.Notification$Builder`, ...) still resolve, because they
|
||||
//! are reachable from the bootstrap loader `FindClass` falls back to --
|
||||
//! `androidx.core.app.NotificationManagerCompat` is not, since it is
|
||||
//! packaged inside this app's own APK. The failure was
|
||||
//! `Error::NoClassDefFound`, logged by `notify::show`'s `LogErrorAndDefault`
|
||||
//! as "failed to resolve Java class ... (class not found or linkage
|
||||
//! error)" -- on a real device this reads as "the notification silently
|
||||
//! never arrives," since the whole call is inside the follow loop and the
|
||||
//! ongoing foreground notification (built on the main thread, in
|
||||
//! `try_start`, before the background thread exists) posts fine either
|
||||
//! way. `remember_class_loader` caches the app's own `ClassLoader` the
|
||||
//! first time any entry point has a `Context` to ask, and every class
|
||||
//! lookup below goes through it explicitly via `LoaderContext::Loader`
|
||||
//! rather than the thread-dependent default -- so it is correct on the
|
||||
//! main thread and on this crate's own background threads alike.
|
||||
|
||||
use jni::Env;
|
||||
use jni::errors::Result;
|
||||
use jni::objects::{JClass, JClassLoader, JObject, JValue, JValueOwned};
|
||||
use jni::refs::{Global, LoaderContext};
|
||||
use jni::signature::{RuntimeFieldSignature, RuntimeMethodSignature};
|
||||
use jni::strings::JNIString;
|
||||
use std::sync::OnceLock;
|
||||
|
||||
static CLASS_LOADER: OnceLock<Global<JClassLoader<'static>>> = OnceLock::new();
|
||||
|
||||
/// Caches `context`'s own `ClassLoader`, the first time this is called.
|
||||
/// Cheap to call from every entry point that has a `Context` on hand
|
||||
/// (`MainActivity`'s and `NotificationService`'s all do): later calls are
|
||||
/// a `OnceLock::get` and nothing else.
|
||||
pub fn remember_class_loader(env: &mut Env, context: &JObject) -> Result<()> {
|
||||
if CLASS_LOADER.get().is_some() {
|
||||
return Ok(());
|
||||
}
|
||||
// context.getClass().getClassLoader() -- resolved via `call_method` on
|
||||
// real objects throughout, so this needs no class-name lookup of its
|
||||
// own and has nothing to bootstrap.
|
||||
let class_obj = call_method(env, context, "getClass", "()Ljava/lang/Class;", &[])?.l()?;
|
||||
let loader_obj = call_method(
|
||||
env,
|
||||
&class_obj,
|
||||
"getClassLoader",
|
||||
"()Ljava/lang/ClassLoader;",
|
||||
&[],
|
||||
)?
|
||||
.l()?;
|
||||
let loader = env.cast_local::<JClassLoader>(loader_obj)?;
|
||||
let global = env.new_global_ref(&loader)?;
|
||||
// Lost the race with another entry point calling this concurrently --
|
||||
// both loaders name the same app, so either one is fine and there is
|
||||
// nothing to reconcile.
|
||||
let _ = CLASS_LOADER.set(global);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Resolves `name` (slash-separated, e.g. `androidx/core/app/NotificationCompat`)
|
||||
/// through the cached app classloader when one has been remembered, and
|
||||
/// through the ordinary default otherwise -- which is every call made
|
||||
/// before any entry point has run, and is also correct for a main-thread
|
||||
/// caller, so there is no case this makes worse.
|
||||
fn resolve_class<'local>(env: &mut Env<'local>, name: &str) -> Result<JClass<'local>> {
|
||||
match CLASS_LOADER.get() {
|
||||
Some(loader) => {
|
||||
let binary_name = name.replace('/', ".");
|
||||
LoaderContext::Loader(loader).load_class(env, JNIString::new(&binary_name), true)
|
||||
}
|
||||
None => env.find_class(JNIString::new(name)),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn find_class<'local>(env: &mut Env<'local>, name: &str) -> Result<JClass<'local>> {
|
||||
resolve_class(env, name)
|
||||
}
|
||||
|
||||
/// A new Java string as a plain `JObject` -- what every call site here
|
||||
/// wants it as (`JValue::Object` takes `&JObject`, not `&JString`, and
|
||||
/// `JString: Into<JObject>` is the documented way across).
|
||||
pub fn jstr_obj<'local>(env: &mut Env<'local>, text: impl AsRef<str>) -> Result<JObject<'local>> {
|
||||
Ok(env.new_string(text)?.into())
|
||||
}
|
||||
|
||||
pub fn new_object<'local>(
|
||||
env: &mut Env<'local>,
|
||||
class: &str,
|
||||
sig: &str,
|
||||
args: &[JValue],
|
||||
) -> Result<JObject<'local>> {
|
||||
let sig = RuntimeMethodSignature::from_str(sig)?;
|
||||
let class = resolve_class(env, class)?;
|
||||
env.new_object(class, sig.method_signature(), args)
|
||||
}
|
||||
|
||||
pub fn call_method<'local>(
|
||||
env: &mut Env<'local>,
|
||||
obj: &JObject,
|
||||
method: &str,
|
||||
sig: &str,
|
||||
args: &[JValue],
|
||||
) -> Result<JValueOwned<'local>> {
|
||||
let sig = RuntimeMethodSignature::from_str(sig)?;
|
||||
env.call_method(obj, JNIString::new(method), sig.method_signature(), args)
|
||||
}
|
||||
|
||||
pub fn call_static_method<'local>(
|
||||
env: &mut Env<'local>,
|
||||
class: &str,
|
||||
method: &str,
|
||||
sig: &str,
|
||||
args: &[JValue],
|
||||
) -> Result<JValueOwned<'local>> {
|
||||
let sig = RuntimeMethodSignature::from_str(sig)?;
|
||||
let class = resolve_class(env, class)?;
|
||||
env.call_static_method(class, JNIString::new(method), sig.method_signature(), args)
|
||||
}
|
||||
|
||||
pub fn get_static_field<'local>(
|
||||
env: &mut Env<'local>,
|
||||
class: &str,
|
||||
field: &str,
|
||||
sig: &str,
|
||||
) -> Result<JValueOwned<'local>> {
|
||||
let sig = RuntimeFieldSignature::from_str(sig)?;
|
||||
let class = resolve_class(env, class)?;
|
||||
env.get_static_field(class, JNIString::new(field), sig.field_signature())
|
||||
}
|
||||
@@ -0,0 +1,131 @@
|
||||
//! The JNI bridge behind E3's two Java stub classes. See `Cargo.toml`'s
|
||||
//! package comment for what this crate is and RUST.md's E3 entry for the
|
||||
//! design decisions.
|
||||
//!
|
||||
//! Each native method is declared with `jni`'s [`native_method!`] macro
|
||||
//! rather than a hand-written `#[no_mangle] extern "system" fn Java_...`:
|
||||
//! the macro derives the mangled export name and the JNI signature from the
|
||||
//! Rust function itself, so the two cannot drift apart the way a
|
||||
//! hand-typed name string and a hand-typed `"(Landroid/...;)V"` signature
|
||||
//! routinely do. `error_policy = LogErrorAndDefault` matches
|
||||
//! `Notifications.kt`'s own posture: a failure here (a lost connection, a
|
||||
//! JNI call that threw) is reported to logcat, not thrown back into Java
|
||||
//! as an exception that would crash the app over something recoverable.
|
||||
//!
|
||||
//! Each `const _: NativeMethod = native_method! { ... };` binding is
|
||||
//! otherwise unused by name -- `_` is the idiomatic way to keep a
|
||||
//! side-effecting const (here, generating the `#[export_name]`d function
|
||||
//! the JVM resolves by the JNI naming convention) without a `dead_code`
|
||||
//! warning for a binding nothing reads.
|
||||
|
||||
mod jcall;
|
||||
mod notify;
|
||||
mod settings;
|
||||
mod share;
|
||||
|
||||
use jni::errors::LogErrorAndDefault;
|
||||
use jni::objects::{JClass, JObject};
|
||||
use jni::sys::jint;
|
||||
use jni::{Env, NativeMethod, native_method};
|
||||
|
||||
/// Installs the `log` backend that routes to logcat, once per process.
|
||||
/// Without it, `LogErrorAndDefault` (every native method below) and any
|
||||
/// `log::error!` inside `jni` itself (e.g. `JString`'s `Display` fallback)
|
||||
/// call into the `log` facade's default no-op logger, and a real failure
|
||||
/// vanishes with nothing on logcat to say so -- silently *more* wrong than
|
||||
/// crashing, since nothing on screen or in the log says a notification was
|
||||
/// dropped. Called from every entry point below rather than a Java-side
|
||||
/// `Application.onCreate`, since this crate deliberately has no such class
|
||||
/// to hook (see RUST.md's E3 entry on the two-Java-classes floor).
|
||||
fn ensure_logger() {
|
||||
static ONCE: std::sync::Once = std::sync::Once::new();
|
||||
ONCE.call_once(|| {
|
||||
#[cfg(target_os = "android")]
|
||||
android_logger::init_once(
|
||||
android_logger::Config::default()
|
||||
.with_max_level(log::LevelFilter::Debug)
|
||||
.with_tag("android-shell"),
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
// The parameters are spelled as their Java types, not as `JObject`: the
|
||||
// macro encodes each argument into the exported symbol's JNI signature
|
||||
// (and JNI resolves `Java_...` names *by* that signature), so a generic
|
||||
// `JObject` here would export `(Ljava/lang/Object;...)` against a Java
|
||||
// method actually declared `(Landroid/app/Activity;...)` -- two different
|
||||
// symbols that never resolve to each other, silently, with no compiler
|
||||
// error on either side. `android.app.Activity` etc. have no dedicated
|
||||
// Rust wrapper in this crate, so they fall back to plain `JObject` in the
|
||||
// implementation functions below (the "Built-in Types" note in
|
||||
// `native_method!`'s docs).
|
||||
const _: NativeMethod = native_method! {
|
||||
java_type = "com.example.aiapp.shell.MainActivity",
|
||||
static extern fn native_handle_intent(activity: android.app.Activity, intent: android.content.Intent) -> (),
|
||||
error_policy = LogErrorAndDefault,
|
||||
};
|
||||
|
||||
/// `MainActivity.nativeHandleIntent` -- called from `onCreate` and
|
||||
/// `onNewIntent`. See `share::handle_intent` for what an intent can mean.
|
||||
fn native_handle_intent<'local>(
|
||||
env: &mut Env<'local>,
|
||||
_class: JClass<'local>,
|
||||
activity: JObject<'local>,
|
||||
intent: JObject<'local>,
|
||||
) -> Result<(), jni::errors::Error> {
|
||||
ensure_logger();
|
||||
jcall::remember_class_loader(env, &activity)?;
|
||||
share::handle_intent(env, &activity, &intent)
|
||||
}
|
||||
|
||||
const _: NativeMethod = native_method! {
|
||||
java_type = "com.example.aiapp.shell.NotificationService",
|
||||
static extern fn native_sync(context: android.content.Context) -> (),
|
||||
error_policy = LogErrorAndDefault,
|
||||
};
|
||||
|
||||
/// `NotificationService.nativeSync` -- called both from `MainActivity` (an
|
||||
/// enrollment may have just landed) and from `NotificationService.sync`
|
||||
/// itself. See `notify::sync`.
|
||||
fn native_sync<'local>(
|
||||
env: &mut Env<'local>,
|
||||
_class: JClass<'local>,
|
||||
context: JObject<'local>,
|
||||
) -> Result<(), jni::errors::Error> {
|
||||
ensure_logger();
|
||||
jcall::remember_class_loader(env, &context)?;
|
||||
notify::sync(env, &context)
|
||||
}
|
||||
|
||||
const _: NativeMethod = native_method! {
|
||||
java_type = "com.example.aiapp.shell.NotificationService",
|
||||
static extern fn native_on_start_command(service: android.app.Service) -> jint,
|
||||
error_policy = LogErrorAndDefault,
|
||||
};
|
||||
|
||||
/// `NotificationService.nativeOnStartCommand`. See `notify::on_start_command`.
|
||||
fn native_on_start_command<'local>(
|
||||
env: &mut Env<'local>,
|
||||
_class: JClass<'local>,
|
||||
service: JObject<'local>,
|
||||
) -> Result<jint, jni::errors::Error> {
|
||||
ensure_logger();
|
||||
jcall::remember_class_loader(env, &service)?;
|
||||
Ok(notify::on_start_command(env, service))
|
||||
}
|
||||
|
||||
const _: NativeMethod = native_method! {
|
||||
java_type = "com.example.aiapp.shell.NotificationService",
|
||||
static extern fn native_on_destroy() -> (),
|
||||
error_policy = LogErrorAndDefault,
|
||||
};
|
||||
|
||||
/// `NotificationService.nativeOnDestroy`. See `notify::on_destroy`.
|
||||
fn native_on_destroy<'local>(
|
||||
_env: &mut Env<'local>,
|
||||
_class: JClass<'local>,
|
||||
) -> Result<(), jni::errors::Error> {
|
||||
ensure_logger();
|
||||
notify::on_destroy();
|
||||
Ok(())
|
||||
}
|
||||
@@ -0,0 +1,556 @@
|
||||
//! Where a notification is said, and the foreground service that keeps
|
||||
//! the connection open while the app is closed. Ported from
|
||||
//! `Notifications.kt`'s `NotificationService`, minus the "session on
|
||||
//! screen" / "hand to the app as a banner" branches: those read
|
||||
//! process-wide state that only exists because a screen is drawn to
|
||||
//! register against, and this experiment draws no screen yet (that is
|
||||
//! E4's job, on iris). So every notification here takes the third branch
|
||||
//! Kotlin's `show` already had -- the platform's own drawer -- which is
|
||||
//! also exactly the case E3's pass condition asks for: **a notification
|
||||
//! arrives with the app closed.**
|
||||
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::time::Duration;
|
||||
|
||||
use client_core::api::UreqTransport;
|
||||
use client_core::notifications::{SessionNotification, follow_notifications};
|
||||
use jni::Env;
|
||||
use jni::errors::Result;
|
||||
use jni::objects::{JObject, JValue};
|
||||
use jni::sys::{JNI_TRUE, jint};
|
||||
|
||||
use crate::settings::{self, ServerSettings};
|
||||
|
||||
const ALERT_CHANNEL: &str = "sessions";
|
||||
const ONGOING_CHANNEL: &str = "connection";
|
||||
const ONGOING_ID: i32 = 1;
|
||||
const ALERT_ID: i32 = 2;
|
||||
/// Same backoff as `Notifications.kt`'s `RECONNECT_DELAY_MS`.
|
||||
const RECONNECT_DELAY: Duration = Duration::from_millis(5_000);
|
||||
|
||||
/// Whether the follow-loop thread is already running. **A deviation from
|
||||
/// `Notifications.kt`, found by testing rather than planned**: the Kotlin
|
||||
/// `onStartCommand` spawns a fresh `thread(isDaemon = true) { follow(...) }`
|
||||
/// on *every* call, with nothing to notice a previous one is still going --
|
||||
/// and `sync()` calling `startForegroundService` when the service is
|
||||
/// already running is an ordinary Android start, not a restart, so
|
||||
/// `onStartCommand` runs again. Enrolling from `MainActivity` (which calls
|
||||
/// `sync` once itself, then again inside `handle_enrollment` after saving
|
||||
/// the token) hits exactly this path and was observed opening **two**
|
||||
/// concurrent connections to `/notifications` from one process -- caught
|
||||
/// on this build via `adb logcat` showing two `jni::vm::java_vm: Attached
|
||||
/// thread ai-app-notifications` lines for one enrollment. Guarded here
|
||||
/// rather than left to match Kotlin's behaviour exactly, since duplicating
|
||||
/// a live connection is a resource leak with no upside; worth carrying the
|
||||
/// same guard back to `Notifications.kt` separately.
|
||||
static RUNNING: AtomicBool = AtomicBool::new(false);
|
||||
|
||||
/// Set by `nativeOnDestroy`, checked by the follow loop between
|
||||
/// reconnects. **Known gap, recorded rather than hidden**: unlike
|
||||
/// `HttpURLConnection.disconnect()` in the Kotlin original, nothing here
|
||||
/// can interrupt a `ureq` read already blocked inside one connection --
|
||||
/// `Transport::stream` hands back a plain `Read` with no cancellation
|
||||
/// handle. So a stop lands at the next reconnect, not mid-read. `/notifications`
|
||||
/// is idle between events (a keep-alive, per `server/src/routes.rs`), so in
|
||||
/// practice this is a bounded wait rather than a hang; closing that gap
|
||||
/// for real means adding a cancellation point to `client_core::Transport`,
|
||||
/// which is a decision affecting every caller of that trait, not just this
|
||||
/// one -- left for whoever next depends on prompt shutdown.
|
||||
static STOPPING: AtomicBool = AtomicBool::new(false);
|
||||
|
||||
fn static_int(env: &mut Env, class: &str, field: &str) -> Result<i32> {
|
||||
crate::jcall::get_static_field(env, class, field, "I")?.i()
|
||||
}
|
||||
|
||||
fn notification_manager<'l>(env: &mut Env<'l>, context: &JObject) -> Result<JObject<'l>> {
|
||||
crate::jcall::call_static_method(
|
||||
env,
|
||||
"androidx/core/app/NotificationManagerCompat",
|
||||
"from",
|
||||
"(Landroid/content/Context;)Landroidx/core/app/NotificationManagerCompat;",
|
||||
&[JValue::Object(context)],
|
||||
)?
|
||||
.l()
|
||||
}
|
||||
|
||||
fn create_channel(
|
||||
env: &mut Env,
|
||||
manager: &JObject,
|
||||
id: &str,
|
||||
name: &str,
|
||||
importance: i32,
|
||||
) -> Result<()> {
|
||||
let id_j = crate::jcall::jstr_obj(env, id)?;
|
||||
let builder = crate::jcall::new_object(
|
||||
env,
|
||||
"androidx/core/app/NotificationChannelCompat$Builder",
|
||||
"(Ljava/lang/String;I)V",
|
||||
&[JValue::Object(&id_j), JValue::Int(importance)],
|
||||
)?;
|
||||
let name_j = crate::jcall::jstr_obj(env, name)?;
|
||||
crate::jcall::call_method(
|
||||
env,
|
||||
&builder,
|
||||
"setName",
|
||||
"(Ljava/lang/CharSequence;)Landroidx/core/app/NotificationChannelCompat$Builder;",
|
||||
&[JValue::Object(&name_j)],
|
||||
)?;
|
||||
let channel = crate::jcall::call_method(
|
||||
env,
|
||||
&builder,
|
||||
"build",
|
||||
"()Landroidx/core/app/NotificationChannelCompat;",
|
||||
&[],
|
||||
)?
|
||||
.l()?;
|
||||
crate::jcall::call_method(
|
||||
env,
|
||||
manager,
|
||||
"createNotificationChannel",
|
||||
"(Landroidx/core/app/NotificationChannelCompat;)V",
|
||||
&[JValue::Object(&channel)],
|
||||
)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Two channels, because they are two different things to be told -- see
|
||||
/// `Notifications.kt`'s `createChannels` for the reasoning; the names and
|
||||
/// importances here are copied from it exactly, since a phone that has
|
||||
/// seen both apps should not learn two different vocabularies for the
|
||||
/// same fact.
|
||||
fn create_channels(env: &mut Env, context: &JObject) -> Result<()> {
|
||||
let manager = notification_manager(env, context)?;
|
||||
let default = static_int(
|
||||
env,
|
||||
"androidx/core/app/NotificationManagerCompat",
|
||||
"IMPORTANCE_DEFAULT",
|
||||
)?;
|
||||
let min = static_int(
|
||||
env,
|
||||
"androidx/core/app/NotificationManagerCompat",
|
||||
"IMPORTANCE_MIN",
|
||||
)?;
|
||||
create_channel(
|
||||
env,
|
||||
&manager,
|
||||
ALERT_CHANNEL,
|
||||
"Sessions needing attention",
|
||||
default,
|
||||
)?;
|
||||
create_channel(env, &manager, ONGOING_CHANNEL, "Staying connected", min)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn new_intent_for<'l>(
|
||||
env: &mut Env<'l>,
|
||||
context: &JObject,
|
||||
class_name: &str,
|
||||
) -> Result<JObject<'l>> {
|
||||
let target_class = crate::jcall::find_class(env, class_name)?;
|
||||
crate::jcall::new_object(
|
||||
env,
|
||||
"android/content/Intent",
|
||||
"(Landroid/content/Context;Ljava/lang/Class;)V",
|
||||
&[JValue::Object(context), JValue::Object(&target_class)],
|
||||
)
|
||||
}
|
||||
|
||||
/// The intent a tap on an alert opens -- mirrors `Notifications.kt`'s
|
||||
/// `sessionIntent`, including building the URI through `Uri.Builder`
|
||||
/// rather than string concatenation, for the same reason: an id needing
|
||||
/// escaping must survive the round trip.
|
||||
fn session_intent<'l>(
|
||||
env: &mut Env<'l>,
|
||||
context: &JObject,
|
||||
session_id: &str,
|
||||
) -> Result<JObject<'l>> {
|
||||
let intent = new_intent_for(env, context, "com/example/aiapp/shell/MainActivity")?;
|
||||
let action_view = crate::jcall::jstr_obj(env, "android.intent.action.VIEW")?;
|
||||
crate::jcall::call_method(
|
||||
env,
|
||||
&intent,
|
||||
"setAction",
|
||||
"(Ljava/lang/String;)Landroid/content/Intent;",
|
||||
&[JValue::Object(&action_view)],
|
||||
)?;
|
||||
let builder = crate::jcall::new_object(env, "android/net/Uri$Builder", "()V", &[])?;
|
||||
let scheme = crate::jcall::jstr_obj(env, settings::SCHEME)?;
|
||||
crate::jcall::call_method(
|
||||
env,
|
||||
&builder,
|
||||
"scheme",
|
||||
"(Ljava/lang/String;)Landroid/net/Uri$Builder;",
|
||||
&[JValue::Object(&scheme)],
|
||||
)?;
|
||||
let authority = crate::jcall::jstr_obj(env, "session")?;
|
||||
crate::jcall::call_method(
|
||||
env,
|
||||
&builder,
|
||||
"authority",
|
||||
"(Ljava/lang/String;)Landroid/net/Uri$Builder;",
|
||||
&[JValue::Object(&authority)],
|
||||
)?;
|
||||
let path = crate::jcall::jstr_obj(env, session_id)?;
|
||||
crate::jcall::call_method(
|
||||
env,
|
||||
&builder,
|
||||
"appendPath",
|
||||
"(Ljava/lang/String;)Landroid/net/Uri$Builder;",
|
||||
&[JValue::Object(&path)],
|
||||
)?;
|
||||
let uri = crate::jcall::call_method(env, &builder, "build", "()Landroid/net/Uri;", &[])?.l()?;
|
||||
crate::jcall::call_method(
|
||||
env,
|
||||
&intent,
|
||||
"setData",
|
||||
"(Landroid/net/Uri;)Landroid/content/Intent;",
|
||||
&[JValue::Object(&uri)],
|
||||
)?;
|
||||
Ok(intent)
|
||||
}
|
||||
|
||||
fn pending_activity<'l>(
|
||||
env: &mut Env<'l>,
|
||||
context: &JObject,
|
||||
intent: &JObject,
|
||||
) -> Result<JObject<'l>> {
|
||||
let update_current = static_int(env, "android/app/PendingIntent", "FLAG_UPDATE_CURRENT")?;
|
||||
let immutable = static_int(env, "android/app/PendingIntent", "FLAG_IMMUTABLE")?;
|
||||
crate::jcall::call_static_method(
|
||||
env,
|
||||
"android/app/PendingIntent",
|
||||
"getActivity",
|
||||
"(Landroid/content/Context;ILandroid/content/Intent;I)Landroid/app/PendingIntent;",
|
||||
&[
|
||||
JValue::Object(context),
|
||||
JValue::Int(0),
|
||||
JValue::Object(intent),
|
||||
JValue::Int(update_current | immutable),
|
||||
],
|
||||
)?
|
||||
.l()
|
||||
}
|
||||
|
||||
fn builder_call<'l>(
|
||||
env: &mut Env<'l>,
|
||||
builder: &JObject<'l>,
|
||||
method: &str,
|
||||
sig: &str,
|
||||
args: &[JValue],
|
||||
) -> Result<()> {
|
||||
crate::jcall::call_method(env, builder, method, sig, args)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// The type Android 14+ requires a foreground service to declare, and
|
||||
/// nothing before it -- mirrors `Notifications.kt`'s `foregroundType`.
|
||||
fn foreground_type(env: &mut Env) -> Result<i32> {
|
||||
let sdk = static_int(env, "android/os/Build$VERSION", "SDK_INT")?;
|
||||
let upside_down_cake = static_int(env, "android/os/Build$VERSION_CODES", "UPSIDE_DOWN_CAKE")?;
|
||||
if sdk >= upside_down_cake {
|
||||
static_int(
|
||||
env,
|
||||
"android/content/pm/ServiceInfo",
|
||||
"FOREGROUND_SERVICE_TYPE_SPECIAL_USE",
|
||||
)
|
||||
} else {
|
||||
Ok(0)
|
||||
}
|
||||
}
|
||||
|
||||
fn ongoing_notification<'l>(env: &mut Env<'l>, context: &JObject) -> Result<JObject<'l>> {
|
||||
let channel = crate::jcall::jstr_obj(env, ONGOING_CHANNEL)?;
|
||||
let builder = crate::jcall::new_object(
|
||||
env,
|
||||
"androidx/core/app/NotificationCompat$Builder",
|
||||
"(Landroid/content/Context;Ljava/lang/String;)V",
|
||||
&[JValue::Object(context), JValue::Object(&channel)],
|
||||
)?;
|
||||
let title = crate::jcall::jstr_obj(env, "Watching for sessions that need you")?;
|
||||
builder_call(
|
||||
env,
|
||||
&builder,
|
||||
"setContentTitle",
|
||||
"(Ljava/lang/CharSequence;)Landroidx/core/app/NotificationCompat$Builder;",
|
||||
&[JValue::Object(&title)],
|
||||
)?;
|
||||
let icon = static_int(env, "android/R$drawable", "stat_notify_sync")?;
|
||||
builder_call(
|
||||
env,
|
||||
&builder,
|
||||
"setSmallIcon",
|
||||
"(I)Landroidx/core/app/NotificationCompat$Builder;",
|
||||
&[JValue::Int(icon)],
|
||||
)?;
|
||||
builder_call(
|
||||
env,
|
||||
&builder,
|
||||
"setOngoing",
|
||||
"(Z)Landroidx/core/app/NotificationCompat$Builder;",
|
||||
&[JValue::Bool(JNI_TRUE)],
|
||||
)?;
|
||||
let priority_min = static_int(env, "androidx/core/app/NotificationCompat", "PRIORITY_MIN")?;
|
||||
builder_call(
|
||||
env,
|
||||
&builder,
|
||||
"setPriority",
|
||||
"(I)Landroidx/core/app/NotificationCompat$Builder;",
|
||||
&[JValue::Int(priority_min)],
|
||||
)?;
|
||||
crate::jcall::call_method(env, &builder, "build", "()Landroid/app/Notification;", &[])?.l()
|
||||
}
|
||||
|
||||
/// Starts the service if there is a server to connect to, and stops it
|
||||
/// otherwise -- mirrors `Notifications.kt`'s `NotificationService.sync`.
|
||||
pub fn sync(env: &mut Env, context: &JObject) -> Result<()> {
|
||||
let service_intent =
|
||||
new_intent_for(env, context, "com/example/aiapp/shell/NotificationService")?;
|
||||
if settings::load(env, context)?.is_none() {
|
||||
crate::jcall::call_method(
|
||||
env,
|
||||
context,
|
||||
"stopService",
|
||||
"(Landroid/content/Intent;)Z",
|
||||
&[JValue::Object(&service_intent)],
|
||||
)?;
|
||||
return Ok(());
|
||||
}
|
||||
create_channels(env, context)?;
|
||||
crate::jcall::call_static_method(
|
||||
env,
|
||||
"androidx/core/content/ContextCompat",
|
||||
"startForegroundService",
|
||||
"(Landroid/content/Context;Landroid/content/Intent;)V",
|
||||
&[JValue::Object(context), JValue::Object(&service_intent)],
|
||||
)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// The `Service.onStartCommand` body -- loads settings, starts the
|
||||
/// foreground notification, and spawns the follow-loop thread. Answers the
|
||||
/// platform's `START_STICKY`/`START_NOT_STICKY` constant, read from the
|
||||
/// framework rather than hardcoded so a wrong guess at their values cannot
|
||||
/// silently pick the other behaviour.
|
||||
pub fn on_start_command(env: &mut Env, service: JObject) -> jint {
|
||||
match try_start(env, &service) {
|
||||
Ok(true) => static_int(env, "android/app/Service", "START_STICKY").unwrap_or(1),
|
||||
Ok(false) => {
|
||||
let _ = crate::jcall::call_method(env, &service, "stopSelf", "()V", &[]);
|
||||
static_int(env, "android/app/Service", "START_NOT_STICKY").unwrap_or(2)
|
||||
}
|
||||
Err(e) => {
|
||||
log_error(env, "onStartCommand", &e);
|
||||
static_int(env, "android/app/Service", "START_NOT_STICKY").unwrap_or(2)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn try_start(env: &mut Env, service: &JObject) -> Result<bool> {
|
||||
let Some(settings) = settings::load(env, service)? else {
|
||||
return Ok(false);
|
||||
};
|
||||
let ca = settings::load_pinned_ca(env)?;
|
||||
let notification = ongoing_notification(env, service)?;
|
||||
let fg_type = foreground_type(env)?;
|
||||
crate::jcall::call_static_method(
|
||||
env,
|
||||
"androidx/core/app/ServiceCompat",
|
||||
"startForeground",
|
||||
"(Landroid/app/Service;ILandroid/app/Notification;I)V",
|
||||
&[
|
||||
JValue::Object(service),
|
||||
JValue::Int(ONGOING_ID),
|
||||
JValue::Object(¬ification),
|
||||
JValue::Int(fg_type),
|
||||
],
|
||||
)?;
|
||||
|
||||
// See `RUNNING`'s doc: a second `onStartCommand` while the loop from
|
||||
// the first is still going -- the ordinary case for this service,
|
||||
// since `sync()` is called from more than one place -- must not open
|
||||
// a second connection.
|
||||
if RUNNING.swap(true, Ordering::SeqCst) {
|
||||
return Ok(true);
|
||||
}
|
||||
|
||||
let vm = env.get_java_vm()?;
|
||||
let context = env.new_global_ref(service)?;
|
||||
STOPPING.store(false, Ordering::SeqCst);
|
||||
std::thread::Builder::new()
|
||||
.name("ai-app-notifications".to_string())
|
||||
.spawn(move || {
|
||||
// Requests a *permanent* attachment (detached only when this thread
|
||||
// exits), matching the Kotlin original's `thread(isDaemon = true)`:
|
||||
// this is the long-lived follow loop, not a one-shot callback.
|
||||
let _: jni::errors::Result<()> = vm.attach_current_thread(|env| {
|
||||
follow_loop(env, &context, settings, &ca);
|
||||
Ok(())
|
||||
});
|
||||
})
|
||||
.ok();
|
||||
Ok(true)
|
||||
}
|
||||
|
||||
/// Follows the backend's notification stream, reconnecting until stopped
|
||||
/// -- mirrors `Notifications.kt`'s `follow`. A dropped connection is the
|
||||
/// ordinary case, so it retries quietly and forever; nothing is shown when
|
||||
/// it cannot connect, for the same reason as the Kotlin original: a
|
||||
/// notification saying "I could not tell you whether anything happened" is
|
||||
/// noise about a condition nobody can act on.
|
||||
fn follow_loop(env: &mut Env, context: &JObject, settings: ServerSettings, ca: &[u8]) {
|
||||
while !STOPPING.load(Ordering::SeqCst) {
|
||||
if let Ok(transport) = UreqTransport::new(settings.base_url(), settings.token.clone(), ca) {
|
||||
let _ = follow_notifications(&transport, |notification| {
|
||||
if let Err(e) = show(env, context, ¬ification) {
|
||||
log_error(env, "show", &e);
|
||||
}
|
||||
!STOPPING.load(Ordering::SeqCst)
|
||||
});
|
||||
}
|
||||
if STOPPING.load(Ordering::SeqCst) {
|
||||
return;
|
||||
}
|
||||
std::thread::sleep(RECONNECT_DELAY);
|
||||
}
|
||||
}
|
||||
|
||||
/// One notification per session, replacing that session's previous one --
|
||||
/// mirrors `Notifications.kt`'s `show`, minus the on-screen/banner
|
||||
/// branches this module's doc comment explains.
|
||||
fn show(env: &mut Env, context: &JObject, notification: &SessionNotification) -> Result<()> {
|
||||
let manager = notification_manager(env, context)?;
|
||||
let sdk = static_int(env, "android/os/Build$VERSION", "SDK_INT")?;
|
||||
let tiramisu = static_int(env, "android/os/Build$VERSION_CODES", "TIRAMISU")?;
|
||||
let allowed = if sdk < tiramisu {
|
||||
true
|
||||
} else {
|
||||
let permission = crate::jcall::jstr_obj(env, "android.permission.POST_NOTIFICATIONS")?;
|
||||
let granted = static_int(
|
||||
env,
|
||||
"android/content/pm/PackageManager",
|
||||
"PERMISSION_GRANTED",
|
||||
)?;
|
||||
let result = crate::jcall::call_static_method(
|
||||
env,
|
||||
"androidx/core/content/ContextCompat",
|
||||
"checkSelfPermission",
|
||||
"(Landroid/content/Context;Ljava/lang/String;)I",
|
||||
&[JValue::Object(context), JValue::Object(&permission)],
|
||||
)?
|
||||
.i()?;
|
||||
result == granted
|
||||
};
|
||||
let enabled =
|
||||
crate::jcall::call_method(env, &manager, "areNotificationsEnabled", "()Z", &[])?.z()?;
|
||||
if !allowed || !enabled {
|
||||
return Ok(());
|
||||
}
|
||||
let intent = session_intent(env, context, ¬ification.session_id)?;
|
||||
let pending = pending_activity(env, context, &intent)?;
|
||||
let channel = crate::jcall::jstr_obj(env, ALERT_CHANNEL)?;
|
||||
let builder = crate::jcall::new_object(
|
||||
env,
|
||||
"androidx/core/app/NotificationCompat$Builder",
|
||||
"(Landroid/content/Context;Ljava/lang/String;)V",
|
||||
&[JValue::Object(context), JValue::Object(&channel)],
|
||||
)?;
|
||||
let title = crate::jcall::jstr_obj(env, ¬ification.title)?;
|
||||
builder_call(
|
||||
env,
|
||||
&builder,
|
||||
"setContentTitle",
|
||||
"(Ljava/lang/CharSequence;)Landroidx/core/app/NotificationCompat$Builder;",
|
||||
&[JValue::Object(&title)],
|
||||
)?;
|
||||
let text = crate::jcall::jstr_obj(env, notification.kind.attention_line())?;
|
||||
builder_call(
|
||||
env,
|
||||
&builder,
|
||||
"setContentText",
|
||||
"(Ljava/lang/CharSequence;)Landroidx/core/app/NotificationCompat$Builder;",
|
||||
&[JValue::Object(&text)],
|
||||
)?;
|
||||
let icon = static_int(env, "android/R$drawable", "stat_notify_chat")?;
|
||||
builder_call(
|
||||
env,
|
||||
&builder,
|
||||
"setSmallIcon",
|
||||
"(I)Landroidx/core/app/NotificationCompat$Builder;",
|
||||
&[JValue::Int(icon)],
|
||||
)?;
|
||||
builder_call(
|
||||
env,
|
||||
&builder,
|
||||
"setContentIntent",
|
||||
"(Landroid/app/PendingIntent;)Landroidx/core/app/NotificationCompat$Builder;",
|
||||
&[JValue::Object(&pending)],
|
||||
)?;
|
||||
builder_call(
|
||||
env,
|
||||
&builder,
|
||||
"setAutoCancel",
|
||||
"(Z)Landroidx/core/app/NotificationCompat$Builder;",
|
||||
&[JValue::Bool(JNI_TRUE)],
|
||||
)?;
|
||||
let when = (notification.at * 1000.0) as i64;
|
||||
builder_call(
|
||||
env,
|
||||
&builder,
|
||||
"setWhen",
|
||||
"(J)Landroidx/core/app/NotificationCompat$Builder;",
|
||||
&[JValue::Long(when)],
|
||||
)?;
|
||||
builder_call(
|
||||
env,
|
||||
&builder,
|
||||
"setShowWhen",
|
||||
"(Z)Landroidx/core/app/NotificationCompat$Builder;",
|
||||
&[JValue::Bool(JNI_TRUE)],
|
||||
)?;
|
||||
let built =
|
||||
crate::jcall::call_method(env, &builder, "build", "()Landroid/app/Notification;", &[])?
|
||||
.l()?;
|
||||
let tag = crate::jcall::jstr_obj(env, ¬ification.session_id)?;
|
||||
crate::jcall::call_method(
|
||||
env,
|
||||
&manager,
|
||||
"notify",
|
||||
"(Ljava/lang/String;ILandroid/app/Notification;)V",
|
||||
&[
|
||||
JValue::Object(&tag),
|
||||
JValue::Int(ALERT_ID),
|
||||
JValue::Object(&built),
|
||||
],
|
||||
)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Ends the follow loop -- mirrors `Notifications.kt`'s `onDestroy`, with
|
||||
/// the gap this module's `STOPPING` doc explains.
|
||||
pub fn on_destroy() {
|
||||
STOPPING.store(true, Ordering::SeqCst);
|
||||
// `RUNNING`'s path out. Same race as `STOPPING` itself (this doc's own
|
||||
// comment): the old thread may still be inside a blocked read when a
|
||||
// new `onStartCommand` follows immediately, which would spawn a
|
||||
// second one before the first has actually stopped. Narrower than not
|
||||
// resetting at all -- a service destroyed and never restarted would
|
||||
// otherwise wedge `RUNNING` true forever -- and no worse than the
|
||||
// known gap already accepted above.
|
||||
RUNNING.store(false, Ordering::SeqCst);
|
||||
}
|
||||
|
||||
pub fn log_error(env: &mut Env, where_: &str, error: &jni::errors::Error) {
|
||||
let message = format!("android-shell: {where_}: {error}");
|
||||
let _ = (|| -> Result<()> {
|
||||
let tag = crate::jcall::jstr_obj(env, "android-shell")?;
|
||||
let msg = crate::jcall::jstr_obj(env, &message)?;
|
||||
crate::jcall::call_static_method(
|
||||
env,
|
||||
"android/util/Log",
|
||||
"e",
|
||||
"(Ljava/lang/String;Ljava/lang/String;)I",
|
||||
&[JValue::Object(&tag), JValue::Object(&msg)],
|
||||
)?;
|
||||
Ok(())
|
||||
})();
|
||||
}
|
||||
@@ -0,0 +1,141 @@
|
||||
//! Enrollment: where the backend is, and the Keystore-sealed token to
|
||||
//! reach it. This crate does not reimplement the Android Keystore AES-GCM
|
||||
//! sealing in Rust -- it calls the same `wg-app-link` `ServerStore` Kotlin
|
||||
//! class the production app already uses (see `ServerConfig.kt`), through
|
||||
//! JNI, for two reasons: that code is shared with Dev Updater and already
|
||||
//! tested, and the sealed value on a real phone is keyed to the exact
|
||||
//! Keystore alias that class already uses -- reimplementing the crypto
|
||||
//! here would either duplicate it or invalidate an existing enrollment.
|
||||
|
||||
use jni::Env;
|
||||
use jni::errors::Result;
|
||||
use jni::objects::{JObject, JString, JValue};
|
||||
|
||||
/// Where the backend is and how to authenticate to it -- the Rust twin of
|
||||
/// `wg-app-link`'s `ServerSettings` data class, read back field by field
|
||||
/// rather than kept as a live JNI reference, so it can cross a thread
|
||||
/// boundary (a `JObject` is tied to one `Env`/thread).
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct ServerSettings {
|
||||
pub host: String,
|
||||
pub port: i32,
|
||||
pub token: String,
|
||||
}
|
||||
|
||||
impl ServerSettings {
|
||||
pub fn base_url(&self) -> String {
|
||||
format!("https://{}:{}", self.host, self.port)
|
||||
}
|
||||
}
|
||||
|
||||
/// This experiment's own scheme and Keystore alias -- distinct from the
|
||||
/// production app's (`aiapp` / `aiapp-token-key`) so the two can be
|
||||
/// installed side by side on the same development device without
|
||||
/// colliding over which one a scanned QR or a deep link resolves to. See
|
||||
/// RUST.md's E3 entry for why they are not the same value.
|
||||
pub(crate) const SCHEME: &str = "aiappshell";
|
||||
const KEY_ALIAS: &str = "aiapp-shell-token-key";
|
||||
const STORE_CLASS: &str = "com/example/wgapplink/ServerStore";
|
||||
const SETTINGS_CLASS: &str = "com/example/wgapplink/ServerSettings";
|
||||
|
||||
fn new_store<'l>(env: &mut Env<'l>) -> Result<JObject<'l>> {
|
||||
let scheme = crate::jcall::jstr_obj(env, SCHEME)?;
|
||||
let alias = crate::jcall::jstr_obj(env, KEY_ALIAS)?;
|
||||
crate::jcall::new_object(
|
||||
env,
|
||||
STORE_CLASS,
|
||||
"(Ljava/lang/String;Ljava/lang/String;)V",
|
||||
&[JValue::Object(&scheme), JValue::Object(&alias)],
|
||||
)
|
||||
}
|
||||
|
||||
fn read_settings(env: &mut Env, settings_obj: &JObject) -> Result<ServerSettings> {
|
||||
let host = get_string(env, settings_obj, "getHost")?;
|
||||
let port = crate::jcall::call_method(env, settings_obj, "getPort", "()I", &[])?.i()?;
|
||||
let token = get_string(env, settings_obj, "getToken")?;
|
||||
Ok(ServerSettings { host, port, token })
|
||||
}
|
||||
|
||||
fn get_string(env: &mut Env, obj: &JObject, getter: &str) -> Result<String> {
|
||||
let value = crate::jcall::call_method(env, obj, getter, "()Ljava/lang/String;", &[])?.l()?;
|
||||
let jstr: JString = env.cast_local::<JString>(value)?;
|
||||
jstr.try_to_string(env)
|
||||
}
|
||||
|
||||
/// The stored enrollment, or `None` when there is not one -- mirrors
|
||||
/// `ServerConfig.kt`'s `loadServerSettings`.
|
||||
pub fn load(env: &mut Env, context: &JObject) -> Result<Option<ServerSettings>> {
|
||||
let store = new_store(env)?;
|
||||
let settings_obj = crate::jcall::call_method(
|
||||
env,
|
||||
&store,
|
||||
"load",
|
||||
"(Landroid/content/Context;)Lcom/example/wgapplink/ServerSettings;",
|
||||
&[JValue::Object(context)],
|
||||
)?
|
||||
.l()?;
|
||||
if settings_obj.is_null() {
|
||||
return Ok(None);
|
||||
}
|
||||
Ok(Some(read_settings(env, &settings_obj)?))
|
||||
}
|
||||
|
||||
/// Seals and stores `settings` -- mirrors `ServerConfig.kt`'s `saveServerSettings`.
|
||||
pub fn save(env: &mut Env, context: &JObject, settings: &ServerSettings) -> Result<()> {
|
||||
let store = new_store(env)?;
|
||||
let host = crate::jcall::jstr_obj(env, &settings.host)?;
|
||||
let token = crate::jcall::jstr_obj(env, &settings.token)?;
|
||||
let settings_obj = crate::jcall::new_object(
|
||||
env,
|
||||
SETTINGS_CLASS,
|
||||
"(Ljava/lang/String;ILjava/lang/String;)V",
|
||||
&[
|
||||
JValue::Object(&host),
|
||||
JValue::Int(settings.port),
|
||||
JValue::Object(&token),
|
||||
],
|
||||
)?;
|
||||
crate::jcall::call_method(
|
||||
env,
|
||||
&store,
|
||||
"save",
|
||||
"(Landroid/content/Context;Lcom/example/wgapplink/ServerSettings;)V",
|
||||
&[JValue::Object(context), JValue::Object(&settings_obj)],
|
||||
)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Parses an `aiappshell://enroll?...` URI -- mirrors `ServerConfig.kt`'s
|
||||
/// `parseEnrollmentUri`, asking the same Kotlin code that already owns the
|
||||
/// query-parameter rules rather than re-deriving them here.
|
||||
pub fn parse_enrollment_uri(env: &mut Env, uri: &JObject) -> Result<Option<ServerSettings>> {
|
||||
let store = new_store(env)?;
|
||||
let settings_obj = crate::jcall::call_method(
|
||||
env,
|
||||
&store,
|
||||
"parseEnrollmentUri",
|
||||
"(Landroid/net/Uri;)Lcom/example/wgapplink/ServerSettings;",
|
||||
&[JValue::Object(uri)],
|
||||
)?
|
||||
.l()?;
|
||||
if settings_obj.is_null() {
|
||||
return Ok(None);
|
||||
}
|
||||
Ok(Some(read_settings(env, &settings_obj)?))
|
||||
}
|
||||
|
||||
/// The CA this build pins, generated at build time the same way
|
||||
/// `androidApp`'s `generatePinnedCert` task does (see `build.gradle.kts`)
|
||||
/// but into a plain Java constant, since this module has no Kotlin of its
|
||||
/// own to generate into.
|
||||
pub fn load_pinned_ca(env: &mut Env) -> Result<Vec<u8>> {
|
||||
let value = crate::jcall::get_static_field(
|
||||
env,
|
||||
"com/example/aiapp/shell/PinnedCa",
|
||||
"PINNED_CA_PEM",
|
||||
"Ljava/lang/String;",
|
||||
)?
|
||||
.l()?;
|
||||
let jstr: JString = env.cast_local::<JString>(value)?;
|
||||
Ok(jstr.try_to_string(env)?.into_bytes())
|
||||
}
|
||||
@@ -0,0 +1,183 @@
|
||||
//! Deep links and the share sheet -- ported from `MainActivity.kt`'s
|
||||
//! `handleIntent`/`onNewIntent` and `Share.kt`'s `sharedContent`.
|
||||
//!
|
||||
//! **Scope cut, recorded rather than silent**: only shared *text*
|
||||
//! (`Intent.EXTRA_TEXT`) is attached to a session. `Attachments.kt`'s
|
||||
//! upload path -- `ContentResolver` reads of a shared file/photo URI,
|
||||
//! bitmap downscaling, EXIF rotation -- is real work of its own and is not
|
||||
//! ported here, because `client-core`'s `ApiClient` does not have the
|
||||
//! `/sessions/{id}/attachments` route yet either (see `CLIENT_CORE.md`'s
|
||||
//! "not covered" list). So `ACTION_SEND`/`ACTION_SEND_MULTIPLE` with a
|
||||
//! `content://` stream and no text falls through to a toast saying so,
|
||||
//! rather than silently doing nothing. Closing this gap is the same
|
||||
//! `client-core` work whichever caller needs it next.
|
||||
//!
|
||||
//! **Which session a share lands in** is also a placeholder: with no
|
||||
//! screen drawn yet (E4's job), there is no picker to ask, so this attaches
|
||||
//! to whichever session has the latest `last_activity` -- the one most
|
||||
//! likely to be what somebody meant. Worth revisiting once a real screen
|
||||
//! exists to ask instead of guessing.
|
||||
|
||||
use client_core::api::{ApiClient, UreqTransport};
|
||||
use jni::Env;
|
||||
use jni::errors::Result;
|
||||
use jni::objects::{JObject, JString, JValue};
|
||||
|
||||
use crate::notify;
|
||||
use crate::settings;
|
||||
|
||||
const ACTION_SEND: &str = "android.intent.action.SEND";
|
||||
const ACTION_SEND_MULTIPLE: &str = "android.intent.action.SEND_MULTIPLE";
|
||||
const ACTION_VIEW: &str = "android.intent.action.VIEW";
|
||||
const EXTRA_TEXT: &str = "android.intent.extra.TEXT";
|
||||
|
||||
fn get_string_method(env: &mut Env, obj: &JObject, method: &str) -> Result<Option<String>> {
|
||||
let value = crate::jcall::call_method(env, obj, method, "()Ljava/lang/String;", &[])?.l()?;
|
||||
if value.is_null() {
|
||||
return Ok(None);
|
||||
}
|
||||
let jstr: JString = env.cast_local::<JString>(value)?;
|
||||
Ok(Some(jstr.try_to_string(env)?))
|
||||
}
|
||||
|
||||
fn toast(env: &mut Env, context: &JObject, message: &str) -> Result<()> {
|
||||
let message = crate::jcall::jstr_obj(env, message)?;
|
||||
crate::jcall::call_static_method(
|
||||
env,
|
||||
"com/example/aiapp/shell/MainActivity",
|
||||
"toast",
|
||||
"(Landroid/content/Context;Ljava/lang/String;)V",
|
||||
&[JValue::Object(context), JValue::Object(&message)],
|
||||
)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// The one place an incoming intent is sorted into what it means -- mirrors
|
||||
/// `MainActivity.kt`'s `handleIntent`.
|
||||
pub fn handle_intent(env: &mut Env, activity: &JObject, intent: &JObject) -> Result<()> {
|
||||
let action = get_string_method(env, intent, "getAction")?;
|
||||
if matches!(
|
||||
action.as_deref(),
|
||||
Some(ACTION_SEND) | Some(ACTION_SEND_MULTIPLE)
|
||||
) {
|
||||
return handle_share(env, activity, intent);
|
||||
}
|
||||
if action.as_deref() != Some(ACTION_VIEW) {
|
||||
return Ok(());
|
||||
}
|
||||
let uri = crate::jcall::call_method(env, intent, "getData", "()Landroid/net/Uri;", &[])?.l()?;
|
||||
if uri.is_null() {
|
||||
return Ok(());
|
||||
}
|
||||
let scheme = get_string_method(env, &uri, "getScheme")?;
|
||||
if scheme.as_deref() != Some(settings::SCHEME) {
|
||||
return Ok(());
|
||||
}
|
||||
match get_string_method(env, &uri, "getHost")?.as_deref() {
|
||||
Some("session") => handle_session_open(env, activity, &uri),
|
||||
Some("enroll") => handle_enrollment(env, activity, &uri),
|
||||
_ => Ok(()),
|
||||
}
|
||||
}
|
||||
|
||||
fn handle_session_open(env: &mut Env, activity: &JObject, uri: &JObject) -> Result<()> {
|
||||
let Some(session_id) = get_string_method(env, uri, "getLastPathSegment")? else {
|
||||
return Ok(());
|
||||
};
|
||||
// There is no session screen yet (E4's job); the toast is this
|
||||
// experiment's stand-in proof that the tap was routed to the right
|
||||
// session id.
|
||||
toast(env, activity, &format!("Opened session {session_id}"))
|
||||
}
|
||||
|
||||
fn handle_enrollment(env: &mut Env, activity: &JObject, uri: &JObject) -> Result<()> {
|
||||
match settings::parse_enrollment_uri(env, uri)? {
|
||||
Some(parsed) => {
|
||||
settings::save(env, activity, &parsed)?;
|
||||
notify::sync(env, activity)?;
|
||||
toast(
|
||||
env,
|
||||
activity,
|
||||
&format!("Enrolled with {}", parsed.base_url()),
|
||||
)
|
||||
}
|
||||
None => toast(env, activity, "Not a valid enrollment code"),
|
||||
}
|
||||
}
|
||||
|
||||
/// The share sheet -- mirrors `Share.kt`'s `sharedContent` for what counts
|
||||
/// as a share, and `AttachmentButton`'s upload-then-message pattern for
|
||||
/// what happens to it, minus attachments per this module's doc comment.
|
||||
fn handle_share(env: &mut Env, activity: &JObject, intent: &JObject) -> Result<()> {
|
||||
let extra_text = crate::jcall::jstr_obj(env, EXTRA_TEXT)?;
|
||||
let text = crate::jcall::call_method(
|
||||
env,
|
||||
intent,
|
||||
"getStringExtra",
|
||||
"(Ljava/lang/String;)Ljava/lang/String;",
|
||||
&[JValue::Object(&extra_text)],
|
||||
)?
|
||||
.l()?;
|
||||
let text = if text.is_null() {
|
||||
None
|
||||
} else {
|
||||
let jstr: JString = env.cast_local::<JString>(text)?;
|
||||
Some(jstr.try_to_string(env)?)
|
||||
};
|
||||
let Some(text) = text.filter(|t| !t.trim().is_empty()) else {
|
||||
return toast(
|
||||
env,
|
||||
activity,
|
||||
"Nothing to share -- only shared text is supported so far",
|
||||
);
|
||||
};
|
||||
|
||||
// Network I/O must not run on the calling thread: `handle_intent` is
|
||||
// called from `onCreate`/`onNewIntent`, both on the main thread, and a
|
||||
// blocking socket read there is a `NetworkOnMainThreadException`. So
|
||||
// the actual send happens on a JNI-attached background thread, the
|
||||
// same shape `notify::try_start`'s follow loop uses; `toast` from that
|
||||
// thread is safe because `MainActivity.toast` itself hops back to the
|
||||
// main looper (see that method).
|
||||
let vm = env.get_java_vm()?;
|
||||
let activity_ref = env.new_global_ref(activity)?;
|
||||
std::thread::spawn(move || {
|
||||
let _: jni::errors::Result<()> = vm.attach_current_thread(|env| {
|
||||
share_in_background(env, &activity_ref, text);
|
||||
Ok(())
|
||||
});
|
||||
});
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn share_in_background(env: &mut Env, activity: &JObject, text: String) {
|
||||
let outcome = attach_to_a_session(env, activity, &text);
|
||||
let message = match outcome {
|
||||
Ok(title) => format!("Shared into \"{title}\""),
|
||||
Err(message) => message,
|
||||
};
|
||||
let _ = toast(env, activity, &message);
|
||||
}
|
||||
|
||||
fn attach_to_a_session(
|
||||
env: &mut Env,
|
||||
activity: &JObject,
|
||||
text: &str,
|
||||
) -> std::result::Result<String, String> {
|
||||
let settings = settings::load(env, activity)
|
||||
.map_err(|e| e.to_string())?
|
||||
.ok_or_else(|| "Not enrolled yet".to_string())?;
|
||||
let ca = settings::load_pinned_ca(env).map_err(|e| e.to_string())?;
|
||||
let transport = UreqTransport::new(settings.base_url(), settings.token.clone(), &ca)
|
||||
.map_err(|e| e.to_string())?;
|
||||
let client = ApiClient::new(transport);
|
||||
let sessions = client.fetch_sessions().map_err(|e| e.to_string())?;
|
||||
let target = sessions
|
||||
.into_iter()
|
||||
.max_by(|a, b| a.last_activity.total_cmp(&b.last_activity))
|
||||
.ok_or_else(|| "No session to share into".to_string())?;
|
||||
client
|
||||
.send_message(&target.id, text, &[])
|
||||
.map_err(|e| e.to_string())?;
|
||||
Ok(target.title)
|
||||
}
|
||||
@@ -102,6 +102,16 @@ android {
|
||||
targetSdk = 37
|
||||
versionCode = 1
|
||||
versionName = "1.0"
|
||||
// Read by MainActivity to decide, at startup, whether this is the P0 benchmark build
|
||||
// (docs/RUST.md's P0 box) rather than the app somebody enrolled. False everywhere except
|
||||
// the `bench` build type below, which overrides it.
|
||||
buildConfigField("boolean", "FIXTURE_MODE", "false")
|
||||
}
|
||||
buildFeatures {
|
||||
// Only for FIXTURE_MODE above; nothing else here reaches for generated BuildConfig fields.
|
||||
buildConfig = true
|
||||
// Only for the bench build type's resValue("string", "app_name", ...) below.
|
||||
resValues = true
|
||||
}
|
||||
packaging {
|
||||
resources { excludes += "/META-INF/{AL2.0,LGPL2.1}" }
|
||||
@@ -131,6 +141,35 @@ android {
|
||||
isMinifyEnabled = false
|
||||
if (keystore != null) signingConfig = signingConfigs.getByName("release")
|
||||
}
|
||||
// P0's benchmark build (docs/RUST.md, docs/DECISIONS.md's 2026-09-05 entry): release
|
||||
// optimisations so a frame time measured here means what release means everywhere else in
|
||||
// this project, its own application id so it installs beside a real enrollment rather than
|
||||
// replacing it, and FIXTURE_MODE so MainActivity opens straight onto the fixture session
|
||||
// instead of asking to be enrolled. Signed with the same key as release -- it never talks
|
||||
// to a real backend, so there is no CA of its own to mismatch, and a second keystore would
|
||||
// be one more secret to keep off this machine's shared mount for no benefit.
|
||||
create("bench") {
|
||||
initWith(getByName("release"))
|
||||
// :link (wg-app-link) has no "bench" build type of its own -- it is a library shared
|
||||
// with dev-updater and has no reason to know this project invented one -- so this says
|
||||
// which of its build types to link against instead.
|
||||
matchingFallbacks += listOf("release")
|
||||
applicationIdSuffix = ".bench"
|
||||
// "AI Sessions bench" everywhere the OS shows the app's name (launcher, recents,
|
||||
// Settings): this resValue overrides res/values/strings.xml's app_name for this
|
||||
// build type alone, and AndroidManifest.xml's android:label reads @string/app_name
|
||||
// rather than a literal so a build type can override it without touching the
|
||||
// manifest.
|
||||
resValue("string", "app_name", "AI Sessions bench")
|
||||
buildConfigField("boolean", "FIXTURE_MODE", "true")
|
||||
if (keystore != null) signingConfig = signingConfigs.getByName("release")
|
||||
}
|
||||
}
|
||||
sourceSets {
|
||||
// The fixture both bench builds (this one and iris's) open with; see
|
||||
// app/bench-fixture/README.md. Read directly from its own directory rather than copied
|
||||
// into androidApp/src -- one file to keep in sync with the generator, not two.
|
||||
getByName("bench").assets.directories.add("../bench-fixture/assets")
|
||||
}
|
||||
compileOptions {
|
||||
sourceCompatibility = JavaVersion.VERSION_21
|
||||
|
||||
@@ -25,7 +25,7 @@
|
||||
the fix is a judgement about how this app should look. Drop this
|
||||
suppression when a real icon lands. -->
|
||||
<application
|
||||
android:label="AI Sessions"
|
||||
android:label="@string/app_name"
|
||||
android:allowBackup="true"
|
||||
android:theme="@android:style/Theme.Material.Light.NoActionBar"
|
||||
tools:ignore="MissingApplicationIcon">
|
||||
|
||||
@@ -142,6 +142,20 @@ data class SessionSummary(
|
||||
val keepsOwnTranscript: Boolean,
|
||||
/** How much the session asks before acting; null when it was never set. */
|
||||
val permissionMode: String?,
|
||||
/**
|
||||
* How hard the model thinks, or null for the CLI's own default.
|
||||
*
|
||||
* Null is a level somebody can choose, not only one to start in -- see [EFFORT_LEVELS]. It is
|
||||
* reported rather than assumed for the same reason [permissionMode] is.
|
||||
*/
|
||||
val effort: String?,
|
||||
/**
|
||||
* Whether a thinking level does anything here -- a Claude CLI session, not a llama or echo one.
|
||||
*
|
||||
* Asked of the server rather than worked out from the provider's name, because this is a
|
||||
* property of the driver's *kind* and the phone only has the name.
|
||||
*/
|
||||
val takesEffort: Boolean,
|
||||
/**
|
||||
* Whether this continues a session the machine already had, which changes what deleting means.
|
||||
*/
|
||||
@@ -153,6 +167,24 @@ data class SessionSummary(
|
||||
* itself from a default is one you can turn off while believing you are reading it.
|
||||
*/
|
||||
val notify: Boolean,
|
||||
/**
|
||||
* Whether this session sends itself a message once its account's usage limit lifts, and what
|
||||
* that message says.
|
||||
*
|
||||
* The message is what the server would actually send, with its own default already filled in,
|
||||
* so the field shows the words rather than an empty box standing for them.
|
||||
*/
|
||||
val autoResume: Boolean,
|
||||
val autoResumeMessage: String,
|
||||
/**
|
||||
* When the server next intends to check whether the limit has lifted, in epoch seconds, or null
|
||||
* when nothing is waiting.
|
||||
*
|
||||
* A time to *ask*, not a time to resume: the server checks the meter at that moment and waits
|
||||
* again if the limit is still on. Worded that way wherever it is shown, because a promise this
|
||||
* app cannot keep is worse than no time at all.
|
||||
*/
|
||||
val resumeAt: Double?,
|
||||
/**
|
||||
* The directory the session works in, or null where it was never given one.
|
||||
*
|
||||
@@ -178,8 +210,27 @@ data class SessionSummary(
|
||||
* server because that is where a provider's kind is known.
|
||||
*/
|
||||
val maxImageEdge: Int?,
|
||||
/**
|
||||
* Which of `GET /usage`'s snapshots is about this session, and null where nothing meters it.
|
||||
*
|
||||
* The rate-limit bar answers a question about an *account*, and what decides which account --
|
||||
* if any -- is the provider this session runs, not the machine it runs on. Pairing by machine
|
||||
* alone drew the Claude CLI's five-hour window under every echo session on a machine that also
|
||||
* has the CLI: a quota that session cannot spend and could never run down. Decided by the
|
||||
* server for the same reason [maxImageEdge] is -- it is a fact about the provider's kind, and
|
||||
* this app has only its name.
|
||||
*/
|
||||
val usageProvider: String?,
|
||||
val status: String,
|
||||
val lastActivity: Double,
|
||||
/**
|
||||
* How many subagents this session has, however their own status now reads.
|
||||
*
|
||||
* A directory listing on the server rather than a status read per subagent, so the list stays
|
||||
* cheap; the per-subagent state is only fetched when the card is expanded. Zero on a server
|
||||
* that predates subagents, so this app still opens against one.
|
||||
*/
|
||||
val subagents: Int,
|
||||
)
|
||||
|
||||
private fun parseSession(session: JSONObject) =
|
||||
@@ -192,14 +243,24 @@ private fun parseSession(session: JSONObject) =
|
||||
title = session.getString("title"),
|
||||
model = session.optString("model").ifEmpty { null },
|
||||
permissionMode = session.optString("permissionMode").ifEmpty { null },
|
||||
effort = session.optString("effort").ifEmpty { null },
|
||||
takesEffort = session.optBoolean("takesEffort", false),
|
||||
imported = session.optBoolean("imported", false),
|
||||
notify = session.optBoolean("notify", true),
|
||||
autoResume = session.optBoolean("autoResume", false),
|
||||
// The server sends its own default rather than nothing, so an empty answer means an older
|
||||
// server -- and this app's word for it is the same word.
|
||||
autoResumeMessage =
|
||||
session.optString("autoResumeMessage").ifEmpty { DEFAULT_RESUME_MESSAGE },
|
||||
resumeAt = if (session.has("resumeAt")) session.getDouble("resumeAt") else null,
|
||||
cwd = session.optString("cwd").ifEmpty { null },
|
||||
contextTokens =
|
||||
if (session.has("contextTokens")) session.getLong("contextTokens") else null,
|
||||
maxImageEdge = session.optInt("maxImageEdge", 0).takeIf { it > 0 },
|
||||
usageProvider = session.optString("usageProvider").ifEmpty { null },
|
||||
status = session.getString("status"),
|
||||
lastActivity = session.getDouble("lastActivity"),
|
||||
subagents = session.optInt("subagents", 0),
|
||||
)
|
||||
|
||||
fun fetchSessions(settings: ServerSettings): List<SessionSummary> =
|
||||
@@ -215,6 +276,35 @@ fun fetchSessions(settings: ServerSettings): List<SessionSummary> =
|
||||
fun fetchSession(settings: ServerSettings, sessionId: String): SessionSummary =
|
||||
requestFromServer(settings, "/sessions/$sessionId") { parseSession(it.jsonObject()) }
|
||||
|
||||
/**
|
||||
* One row of `GET /sessions/{id}/subagents`, oldest first.
|
||||
*
|
||||
* A subagent is a second transcript owned by a session -- no process, no controls of its own -- so
|
||||
* this carries only what a card needs to draw and to open it; see SUBAGENTS.md. [status] is
|
||||
* "running", "exited" or "unknown": a subagent whose session is not itself running cannot be
|
||||
* running, and the list says so rather than reporting a state that cannot hold.
|
||||
*/
|
||||
data class SubagentSummary(
|
||||
val id: String,
|
||||
val title: String,
|
||||
val status: String,
|
||||
val created: Double,
|
||||
val lastActivity: Double,
|
||||
)
|
||||
|
||||
fun fetchSubagents(settings: ServerSettings, sessionId: String): List<SubagentSummary> =
|
||||
requestFromServer(settings, "/sessions/$sessionId/subagents") {
|
||||
it.jsonObjects { row ->
|
||||
SubagentSummary(
|
||||
id = row.getString("id"),
|
||||
title = row.getString("title"),
|
||||
status = row.getString("status"),
|
||||
created = row.getDouble("created"),
|
||||
lastActivity = row.getDouble("lastActivity"),
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// What the server offers, so the spawn screen has no hardcoded lists: a setup added to the server's
|
||||
// config.ron appears here with no app rebuild.
|
||||
//
|
||||
@@ -375,6 +465,12 @@ data class SshDetails(
|
||||
* Where files attached from here land on that machine; null for the session's own directory.
|
||||
*/
|
||||
val attachmentsDir: String? = null,
|
||||
/**
|
||||
* Where that machine keeps its GGUF models; null for the same place the backend keeps its own
|
||||
* (`~/.local/share/ai-app/models`, read on that machine). A llama.cpp session serves the file
|
||||
* from the machine it runs on, so this is where its models are looked for and listed.
|
||||
*/
|
||||
val modelsDir: String? = null,
|
||||
)
|
||||
|
||||
private fun SshDetails.toJson() =
|
||||
@@ -382,6 +478,7 @@ private fun SshDetails.toJson() =
|
||||
if (port != null) put("port", port)
|
||||
if (!identityFile.isNullOrBlank()) put("identityFile", identityFile)
|
||||
if (!attachmentsDir.isNullOrBlank()) put("attachmentsDir", attachmentsDir)
|
||||
if (!modelsDir.isNullOrBlank()) put("modelsDir", modelsDir)
|
||||
}
|
||||
|
||||
/** What a machine turns out to have, without saving anything. */
|
||||
@@ -450,6 +547,8 @@ fun spawnSession(
|
||||
model: String? = null,
|
||||
cwd: String? = null,
|
||||
permissionMode: String? = null,
|
||||
/** Null for whatever the server's default is; see [fetchDefaultEffort]. */
|
||||
effort: String? = null,
|
||||
params: Map<String, String> = emptyMap(),
|
||||
/** Continue this Claude Code session instead of starting an empty one. */
|
||||
import: String? = null,
|
||||
@@ -467,6 +566,7 @@ fun spawnSession(
|
||||
if (!model.isNullOrBlank()) put("model", model)
|
||||
if (!cwd.isNullOrBlank()) put("cwd", cwd)
|
||||
if (!permissionMode.isNullOrBlank()) put("permissionMode", permissionMode)
|
||||
if (!effort.isNullOrBlank()) put("effort", effort)
|
||||
if (!import.isNullOrBlank()) put("import", import)
|
||||
if (params.isNotEmpty()) {
|
||||
put("params", JSONObject(params.toMap<String, Any>()))
|
||||
@@ -906,7 +1006,7 @@ fun startImport(
|
||||
*/
|
||||
fun fetchTranscript(
|
||||
settings: ServerSettings,
|
||||
sessionId: String,
|
||||
address: TranscriptAddress,
|
||||
before: Long? = null,
|
||||
limit: Int = 80,
|
||||
// Count [limit] in rows, not events, joining a reply's streamed deltas into one -- so a page of
|
||||
@@ -925,7 +1025,7 @@ fun fetchTranscript(
|
||||
if (coalesce) append("&coalesce=true")
|
||||
if (after != null) append("&after=").append(after)
|
||||
}
|
||||
return requestFromServer(settings, "/sessions/$sessionId/transcript$query") { connection ->
|
||||
return requestFromServer(settings, "/${address.urlPath}/transcript$query") { connection ->
|
||||
val body = JSONArray(connection.inputStream.bufferedReader().readText())
|
||||
// The text as well as the event: the transcript cache stores the one and the fold needs the
|
||||
// other, and they have to be the same line.
|
||||
@@ -973,6 +1073,56 @@ fun setSessionModel(settings: ServerSettings, sessionId: String, model: String)
|
||||
*/
|
||||
val PERMISSION_MODES = listOf("manual", "acceptEdits", "auto", "bypassPermissions", "plan")
|
||||
|
||||
/**
|
||||
* What a new session's thinking level is when nothing chose one, or null for the CLI's own.
|
||||
*
|
||||
* Held by the server rather than by this phone, because a second device would otherwise spawn
|
||||
* sessions at a level the first one's owner never picked.
|
||||
*/
|
||||
fun fetchDefaultEffort(settings: ServerSettings): String? =
|
||||
requestFromServer(settings, "/defaults") {
|
||||
it.jsonObject().optString("effort").ifEmpty { null }
|
||||
}
|
||||
|
||||
/** Sets what new sessions start at. Nothing already running changes. */
|
||||
fun setDefaultEffort(settings: ServerSettings, level: String?) {
|
||||
requestFromServer(
|
||||
settings,
|
||||
"/defaults",
|
||||
method = "POST",
|
||||
jsonBody = JSONObject().put("effort", level ?: JSONObject.NULL).toString(),
|
||||
) {}
|
||||
}
|
||||
|
||||
/**
|
||||
* How hard the model thinks, as `claude --effort` takes them, cheapest first.
|
||||
*
|
||||
* Not offered alongside the model and the permission mode on the session's own bar, because it does
|
||||
* not behave like them: the CLI has a control request for those two and none for this (checked
|
||||
* against 2.1.258), so a level is settled when the process is launched. Changing it therefore stops
|
||||
* the process, which is what the working directory beside it in this dialog does, and why it is
|
||||
* here rather than on a bar whose other controls take effect mid-turn.
|
||||
*/
|
||||
val EFFORT_LEVELS = listOf("low", "medium", "high", "xhigh", "max")
|
||||
|
||||
/** What the picker shows, and sends as null, for a session that has chosen no level. */
|
||||
const val DEFAULT_EFFORT = "default"
|
||||
|
||||
/**
|
||||
* Records how hard a session thinks and **stops its process**, since the level is read when the
|
||||
* process is launched. The next message, or Start, runs one that has it.
|
||||
*
|
||||
* [level] is null for the CLI's own default.
|
||||
*/
|
||||
fun setSessionEffort(settings: ServerSettings, sessionId: String, level: String?) {
|
||||
requestFromServer(
|
||||
settings,
|
||||
"/sessions/$sessionId/effort",
|
||||
method = "POST",
|
||||
jsonBody = JSONObject().put("effort", level ?: JSONObject.NULL).toString(),
|
||||
) {}
|
||||
}
|
||||
|
||||
/** Switches how much a running session asks before acting, also in place. */
|
||||
fun setSessionPermissionMode(settings: ServerSettings, sessionId: String, mode: String) {
|
||||
requestFromServer(
|
||||
@@ -984,6 +1134,36 @@ fun setSessionPermissionMode(settings: ServerSettings, sessionId: String, mode:
|
||||
}
|
||||
|
||||
/** Turns this session's notifications on or off. Stored on the backend -- see `SessionConfig`. */
|
||||
/**
|
||||
* What an auto-resume says when nothing else was typed. Mirrors the server's own default, so a
|
||||
* cleared field shows the word that would actually be sent instead of going blank.
|
||||
*/
|
||||
const val DEFAULT_RESUME_MESSAGE = "continue"
|
||||
|
||||
/**
|
||||
* Turns auto-resume on or off and sets what it would say, in one request because they are one
|
||||
* decision -- see the server's `/sessions/{id}/auto-resume`.
|
||||
*/
|
||||
fun setSessionAutoResume(
|
||||
settings: ServerSettings,
|
||||
sessionId: String,
|
||||
autoResume: Boolean,
|
||||
message: String?,
|
||||
) {
|
||||
requestFromServer(
|
||||
settings,
|
||||
"/sessions/$sessionId/auto-resume",
|
||||
method = "POST",
|
||||
jsonBody =
|
||||
JSONObject()
|
||||
.put("autoResume", autoResume)
|
||||
// Empty means the server's default rather than a session poked with nothing to
|
||||
// read, which is the same rule the server applies to the field.
|
||||
.put("message", message?.trim()?.ifEmpty { null } ?: JSONObject.NULL)
|
||||
.toString(),
|
||||
) {}
|
||||
}
|
||||
|
||||
fun setSessionNotify(settings: ServerSettings, sessionId: String, notify: Boolean) {
|
||||
requestFromServer(
|
||||
settings,
|
||||
@@ -1074,6 +1254,26 @@ private fun parseDownload(o: JSONObject) =
|
||||
error = if (o.has("error")) o.getString("error") else null,
|
||||
)
|
||||
|
||||
/**
|
||||
* The models on one machine, which is the list a llama.cpp session there can choose from.
|
||||
*
|
||||
* Not [fetchModels], which is what the *backend* has downloaded. A session serves its model from
|
||||
* the machine it runs on, so for a machine reached over ssh those are two different lists -- and
|
||||
* offering the backend's would name files that are not there, turning a choice that cannot work
|
||||
* into a session that fails when it tries to load one.
|
||||
*/
|
||||
fun fetchSetupModels(settings: ServerSettings, setupId: String): List<LocalModel> =
|
||||
requestFromServer(settings, "/setups/${setupId.urlEncoded()}/models") { connection ->
|
||||
JSONArray(connection.inputStream.bufferedReader().readText()).mapObjects { m ->
|
||||
LocalModel(
|
||||
key = m.getString("key"),
|
||||
repo = m.getString("repo"),
|
||||
file = m.getString("file"),
|
||||
bytes = m.getLong("bytes"),
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
fun fetchModels(settings: ServerSettings): Models =
|
||||
requestFromServer(settings, "/models") { connection ->
|
||||
val body = JSONObject(connection.inputStream.bufferedReader().readText())
|
||||
|
||||
@@ -1,7 +1,9 @@
|
||||
package com.example.aiapp
|
||||
|
||||
import androidx.activity.compose.BackHandler
|
||||
import androidx.compose.foundation.background
|
||||
import androidx.compose.foundation.layout.Box
|
||||
import androidx.compose.foundation.layout.fillMaxSize
|
||||
import androidx.compose.foundation.layout.imePadding
|
||||
import androidx.compose.foundation.layout.padding
|
||||
import androidx.compose.material3.AlertDialog
|
||||
@@ -34,7 +36,23 @@ import kotlinx.coroutines.withContext
|
||||
* session, spawning one, and settings.
|
||||
*/
|
||||
private sealed class Screen {
|
||||
data object Main : Screen()
|
||||
/**
|
||||
* The session list, with a subagent's own transcript over it when [subagent] is set.
|
||||
*
|
||||
* A layer on this screen rather than a screen of its own, for the same reason [Session.files]
|
||||
* is: [SessionListScreen] owns which cards are expanded and what each expansion fetched, kept
|
||||
* in `remember`, and a subagent is opened from a card's expander. As a sibling `Screen` it was
|
||||
* disposed and recreated on every return, which lost that state -- an expanded card collapsed
|
||||
* itself the moment its own subagent's view was closed.
|
||||
*/
|
||||
data class Main(val subagent: SubagentTarget? = null) : Screen()
|
||||
|
||||
/**
|
||||
* One subagent's own transcript, read-only. See [SessionScreen]'s `subagent` parameter and
|
||||
* SUBAGENTS.md's "Phone". Closing it returns to [Main] under it, not to [Session]: a subagent
|
||||
* is opened from the session list's card rather than from inside the session it belongs to.
|
||||
*/
|
||||
data class SubagentTarget(val summary: SessionSummary, val subagent: SubagentSummary)
|
||||
|
||||
/**
|
||||
* One session, with the file explorer over it when [files] is set.
|
||||
@@ -81,7 +99,7 @@ fun AppRoot(
|
||||
val context = LocalContext.current
|
||||
val scope = rememberCoroutineScope()
|
||||
var settings by remember(settingsVersion) { mutableStateOf(loadServerSettings(context)) }
|
||||
var screen by remember { mutableStateOf<Screen>(Screen.Main) }
|
||||
var screen by remember { mutableStateOf<Screen>(Screen.Main()) }
|
||||
// A notification tap this could not follow, and why. Null both before one is asked for and
|
||||
// after one succeeds, since success is a screen rather than a message.
|
||||
var failedOpen by remember { mutableStateOf<FailedOpen?>(null) }
|
||||
@@ -96,7 +114,7 @@ fun AppRoot(
|
||||
share = shareRequest
|
||||
// A session already open takes it. Otherwise the list is where the choice is made,
|
||||
// whatever screen was showing: Spawn and Settings have nowhere to put a file.
|
||||
if (screen !is Screen.Session) screen = Screen.Main
|
||||
if (screen !is Screen.Session) screen = Screen.Main()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -123,7 +141,7 @@ fun AppRoot(
|
||||
existing = null,
|
||||
onSaved = { saved ->
|
||||
settings = saved
|
||||
screen = Screen.Main
|
||||
screen = Screen.Main()
|
||||
},
|
||||
onBack = null,
|
||||
)
|
||||
@@ -136,7 +154,7 @@ fun AppRoot(
|
||||
// shows, so it always refetches.
|
||||
val goToMain = {
|
||||
reloadToken++
|
||||
screen = Screen.Main
|
||||
screen = Screen.Main()
|
||||
}
|
||||
if (screen !is Screen.Main) {
|
||||
BackHandler(onBack = goToMain)
|
||||
@@ -185,6 +203,9 @@ fun AppRoot(
|
||||
reloadToken = reloadToken,
|
||||
share = share,
|
||||
onOpen = { screen = Screen.Session(it) },
|
||||
onOpenSubagent = { summary, subagent ->
|
||||
screen = here.copy(subagent = Screen.SubagentTarget(summary, subagent))
|
||||
},
|
||||
onSpawn = { screen = Screen.Spawn },
|
||||
onImported = { imported ->
|
||||
reloadToken++
|
||||
@@ -192,6 +213,27 @@ fun AppRoot(
|
||||
},
|
||||
onSettings = { screen = Screen.Settings },
|
||||
)
|
||||
// Its own back handler is registered after MainScreen's, so it is the one the
|
||||
// platform asks first while a subagent is open -- the same rule the files
|
||||
// explorer's handler follows over its session, below.
|
||||
here.subagent?.let { target ->
|
||||
BackHandler { screen = here.copy(subagent = null) }
|
||||
// Its own opaque background: this screen was always the sole content under
|
||||
// the theme's own Surface before, so it never had to paint one -- stacked over
|
||||
// the list here, the space between its own cards let the list underneath show
|
||||
// through without this. The same fix FilesScreen needed over its session.
|
||||
Box(Modifier.fillMaxSize().background(MaterialTheme.colorScheme.background)) {
|
||||
key(target.summary.id, target.subagent.id) {
|
||||
SessionScreen(
|
||||
settings = current,
|
||||
summary = target.summary,
|
||||
onBack = { screen = here.copy(subagent = null) },
|
||||
onFiles = {},
|
||||
subagent = target.subagent,
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
is Screen.Session ->
|
||||
// Keyed on the id, because a different session is a different screen rather than this
|
||||
|
||||
@@ -0,0 +1,108 @@
|
||||
package com.example.aiapp
|
||||
|
||||
import android.content.Context
|
||||
import java.util.concurrent.CopyOnWriteArrayList
|
||||
|
||||
/**
|
||||
* P0's benchmark gate (see docs/RUST.md and docs/DECISIONS.md's 2026-09-05 entry): an in-process
|
||||
* fake of the backend, so the `bench` build type can drive a real session screen -- the real
|
||||
* [TranscriptSource], the real fold, the real paging -- with no server and no network permission.
|
||||
*
|
||||
* Only ever installed when [BuildConfig.FIXTURE_MODE] is true (see [MainActivity]); everything else
|
||||
* in this build compiles it in but never calls it, since Kotlin has no per-build-type source set
|
||||
* that both [MainActivity] (which every variant compiles) and this can share without one.
|
||||
*
|
||||
* The design: [requestFromServer] and [Sse] talk to `https://$FIXTURE_HOST:$FIXTURE_PORT` through
|
||||
* ordinary `java.net.URL`, exactly as they would talk to a real server. A
|
||||
* [java.net.URLStreamHandlerFactory] registered once for the whole process intercepts every
|
||||
* `https://` connection to that host and answers from this object's in-memory event log instead of
|
||||
* opening a socket -- see BenchNetwork.kt. Everything above that (TranscriptSource, SessionScreen,
|
||||
* the fold, uniqueItems) never learns the difference.
|
||||
*/
|
||||
object BenchFixture {
|
||||
const val FIXTURE_HOST = "bench.fixture.invalid"
|
||||
const val FIXTURE_PORT = 1
|
||||
|
||||
/** How many of the fixture's events are the opening backlog; see bench-fixture/README.md. */
|
||||
private const val BACKLOG_COUNT = 3200
|
||||
|
||||
val settings = ServerSettings(FIXTURE_HOST, FIXTURE_PORT, "bench")
|
||||
|
||||
/** The session id every bench run opens; nothing else in this build ever mints one. */
|
||||
const val SESSION_ID = "bench-fixture-session"
|
||||
|
||||
/**
|
||||
* The whole transcript, seq order, growing as [pushLive] is called during the streaming phase.
|
||||
* Read by both the REST page handler and the SSE handler, so a page requested mid- stream and a
|
||||
* live frame agree on what has "already happened" -- the same thing a real server's own
|
||||
* transcript file guarantees.
|
||||
*/
|
||||
private val log = CopyOnWriteArrayList<Pair<String, SeqEvent>>()
|
||||
|
||||
/** The events not yet appended to [log] -- the streaming phase's own source. */
|
||||
private var streamTail: List<Pair<String, SeqEvent>> = emptyList()
|
||||
|
||||
private val images = mutableMapOf<String, ByteArray>()
|
||||
|
||||
@Volatile private var loaded = false
|
||||
|
||||
/**
|
||||
* Parses the bundled fixture once. Safe to call more than once; only the first does anything.
|
||||
*/
|
||||
@Synchronized
|
||||
fun ensureLoaded(context: Context) {
|
||||
if (loaded) return
|
||||
val lines =
|
||||
context.assets.open("transcript.jsonl").bufferedReader().readLines().filter {
|
||||
it.isNotBlank()
|
||||
}
|
||||
val parsed = lines.map { it to parseSeqEvent(it) }
|
||||
log.addAll(parsed.take(BACKLOG_COUNT))
|
||||
streamTail = parsed.drop(BACKLOG_COUNT)
|
||||
for (name in listOf("bench1.png", "bench2.png")) {
|
||||
images[name] = context.assets.open(name).readBytes()
|
||||
}
|
||||
loaded = true
|
||||
}
|
||||
|
||||
/** The events the streaming phase has left to send. */
|
||||
fun remainingStreamEvents(): Int = streamTail.size
|
||||
|
||||
/** Sends the next fixture event onto the live log, as a real SSE frame would arrive. */
|
||||
fun pushNextLiveEvent(): Boolean {
|
||||
val next = streamTail.firstOrNull() ?: return false
|
||||
streamTail = streamTail.drop(1)
|
||||
log.add(next)
|
||||
return true
|
||||
}
|
||||
|
||||
/** Undoes [pushNextLiveEvent] and reloads the opening backlog, for running the bench twice. */
|
||||
@Synchronized
|
||||
fun resetToBacklog(context: Context) {
|
||||
loaded = false
|
||||
log.clear()
|
||||
ensureLoaded(context)
|
||||
}
|
||||
|
||||
fun fileBytes(name: String): ByteArray? = images[name]
|
||||
|
||||
/**
|
||||
* Raw JSON lines with seq > [after], in order -- what an `/events?after=` connection replays.
|
||||
*/
|
||||
fun linesAfter(after: Long): List<String> =
|
||||
log.filter { it.second.seq > after }.map { it.first }
|
||||
|
||||
/**
|
||||
* One REST page: [fetchTranscript]'s `before`/`limit`/`after`, against the growing log. Ignores
|
||||
* `coalesce` -- the fixture's own deltas are already split the way a real reply streams, and
|
||||
* what the benchmark exercises is the fold and the paging, not the server's row-joining, which
|
||||
* client-core's own port tracks separately (CLIENT_CORE.md).
|
||||
*/
|
||||
fun page(before: Long?, limit: Int, after: Long?): List<String> {
|
||||
val upper = before ?: (log.lastOrNull()?.second?.seq?.plus(1) ?: 1L)
|
||||
val candidates = log.filter {
|
||||
it.second.seq < upper && (after == null || it.second.seq > after)
|
||||
}
|
||||
return candidates.takeLast(limit).map { it.first }
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,181 @@
|
||||
package com.example.aiapp
|
||||
|
||||
import java.io.ByteArrayInputStream
|
||||
import java.io.IOException
|
||||
import java.io.InputStream
|
||||
import java.io.PipedInputStream
|
||||
import java.io.PipedOutputStream
|
||||
import java.net.HttpURLConnection
|
||||
import java.net.URL
|
||||
import java.net.URLStreamHandler
|
||||
import java.net.URLStreamHandlerFactory
|
||||
import java.security.Principal
|
||||
import java.security.cert.Certificate
|
||||
import javax.net.ssl.HttpsURLConnection
|
||||
import javax.net.ssl.SSLPeerUnverifiedException
|
||||
import org.json.JSONArray
|
||||
|
||||
/**
|
||||
* Installs the process-wide interception [BenchFixture] needs. Idempotent and safe to call more
|
||||
* than once; the JDK only allows [URL.setURLStreamHandlerFactory] to be called successfully once
|
||||
* per process, and a second real call throws -- so this guards it rather than relying on every
|
||||
* caller to remember.
|
||||
*
|
||||
* Scoped to [BenchFixture.FIXTURE_HOST]: any other `https://` URL falls through to the platform's
|
||||
* ordinary handler, so this only ever changes behaviour for the one host the bench build invents.
|
||||
*/
|
||||
@Synchronized
|
||||
fun installFixtureNetworkOnce() {
|
||||
if (installed) return
|
||||
installed = true
|
||||
URL.setURLStreamHandlerFactory(
|
||||
URLStreamHandlerFactory { protocol ->
|
||||
if (protocol != "https") null
|
||||
else
|
||||
object : URLStreamHandler() {
|
||||
override fun openConnection(url: URL): HttpURLConnection =
|
||||
if (url.host == BenchFixture.FIXTURE_HOST) FixtureConnection(url)
|
||||
else
|
||||
// The bench build makes no other https call -- this factory is
|
||||
// installed only in FIXTURE_MODE (MainActivity) -- so there is
|
||||
// deliberately no delegate to a platform handler here: once a
|
||||
// URLStreamHandlerFactory is installed there is no supported way to
|
||||
// ask the JDK for its own default handler back, and re-entering this
|
||||
// same factory for the fallback would recurse forever rather than
|
||||
// reach one.
|
||||
throw java.io.IOException(
|
||||
"bench build's fixture network has no route to https host " +
|
||||
"${url.host} -- only ${BenchFixture.FIXTURE_HOST} is served"
|
||||
)
|
||||
}
|
||||
}
|
||||
)
|
||||
}
|
||||
|
||||
private var installed = false
|
||||
|
||||
/**
|
||||
* Answers one request against [BenchFixture] instead of opening a socket. Implements just enough of
|
||||
* [HttpsURLConnection] for [requestFromServer] and [Sse] to work unmodified: both only call
|
||||
* `connect`/`disconnect`, set a handful of request properties they never need answered, and read
|
||||
* `responseCode` and `inputStream`.
|
||||
*/
|
||||
private class FixtureConnection(url: URL) : HttpsURLConnection(url) {
|
||||
private var input: InputStream? = null
|
||||
private var writer: Thread? = null
|
||||
|
||||
override fun connect() {
|
||||
if (input != null) return
|
||||
input = route(url.path, url.query)
|
||||
}
|
||||
|
||||
override fun disconnect() {
|
||||
writer?.interrupt()
|
||||
try {
|
||||
input?.close()
|
||||
} catch (_: IOException) {}
|
||||
}
|
||||
|
||||
override fun usingProxy() = false
|
||||
|
||||
override fun getResponseCode(): Int {
|
||||
connect()
|
||||
return 200
|
||||
}
|
||||
|
||||
override fun getInputStream(): InputStream {
|
||||
connect()
|
||||
return input!!
|
||||
}
|
||||
|
||||
override fun getErrorStream(): InputStream? = null
|
||||
|
||||
// Nothing here reads any of these; implemented only because HttpsURLConnection declares them
|
||||
// abstract. A fixture never negotiates real TLS, so each says exactly that rather than
|
||||
// fabricating a plausible-looking certificate.
|
||||
override fun getCipherSuite() = "none (bench fixture, no TLS)"
|
||||
|
||||
override fun getLocalCertificates(): Array<Certificate>? = null
|
||||
|
||||
override fun getServerCertificates(): Array<Certificate> =
|
||||
throw SSLPeerUnverifiedException("bench fixture connection presents no certificate")
|
||||
|
||||
override fun getPeerPrincipal(): Principal =
|
||||
throw SSLPeerUnverifiedException("bench fixture connection presents no certificate")
|
||||
|
||||
override fun getLocalPrincipal(): Principal? = null
|
||||
|
||||
/**
|
||||
* [path] is `/sessions/{id}/...`; everything else this build's fixture is asked for is a bug.
|
||||
*/
|
||||
private fun route(path: String, query: String?): InputStream {
|
||||
val params =
|
||||
(query ?: "")
|
||||
.split("&")
|
||||
.filter { it.contains('=') }
|
||||
.associate {
|
||||
val (k, v) = it.split("=", limit = 2)
|
||||
k to java.net.URLDecoder.decode(v, "UTF-8")
|
||||
}
|
||||
return when {
|
||||
path.endsWith("/transcript") -> {
|
||||
val lines =
|
||||
BenchFixture.page(
|
||||
before = params["before"]?.toLongOrNull(),
|
||||
limit = params["limit"]?.toIntOrNull() ?: 80,
|
||||
after = params["after"]?.toLongOrNull(),
|
||||
)
|
||||
val body = JSONArray(lines.map { org.json.JSONObject(it) })
|
||||
ByteArrayInputStream(body.toString().toByteArray())
|
||||
}
|
||||
path.endsWith("/events") -> openEventsStream(params["after"]?.toLongOrNull() ?: 0L)
|
||||
path.contains("/files/") -> {
|
||||
val name = path.substringAfterLast("/files/")
|
||||
val bytes =
|
||||
BenchFixture.fileBytes(name)
|
||||
?: throw IOException("bench fixture has no file named $name")
|
||||
ByteArrayInputStream(bytes)
|
||||
}
|
||||
else -> throw IOException("bench fixture has no route for $path")
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A live SSE body: [BenchFixture.linesAfter] replayed immediately, then polled every 50ms for
|
||||
* anything [BenchFixture.pushNextLiveEvent] has added since -- the same shape a real backend's
|
||||
* backlog-then-follow gives [Sse], just polled instead of woken, which is a fixture's business
|
||||
* rather than something worth a condition variable for.
|
||||
*/
|
||||
private fun openEventsStream(after: Long): InputStream {
|
||||
val pipeIn = PipedInputStream(1 shl 16)
|
||||
val pipeOut = PipedOutputStream(pipeIn)
|
||||
var sent = after
|
||||
val thread = Thread {
|
||||
try {
|
||||
while (!Thread.currentThread().isInterrupted) {
|
||||
val fresh = BenchFixture.linesAfter(sent)
|
||||
for (line in fresh) {
|
||||
pipeOut.write("data: $line\n\n".toByteArray())
|
||||
pipeOut.flush()
|
||||
sent = org.json.JSONObject(line).getLong("seq")
|
||||
}
|
||||
Thread.sleep(50)
|
||||
}
|
||||
} catch (_: InterruptedException) {
|
||||
// disconnect() -- the ordinary way this ends.
|
||||
} catch (_: IOException) {
|
||||
// The reader side (Sse) closed its end.
|
||||
} finally {
|
||||
try {
|
||||
pipeOut.close()
|
||||
} catch (_: IOException) {}
|
||||
}
|
||||
}
|
||||
.also {
|
||||
it.isDaemon = true
|
||||
it.start()
|
||||
}
|
||||
writer = thread
|
||||
return pipeIn
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,325 @@
|
||||
package com.example.aiapp
|
||||
|
||||
import android.content.Context
|
||||
import android.os.BatteryManager
|
||||
import android.os.Process
|
||||
import android.view.View
|
||||
import androidx.compose.foundation.gestures.FlingBehavior
|
||||
import androidx.compose.foundation.lazy.LazyListState
|
||||
import androidx.compose.ui.focus.FocusRequester
|
||||
import androidx.core.view.ViewCompat
|
||||
import androidx.core.view.WindowInsetsCompat
|
||||
import androidx.core.view.WindowInsetsControllerCompat
|
||||
import java.io.File
|
||||
import kotlinx.coroutines.CoroutineScope
|
||||
import kotlinx.coroutines.delay
|
||||
import kotlinx.coroutines.isActive
|
||||
import kotlinx.coroutines.launch
|
||||
|
||||
/**
|
||||
* P0's scripted benchmark, run in-process instead of by a shell script: the phone has no usable
|
||||
* system tracing (this-machine-android's skill) and no agent can drive it, so the same scroll loop
|
||||
* and streaming phase `transcript-bench.sh`/`stream-bench.sh` drive over `ui-trace` are reproduced
|
||||
* here against [LazyListState] and [BenchFixture] directly. Only reachable from the `bench` build
|
||||
* (see [SessionSettingsDialog]'s `onRunBenchmark`), but compiled into every build for the reason
|
||||
* [BenchFixture]'s doc comment gives.
|
||||
*
|
||||
* **v2 (2026-09-06)**, asked for by Iris because the v1 fling was too gentle to stress-test the
|
||||
* scroll path and said nothing about typing or the keyboard. Four phases now, each a slice of the
|
||||
* same [FrameStats] recording ([FrameStats.markPhase]/[FrameStats.phaseLines] -- one recorder, not
|
||||
* two): **fling** (real `FlingBehavior`, not `animateScrollBy`), **stream** (unchanged from v1),
|
||||
* **type** (600 fixed characters into the real composer `TextFieldValue`, then deleted), and
|
||||
* **keyboard** (five show/hide cycles). The exact constants below are also written into
|
||||
* `docs/RUST.md`'s P0 box, "Benchmark v2 (2026-09-06)", so the iris half implements the identical
|
||||
* spec -- changing a number here without updating that box makes the two apps measure different
|
||||
* things while looking like the same benchmark.
|
||||
*/
|
||||
object BenchRun {
|
||||
/** transcript-bench.sh's default: 6 cycles of 4 swipes each, kept as the pre-v2 comparison. */
|
||||
private const val CYCLES = 6
|
||||
private const val SWIPE_PX = 900f
|
||||
private const val SWIPE_MS = 200
|
||||
private const val SWIPE_PAUSE_MS = 500L
|
||||
|
||||
/**
|
||||
* Fling phase (v2): a real fling through the list's own [FlingBehavior], not `animateScrollBy`
|
||||
* -- Iris's ask was that it "travel way faster" than the old tween-based swipe, and a tween can
|
||||
* never exceed the distance it is told to cover in the time it is given, while a real fling
|
||||
* decays from an initial velocity the way a finger flick does. 12,000 px/s is roughly a hard,
|
||||
* fast flick on a ~420dp/in device (about 30 dp/ms-equivalent initial speed); chosen well above
|
||||
* the ~4,500 px/s a moderate `animateScrollBy` swipe implies, so this phase exercises the fast
|
||||
* end of what the platform's fling decay produces rather than the gentle one v1 measured.
|
||||
*/
|
||||
private const val FLING_VELOCITY_PX_S = 12_000f
|
||||
|
||||
private const val FLING_COUNT = 8
|
||||
private const val FLING_SETTLE_CAP_MS = 3_000L
|
||||
private const val FLING_PAUSE_MS = 300L
|
||||
|
||||
/** stream-bench.sh's shape: a real reply arrives as many small deltas, not one big write. */
|
||||
private const val STREAM_EVENTS_PER_SEC = 20
|
||||
private const val STREAM_SECONDS = 20
|
||||
|
||||
/**
|
||||
* Type phase (v2): sentences built from long, multisyllabic words so the composer actually
|
||||
* wraps across lines rather than fitting one, and long enough (600 chars) that the composer's
|
||||
* own height grows over several frames, pushing the transcript above it upward the same way a
|
||||
* real long message does. Exactly this string is also in `docs/RUST.md`'s P0 box so the iris
|
||||
* half types the identical content.
|
||||
*/
|
||||
const val TYPE_TEXT =
|
||||
"Benchmarking this transcript screen requires unusually long, multisyllabic words so " +
|
||||
"wrapping and reflow are properly exercised: internationalization, " +
|
||||
"counterproductiveness, disproportionately, incomprehensibility, " +
|
||||
"deinstitutionalization, uncharacteristically, overenthusiastically, " +
|
||||
"misunderstanding, straightforwardness, telecommunications, and interdisciplinary " +
|
||||
"collaboration all push a narrow composer field to wrap across several lines while " +
|
||||
"the transcript above is pushed upward by the growing keyboard-adjacent box, which " +
|
||||
"is exactly what a real reader typing a long message sees happening now!!!"
|
||||
|
||||
private const val TYPE_CHAR_DELAY_MS = 50L
|
||||
|
||||
/**
|
||||
* Keyboard phase (v2): five show/hide cycles, a second apart, is enough to see whether the
|
||||
* transition is ever actually observed rather than being a one-off fluke either way.
|
||||
*/
|
||||
private const val KEYBOARD_CYCLES = 5
|
||||
private const val KEYBOARD_SHOW_WAIT_MS = 1_000L
|
||||
private const val KEYBOARD_HIDE_WAIT_MS = 1_000L
|
||||
|
||||
/**
|
||||
* Scrolls, flings, streams, types and toggles the keyboard, then returns the extra report lines
|
||||
* P0 asked for (per-phase travel/typing/keyboard counts, plus CPU time, peak RSS, battery
|
||||
* current) -- [FrameStats] and [DebugStats] are reset first, exactly as `copyRenderReport`
|
||||
* resets them, so the two accountings cover the same stretch of work.
|
||||
*/
|
||||
suspend fun run(
|
||||
context: Context,
|
||||
scope: CoroutineScope,
|
||||
listState: LazyListState,
|
||||
flingBehavior: FlingBehavior,
|
||||
composerFocus: FocusRequester,
|
||||
setComposerText: (String) -> Unit,
|
||||
view: View,
|
||||
): List<String> {
|
||||
FrameStats.reset()
|
||||
DebugStats.reset()
|
||||
val cpuStartMs = Process.getElapsedCpuTime()
|
||||
|
||||
val battery = BatterySampler(context)
|
||||
// Launched in the caller's scope rather than a fresh coroutineScope{} here, which would
|
||||
// suspend this function until the sampler job ended -- and it only ends when told to.
|
||||
val samplerJob = scope.launch {
|
||||
while (isActive) {
|
||||
battery.sample()
|
||||
delay(1000)
|
||||
}
|
||||
}
|
||||
|
||||
val travel = runFlingPhase(listState, flingBehavior)
|
||||
val sent = runStreamPhase()
|
||||
runTypePhase(listState, composerFocus, setComposerText, view)
|
||||
val keyboard = runKeyboardPhase(context, view)
|
||||
|
||||
samplerJob.cancel()
|
||||
val cpuMs = Process.getElapsedCpuTime() - cpuStartMs
|
||||
val rssLine = peakRssLine()
|
||||
val batteryLine = battery.finish()
|
||||
|
||||
return listOf(
|
||||
" fling: $FLING_COUNT flings out + $FLING_COUNT back at" +
|
||||
" ${FLING_VELOCITY_PX_S.toInt()}px/s, travel $travel",
|
||||
" scroll: $CYCLES cycles (${CYCLES * 4} swipes, legacy tween), " +
|
||||
"streamed $sent/${STREAM_EVENTS_PER_SEC * STREAM_SECONDS} fixture events",
|
||||
" type: ${TYPE_TEXT.length} characters inserted then deleted, one per" +
|
||||
" ${TYPE_CHAR_DELAY_MS}ms",
|
||||
keyboard,
|
||||
" process CPU time over this run: ${cpuMs}ms",
|
||||
rssLine,
|
||||
batteryLine,
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* Phase 1: starting pinned at the newest end, [FLING_COUNT] flings away from it (toward older
|
||||
* messages) through the list's real fling path, then [FLING_COUNT] back. Positive velocity here
|
||||
* matches this list's existing scroll-offset convention (`TranscriptList`'s `reverseLayout`
|
||||
* pins index 0 -- the newest item -- at the bottom; a positive scroll offset moves the viewport
|
||||
* toward higher indices, i.e. away from the newest end and toward older content), the same sign
|
||||
* the pre-v2 swipe loop below already used for its first two swipes.
|
||||
*/
|
||||
private suspend fun runFlingPhase(
|
||||
listState: LazyListState,
|
||||
flingBehavior: FlingBehavior,
|
||||
): String {
|
||||
FrameStats.markPhase("fling")
|
||||
listState.scrollToItem(0)
|
||||
val start = position(listState)
|
||||
repeat(FLING_COUNT) {
|
||||
listState.scroll { with(flingBehavior) { performFling(FLING_VELOCITY_PX_S) } }
|
||||
waitForSettle(listState)
|
||||
delay(FLING_PAUSE_MS)
|
||||
}
|
||||
val outward = position(listState)
|
||||
repeat(FLING_COUNT) {
|
||||
listState.scroll { with(flingBehavior) { performFling(-FLING_VELOCITY_PX_S) } }
|
||||
waitForSettle(listState)
|
||||
delay(FLING_PAUSE_MS)
|
||||
}
|
||||
val back = position(listState)
|
||||
return "start=$start outward=$outward end=$back"
|
||||
}
|
||||
|
||||
private fun position(listState: LazyListState) =
|
||||
"idx=${listState.firstVisibleItemIndex}/off=${listState.firstVisibleItemScrollOffset}px"
|
||||
|
||||
/** Belt-and-suspenders on top of `performFling` already suspending until its own decay ends. */
|
||||
private suspend fun waitForSettle(listState: LazyListState) {
|
||||
val startedAt = System.currentTimeMillis()
|
||||
while (
|
||||
listState.isScrollInProgress &&
|
||||
System.currentTimeMillis() - startedAt < FLING_SETTLE_CAP_MS
|
||||
) {
|
||||
delay(16)
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Phase 2 (unchanged from v1): pinned to the newest end before streaming starts, the way
|
||||
* stream-bench.sh's "Jump to latest" tap is -- a reply streamed into a list parked further back
|
||||
* arrives off-screen and the report would show nothing happened.
|
||||
*/
|
||||
private suspend fun runStreamPhase(): Int {
|
||||
FrameStats.markPhase("stream")
|
||||
var sent = 0
|
||||
val total = STREAM_EVENTS_PER_SEC * STREAM_SECONDS
|
||||
while (sent < total && BenchFixture.remainingStreamEvents() > 0) {
|
||||
BenchFixture.pushNextLiveEvent()
|
||||
sent++
|
||||
delay(1000L / STREAM_EVENTS_PER_SEC)
|
||||
}
|
||||
// Lets the last few deltas land and draw before the next phase starts.
|
||||
delay(300)
|
||||
return sent
|
||||
}
|
||||
|
||||
/**
|
||||
* Phase 3: focuses the real composer, shows the keyboard if the platform allows it, then types
|
||||
* [TYPE_TEXT] one character at a time through the same `TextFieldValue` state a real keystroke
|
||||
* updates, and deletes it the same way -- this is what exercises wrapping and the transcript
|
||||
* being pushed upward, not a single big write.
|
||||
*/
|
||||
private suspend fun runTypePhase(
|
||||
listState: LazyListState,
|
||||
composerFocus: FocusRequester,
|
||||
setComposerText: (String) -> Unit,
|
||||
view: View,
|
||||
) {
|
||||
FrameStats.markPhase("type")
|
||||
listState.scrollToItem(0)
|
||||
composerFocus.requestFocus()
|
||||
showIme(view.context, view)
|
||||
// Lets focus and the keyboard's opening animation land before typing starts, so the frames
|
||||
// this phase records are the wrap/reflow it is measuring, not the keyboard opening.
|
||||
delay(300)
|
||||
var typed = ""
|
||||
for (ch in TYPE_TEXT) {
|
||||
typed += ch
|
||||
setComposerText(typed)
|
||||
delay(TYPE_CHAR_DELAY_MS)
|
||||
}
|
||||
delay(200)
|
||||
while (typed.isNotEmpty()) {
|
||||
typed = typed.dropLast(1)
|
||||
setComposerText(typed)
|
||||
delay(TYPE_CHAR_DELAY_MS)
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Phase 4: [KEYBOARD_CYCLES] show/hide cycles through the same [WindowInsetsControllerCompat]
|
||||
* path a real IME toggle goes through, reporting how many of each were actually confirmed by
|
||||
* [android.view.WindowInsets.isVisible] rather than assumed from having asked -- UI_RULES:
|
||||
* never present an inferred value as a measured one. If the platform never shows it even once,
|
||||
* this says so in words rather than reporting a phase with no keyboard in it.
|
||||
*/
|
||||
private suspend fun runKeyboardPhase(context: Context, view: View): String {
|
||||
FrameStats.markPhase("keyboard")
|
||||
var shown = 0
|
||||
var hidden = 0
|
||||
repeat(KEYBOARD_CYCLES) {
|
||||
showIme(context, view)
|
||||
delay(KEYBOARD_SHOW_WAIT_MS)
|
||||
if (imeVisible(view)) shown++
|
||||
hideIme(context, view)
|
||||
delay(KEYBOARD_HIDE_WAIT_MS)
|
||||
if (!imeVisible(view)) hidden++
|
||||
}
|
||||
return if (shown == 0) {
|
||||
" keyboard: could not be shown ($KEYBOARD_CYCLES attempts, 0 confirmed visible)"
|
||||
} else {
|
||||
" keyboard: shown $shown/$KEYBOARD_CYCLES, hidden $hidden/$KEYBOARD_CYCLES" +
|
||||
" (confirmed via isImeVisible)"
|
||||
}
|
||||
}
|
||||
|
||||
private fun controller(context: Context, view: View): WindowInsetsControllerCompat? {
|
||||
val window = context.activity()?.window ?: return null
|
||||
return WindowInsetsControllerCompat(window, view)
|
||||
}
|
||||
|
||||
private fun showIme(context: Context, view: View) {
|
||||
controller(context, view)?.show(WindowInsetsCompat.Type.ime())
|
||||
}
|
||||
|
||||
private fun hideIme(context: Context, view: View) {
|
||||
controller(context, view)?.hide(WindowInsetsCompat.Type.ime())
|
||||
}
|
||||
|
||||
private fun imeVisible(view: View): Boolean =
|
||||
ViewCompat.getRootWindowInsets(view)?.isVisible(WindowInsetsCompat.Type.ime()) ?: false
|
||||
|
||||
/** VmHWM from /proc/self/status: the process's high-water mark, in kB, since it started. */
|
||||
private fun peakRssLine(): String {
|
||||
val kb =
|
||||
try {
|
||||
File("/proc/self/status")
|
||||
.readLines()
|
||||
.firstOrNull { it.startsWith("VmHWM:") }
|
||||
?.trim()
|
||||
?.removePrefix("VmHWM:")
|
||||
?.trim()
|
||||
?.removeSuffix("kB")
|
||||
?.trim()
|
||||
?.toLongOrNull()
|
||||
} catch (_: Exception) {
|
||||
null
|
||||
}
|
||||
return " peak RSS: " +
|
||||
(kb?.let { "${it}kB" } ?: "unavailable (/proc/self/status unreadable)")
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Samples [BatteryManager.BATTERY_PROPERTY_CURRENT_NOW] (microamps) once a second for the length of
|
||||
* a run. The property returns `Int.MIN_VALUE` on hardware that does not support it -- most
|
||||
* emulators -- and that is reported as "unavailable" rather than folded into an average with the
|
||||
* real samples, which would silently understate every number after it. See UI_RULES: never present
|
||||
* an inferred value as a measured one.
|
||||
*/
|
||||
private class BatterySampler(context: Context) {
|
||||
private val manager = context.getSystemService(BatteryManager::class.java)
|
||||
private val samples = mutableListOf<Int>()
|
||||
|
||||
fun sample() {
|
||||
val value = manager?.getIntProperty(BatteryManager.BATTERY_PROPERTY_CURRENT_NOW)
|
||||
if (value != null && value != Int.MIN_VALUE) samples.add(value)
|
||||
}
|
||||
|
||||
fun finish(): String {
|
||||
if (samples.isEmpty()) return " battery current: unavailable on this device"
|
||||
val meanUa = samples.sum() / samples.size
|
||||
return " battery current: mean ${meanUa}µA over ${samples.size} samples" +
|
||||
" (min ${samples.min()}, max ${samples.max()})"
|
||||
}
|
||||
}
|
||||
@@ -127,6 +127,20 @@ fun debugReport(
|
||||
frames: List<String>,
|
||||
accounting: List<String>,
|
||||
crash: String?,
|
||||
/**
|
||||
* P0's benchmark-only measurements (process CPU time, peak RSS, battery current) -- empty on
|
||||
* every path but [BenchRun.runP0Benchmark], which is the only caller that has them. A section
|
||||
* heading only appears when there is something to put under it, so an ordinary copy from the
|
||||
* render-report button reads exactly as it did before this existed.
|
||||
*/
|
||||
extra: List<String> = emptyList(),
|
||||
/**
|
||||
* Bench v2's per-phase frame accounting ([FrameStats.phaseLines]) --
|
||||
* fling/stream/type/keyboard, each a slice of the same frames the whole-run sections below
|
||||
* still cover in full. Empty on every path but the scripted bench run, same reasoning as
|
||||
* [extra].
|
||||
*/
|
||||
phaseFrames: List<String> = emptyList(),
|
||||
): String = buildString {
|
||||
appendLine("ai-app render report")
|
||||
appendLine(device)
|
||||
@@ -141,6 +155,11 @@ fun debugReport(
|
||||
appendLine("transcript:")
|
||||
transcript.forEach { appendLine(it) }
|
||||
appendLine()
|
||||
if (phaseFrames.isNotEmpty()) {
|
||||
appendLine("per phase:")
|
||||
phaseFrames.forEach { appendLine(it) }
|
||||
appendLine()
|
||||
}
|
||||
appendLine("frames:")
|
||||
frames.forEach { appendLine(it) }
|
||||
appendLine()
|
||||
@@ -152,6 +171,11 @@ fun debugReport(
|
||||
appendLine("work since this was last copied:")
|
||||
val work = DebugStats.lines()
|
||||
if (work.isEmpty()) appendLine(" nothing recorded") else work.forEach { appendLine(it) }
|
||||
if (extra.isNotEmpty()) {
|
||||
appendLine()
|
||||
appendLine("bench:")
|
||||
extra.forEach { appendLine(it) }
|
||||
}
|
||||
}
|
||||
|
||||
/** Puts [text] on the clipboard under [label], which is what the system offers as its name. */
|
||||
|
||||
@@ -12,6 +12,10 @@ import androidx.compose.ui.Alignment
|
||||
import androidx.compose.ui.Modifier
|
||||
import androidx.compose.ui.graphics.Color
|
||||
import androidx.compose.ui.unit.dp
|
||||
import java.time.Instant
|
||||
import java.time.ZoneId
|
||||
import java.time.format.DateTimeFormatter
|
||||
import java.time.format.FormatStyle
|
||||
|
||||
/**
|
||||
* A line across the transcript saying what left the session's context.
|
||||
@@ -47,3 +51,40 @@ fun TranscriptDivider(text: String, color: Color, modifier: Modifier = Modifier)
|
||||
fun ClearedRow(modifier: Modifier = Modifier) {
|
||||
TranscriptDivider("Context cleared", clearedColor, modifier)
|
||||
}
|
||||
|
||||
/**
|
||||
* The mark running out of quota leaves.
|
||||
*
|
||||
* The same red the usage bar takes when a window is spent, because it is the same fact in a second
|
||||
* place: colour by consequence, so "there is nothing left to spend" is learned once.
|
||||
*
|
||||
* A time rather than a countdown. The row is folded once and never re-measured, so a span would go
|
||||
* stale on screen the moment it was drawn; and this is when the *account* said it would reset,
|
||||
* which is not a promise about when the session picks back up. A limit the session was told no
|
||||
* reset time for says nothing about one -- that state has its own words rather than a plausible
|
||||
* number.
|
||||
*/
|
||||
@Composable
|
||||
fun LimitRow(item: TranscriptItem.LimitNote, modifier: Modifier = Modifier) {
|
||||
TranscriptDivider(limitSummary(item.resetsAt, ZoneId.systemDefault()), overLimitColor, modifier)
|
||||
}
|
||||
|
||||
/**
|
||||
* What the row says. Split out so the wording is testable without a screen, since the two states it
|
||||
* has to keep apart -- a reset time that arrived and one that never did -- are exactly the pair
|
||||
* that reads the same when it goes wrong.
|
||||
*
|
||||
* [zone] is a parameter rather than read here so a test says the same thing wherever it runs.
|
||||
*/
|
||||
fun limitSummary(resetsAt: Double?, zone: ZoneId): String {
|
||||
val at = resetsAt?.let {
|
||||
try {
|
||||
DateTimeFormatter.ofLocalizedTime(FormatStyle.SHORT)
|
||||
.withZone(zone)
|
||||
.format(Instant.ofEpochSecond(it.toLong()))
|
||||
} catch (_: Exception) {
|
||||
null
|
||||
}
|
||||
}
|
||||
return if (at == null) "Usage limit reached" else "Usage limit reached • resets $at"
|
||||
}
|
||||
@@ -14,7 +14,7 @@ private const val RESET_EVENT = "reset"
|
||||
* mean. [close] from any thread ends it, and the caller owns reconnecting -- with the last seq it
|
||||
* saw as the new cursor.
|
||||
*/
|
||||
class EventStream(settings: ServerSettings, private val sessionId: String) {
|
||||
class EventStream(settings: ServerSettings, private val address: TranscriptAddress) {
|
||||
private val stream = Sse(settings)
|
||||
|
||||
fun close() = stream.close()
|
||||
@@ -35,7 +35,7 @@ class EventStream(settings: ServerSettings, private val sessionId: String) {
|
||||
// one and the screen folds the other, and they have to be the same line.
|
||||
onEvent: (raw: String, event: SeqEvent) -> Unit,
|
||||
) {
|
||||
stream.run("/sessions/$sessionId/events?after=$after", onOpen) { name, data ->
|
||||
stream.run("/${address.urlPath}/events?after=$after", onOpen) { name, data ->
|
||||
// A named frame carries no payload and a data frame has no name.
|
||||
if (name == RESET_EVENT) onReset()
|
||||
else if (data.isNotEmpty()) onEvent(data, parseSeqEvent(data))
|
||||
|
||||
@@ -157,6 +157,18 @@ sealed class SessionEvent {
|
||||
*/
|
||||
data object Cleared : SessionEvent()
|
||||
|
||||
/**
|
||||
* The session stopped because its account's usage limit was reached.
|
||||
*
|
||||
* Its own event rather than an [Error] carrying the CLI's sentence, because it is a state
|
||||
* rather than something that went wrong -- and because the raw sentence is `Claude AI usage
|
||||
* limit reached|1788546972`, which is not readable by the person it is shown to.
|
||||
*
|
||||
* [resetsAt] is epoch seconds and null where the session was told nothing. Only the server acts
|
||||
* on it; what this draws it as is a time, not a countdown, because nothing here re-measures it.
|
||||
*/
|
||||
data class LimitReached(val resetsAt: Double?) : SessionEvent()
|
||||
|
||||
data class Error(val message: String) : SessionEvent()
|
||||
|
||||
/**
|
||||
@@ -261,6 +273,10 @@ fun parseSeqEvent(json: String): SeqEvent {
|
||||
trigger = body.optString("trigger").ifEmpty { null },
|
||||
)
|
||||
"cleared" -> SessionEvent.Cleared
|
||||
"limitReached" ->
|
||||
SessionEvent.LimitReached(
|
||||
if (body.has("resetsAt")) body.getDouble("resetsAt") else null
|
||||
)
|
||||
"error" -> SessionEvent.Error(body.getString("message"))
|
||||
else -> SessionEvent.Unknown(type)
|
||||
}
|
||||
|
||||
@@ -42,6 +42,21 @@ object FrameStats {
|
||||
private val gpu = ArrayList<Long>()
|
||||
private var since = System.currentTimeMillis()
|
||||
|
||||
/**
|
||||
* Where a named phase of a scripted run (bench v2's fling/stream/type/keyboard) started, as an
|
||||
* index into [total] and a wall-clock time -- not a second recorder, just a mark on this one,
|
||||
* so a phase's frames are the same [FrameMetrics] the whole-run report already has, sliced.
|
||||
*/
|
||||
private data class PhaseMark(val name: String, val startIndex: Int, val startMs: Long)
|
||||
|
||||
private val phaseMarks = ArrayList<PhaseMark>()
|
||||
|
||||
/** Call at the start of each named phase of a scripted run; see [BenchRun]. */
|
||||
@Synchronized
|
||||
fun markPhase(name: String) {
|
||||
phaseMarks += PhaseMark(name, total.size, System.currentTimeMillis())
|
||||
}
|
||||
|
||||
@Synchronized
|
||||
fun add(metrics: FrameMetrics) {
|
||||
// The first frame after a window opens includes inflating it and is nobody's scroll.
|
||||
@@ -69,6 +84,7 @@ object FrameStats {
|
||||
listOf(total, waited, input, animation, layout, draw, sync, issue, swap, gpu).forEach {
|
||||
it.clear()
|
||||
}
|
||||
phaseMarks.clear()
|
||||
since = System.currentTimeMillis()
|
||||
}
|
||||
|
||||
@@ -95,6 +111,38 @@ object FrameStats {
|
||||
) + if (gpu.isEmpty()) emptyList() else listOf(phase("gpu ", gpu))
|
||||
}
|
||||
|
||||
/**
|
||||
* One block per [markPhase] call: how many frames landed between that mark and the next (or the
|
||||
* end of the run, for the last one), how many were late, the total/p50/p90/p99, the worst
|
||||
* single frame, and how long the phase actually ran. Marks with no frames between them (a phase
|
||||
* that finished before a frame was drawn) still get a line rather than being silently dropped
|
||||
* -- UI_RULES' "say what you don't know" applies to a phase as much as to a single number.
|
||||
*/
|
||||
@Synchronized
|
||||
fun phaseLines(refreshHz: Float): List<String> {
|
||||
if (phaseMarks.isEmpty()) return emptyList()
|
||||
val budget = if (refreshHz > 0) 1000.0 / refreshHz else 16.7
|
||||
val lines = ArrayList<String>()
|
||||
phaseMarks.forEachIndexed { i, mark ->
|
||||
val endIndex = if (i + 1 < phaseMarks.size) phaseMarks[i + 1].startIndex else total.size
|
||||
val endMs =
|
||||
if (i + 1 < phaseMarks.size) phaseMarks[i + 1].startMs
|
||||
else System.currentTimeMillis()
|
||||
val samples = total.subList(mark.startIndex, endIndex)
|
||||
val seconds = (endMs - mark.startMs) / 1000.0
|
||||
lines += " ${mark.name}: ${samples.size} frames over ${"%.1f".format(seconds)}s"
|
||||
if (samples.isEmpty()) {
|
||||
lines += " no frames recorded in this phase"
|
||||
} else {
|
||||
val late = samples.count { it / 1_000_000.0 > budget }
|
||||
lines += " late: $late (${percent(late, samples.size)})"
|
||||
lines += " " + phase("total ", samples)
|
||||
lines += " worst ${"%.1fms".format(samples.max() / 1_000_000.0)}"
|
||||
}
|
||||
}
|
||||
return lines
|
||||
}
|
||||
|
||||
/** How long the frames recorded here spent in their draw phase, and how many there were. */
|
||||
@Synchronized fun drawPhase(): Pair<Long, Int> = draw.sum() to draw.size
|
||||
|
||||
|
||||
@@ -66,6 +66,35 @@ class MainActivity : ComponentActivity() {
|
||||
// Transparent status bar on every version; the Surface below paints through underneath it
|
||||
// and content insets itself. Same reasoning as dev-updater's MainActivity.
|
||||
enableEdgeToEdge()
|
||||
|
||||
// The `bench` build's entire purpose (P0, docs/RUST.md): open straight onto the session
|
||||
// screen against BenchFixture's in-process fake backend, with no enrollment, no network
|
||||
// permission, and no notification prompt -- none of them mean anything with no server and
|
||||
// no real device to notify. See BenchFixture.kt and BenchNetwork.kt for how a screen built
|
||||
// to talk to a real backend is made to talk to this instead. Still needs the same
|
||||
// status/navigation-bar padding the ordinary flow below applies: edge-to-edge is the
|
||||
// platform's own default from Android 15 on this app's targetSdk, with or without the call
|
||||
// above, so skipping the padding here put the header's own buttons under the status bar --
|
||||
// there to look at, but not there for `ui-trace`'s tap-by-label to land on.
|
||||
if (BuildConfig.FIXTURE_MODE) {
|
||||
installFixtureNetworkOnce()
|
||||
BenchFixture.ensureLoaded(this)
|
||||
setContent {
|
||||
MaterialTheme(colorScheme = AiAppColors) {
|
||||
Surface(modifier = Modifier.fillMaxSize()) {
|
||||
Box(Modifier.fillMaxSize().statusBarsPadding().navigationBarsPadding()) {
|
||||
SessionScreen(
|
||||
settings = BenchFixture.settings,
|
||||
summary = benchSessionSummary(),
|
||||
onBack = { finish() },
|
||||
onFiles = {},
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return
|
||||
}
|
||||
// Dark status-bar icons only over a light background, decided from the scheme rather than
|
||||
// fixed. It was hardcoded to `true`, which was right against the default light surface and
|
||||
// became unreadable the moment the app wore Catppuccin Mocha.
|
||||
@@ -147,6 +176,33 @@ class MainActivity : ComponentActivity() {
|
||||
}
|
||||
}
|
||||
|
||||
/** The one session the `bench` build ever shows -- BenchFixture's session id, nothing else. */
|
||||
private fun benchSessionSummary() =
|
||||
SessionSummary(
|
||||
id = BenchFixture.SESSION_ID,
|
||||
setup = "bench",
|
||||
setupName = "bench",
|
||||
provider = "bench",
|
||||
title = "P0 benchmark",
|
||||
model = null,
|
||||
keepsOwnTranscript = false,
|
||||
permissionMode = null,
|
||||
effort = null,
|
||||
takesEffort = false,
|
||||
imported = false,
|
||||
notify = false,
|
||||
autoResume = false,
|
||||
autoResumeMessage = "",
|
||||
resumeAt = null,
|
||||
cwd = null,
|
||||
contextTokens = null,
|
||||
maxImageEdge = null,
|
||||
usageProvider = null,
|
||||
status = "idle",
|
||||
lastActivity = 0.0,
|
||||
subagents = 0,
|
||||
)
|
||||
|
||||
// launchMode="singleTop": an enrollment scan, or a notification tapped while the app is open,
|
||||
// lands here rather than in a second activity instance.
|
||||
override fun onNewIntent(intent: Intent) {
|
||||
|
||||
@@ -48,6 +48,8 @@ fun MainScreen(
|
||||
/** What another app shared in and no session has taken yet; see [ShareRequest]. */
|
||||
share: ShareRequest? = null,
|
||||
onOpen: (SessionSummary) -> Unit,
|
||||
/** Opens one session's subagent, from the expander under its card. */
|
||||
onOpenSubagent: (SessionSummary, SubagentSummary) -> Unit,
|
||||
onSpawn: () -> Unit,
|
||||
onImported: (SessionSummary) -> Unit,
|
||||
onSettings: () -> Unit,
|
||||
@@ -139,6 +141,7 @@ fun MainScreen(
|
||||
settings = settings,
|
||||
reloadToken = token,
|
||||
onOpen = onOpen,
|
||||
onOpenSubagent = onOpenSubagent,
|
||||
onSpawn = onSpawn,
|
||||
)
|
||||
MainTab.Import ->
|
||||
|
||||
@@ -1,7 +1,9 @@
|
||||
package com.example.aiapp
|
||||
|
||||
import androidx.compose.foundation.ExperimentalFoundationApi
|
||||
import androidx.compose.foundation.clickable
|
||||
import androidx.compose.foundation.combinedClickable
|
||||
import androidx.compose.foundation.layout.Arrangement
|
||||
import androidx.compose.foundation.layout.Box
|
||||
import androidx.compose.foundation.layout.Column
|
||||
import androidx.compose.foundation.layout.Row
|
||||
@@ -9,6 +11,7 @@ import androidx.compose.foundation.layout.Spacer
|
||||
import androidx.compose.foundation.layout.fillMaxSize
|
||||
import androidx.compose.foundation.layout.fillMaxWidth
|
||||
import androidx.compose.foundation.layout.height
|
||||
import androidx.compose.foundation.layout.heightIn
|
||||
import androidx.compose.foundation.layout.padding
|
||||
import androidx.compose.foundation.layout.width
|
||||
import androidx.compose.foundation.lazy.LazyColumn
|
||||
@@ -17,6 +20,7 @@ import androidx.compose.material3.Card
|
||||
import androidx.compose.material3.CircularProgressIndicator
|
||||
import androidx.compose.material3.FloatingActionButton
|
||||
import androidx.compose.material3.MaterialTheme
|
||||
import androidx.compose.material3.OutlinedCard
|
||||
import androidx.compose.material3.Switch
|
||||
import androidx.compose.material3.Text
|
||||
import androidx.compose.material3.TextButton
|
||||
@@ -30,6 +34,8 @@ import androidx.compose.runtime.setValue
|
||||
import androidx.compose.ui.Alignment
|
||||
import androidx.compose.ui.Modifier
|
||||
import androidx.compose.ui.platform.LocalContext
|
||||
import androidx.compose.ui.semantics.contentDescription
|
||||
import androidx.compose.ui.semantics.semantics
|
||||
import androidx.compose.ui.unit.dp
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.launch
|
||||
@@ -47,12 +53,40 @@ fun SessionListScreen(
|
||||
settings: ServerSettings,
|
||||
reloadToken: Int,
|
||||
onOpen: (SessionSummary) -> Unit,
|
||||
/** Opens one session's subagent, from the expander under its card. */
|
||||
onOpenSubagent: (SessionSummary, SubagentSummary) -> Unit,
|
||||
onSpawn: () -> Unit,
|
||||
) {
|
||||
val scope = rememberCoroutineScope()
|
||||
var listState by remember { mutableStateOf<LoadState<List<SessionSummary>>>(LoadState.Loading) }
|
||||
var confirmingDelete by remember { mutableStateOf<SessionSummary?>(null) }
|
||||
|
||||
// Which session cards are expanded to show their subagents, and what each expansion fetched.
|
||||
// Ids rather than a flag on the row for the same reason `deleting` is: the rows are rebuilt
|
||||
// from
|
||||
// whatever the server last said, and this belongs to the reader's own choice, which survives a
|
||||
// refresh.
|
||||
var expandedSessions by remember { mutableStateOf(setOf<String>()) }
|
||||
var subagentLoads by remember {
|
||||
mutableStateOf(mapOf<String, LoadState<List<SubagentSummary>>>())
|
||||
}
|
||||
|
||||
fun loadSubagents(sessionId: String) {
|
||||
subagentLoads = subagentLoads + (sessionId to LoadState.Loading)
|
||||
scope.launch {
|
||||
subagentLoads =
|
||||
subagentLoads +
|
||||
(sessionId to
|
||||
try {
|
||||
LoadState.Loaded(
|
||||
withContext(Dispatchers.IO) { fetchSubagents(settings, sessionId) }
|
||||
)
|
||||
} catch (e: ApiException) {
|
||||
LoadState.failed(e)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Failures that belong to one session rather than to the list, keyed by its id and shown on its
|
||||
// own card. The two scopes are decided by whether the server answered: it answered and refused,
|
||||
// so this says nothing about the other rows.
|
||||
@@ -84,6 +118,14 @@ fun SessionListScreen(
|
||||
withContext(Dispatchers.IO) {
|
||||
transcriptCache.retainOnly(loaded.value.map { it.id }.toSet())
|
||||
}
|
||||
// A session gone from this answer cannot still be expanded, and an expanded one
|
||||
// that is still here asks again -- its subagents may have changed since the
|
||||
// last
|
||||
// fetch.
|
||||
val ids = loaded.value.map { it.id }.toSet()
|
||||
expandedSessions = expandedSessions intersect ids
|
||||
subagentLoads = subagentLoads.filterKeys { it in ids }
|
||||
expandedSessions.forEach(::loadSubagents)
|
||||
loaded
|
||||
} catch (e: ApiException) {
|
||||
LoadState.failed(e)
|
||||
@@ -127,6 +169,17 @@ fun SessionListScreen(
|
||||
deleting = session.id in deleting,
|
||||
onOpen = { onOpen(session) },
|
||||
onLongPress = { confirmingDelete = session },
|
||||
expanded = session.id in expandedSessions,
|
||||
subagents = subagentLoads[session.id],
|
||||
onToggleSubagents = {
|
||||
if (session.id in expandedSessions) {
|
||||
expandedSessions = expandedSessions - session.id
|
||||
} else {
|
||||
expandedSessions = expandedSessions + session.id
|
||||
loadSubagents(session.id)
|
||||
}
|
||||
},
|
||||
onOpenSubagent = { subagent -> onOpenSubagent(session, subagent) },
|
||||
)
|
||||
Spacer(Modifier.height(12.dp))
|
||||
}
|
||||
@@ -225,7 +278,7 @@ fun SessionListScreen(
|
||||
deleteSession(settings, session.id, alsoDeleteForeign)
|
||||
// After it succeeded, not before: a refused delete leaves the
|
||||
// session exactly as it was, and its transcript with it.
|
||||
transcriptCache.session(session.id).purge()
|
||||
transcriptCache.session(TranscriptAddress(session.id)).purge()
|
||||
}
|
||||
// Only this row, and only what changed. Refetching the list instead
|
||||
// put every other session back through loading and handed the
|
||||
@@ -276,6 +329,12 @@ private fun SessionCard(
|
||||
deleting: Boolean,
|
||||
onOpen: () -> Unit,
|
||||
onLongPress: () -> Unit,
|
||||
/** Whether the expander below is open. Collapsed by default; see [SessionListScreen]. */
|
||||
expanded: Boolean,
|
||||
/** What the expander's own fetch answered, or null before it has been asked. */
|
||||
subagents: LoadState<List<SubagentSummary>>?,
|
||||
onToggleSubagents: () -> Unit,
|
||||
onOpenSubagent: (SubagentSummary) -> Unit,
|
||||
) {
|
||||
BusyItem(label = if (deleting) "deleting" else null) {
|
||||
Card(
|
||||
@@ -332,11 +391,105 @@ private fun SessionCard(
|
||||
color = MaterialTheme.colorScheme.error,
|
||||
)
|
||||
}
|
||||
// Nothing at all for a card with no subagents: a disabled expander here would be
|
||||
// noise on every ordinary session's card. Its own row at the bottom rather than
|
||||
// beside the title or the machine line, so opening it never displaces text that was
|
||||
// already on screen -- see UI_RULES on a control not displacing the text beside it.
|
||||
if (session.subagents > 0) {
|
||||
Spacer(Modifier.height(8.dp))
|
||||
// The platform's minimum touch height, not the chevron's own ten or so dp:
|
||||
// at the chevron's height a tap meant for it landed on the first subcard
|
||||
// beneath and opened a subagent instead.
|
||||
Row(
|
||||
horizontalArrangement = Arrangement.Center,
|
||||
verticalAlignment = Alignment.CenterVertically,
|
||||
modifier =
|
||||
Modifier.fillMaxWidth()
|
||||
.heightIn(min = 48.dp)
|
||||
.clickable(enabled = !deleting, onClick = onToggleSubagents)
|
||||
.semantics {
|
||||
contentDescription =
|
||||
if (expanded) "Collapse subagents" else "Expand subagents"
|
||||
},
|
||||
) {
|
||||
Chevron(if (expanded) Pointing.Up else Pointing.Down)
|
||||
}
|
||||
if (expanded) {
|
||||
Spacer(Modifier.height(4.dp))
|
||||
Column(verticalArrangement = Arrangement.spacedBy(8.dp)) {
|
||||
when (subagents) {
|
||||
null,
|
||||
is LoadState.Loading ->
|
||||
CircularProgressIndicator(
|
||||
modifier = Modifier.width(20.dp).height(20.dp),
|
||||
strokeWidth = 2.dp,
|
||||
)
|
||||
is LoadState.Error ->
|
||||
// Said here rather than left silent: a fetch that failed and an
|
||||
// expander that simply found nothing must not look the same --
|
||||
// see UI_RULES on designing the unknown state first.
|
||||
Text(
|
||||
subagents.message,
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
color = MaterialTheme.colorScheme.error,
|
||||
)
|
||||
is LoadState.Loaded ->
|
||||
subagents.value.forEach { subagent ->
|
||||
SubagentCard(
|
||||
subagent,
|
||||
onClick = { onOpenSubagent(subagent) },
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* One subagent, indented inside its session's card -- the way dev-updater draws a project's
|
||||
* components (`ComponentCard`, `UpdaterScreen.kt`): an outlined card, not the session card's own
|
||||
* filled one, so the nesting reads as one step rather than as another session.
|
||||
*/
|
||||
@Composable
|
||||
private fun SubagentCard(subagent: SubagentSummary, onClick: () -> Unit) {
|
||||
OutlinedCard(Modifier.fillMaxWidth().clickable(onClick = onClick)) {
|
||||
Column(Modifier.padding(horizontal = 12.dp, vertical = 8.dp)) {
|
||||
Text(subagent.title, style = MaterialTheme.typography.titleSmall)
|
||||
Spacer(Modifier.height(2.dp))
|
||||
Row(modifier = Modifier.fillMaxWidth()) {
|
||||
Text(
|
||||
subagentStatusLabel(subagent.status),
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
modifier = Modifier.weight(1f),
|
||||
)
|
||||
Text(
|
||||
relativeTime(subagent.lastActivity),
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The subcard's word for a subagent's status -- see SUBAGENTS.md's "Wire shape". Its own function
|
||||
* rather than a branch inside [StatusText], because a subagent's three states are not that
|
||||
* composable's five: "exited" reads as "finished" here, since its process was always its parent's
|
||||
* and never something of its own to have merely stopped.
|
||||
*/
|
||||
private fun subagentStatusLabel(status: String) =
|
||||
when (status) {
|
||||
"running" -> "running"
|
||||
"exited" -> "finished"
|
||||
else -> "unknown"
|
||||
}
|
||||
|
||||
@Composable
|
||||
fun StatusText(status: String) {
|
||||
val (label, color) =
|
||||
|
||||
@@ -12,6 +12,7 @@ import androidx.activity.result.PickVisualMediaRequest
|
||||
import androidx.activity.result.contract.ActivityResultContracts
|
||||
import androidx.compose.foundation.background
|
||||
import androidx.compose.foundation.clickable
|
||||
import androidx.compose.foundation.gestures.ScrollableDefaults
|
||||
import androidx.compose.foundation.gestures.awaitEachGesture
|
||||
import androidx.compose.foundation.gestures.awaitFirstDown
|
||||
import androidx.compose.foundation.layout.Box
|
||||
@@ -64,12 +65,15 @@ import androidx.compose.runtime.snapshots.Snapshot
|
||||
import androidx.compose.ui.Alignment
|
||||
import androidx.compose.ui.Modifier
|
||||
import androidx.compose.ui.draw.drawWithContent
|
||||
import androidx.compose.ui.focus.FocusRequester
|
||||
import androidx.compose.ui.focus.focusRequester
|
||||
import androidx.compose.ui.graphics.graphicsLayer
|
||||
import androidx.compose.ui.input.pointer.PointerEventPass
|
||||
import androidx.compose.ui.input.pointer.pointerInput
|
||||
import androidx.compose.ui.layout.onSizeChanged
|
||||
import androidx.compose.ui.platform.LocalContext
|
||||
import androidx.compose.ui.platform.LocalDensity
|
||||
import androidx.compose.ui.platform.LocalView
|
||||
import androidx.compose.ui.semantics.contentDescription
|
||||
import androidx.compose.ui.semantics.semantics
|
||||
import androidx.compose.ui.text.TextRange
|
||||
@@ -216,16 +220,33 @@ fun SessionScreen(
|
||||
share: ShareRequest? = null,
|
||||
/** Said once [share] has been attached here, so it is not attached again. */
|
||||
onShareTaken: () -> Unit = {},
|
||||
/**
|
||||
* Draws this screen read-only, on a subagent's own transcript instead of the session's.
|
||||
*
|
||||
* A subagent has no process and no controls of its own -- see SUBAGENTS.md's "Phone" -- so
|
||||
* every gate below keyed on this switches off the composer, the files button, the settings cog,
|
||||
* the usage bar and notifications, while everything that draws a transcript (paging, cache,
|
||||
* selection, images, the status row, stream reconnects) is reused unchanged, pointed at
|
||||
* [address] instead of the session's own.
|
||||
*/
|
||||
subagent: SubagentSummary? = null,
|
||||
) {
|
||||
DebugStats.count("session screen recomposed")
|
||||
val isSubagent = subagent != null
|
||||
val address = TranscriptAddress(summary.id, subagent?.id)
|
||||
val scope = rememberCoroutineScope()
|
||||
val topEdgeHeld = remember { TopEdgeHold() }
|
||||
var items by remember { mutableStateOf(listOf<TranscriptItem>()) }
|
||||
var status by remember { mutableStateOf(summary.status) }
|
||||
var status by remember { mutableStateOf(subagent?.status ?: summary.status) }
|
||||
// Seeded from the row this screen was opened from, so a conversation already under way says how
|
||||
// much it is holding before any turn happens here. Null is "nobody has measured it", which is a
|
||||
// different answer from an empty context and is drawn differently.
|
||||
var contextTokens by remember(summary.id) { mutableStateOf(summary.contextTokens) }
|
||||
//
|
||||
// A subagent has no context measurement of its own, so it always starts unmeasured rather than
|
||||
// borrowing the parent session's figure -- see UI_RULES on not showing an inferred value as one
|
||||
// that was measured.
|
||||
var contextTokens by
|
||||
remember(address) { mutableStateOf(if (isSubagent) null else summary.contextTokens) }
|
||||
// When the current compaction started. The moment comes off the `compacting` status event
|
||||
// itself -- the server timestamps every transcript line -- rather than off this device noticing
|
||||
// one, which is what makes it survive leaving the session and reopening it.
|
||||
@@ -241,7 +262,13 @@ fun SessionScreen(
|
||||
val context = LocalContext.current
|
||||
// Seeded from what was left in the box last time and written back on every keystroke, so
|
||||
// leaving the screen does not throw away a half-typed message. See `Drafts.kt`.
|
||||
var input by remember(summary.id) { mutableStateOf(atEnd(loadDraft(context, summary.id))) }
|
||||
//
|
||||
// A subagent has no box to type into, so it never touches a draft at all -- not this session's,
|
||||
// which is what reading one keyed only by `summary.id` would do here.
|
||||
var input by
|
||||
remember(summary.id) {
|
||||
mutableStateOf(if (isSubagent) atEnd("") else atEnd(loadDraft(context, summary.id)))
|
||||
}
|
||||
// A model the reader has chosen and not yet confirmed. See [ModelSwitchWarning]: switching
|
||||
// makes the session re-read the whole conversation.
|
||||
var pendingModel by remember { mutableStateOf<String?>(null) }
|
||||
@@ -294,27 +321,26 @@ fun SessionScreen(
|
||||
// Reload throws away what it was reading from.
|
||||
val cache = remember(settings) { TranscriptCache(cacheRoot(context, settings)) }
|
||||
val source =
|
||||
remember(summary.id, epoch) {
|
||||
TranscriptSource(settings, summary.id, cache.session(summary.id))
|
||||
}
|
||||
remember(address, epoch) { TranscriptSource(settings, address, cache.session(address)) }
|
||||
// Whether the cached tail has been shown to still be the server's own line. Nothing is resumed
|
||||
// from a cached cursor until it has, and a probe that could not be made leaves this false for
|
||||
// the stream loop to try again.
|
||||
var probePassed by remember(summary.id, epoch) { mutableStateOf(false) }
|
||||
var probePassed by remember(address, epoch) { mutableStateOf(false) }
|
||||
// Whether the opening effect is still settling that question. It draws the cached rows and
|
||||
// lifts [ready] before the answer arrives, which is the point of the cache -- so the stream
|
||||
// below waits for this rather than for `ready`, or it asks the same question twice.
|
||||
var probing by remember(summary.id, epoch) { mutableStateOf(true) }
|
||||
var probing by remember(address, epoch) { mutableStateOf(true) }
|
||||
// The oldest sequence number loaded, and whether there is more behind it. Paging backwards is
|
||||
// what keeps opening a long session cheap.
|
||||
var oldestSeq by remember { mutableLongStateOf(0L) }
|
||||
// Where this session was last being read, from this device's own store. Read once, because the
|
||||
// answer stops being interesting the moment the list is on screen.
|
||||
val savedAnchor = remember(summary.id, epoch) { loadScrollAnchor(context, summary.id) }
|
||||
// Where this transcript was last being read, from this device's own store, keyed by the address
|
||||
// rather than the session id so a subagent's saved position cannot collide with its session's.
|
||||
// Read once, because the answer stops being interesting the moment the list is on screen.
|
||||
val savedAnchor = remember(address, epoch) { loadScrollAnchor(context, address.cachePath) }
|
||||
// Whether the saved position is still being put back. Nothing is drawn while it is: opening at
|
||||
// the newest end and then travelling to the anchor is exactly the journey a reader must never
|
||||
// see.
|
||||
var restoring by remember(summary.id, epoch) { mutableStateOf(savedAnchor != null) }
|
||||
var restoring by remember(address, epoch) { mutableStateOf(savedAnchor != null) }
|
||||
// Messages the server has taken and the session has not read yet, by the id that will resolve
|
||||
// them. From the event stream rather than from what this screen sent, so they survive leaving
|
||||
// the session -- and a message sent from another device is drawn waiting on this one too.
|
||||
@@ -327,11 +353,22 @@ fun SessionScreen(
|
||||
var loadingHistory by remember { mutableStateOf(false) }
|
||||
var ready by remember { mutableStateOf(false) }
|
||||
// Replies parsed ahead of the rows that draw them; see [ParsedReplies].
|
||||
val replies = remember(summary.id) { ParsedReplies() }
|
||||
// Keyed like everything else describing one session's transcript. `rememberLazyListState` saves
|
||||
// through `rememberSaveable`, and this screen restores by its own anchor instead -- two
|
||||
// restores would fight over the first frame.
|
||||
val listState = remember(summary.id) { LazyListState() }
|
||||
val replies = remember(address) { ParsedReplies() }
|
||||
// Keyed like everything else describing one transcript. `rememberLazyListState` saves through
|
||||
// `rememberSaveable`, and this screen restores by its own anchor instead -- two restores would
|
||||
// fight over the first frame.
|
||||
val listState = remember(address) { LazyListState() }
|
||||
// The list's own fling path -- what a real flick decays through -- captured here so BenchRun's
|
||||
// fling phase can drive `LazyListState.scroll` through exactly the `FlingBehavior` this
|
||||
// screen's
|
||||
// `TranscriptList` already uses by not overriding it (its `LazyColumn` takes no `flingBehavior`
|
||||
// argument, so this is the same default it gets).
|
||||
val flingBehavior = ScrollableDefaults.flingBehavior()
|
||||
// Where BenchRun's type phase focuses before it types, and the view it toggles the keyboard on
|
||||
// -- both bench-only, but cheap enough (a remembered object, a CompositionLocal read) to hold
|
||||
// unconditionally rather than behind a second code path only the bench build compiles.
|
||||
val composerFocus = remember { FocusRequester() }
|
||||
val view = LocalView.current
|
||||
// Whether the newest message is on screen right now. The list is reversed, so the newest end is
|
||||
// the scrolling start: nothing behind you is exactly being at the bottom. Asked of the scroll
|
||||
// state rather than of item indices, because a zero-height first item makes an index ambiguous.
|
||||
@@ -637,7 +674,7 @@ fun SessionScreen(
|
||||
// ended and carries live events only. The window comes from this phone's own copy when there is
|
||||
// one, and then costs a single request to check that the server's transcript is still the one
|
||||
// it came from. See TRANSCRIPT_CACHE.md.
|
||||
LaunchedEffect(summary.id, epoch) {
|
||||
LaunchedEffect(address, epoch) {
|
||||
/**
|
||||
* One opening window onto the screen, whichever side it came from.
|
||||
*
|
||||
@@ -667,11 +704,16 @@ fun SessionScreen(
|
||||
// A replay is as old as the last visit; the row this screen was opened from was
|
||||
// fetched moments ago. So the transcript comes from the cache and everything that
|
||||
// is not the transcript comes from the summary -- otherwise a session that finished
|
||||
// an hour ago opens saying "working" until the stream connects.
|
||||
status = summary.status
|
||||
model = summary.model
|
||||
permissionMode = summary.permissionMode ?: "auto"
|
||||
if (summary.status != "compacting") compactingSince = null
|
||||
// an hour ago opens saying "working" until the stream connects. A subagent's status
|
||||
// comes from its own summary, never the parent session's: they are two different
|
||||
// things running or not, and the parent's model and permission mode do not apply to
|
||||
// it at all.
|
||||
status = subagent?.status ?: summary.status
|
||||
if (!isSubagent) {
|
||||
model = summary.model
|
||||
permissionMode = summary.permissionMode ?: "auto"
|
||||
}
|
||||
if (status != "compacting") compactingSince = null
|
||||
// Nothing to put back, so these rows are the screen and the probe can return under
|
||||
// them. A restore still has history to fetch and is gated below.
|
||||
if (savedAnchor == null) ready = true
|
||||
@@ -798,7 +840,7 @@ fun SessionScreen(
|
||||
// at the top on their return. Switching apps is a choice somebody made, not a fault to report.
|
||||
// Stopping the stream deliberately makes the drop a close rather than an error, and resuming
|
||||
// reconnects from the same cursor.
|
||||
LaunchedEffect(summary.id, ready, epoch, lifecycleOwner) {
|
||||
LaunchedEffect(address, ready, epoch, lifecycleOwner) {
|
||||
if (!ready) return@LaunchedEffect
|
||||
// The opening effect draws cached rows and lifts `ready` *before* it has checked that the
|
||||
// cursor under them is still the server's, so `ready` is no longer the whole gate. Without
|
||||
@@ -868,17 +910,22 @@ fun SessionScreen(
|
||||
// The screen going away entirely, which the lifecycle scope above does not cover: a composable
|
||||
// can leave the composition while the activity stays started. Keyed on the epoch as well, so
|
||||
// Reload's replacement source is the one a later disposal closes.
|
||||
DisposableEffect(summary.id, epoch) { onDispose { source.close() } }
|
||||
DisposableEffect(address, epoch) { onDispose { source.close() } }
|
||||
|
||||
// Nothing gets announced about the session somebody is reading; see NotificationService.
|
||||
// RESUMED rather than STARTED because "looking at it" means the foreground.
|
||||
LaunchedEffect(summary.id, lifecycleOwner) {
|
||||
lifecycleOwner.repeatOnLifecycle(Lifecycle.State.RESUMED) {
|
||||
NotificationService.showing(context, summary.id)
|
||||
try {
|
||||
awaitCancellation()
|
||||
} finally {
|
||||
NotificationService.stoppedShowing(summary.id)
|
||||
//
|
||||
// Not for a subagent: it has no notifications of its own, and it is not the session this would
|
||||
// otherwise mark as being read.
|
||||
if (!isSubagent) {
|
||||
LaunchedEffect(summary.id, lifecycleOwner) {
|
||||
lifecycleOwner.repeatOnLifecycle(Lifecycle.State.RESUMED) {
|
||||
NotificationService.showing(context, summary.id)
|
||||
try {
|
||||
awaitCancellation()
|
||||
} finally {
|
||||
NotificationService.stoppedShowing(summary.id)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -924,7 +971,7 @@ fun SessionScreen(
|
||||
val (index, offset, awayFromNewest) = settled
|
||||
saveScrollAnchor(
|
||||
context,
|
||||
summary.id,
|
||||
address.cachePath,
|
||||
// Nothing to restore at the newest end, which is where a session with no anchor
|
||||
// opens anyway. One *before* the index, because item zero is the "below" slot.
|
||||
if (!awayFromNewest) null
|
||||
@@ -947,7 +994,7 @@ fun SessionScreen(
|
||||
//
|
||||
// There is no correction beside this one. Following the newest message is not an effect: the
|
||||
// list is reversed, so an arriving message extends the end the viewport is pinned to.
|
||||
val unitSizes = remember(summary.id) { HashMap<Any, Int>() }
|
||||
val unitSizes = remember(address) { HashMap<Any, Int>() }
|
||||
LaunchedEffect(listState, moreHistory) {
|
||||
snapshotFlow { listState.layoutInfo }
|
||||
.collect { info ->
|
||||
@@ -983,21 +1030,25 @@ fun SessionScreen(
|
||||
}
|
||||
}
|
||||
|
||||
LaunchedEffect(summary.setupName, summary.provider) {
|
||||
offeredModels =
|
||||
try {
|
||||
withContext(Dispatchers.IO) {
|
||||
fetchSetups(settings)
|
||||
.firstOrNull { it.name == summary.setupName }
|
||||
?.providers
|
||||
?.firstOrNull { it.name == summary.provider }
|
||||
?.models
|
||||
.orEmpty()
|
||||
// Only for the model picker, which a subagent does not have.
|
||||
if (!isSubagent) {
|
||||
LaunchedEffect(summary.setupName, summary.provider) {
|
||||
offeredModels =
|
||||
try {
|
||||
withContext(Dispatchers.IO) {
|
||||
fetchSetups(settings)
|
||||
.firstOrNull { it.name == summary.setupName }
|
||||
?.providers
|
||||
?.firstOrNull { it.name == summary.provider }
|
||||
?.models
|
||||
.orEmpty()
|
||||
}
|
||||
} catch (_: Exception) {
|
||||
// Not worth reporting: the picker simply has nothing to offer, which is
|
||||
// visible.
|
||||
emptyList()
|
||||
}
|
||||
} catch (_: Exception) {
|
||||
// Not worth reporting: the picker simply has nothing to offer, which is visible.
|
||||
emptyList()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1148,8 +1199,9 @@ fun SessionScreen(
|
||||
}
|
||||
|
||||
// One poll for the machines' limits, read by everything on this screen that reports them.
|
||||
val usageFeed = rememberUsageFeed(settings)
|
||||
val usage = usageFeed.forSetup(summary.setup)
|
||||
// Nothing meters a subagent -- it has no account of its own -- so it never starts this poll.
|
||||
val usageFeed = if (isSubagent) null else rememberUsageFeed(settings)
|
||||
val usage = usageFeed?.forSession(summary) ?: SessionUsage.NotMetered
|
||||
RecordFrames()
|
||||
var usageOpen by remember { mutableStateOf(false) }
|
||||
var settingsOpen by remember { mutableStateOf(false) }
|
||||
@@ -1191,7 +1243,12 @@ fun SessionScreen(
|
||||
// the bench scripts keep working when this moves again. They pressed it at a hand-measured
|
||||
// coordinate until 2026-09-03, and anything that moved the header made that tap land on
|
||||
// whatever now sat there -- reporting a number that was never measured.
|
||||
val copyRenderReport = {
|
||||
// Shared by the ordinary "Copy" button and (bench build only) "Run benchmark": what differs
|
||||
// between them is only whether there is a [extra] section, built by BenchRun.run beforehand --
|
||||
// everything about assembling, copying and logging the report is exactly the same act either
|
||||
// way, and a second copy of it beside `onRunBenchmark` below would be the two silently
|
||||
// disagreeing about what "the report" contains the first time either one changed.
|
||||
fun buildAndCopyReport(extra: List<String> = emptyList()) {
|
||||
val report =
|
||||
debugReport(
|
||||
device =
|
||||
@@ -1213,6 +1270,11 @@ fun SessionScreen(
|
||||
accounting =
|
||||
FrameStats.drawPhase().let { (nanos, count) -> drawAccounting(nanos, count) },
|
||||
crash = lastCrash(context),
|
||||
extra = extra,
|
||||
// Empty outside a BenchRun.run pass -- copyRenderReport's own reset below clears
|
||||
// the
|
||||
// marks along with everything else, so an ordinary copy never has any to show.
|
||||
phaseFrames = FrameStats.phaseLines(context.refreshHz()),
|
||||
)
|
||||
context.copyToClipboard("ai-app render report", report)
|
||||
// Also to the log, so a session driving the app over adb can read the same report the
|
||||
@@ -1226,6 +1288,29 @@ fun SessionScreen(
|
||||
DebugStats.reset()
|
||||
Toast.makeText(context, "Copied render report", Toast.LENGTH_SHORT).show()
|
||||
}
|
||||
val copyRenderReport = { buildAndCopyReport() }
|
||||
// Bench build only: P0's scripted fling/stream/type/keyboard benchmark (BenchRun.kt), against
|
||||
// the fixture session opened below instead of a real server. Null everywhere else -- see
|
||||
// [SessionSettingsDialog]'s onRunBenchmark.
|
||||
val runBenchmark: (() -> Unit)? =
|
||||
if (BuildConfig.FIXTURE_MODE) {
|
||||
{
|
||||
settingsOpen = false
|
||||
scope.launch {
|
||||
val extra =
|
||||
BenchRun.run(
|
||||
context = context,
|
||||
scope = scope,
|
||||
listState = listState,
|
||||
flingBehavior = flingBehavior,
|
||||
composerFocus = composerFocus,
|
||||
setComposerText = { text -> input = atEnd(text) },
|
||||
view = view,
|
||||
)
|
||||
buildAndCopyReport(extra)
|
||||
}
|
||||
}
|
||||
} else null
|
||||
Box(Modifier.fillMaxSize()) {
|
||||
Column(Modifier.fillMaxSize()) {
|
||||
Row(
|
||||
@@ -1236,20 +1321,35 @@ fun SessionScreen(
|
||||
// A ring's worth, which is what the arrow already keeps on its other three sides.
|
||||
Spacer(Modifier.width(GLYPH_BUTTON_MARGIN))
|
||||
Column(Modifier.weight(1f)) {
|
||||
Text(title, style = MaterialTheme.typography.titleMedium)
|
||||
// Machine first, then what runs on it -- the same order and the same wording
|
||||
// everywhere this pair appears, so it reads as one fact rather than two
|
||||
// sentences with different grammar.
|
||||
//
|
||||
// No model. The picker in the footer already shows what this session is set to,
|
||||
// and showing it twice means two things to keep in step -- they disagreed for a
|
||||
// moment on every model change.
|
||||
Text(
|
||||
"${summary.setupName} · ${summary.provider}",
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
)
|
||||
// A subagent's own title, with the session's beneath it in a smaller style --
|
||||
// the header says whose conversation this is as well as what it is. Otherwise
|
||||
// just the session's title, as before.
|
||||
if (subagent != null) {
|
||||
Text(subagent.title, style = MaterialTheme.typography.titleMedium)
|
||||
Text(
|
||||
title,
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
)
|
||||
} else {
|
||||
Text(title, style = MaterialTheme.typography.titleMedium)
|
||||
// Machine first, then what runs on it -- the same order and the same
|
||||
// wording everywhere this pair appears, so it reads as one fact rather than
|
||||
// two sentences with different grammar.
|
||||
//
|
||||
// No model. The picker in the footer already shows what this session is set
|
||||
// to, and showing it twice means two things to keep in step -- they
|
||||
// disagreed for a moment on every model change.
|
||||
Text(
|
||||
"${summary.setupName} · ${summary.provider}",
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
)
|
||||
}
|
||||
}
|
||||
// None of this is a subagent's: it has no files of its own to browse, no settings,
|
||||
// and nothing meters it -- see SUBAGENTS.md's "Phone".
|
||||
//
|
||||
// Beside the provider it reports on, which is the line directly to its left. Its
|
||||
// real home is this provider's settings, which do not exist yet. A session on a
|
||||
// provider with no such service gets an honest "unavailable" rather than a hidden
|
||||
@@ -1264,42 +1364,47 @@ fun SessionScreen(
|
||||
// Usage, files, settings -- widest scope first, narrowing to the right, so the cog
|
||||
// stays at the end where every other screen keeps it. Asked for in this order by
|
||||
// Iris on 2026-09-03.
|
||||
Row {
|
||||
GlyphButton(
|
||||
USAGE_GLYPH,
|
||||
"Usage",
|
||||
{ usageOpen = true },
|
||||
colour = usageGlyphColour(usage),
|
||||
)
|
||||
// The machine's files, which is where the answer to "what did it actually
|
||||
// change" is. It opens *over* this screen rather than replacing it.
|
||||
GlyphButton(
|
||||
FOLDER_GLYPH,
|
||||
"Files",
|
||||
onClick = {
|
||||
onFiles(
|
||||
FilesTarget(
|
||||
setup = summary.setup,
|
||||
setupName = summary.setupName,
|
||||
// Where this session works, and the machine's own home when it
|
||||
// was never given a directory -- resolved there rather than
|
||||
// guessed at here, since this app does not know that home.
|
||||
start = summary.cwd?.takeIf { it.isNotBlank() } ?: "~",
|
||||
if (!isSubagent) {
|
||||
Row {
|
||||
GlyphButton(
|
||||
USAGE_GLYPH,
|
||||
"Usage",
|
||||
{ usageOpen = true },
|
||||
colour = usageGlyphColour(usage),
|
||||
)
|
||||
// The machine's files, which is where the answer to "what did it actually
|
||||
// change" is. It opens *over* this screen rather than replacing it.
|
||||
GlyphButton(
|
||||
FOLDER_GLYPH,
|
||||
"Files",
|
||||
onClick = {
|
||||
onFiles(
|
||||
FilesTarget(
|
||||
setup = summary.setup,
|
||||
setupName = summary.setupName,
|
||||
// Where this session works, and the machine's own home when
|
||||
// it was never given a directory -- resolved there rather
|
||||
// than guessed at here, since this app does not know that
|
||||
// home.
|
||||
start = summary.cwd?.takeIf { it.isNotBlank() } ?: "~",
|
||||
)
|
||||
)
|
||||
)
|
||||
},
|
||||
)
|
||||
// What it opens is about this session, so it sits at the end of the session's
|
||||
// own row. A cog and not a word because there will be more, and a bar of words
|
||||
// has nowhere to put it.
|
||||
GlyphButton(SETTINGS_GLYPH, "Session settings", { settingsOpen = true })
|
||||
},
|
||||
)
|
||||
// What it opens is about this session, so it sits at the end of the
|
||||
// session's own row. A cog and not a word because there will be more, and a
|
||||
// bar of words has nowhere to put it.
|
||||
GlyphButton(SETTINGS_GLYPH, "Session settings", { settingsOpen = true })
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Under the header, above everything the session itself says: it is a fact about the
|
||||
// machine rather than a turn in the conversation, and it is the number that decides
|
||||
// whether to keep going.
|
||||
SessionUsageBar(usage)
|
||||
// whether to keep going. Nothing meters a subagent.
|
||||
if (!isSubagent) {
|
||||
SessionUsageBar(usage)
|
||||
}
|
||||
|
||||
(streamError ?: actionError)?.let { message ->
|
||||
Text(
|
||||
@@ -1545,6 +1650,7 @@ fun SessionScreen(
|
||||
is TranscriptItem.ClearedNote -> ClearedRow()
|
||||
is TranscriptItem.CompactedNote ->
|
||||
CompactedRow(item)
|
||||
is TranscriptItem.LimitNote -> LimitRow(item)
|
||||
// Never reached: a peer message is flattened into
|
||||
// its own units. Here because a `when` over the
|
||||
// item kinds has to stay exhaustive.
|
||||
@@ -1643,185 +1749,211 @@ fun SessionScreen(
|
||||
)
|
||||
}
|
||||
|
||||
// Kept for a subagent -- see SUBAGENTS.md's "Phone" -- with the wording that turns
|
||||
// "exited" into "finished" for one, since it has no process to leave running or stop.
|
||||
SessionStatusRow(
|
||||
status = status,
|
||||
compactingFor = compactingFor,
|
||||
contextTokens = contextTokens,
|
||||
subagent = isSubagent,
|
||||
)
|
||||
|
||||
// Between the transcript and the box: above what is being typed, so the list does not
|
||||
// cover the thing the command is about, and below everything that explains it.
|
||||
CommandSuggestions(
|
||||
// Nothing to suggest about a suggestion that was just taken. `/compact` is a whole
|
||||
// command *and* a prefix of itself, so picking it left the list standing there with
|
||||
// the one row already chosen. Held by what was picked rather than by a flag, so
|
||||
// typing anything else brings the list back without a second thing to reset.
|
||||
commands = if (input.text == picked) emptyList() else suggestedCommands(input.text),
|
||||
onPick = { command ->
|
||||
// At the end of what was inserted, which is where the reader carries on typing:
|
||||
// a command with an argument is put in the box half-written, and a cursor left
|
||||
// at the front makes the next keystroke the first character of "/rename".
|
||||
input = atEnd(command.typed())
|
||||
picked = command.typed()
|
||||
},
|
||||
)
|
||||
|
||||
// Always enabled -- a send while the session is running becomes a steering message
|
||||
// injected at the next tool boundary, which is the point of the whole app.
|
||||
//
|
||||
// The field gets a row of its own, above the buttons: sharing one put the full width
|
||||
// behind three controls, so the thing being typed into was the narrowest on the row.
|
||||
Column(Modifier.fillMaxWidth().padding(8.dp)) {
|
||||
// Directly above the box they will be sent from, so what is attached is visible
|
||||
// rather than counted: the "+2" on the button below said how many and never which.
|
||||
PendingAttachments(
|
||||
settings = settings,
|
||||
sessionId = summary.id,
|
||||
refs = pendingAttachments,
|
||||
onRemove = { pendingAttachments = pendingAttachments - it },
|
||||
)
|
||||
OutlinedTextField(
|
||||
value = input,
|
||||
onValueChange = {
|
||||
input = it
|
||||
saveDraft(context, summary.id, it.text)
|
||||
// Everything from here down is the composer: a subagent cannot be messaged, so none of
|
||||
// it applies -- see SUBAGENTS.md's "Phone".
|
||||
if (!isSubagent) {
|
||||
// Between the transcript and the box: above what is being typed, so the list does
|
||||
// not cover the thing the command is about, and below everything that explains it.
|
||||
CommandSuggestions(
|
||||
// Nothing to suggest about a suggestion that was just taken. `/compact` is a
|
||||
// whole command *and* a prefix of itself, so picking it left the list standing
|
||||
// there with the one row already chosen. Held by what was picked rather than by
|
||||
// a flag, so typing anything else brings the list back without a second thing
|
||||
// to
|
||||
// reset.
|
||||
commands =
|
||||
if (input.text == picked) emptyList() else suggestedCommands(input.text),
|
||||
onPick = { command ->
|
||||
// At the end of what was inserted, which is where the reader carries on
|
||||
// typing: a command with an argument is put in the box half-written, and a
|
||||
// cursor left at the front makes the next keystroke the first character of
|
||||
// "/rename".
|
||||
input = atEnd(command.typed())
|
||||
picked = command.typed()
|
||||
},
|
||||
modifier = Modifier.fillMaxWidth(),
|
||||
// No longer "(+image)": the images are on screen above this, and a placeholder
|
||||
// saying so said it in words beside the thing itself.
|
||||
placeholder = { Text("Message") },
|
||||
maxLines = 4,
|
||||
)
|
||||
Row(
|
||||
verticalAlignment = Alignment.CenterVertically,
|
||||
modifier = Modifier.fillMaxWidth(),
|
||||
) {
|
||||
// Photo or file, asked here rather than by two buttons: the row is full, and
|
||||
// attaching is one action whichever picker answers it.
|
||||
var attaching by remember { mutableStateOf(false) }
|
||||
Box {
|
||||
// Just "+". The count it used to carry was standing in for showing them.
|
||||
BubbleButton(onClick = { attaching = true }) { Text("+") }
|
||||
DropdownMenu(
|
||||
expanded = attaching,
|
||||
onDismissRequest = { attaching = false },
|
||||
// See PickerButton: without this the menu opens a status bar's height
|
||||
// away from the button in an edge-to-edge activity.
|
||||
properties = PopupProperties(clippingEnabled = false),
|
||||
shape = BubbleMenuShape,
|
||||
) {
|
||||
DropdownMenuItem(
|
||||
text = { Text("Photo") },
|
||||
onClick = {
|
||||
attaching = false
|
||||
pickImage.launch(
|
||||
PickVisualMediaRequest(
|
||||
ActivityResultContracts.PickVisualMedia.ImageOnly
|
||||
)
|
||||
)
|
||||
},
|
||||
)
|
||||
DropdownMenuItem(
|
||||
text = { Text("File") },
|
||||
onClick = {
|
||||
attaching = false
|
||||
pickFile.launch(arrayOf("*/*"))
|
||||
},
|
||||
)
|
||||
}
|
||||
}
|
||||
// The settings share what is left after the actions have taken what they need.
|
||||
// A Row hands out intrinsic widths in order and clips whatever runs past the
|
||||
// edge, so with these laid out first the arrival of Stop pushed Send off the
|
||||
// screen entirely -- the app's central control, gone at the moment it is most
|
||||
// in use.
|
||||
|
||||
// Always enabled -- a send while the session is running becomes a steering message
|
||||
// injected at the next tool boundary, which is the point of the whole app.
|
||||
//
|
||||
// The field gets a row of its own, above the buttons: sharing one put the full
|
||||
// width
|
||||
// behind three controls, so the thing being typed into was the narrowest on the
|
||||
// row.
|
||||
Column(Modifier.fillMaxWidth().padding(8.dp)) {
|
||||
// Directly above the box they will be sent from, so what is attached is visible
|
||||
// rather than counted: the "+2" on the button below said how many and never
|
||||
// which.
|
||||
PendingAttachments(
|
||||
settings = settings,
|
||||
sessionId = summary.id,
|
||||
refs = pendingAttachments,
|
||||
onRemove = { pendingAttachments = pendingAttachments - it },
|
||||
)
|
||||
OutlinedTextField(
|
||||
value = input,
|
||||
onValueChange = {
|
||||
input = it
|
||||
saveDraft(context, summary.id, it.text)
|
||||
},
|
||||
// BenchRun's type phase requests focus on this exact field
|
||||
// (`composerFocus`)
|
||||
// so it types through the real composer rather than a stand-in.
|
||||
modifier = Modifier.fillMaxWidth().focusRequester(composerFocus),
|
||||
// No longer "(+image)": the images are on screen above this, and a
|
||||
// placeholder saying so said it in words beside the thing itself.
|
||||
placeholder = { Text("Message") },
|
||||
maxLines = 4,
|
||||
)
|
||||
Row(
|
||||
verticalAlignment = Alignment.CenterVertically,
|
||||
modifier = Modifier.weight(1f),
|
||||
modifier = Modifier.fillMaxWidth(),
|
||||
) {
|
||||
if (offeredModels.isNotEmpty()) {
|
||||
// Photo or file, asked here rather than by two buttons: the row is full,
|
||||
// and
|
||||
// attaching is one action whichever picker answers it.
|
||||
var attaching by remember { mutableStateOf(false) }
|
||||
Box {
|
||||
// Just "+". The count it used to carry was standing in for showing
|
||||
// them.
|
||||
BubbleButton(onClick = { attaching = true }) { Text("+") }
|
||||
DropdownMenu(
|
||||
expanded = attaching,
|
||||
onDismissRequest = { attaching = false },
|
||||
// See PickerButton: without this the menu opens a status bar's
|
||||
// height away from the button in an edge-to-edge activity.
|
||||
properties = PopupProperties(clippingEnabled = false),
|
||||
shape = BubbleMenuShape,
|
||||
) {
|
||||
DropdownMenuItem(
|
||||
text = { Text("Photo") },
|
||||
onClick = {
|
||||
attaching = false
|
||||
pickImage.launch(
|
||||
PickVisualMediaRequest(
|
||||
ActivityResultContracts.PickVisualMedia.ImageOnly
|
||||
)
|
||||
)
|
||||
},
|
||||
)
|
||||
DropdownMenuItem(
|
||||
text = { Text("File") },
|
||||
onClick = {
|
||||
attaching = false
|
||||
pickFile.launch(arrayOf("*/*"))
|
||||
},
|
||||
)
|
||||
}
|
||||
}
|
||||
// The settings share what is left after the actions have taken what they
|
||||
// need. A Row hands out intrinsic widths in order and clips whatever runs
|
||||
// past the edge, so with these laid out first the arrival of Stop pushed
|
||||
// Send off the screen entirely -- the app's central control, gone at the
|
||||
// moment it is most in use.
|
||||
Row(
|
||||
verticalAlignment = Alignment.CenterVertically,
|
||||
modifier = Modifier.weight(1f),
|
||||
) {
|
||||
if (offeredModels.isNotEmpty()) {
|
||||
PickerButton(
|
||||
current = modelLabel(model),
|
||||
// What the machine offers, plus the state a session is in when
|
||||
// it has chosen none of them. The button has always been able
|
||||
// to
|
||||
// say "default"; until this the list could not, so leaving it
|
||||
// was a one-way trip.
|
||||
options = listOf(DEFAULT_MODEL) + offeredModels,
|
||||
// Not set here. The button follows what the session reports it
|
||||
// is set to, which arrives a moment later and is sometimes a
|
||||
// different answer -- a name the CLI resolved, or no change at
|
||||
// all on a provider whose model is fixed. Asked about first,
|
||||
// unless there is nothing to lose by it -- see
|
||||
// [ModelSwitchWarning].
|
||||
onPick = { chosen ->
|
||||
if (
|
||||
modelLabel(chosen) == modelLabel(model) ||
|
||||
!worthWarningAbout(status, contextTokens, items)
|
||||
) {
|
||||
act { setSessionModel(settings, summary.id, chosen) }
|
||||
} else {
|
||||
pendingModel = chosen
|
||||
}
|
||||
},
|
||||
)
|
||||
}
|
||||
PickerButton(
|
||||
current = modelLabel(model),
|
||||
// What the machine offers, plus the state a session is in when it
|
||||
// has chosen none of them. The button has always been able to say
|
||||
// "default"; until this the list could not, so leaving it was a
|
||||
// one-way trip.
|
||||
options = listOf(DEFAULT_MODEL) + offeredModels,
|
||||
// Not set here. The button follows what the session reports it is
|
||||
// set to, which arrives a moment later and is sometimes a different
|
||||
// answer -- a name the CLI resolved, or no change at all on a
|
||||
// provider whose model is fixed. Asked about first, unless there is
|
||||
// nothing to lose by it -- see [ModelSwitchWarning].
|
||||
current = permissionMode,
|
||||
options = PERMISSION_MODES,
|
||||
onPick = { chosen ->
|
||||
if (
|
||||
modelLabel(chosen) == modelLabel(model) ||
|
||||
!worthWarningAbout(status, contextTokens, items)
|
||||
) {
|
||||
act { setSessionModel(settings, summary.id, chosen) }
|
||||
} else {
|
||||
pendingModel = chosen
|
||||
}
|
||||
act { setSessionPermissionMode(settings, summary.id, chosen) }
|
||||
},
|
||||
)
|
||||
}
|
||||
PickerButton(
|
||||
current = permissionMode,
|
||||
options = PERMISSION_MODES,
|
||||
onPick = { chosen ->
|
||||
act { setSessionPermissionMode(settings, summary.id, chosen) }
|
||||
},
|
||||
)
|
||||
}
|
||||
// The same filled shape as the button beside it, not an outlined one: these are
|
||||
// two things you can do about the session, and weighting one as secondary said
|
||||
// they were a primary action and its qualifier. What separates them is the
|
||||
// colour and the mark, which is what they mean.
|
||||
//
|
||||
// Always here, rather than arriving with the turn as it used to. A control that
|
||||
// comes and goes makes its own presence the signal, and a button always in the
|
||||
// same place also cannot push Send off the end of the row by turning up.
|
||||
val process =
|
||||
when {
|
||||
running -> ProcessAction.Pause
|
||||
status == "exited" -> ProcessAction.Start
|
||||
else -> ProcessAction.Stop
|
||||
}
|
||||
Button(
|
||||
onClick = {
|
||||
processInFlight = true
|
||||
act(onDone = { processInFlight = false }) {
|
||||
process.perform(settings, summary.id)
|
||||
// The same filled shape as the button beside it, not an outlined one: these
|
||||
// are two things you can do about the session, and weighting one as
|
||||
// secondary said they were a primary action and its qualifier. What
|
||||
// separates them is the colour and the mark, which is what they mean.
|
||||
//
|
||||
// Always here, rather than arriving with the turn as it used to. A control
|
||||
// that comes and goes makes its own presence the signal, and a button
|
||||
// always
|
||||
// in the same place also cannot push Send off the end of the row by turning
|
||||
// up.
|
||||
val process =
|
||||
when {
|
||||
running -> ProcessAction.Pause
|
||||
status == "exited" -> ProcessAction.Start
|
||||
else -> ProcessAction.Stop
|
||||
}
|
||||
},
|
||||
enabled = !processInFlight,
|
||||
colors = actionButtonColors(process.colour()),
|
||||
) {
|
||||
Glyph(
|
||||
process.glyph,
|
||||
colour = LocalContentColor.current,
|
||||
modifier = Modifier.semantics { contentDescription = process.label },
|
||||
)
|
||||
}
|
||||
Spacer(Modifier.width(8.dp))
|
||||
// The paper plane, with a clock on it while a turn is in flight: sending then
|
||||
// queues the message for the next tool boundary rather than starting a turn of
|
||||
// its own, and the two have to be told apart at a glance. The label says the
|
||||
// same thing to a screen reader.
|
||||
//
|
||||
// Disabled while there is nothing to send, rather than pressable and silent:
|
||||
// `send` has always returned early on an empty composer, so the button promised
|
||||
// something it would not do. Disabled and not hidden, for the reason above.
|
||||
Button(
|
||||
onClick = { send() },
|
||||
enabled = input.text.isNotBlank() || pendingAttachments.isNotEmpty(),
|
||||
colors = actionButtonColors(if (running) queueColor else sendColor),
|
||||
) {
|
||||
Glyph(
|
||||
if (running) QUEUE_GLYPH else SEND_GLYPH,
|
||||
colour = LocalContentColor.current,
|
||||
modifier =
|
||||
Modifier.semantics { contentDescription = sendLabel(running) },
|
||||
)
|
||||
Button(
|
||||
onClick = {
|
||||
processInFlight = true
|
||||
act(onDone = { processInFlight = false }) {
|
||||
process.perform(settings, summary.id)
|
||||
}
|
||||
},
|
||||
enabled = !processInFlight,
|
||||
colors = actionButtonColors(process.colour()),
|
||||
) {
|
||||
Glyph(
|
||||
process.glyph,
|
||||
colour = LocalContentColor.current,
|
||||
modifier =
|
||||
Modifier.semantics { contentDescription = process.label },
|
||||
)
|
||||
}
|
||||
Spacer(Modifier.width(8.dp))
|
||||
// The paper plane, with a clock on it while a turn is in flight: sending
|
||||
// then queues the message for the next tool boundary rather than starting a
|
||||
// turn of its own, and the two have to be told apart at a glance. The label
|
||||
// says the same thing to a screen reader.
|
||||
//
|
||||
// Disabled while there is nothing to send, rather than pressable and
|
||||
// silent:
|
||||
// `send` has always returned early on an empty composer, so the button
|
||||
// promised something it would not do. Disabled and not hidden, for the
|
||||
// reason above.
|
||||
Button(
|
||||
onClick = { send() },
|
||||
enabled = input.text.isNotBlank() || pendingAttachments.isNotEmpty(),
|
||||
colors = actionButtonColors(if (running) queueColor else sendColor),
|
||||
) {
|
||||
Glyph(
|
||||
if (running) QUEUE_GLYPH else SEND_GLYPH,
|
||||
colour = LocalContentColor.current,
|
||||
modifier =
|
||||
Modifier.semantics { contentDescription = sendLabel(running) },
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1832,7 +1964,7 @@ fun SessionScreen(
|
||||
// is the screen's business rather than any row's. See [SessionImageViewer].
|
||||
fullImage?.let { ref -> SessionImageViewer(settings, summary.id, ref) { fullImage = null } }
|
||||
if (usageOpen) {
|
||||
UsageDialog(feed = usageFeed, onDismiss = { usageOpen = false })
|
||||
usageFeed?.let { UsageDialog(feed = it, onDismiss = { usageOpen = false }) }
|
||||
}
|
||||
if (settingsOpen) {
|
||||
// Measured when the dialog opens rather than kept up to date: what the reader is being told
|
||||
@@ -1846,6 +1978,8 @@ fun SessionScreen(
|
||||
settings = settings,
|
||||
sessionId = summary.id,
|
||||
title = title,
|
||||
effort = summary.effort.takeIf { summary.takesEffort },
|
||||
takesEffort = summary.takesEffort,
|
||||
cachedBytes = cachedBytes,
|
||||
// The purge finishes before the epoch moves, because the relaunched opening effect
|
||||
// reads the same directory and would otherwise draw what is about to be deleted. The
|
||||
@@ -1869,6 +2003,7 @@ fun SessionScreen(
|
||||
},
|
||||
onDismiss = { settingsOpen = false },
|
||||
onCopyRenderReport = copyRenderReport,
|
||||
onRunBenchmark = runBenchmark,
|
||||
)
|
||||
}
|
||||
}
|
||||
@@ -2122,6 +2257,13 @@ private fun SessionStatusRow(
|
||||
/** Context the session is holding, or null where nothing has measured it. */
|
||||
contextTokens: Long?,
|
||||
modifier: Modifier = Modifier,
|
||||
/**
|
||||
* Whether this row is for a subagent rather than a session, which changes only one word:
|
||||
* "exited" reads as "finished" there too, the same as the subagent list's own card -- a
|
||||
* subagent's process was always its parent's, so "exited" would read as a fault rather than the
|
||||
* ordinary way one of these ends.
|
||||
*/
|
||||
subagent: Boolean = false,
|
||||
) {
|
||||
DebugStats.count("status row recomposed")
|
||||
Row(
|
||||
@@ -2178,7 +2320,7 @@ private fun SessionStatusRow(
|
||||
Text(
|
||||
when (status) {
|
||||
"idle" -> "idle"
|
||||
"exited" -> "exited"
|
||||
"exited" -> if (subagent) "finished" else "exited"
|
||||
"awaitingInput" -> "your turn"
|
||||
"unknown" -> "can't tell"
|
||||
else -> status
|
||||
@@ -2249,7 +2391,7 @@ private const val ONE_TAP_MS = 250L
|
||||
* session is set to without spending a second line on saying it.
|
||||
*/
|
||||
@Composable
|
||||
private fun PickerButton(current: String, options: List<String>, onPick: (String) -> Unit) {
|
||||
fun PickerButton(current: String, options: List<String>, onPick: (String) -> Unit) {
|
||||
var open by remember { mutableStateOf(false) }
|
||||
// When an outside touch last closed the menu.
|
||||
//
|
||||
|
||||
@@ -6,8 +6,10 @@ import androidx.compose.foundation.layout.Spacer
|
||||
import androidx.compose.foundation.layout.fillMaxWidth
|
||||
import androidx.compose.foundation.layout.height
|
||||
import androidx.compose.foundation.layout.width
|
||||
import androidx.compose.foundation.rememberScrollState
|
||||
import androidx.compose.foundation.text.KeyboardActions
|
||||
import androidx.compose.foundation.text.KeyboardOptions
|
||||
import androidx.compose.foundation.verticalScroll
|
||||
import androidx.compose.material3.AlertDialog
|
||||
import androidx.compose.material3.CircularProgressIndicator
|
||||
import androidx.compose.material3.MaterialTheme
|
||||
@@ -26,6 +28,10 @@ import androidx.compose.ui.Alignment
|
||||
import androidx.compose.ui.Modifier
|
||||
import androidx.compose.ui.text.input.ImeAction
|
||||
import androidx.compose.ui.unit.dp
|
||||
import java.time.Instant
|
||||
import java.time.ZoneId
|
||||
import java.time.format.DateTimeFormatter
|
||||
import java.time.format.FormatStyle
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.launch
|
||||
import kotlinx.coroutines.withContext
|
||||
@@ -54,6 +60,16 @@ fun SessionSettingsDialog(
|
||||
*/
|
||||
title: String,
|
||||
onRenamed: (String) -> Unit,
|
||||
/**
|
||||
* How hard the model thinks, as the session reports it, or null for the CLI's own default.
|
||||
*
|
||||
* Taken from the row this dialog was opened over rather than fetched, because unlike the
|
||||
* notification switch there is nothing else that changes it: the level is this app's to set and
|
||||
* the server does not resolve it into something else.
|
||||
*/
|
||||
effort: String?,
|
||||
/** Whether a level does anything here; the row is left out entirely where it does not. */
|
||||
takesEffort: Boolean,
|
||||
/**
|
||||
* What this phone is holding of the conversation, or null while that is being measured -- see
|
||||
* the Reload row below, which is what would discard it.
|
||||
@@ -66,9 +82,18 @@ fun SessionSettingsDialog(
|
||||
* measures is that screen's own state.
|
||||
*/
|
||||
onCopyRenderReport: () -> Unit,
|
||||
/**
|
||||
* Runs P0's scripted scroll-and-stream benchmark and copies the extended report, or null on
|
||||
* every build but `bench` -- see [BuildConfig.FIXTURE_MODE] and BenchRun.kt. Null rather than
|
||||
* always-present-but-disabled: this has no meaning at all outside the bench build, and a
|
||||
* control with nothing behind it on every other build is not a state worth drawing.
|
||||
*/
|
||||
onRunBenchmark: (() -> Unit)? = null,
|
||||
) {
|
||||
val scope = rememberCoroutineScope()
|
||||
var name by remember(sessionId) { mutableStateOf(title) }
|
||||
var level by remember(sessionId) { mutableStateOf(effort) }
|
||||
var effortError by remember { mutableStateOf<String?>(null) }
|
||||
var saving by remember { mutableStateOf(false) }
|
||||
var error by remember { mutableStateOf<String?>(null) }
|
||||
// Null until the server has been asked. The row this dialog was opened over is a snapshot of
|
||||
@@ -77,6 +102,16 @@ fun SessionSettingsDialog(
|
||||
// and a spinner sits beside it, which is what not knowing looks like.
|
||||
var notify by remember(sessionId) { mutableStateOf<Boolean?>(null) }
|
||||
var notifyError by remember { mutableStateOf<String?>(null) }
|
||||
// The same three-state shape the notification switch has, for the same reason: until the
|
||||
// server has answered, the switch is disabled rather than showing a position nothing confirmed.
|
||||
var autoResume by remember(sessionId) { mutableStateOf<Boolean?>(null) }
|
||||
var resumeMessage by remember(sessionId) { mutableStateOf(DEFAULT_RESUME_MESSAGE) }
|
||||
// When the server next intends to ask whether the limit has lifted, or null when nothing is
|
||||
// waiting. Read once with everything else: it moves on the server's schedule, not this
|
||||
// screen's, and a figure that redrew itself here would be this app re-measuring what it was
|
||||
// told.
|
||||
var resumeAt by remember(sessionId) { mutableStateOf<Double?>(null) }
|
||||
var resumeError by remember { mutableStateOf<String?>(null) }
|
||||
// Where the session works. Null until the server has been asked, for the same reason the switch
|
||||
// above is. An empty answer is a session that was never given a directory, which is not the
|
||||
// same as one whose directory is unknown -- the field is only enabled once one of those is
|
||||
@@ -90,6 +125,9 @@ fun SessionSettingsDialog(
|
||||
try {
|
||||
val fresh = withContext(Dispatchers.IO) { fetchSession(settings, sessionId) }
|
||||
notify = fresh.notify
|
||||
autoResume = fresh.autoResume
|
||||
resumeMessage = fresh.autoResumeMessage
|
||||
resumeAt = fresh.resumeAt
|
||||
cwd = fresh.cwd.orEmpty()
|
||||
typedCwd = fresh.cwd.orEmpty()
|
||||
} catch (e: ApiException) {
|
||||
@@ -97,6 +135,8 @@ fun SessionSettingsDialog(
|
||||
// instead of offering a position nothing confirmed.
|
||||
notifyError = e.message
|
||||
notify = null
|
||||
resumeError = e.message
|
||||
autoResume = null
|
||||
}
|
||||
}
|
||||
|
||||
@@ -125,6 +165,26 @@ fun SessionSettingsDialog(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Chooses a thinking level, which ends the process the old level was launched with.
|
||||
*
|
||||
* Put back if the request is refused, for the reason the notification switch below gives: a
|
||||
* control that stays where it was put after a refusal is stating something untrue.
|
||||
*/
|
||||
fun setEffort(chosen: String?) {
|
||||
val was = level
|
||||
level = chosen
|
||||
effortError = null
|
||||
scope.launch {
|
||||
try {
|
||||
withContext(Dispatchers.IO) { setSessionEffort(settings, sessionId, chosen) }
|
||||
} catch (e: ApiException) {
|
||||
level = was
|
||||
effortError = e.message
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Moved optimistically so the switch answers the finger that moved it, and put back if the
|
||||
// request is refused -- a switch that waits for a round trip reads as broken on a slow tunnel,
|
||||
// and one that stays moved after a refusal lies.
|
||||
@@ -142,6 +202,39 @@ fun SessionSettingsDialog(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Turns auto-resume on or off, or changes what it would say.
|
||||
*
|
||||
* One request for both, because the server takes one: switching it on and typing the message
|
||||
* are two halves of the same decision, and sending them separately would leave a moment where
|
||||
* the session is armed with the old words.
|
||||
*
|
||||
* Put back if refused, like the notification switch. Turning it off also clears what was
|
||||
* scheduled -- said here rather than only on the server, or the row would go on naming a time
|
||||
* that no longer exists.
|
||||
*/
|
||||
fun setAutoResume(on: Boolean, message: String) {
|
||||
val wasOn = autoResume
|
||||
val wasMessage = resumeMessage
|
||||
val wasAt = resumeAt
|
||||
autoResume = on
|
||||
resumeMessage = message
|
||||
if (!on) resumeAt = null
|
||||
resumeError = null
|
||||
scope.launch {
|
||||
try {
|
||||
withContext(Dispatchers.IO) {
|
||||
setSessionAutoResume(settings, sessionId, on, message)
|
||||
}
|
||||
} catch (e: ApiException) {
|
||||
autoResume = wasOn
|
||||
resumeMessage = wasMessage
|
||||
resumeAt = wasAt
|
||||
resumeError = e.message
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Nothing to do when the name has not changed, so the button says so rather than sending a
|
||||
// request whose success would look exactly like the failure of having typed nothing.
|
||||
val changed = name.trim().isNotEmpty() && name.trim() != title
|
||||
@@ -168,7 +261,10 @@ fun SessionSettingsDialog(
|
||||
onDismissRequest = onDismiss,
|
||||
title = { Text("Session settings") },
|
||||
text = {
|
||||
Column {
|
||||
// Scrollable, because this dialog grew past a screenful: a Material dialog constrains
|
||||
// its own height and clips what does not fit, so the last control on the list is one
|
||||
// large system font away from being unreachable with nothing on screen to say so.
|
||||
Column(Modifier.verticalScroll(rememberScrollState())) {
|
||||
OutlinedTextField(
|
||||
value = name,
|
||||
onValueChange = { name = it },
|
||||
@@ -212,6 +308,70 @@ fun SessionSettingsDialog(
|
||||
)
|
||||
}
|
||||
Spacer(Modifier.height(8.dp))
|
||||
Row(
|
||||
verticalAlignment = Alignment.CenterVertically,
|
||||
modifier = Modifier.fillMaxWidth(),
|
||||
) {
|
||||
Text("Resume after a usage limit", modifier = Modifier.weight(1f))
|
||||
if (autoResume == null && resumeError == null) {
|
||||
CircularProgressIndicator(
|
||||
modifier = Modifier.width(16.dp).height(16.dp),
|
||||
strokeWidth = 2.dp,
|
||||
)
|
||||
Spacer(Modifier.width(8.dp))
|
||||
}
|
||||
Switch(
|
||||
checked = autoResume == true,
|
||||
onCheckedChange = { setAutoResume(it, resumeMessage) },
|
||||
enabled = autoResume != null,
|
||||
)
|
||||
}
|
||||
// Disabled rather than hidden while the switch is off: a field that comes and goes
|
||||
// makes its own presence the signal, and a visible one teaches what the switch will
|
||||
// do. Committed on the keyboard's Done rather than on every keystroke, so typing a
|
||||
// sentence is one request instead of one per letter.
|
||||
OutlinedTextField(
|
||||
value = resumeMessage,
|
||||
onValueChange = { resumeMessage = it },
|
||||
label = { Text("Message to send") },
|
||||
// What an empty field means, in the field: the server's own word rather than a
|
||||
// session poked with nothing to read.
|
||||
placeholder = { Text(DEFAULT_RESUME_MESSAGE) },
|
||||
singleLine = true,
|
||||
enabled = autoResume == true,
|
||||
modifier = Modifier.fillMaxWidth(),
|
||||
keyboardOptions = KeyboardOptions(imeAction = ImeAction.Done),
|
||||
keyboardActions =
|
||||
KeyboardActions(onDone = { setAutoResume(true, resumeMessage) }),
|
||||
)
|
||||
// What it does and what it costs, in the order it happens. The last sentence is the
|
||||
// one that matters: the time below is when the server will *ask*, not a promise
|
||||
// about when the session speaks.
|
||||
Text(
|
||||
"When this session stops because the account is out of quota, the server " +
|
||||
"checks the limit and sends this message once it has lifted. It checks " +
|
||||
"again if the limit is still on.",
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
)
|
||||
// Only where something is actually waiting. Absent is not a state worth a row: a
|
||||
// session that has not hit a limit has nothing scheduled, which the reader can see
|
||||
// from the switch.
|
||||
resumeAt?.let { at ->
|
||||
Text(
|
||||
"Waiting now -- next check ${formatCheckTime(at)}.",
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
)
|
||||
}
|
||||
resumeError?.let {
|
||||
Text(
|
||||
it,
|
||||
color = MaterialTheme.colorScheme.error,
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
)
|
||||
}
|
||||
Spacer(Modifier.height(8.dp))
|
||||
Row(
|
||||
verticalAlignment = Alignment.CenterVertically,
|
||||
modifier = Modifier.fillMaxWidth(),
|
||||
@@ -257,6 +417,44 @@ fun SessionSettingsDialog(
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
)
|
||||
}
|
||||
// Left out rather than disabled, the one place this dialog does that: a disabled
|
||||
// control teaches what the thing can do, and a llama session cannot do this at all
|
||||
// -- the row would be teaching something false about it.
|
||||
if (takesEffort) {
|
||||
Spacer(Modifier.height(8.dp))
|
||||
Row(
|
||||
verticalAlignment = Alignment.CenterVertically,
|
||||
modifier = Modifier.fillMaxWidth(),
|
||||
) {
|
||||
Text("Thinking", modifier = Modifier.weight(1f))
|
||||
PickerButton(
|
||||
current = level ?: DEFAULT_EFFORT,
|
||||
// The level the CLI picks for itself is in the list as well as in the
|
||||
// button, so leaving a level is not a one-way trip -- the same
|
||||
// correction the model picker carries.
|
||||
options = listOf(DEFAULT_EFFORT) + EFFORT_LEVELS,
|
||||
onPick = { chosen ->
|
||||
setEffort(chosen.takeIf { it != DEFAULT_EFFORT })
|
||||
},
|
||||
)
|
||||
}
|
||||
// What it costs, said where it is about to be pressed, like Move above: the
|
||||
// CLI reads the level when it launches and has no control request for
|
||||
// changing one.
|
||||
Text(
|
||||
"Changing this stops the session's process. It starts again with the " +
|
||||
"next message, or with Start.",
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
)
|
||||
effortError?.let {
|
||||
Text(
|
||||
it,
|
||||
color = MaterialTheme.colorScheme.error,
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
)
|
||||
}
|
||||
}
|
||||
Spacer(Modifier.height(8.dp))
|
||||
Row(
|
||||
verticalAlignment = Alignment.CenterVertically,
|
||||
@@ -317,6 +515,21 @@ fun SessionSettingsDialog(
|
||||
Text("Render timings", modifier = Modifier.weight(1f))
|
||||
TextButton(onClick = onCopyRenderReport) { Text("Copy") }
|
||||
}
|
||||
// Bench-build only: see [onRunBenchmark]. Named exactly "Run benchmark" because
|
||||
// ui-trace and the emulator smoke run find it by that label, the same way every
|
||||
// other control here is found -- see AGENTS.md's "Driving the UI".
|
||||
onRunBenchmark?.let { run ->
|
||||
Spacer(Modifier.height(8.dp))
|
||||
Row(
|
||||
verticalAlignment = Alignment.CenterVertically,
|
||||
modifier = Modifier.fillMaxWidth(),
|
||||
) {
|
||||
Glyph(SPEED_GLYPH, colour = MaterialTheme.colorScheme.onSurface)
|
||||
Spacer(Modifier.width(8.dp))
|
||||
Text("P0 benchmark", modifier = Modifier.weight(1f))
|
||||
TextButton(onClick = run) { Text("Run benchmark") }
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
// Disabled rather than absent while there is nothing to save: a button that comes and goes
|
||||
@@ -329,3 +542,21 @@ fun SessionSettingsDialog(
|
||||
dismissButton = { TextButton(onClick = onDismiss) { Text("Close") } },
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* When the server will next look, as a local time.
|
||||
*
|
||||
* A time rather than a countdown, for the reason the transcript's own limit row gives: this screen
|
||||
* reads the figure once, and a span drawn from a value nothing refreshes goes stale while somebody
|
||||
* is looking at it.
|
||||
*/
|
||||
private fun formatCheckTime(epochSeconds: Double): String =
|
||||
try {
|
||||
DateTimeFormatter.ofLocalizedTime(FormatStyle.SHORT)
|
||||
.withZone(ZoneId.systemDefault())
|
||||
.format(Instant.ofEpochSecond(epochSeconds.toLong()))
|
||||
} catch (_: Exception) {
|
||||
// A time that cannot be read is not a time to show: the sentence above still says a check
|
||||
// is coming, which is the part the reader can act on.
|
||||
"soon"
|
||||
}
|
||||
@@ -70,13 +70,22 @@ class UsageFeed(
|
||||
/** Ask the backend again now. The dialog's refresh button; the poll does it on its own. */
|
||||
val refresh: () -> Unit,
|
||||
) {
|
||||
/** What [setup]'s own limits came back as. See [usageFor] for why the states are these. */
|
||||
fun forSetup(setup: String): SessionUsage =
|
||||
when (val state = snapshots) {
|
||||
/**
|
||||
* What meters [session], and what that meter came back as. See [usageFor] for the states.
|
||||
*
|
||||
* A session rather than a machine, because a machine is not what is metered: one machine runs
|
||||
* the Claude CLI and an echo session side by side, and only the first of them spends anything.
|
||||
*/
|
||||
fun forSession(session: SessionSummary): SessionUsage {
|
||||
// Settled without asking anybody: a session nothing meters has nothing to check, and
|
||||
// "checking" is what the fetch's own states would say about it for as long as one is out.
|
||||
val provider = session.usageProvider ?: return SessionUsage.NotMetered
|
||||
return when (val state = snapshots) {
|
||||
is LoadState.Loading -> SessionUsage.Waiting
|
||||
is LoadState.Error -> SessionUsage.Unavailable(state.message)
|
||||
is LoadState.Loaded -> usageFor(state.value, setup)
|
||||
is LoadState.Loaded -> usageFor(state.value, session.setup, provider)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -158,9 +167,15 @@ fun SessionUsageBar(usage: SessionUsage, modifier: Modifier = Modifier) {
|
||||
}
|
||||
}
|
||||
|
||||
// Nothing at all for a machine that meters nothing: a row saying "unknown" there would report a
|
||||
// problem about a setup somebody chose, on every screen, forever.
|
||||
if (usage is SessionUsage.NotMetered) {
|
||||
// Nothing at all for a session that meters nothing: a row saying "unknown" there would report
|
||||
// a problem about a setup somebody chose, on every screen, forever.
|
||||
//
|
||||
// And nothing while the first fetch is out, which is a different silence. A request in flight
|
||||
// is not a state to report -- and the session that meters nothing is exactly the one this
|
||||
// cannot yet tell apart, so "5-hour usage: checking" appeared under an echo session for half a
|
||||
// second and was then taken away. A row that has to be withdrawn is worse than one that
|
||||
// arrives late.
|
||||
if (usage is SessionUsage.NotMetered || usage is SessionUsage.Waiting) {
|
||||
return
|
||||
}
|
||||
|
||||
@@ -171,9 +186,10 @@ fun SessionUsageBar(usage: SessionUsage, modifier: Modifier = Modifier) {
|
||||
// Words, not a colour and not an empty bar: every one of these is a different kind of
|
||||
// answer from "this much is used", and only words carry a difference in kind.
|
||||
when (val state = usage) {
|
||||
SessionUsage.NotMetered -> Unit
|
||||
// Both handled above, before the row exists at all.
|
||||
SessionUsage.NotMetered,
|
||||
SessionUsage.Waiting -> Unit
|
||||
is SessionUsage.Unavailable -> UsageNote("5-hour usage unknown -- ${state.why}")
|
||||
SessionUsage.Waiting -> UsageNote("5-hour usage: checking")
|
||||
is SessionUsage.Known -> {
|
||||
val window = state.windows.firstOrNull { it.kind == "session" }
|
||||
if (window == null) {
|
||||
@@ -234,16 +250,22 @@ private fun fiveHourLabel(window: UsageWindow, now: OffsetDateTime): String {
|
||||
}
|
||||
|
||||
/**
|
||||
* One machine's snapshot, out of every machine's.
|
||||
* One meter's snapshot, out of every machine's: [setup]'s row for [provider].
|
||||
*
|
||||
* Both halves are needed to pick it. A machine can hold more than one meter -- the Claude CLI's
|
||||
* account and, while a test has one set, an echo session's invented one -- and a snapshot is one
|
||||
* service on one machine.
|
||||
*
|
||||
* Every way of having *failed* to get numbers is [SessionUsage.Unavailable] with the reason in it.
|
||||
* None of them may look like zero, and none may look like [SessionUsage.NotMetered], which is the
|
||||
* machine having no quota rather than the question going unanswered.
|
||||
*/
|
||||
fun usageFor(snapshots: List<UsageSnapshot>, setup: String): SessionUsage {
|
||||
// No snapshot at all means the backend never asked, which it only does for a machine with
|
||||
// nothing metered on it. That is a different answer from having asked and failed.
|
||||
val mine = snapshots.firstOrNull { it.setup == setup } ?: return SessionUsage.NotMetered
|
||||
fun usageFor(snapshots: List<UsageSnapshot>, setup: String, provider: String): SessionUsage {
|
||||
// No snapshot at all means the backend never asked, which it only does where there is nothing
|
||||
// to ask about. That is a different answer from having asked and failed.
|
||||
val mine =
|
||||
snapshots.firstOrNull { it.setup == setup && it.provider == provider }
|
||||
?: return SessionUsage.NotMetered
|
||||
if (mine.state != "ok") {
|
||||
return SessionUsage.Unavailable(mine.detail ?: mine.state)
|
||||
}
|
||||
|
||||
@@ -236,6 +236,7 @@ private fun AddSetupDialog(
|
||||
var address by remember { mutableStateOf("") }
|
||||
var identity by remember { mutableStateOf("") }
|
||||
var attachmentsDir by remember { mutableStateOf("") }
|
||||
var modelsDir by remember { mutableStateOf("") }
|
||||
var tested by remember { mutableStateOf<String?>(null) }
|
||||
var testing by remember { mutableStateOf(false) }
|
||||
|
||||
@@ -250,6 +251,7 @@ private fun AddSetupDialog(
|
||||
port = typedPort,
|
||||
identityFile = identity.trim().ifEmpty { null },
|
||||
attachmentsDir = attachmentsDir.trim().ifEmpty { null },
|
||||
modelsDir = modelsDir.trim().ifEmpty { null },
|
||||
)
|
||||
}
|
||||
|
||||
@@ -293,6 +295,14 @@ private fun AddSetupDialog(
|
||||
label = { Text("Folder for attached files (optional)") },
|
||||
singleLine = true,
|
||||
)
|
||||
// Where that machine's GGUFs are, for a llama.cpp session on it. Blank means
|
||||
// the same place this backend keeps its own downloads, read on that machine.
|
||||
OutlinedTextField(
|
||||
value = modelsDir,
|
||||
onValueChange = { modelsDir = it },
|
||||
label = { Text("Folder for models (optional)") },
|
||||
singleLine = true,
|
||||
)
|
||||
tested?.let {
|
||||
Spacer(Modifier.height(8.dp))
|
||||
Text(it, style = MaterialTheme.typography.bodySmall)
|
||||
|
||||
@@ -61,18 +61,31 @@ fun SpawnScreen(
|
||||
// "auto" rather than "manual": on a phone every ask is a round trip to a question card, and
|
||||
// answering "allow Bash?" dozens of times per task is what this app exists to avoid.
|
||||
var permissionMode by remember { mutableStateOf("auto") }
|
||||
// Null until the server has been asked, and null again if it answers "no level chosen" -- the
|
||||
// two are told apart by [defaultsAsked], because a picker that shows a level before the answer
|
||||
// arrives is one you can spawn at without having chosen it.
|
||||
var effort by remember { mutableStateOf<String?>(null) }
|
||||
var defaultsAsked by remember { mutableStateOf(false) }
|
||||
var busy by remember { mutableStateOf(false) }
|
||||
// Only the spawn's own failure. The fetch's lives in `options`: this one leaves a filled-in
|
||||
// form worth keeping, and that one leaves nothing to fill in.
|
||||
var spawnError by remember { mutableStateOf<String?>(null) }
|
||||
// Downloaded models, for a llama provider to choose between. Kept separate from the setups: a
|
||||
// Claude session needs none, so failing to list them must not stop the screen rendering.
|
||||
// The models on the *chosen machine*, for a llama provider to choose between. Kept separate
|
||||
// from the setups: a Claude session needs none, so failing to list them must not stop the
|
||||
// screen rendering. Refetched when the machine changes, because a model is a file on one
|
||||
// machine -- see [fetchSetupModels].
|
||||
var models by remember { mutableStateOf<List<LocalModel>>(emptyList()) }
|
||||
var modelKey by remember { mutableStateOf<String?>(null) }
|
||||
var contextSize by remember { mutableStateOf("") }
|
||||
var temperature by remember { mutableStateOf("") }
|
||||
|
||||
LaunchedEffect(Unit) {
|
||||
// Separate from the setups fetch below and deliberately not fatal: failing to learn the
|
||||
// default must leave a screen you can still spawn from, so the picker stays on "default"
|
||||
// and says so rather than the whole form refusing to draw.
|
||||
runCatching { withContext(Dispatchers.IO) { fetchDefaultEffort(settings) } }
|
||||
.onSuccess { effort = it }
|
||||
defaultsAsked = true
|
||||
options =
|
||||
try {
|
||||
val fetched = withContext(Dispatchers.IO) { fetchSetups(settings) }
|
||||
@@ -83,9 +96,6 @@ fun SpawnScreen(
|
||||
} catch (e: ApiException) {
|
||||
LoadState.failed(e)
|
||||
}
|
||||
models =
|
||||
runCatching { withContext(Dispatchers.IO) { fetchModels(settings).local } }
|
||||
.getOrDefault(emptyList())
|
||||
}
|
||||
|
||||
Column(Modifier.fillMaxSize().verticalScroll(rememberScrollState()).padding(16.dp)) {
|
||||
@@ -115,6 +125,17 @@ fun SpawnScreen(
|
||||
is LoadState.Loaded -> state.value
|
||||
}
|
||||
val setup = setups.firstOrNull { it.name == setupName }
|
||||
// Whichever machine is chosen now, asked again when that changes. The old machine's list
|
||||
// is dropped first rather than left on screen: a file name from another machine looks
|
||||
// exactly like one from this one.
|
||||
LaunchedEffect(setup?.id) {
|
||||
models = emptyList()
|
||||
modelKey = null
|
||||
val id = setup?.id ?: return@LaunchedEffect
|
||||
models =
|
||||
runCatching { withContext(Dispatchers.IO) { fetchSetupModels(settings, id) } }
|
||||
.getOrDefault(emptyList())
|
||||
}
|
||||
val current = setup?.providers?.firstOrNull { it.name == providerName }
|
||||
// Only the Claude CLI has models, a working directory and permission modes; keying the
|
||||
// extra fields on the kind rather than the provider name keeps a second Claude provider
|
||||
@@ -175,12 +196,13 @@ fun SpawnScreen(
|
||||
)
|
||||
|
||||
if (isLlama) {
|
||||
// A llama session names one of the models this backend has downloaded, so the choice is
|
||||
// that list rather than free text -- a name that is not on disk is a session that
|
||||
// cannot start.
|
||||
// A llama session names one of the models on the machine it will run on, so the
|
||||
// choice is that list rather than free text -- a name that is not on that machine's
|
||||
// disk is a session that cannot start.
|
||||
if (models.isEmpty()) {
|
||||
Text(
|
||||
"No models downloaded yet. Get one from the Models screen first.",
|
||||
"No models on ${setup?.name ?: "this machine"}. The Models screen downloads " +
|
||||
"to the backend; another machine needs the file put there itself.",
|
||||
style = MaterialTheme.typography.bodyMedium,
|
||||
color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
)
|
||||
@@ -251,6 +273,19 @@ fun SpawnScreen(
|
||||
selected = permissionMode,
|
||||
onSelect = { permissionMode = it },
|
||||
)
|
||||
Spacer(Modifier.height(16.dp))
|
||||
|
||||
// Says what it does to *later* spawns as well, because it does: the level chosen here
|
||||
// is stored as the default, which is the whole way that default is set. A picker that
|
||||
// quietly changed a global would be the same control with the fact left out.
|
||||
ChipGroup(
|
||||
label = "Thinking (kept as the default for new sessions)",
|
||||
options = listOf(DEFAULT_EFFORT) + EFFORT_LEVELS,
|
||||
// The CLI's own default is a level in the list, so this cannot be a one-way trip.
|
||||
// Disabled-looking until the server has answered, for the reason above.
|
||||
selected = if (defaultsAsked) effort ?: DEFAULT_EFFORT else null,
|
||||
onSelect = { chosen -> effort = chosen.takeIf { it != DEFAULT_EFFORT } },
|
||||
)
|
||||
}
|
||||
Spacer(Modifier.height(24.dp))
|
||||
|
||||
@@ -268,6 +303,13 @@ fun SpawnScreen(
|
||||
try {
|
||||
val spawned =
|
||||
withContext(Dispatchers.IO) {
|
||||
// Stored before the spawn and not after it: choosing a level is
|
||||
// an intent about new sessions in general, so a spawn that then
|
||||
// fails must not also lose the choice. Non-fatal for the same
|
||||
// reason the fetch above is -- the session is what was asked for.
|
||||
if (isClaude) {
|
||||
runCatching { setDefaultEffort(settings, effort) }
|
||||
}
|
||||
spawnSession(
|
||||
settings,
|
||||
// The id, not the label: labels are editable and the server
|
||||
@@ -280,6 +322,7 @@ fun SpawnScreen(
|
||||
if (isLlama) modelKey else model.trim().takeIf { isClaude },
|
||||
cwd = cwd.trim().takeIf { isClaude },
|
||||
permissionMode = permissionMode.takeIf { isClaude },
|
||||
effort = effort.takeIf { isClaude },
|
||||
// Sent only when set, so blank means "whatever llama.cpp does
|
||||
// by default" rather than a zero.
|
||||
params =
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
package com.example.aiapp
|
||||
|
||||
/**
|
||||
* Where one transcript lives: a session's own, or one of its subagents'.
|
||||
*
|
||||
* The single mechanism [fetchTranscript], [EventStream], [TranscriptSource] and
|
||||
* [TranscriptCache.session] all take, rather than each growing its own branch between a session and
|
||||
* a subagent -- see SUBAGENTS.md's "Phone" and "Wire shape". A caller that has only a session id
|
||||
* builds one with the one-argument constructor; a subagent's screen supplies both ids.
|
||||
*/
|
||||
data class TranscriptAddress(val sessionId: String, val subagentId: String? = null) {
|
||||
/** The URL segment naming this transcript, before `/transcript` or `/events`. */
|
||||
val urlPath: String
|
||||
get() =
|
||||
if (subagentId == null) "sessions/$sessionId"
|
||||
else "sessions/$sessionId/subagents/$subagentId"
|
||||
|
||||
/**
|
||||
* Where this transcript's cache lives on the phone, relative to the cache root.
|
||||
*
|
||||
* A subagent's nests under its session's directory rather than sitting beside it, so deleting a
|
||||
* session's cache directory takes its subagents' with it -- the same one-way door the server's
|
||||
* own storage describes.
|
||||
*/
|
||||
val cachePath: String
|
||||
get() = if (subagentId == null) sessionId else "$sessionId/subagents/$subagentId"
|
||||
}
|
||||
@@ -35,8 +35,15 @@ class TranscriptCache(
|
||||
private val root: File,
|
||||
private val warn: (String) -> Unit = { Log.w("ai-app", it) },
|
||||
) {
|
||||
/** The cache for one session, whether or not anything has been stored for it yet. */
|
||||
fun session(id: String): SessionCache = SessionCache(File(root, id), warn)
|
||||
/**
|
||||
* The cache for one transcript, whether or not anything has been stored for it yet.
|
||||
*
|
||||
* A subagent's [TranscriptAddress.cachePath] nests it under its session's directory, so
|
||||
* deleting the session (below) takes its subagents' caches with it -- there is no separate
|
||||
* purge for one.
|
||||
*/
|
||||
fun session(address: TranscriptAddress): SessionCache =
|
||||
SessionCache(File(root, address.cachePath), warn)
|
||||
|
||||
/**
|
||||
* Deletes every session directory not in [ids], called after a successful list fetch. The path
|
||||
|
||||
@@ -175,6 +175,18 @@ sealed class TranscriptItem {
|
||||
val preTokens: Long?,
|
||||
val postTokens: Long?,
|
||||
) : TranscriptItem()
|
||||
|
||||
/**
|
||||
* The account ran out of quota, so the turn stopped here.
|
||||
*
|
||||
* A divider rather than an error: nothing failed, and what a reader scrolling back needs from
|
||||
* it is the same thing a clear or a compaction gives them -- why the conversation stops at this
|
||||
* line.
|
||||
*
|
||||
* [resetsAt] is epoch seconds and null where the session was told nothing, which is a state the
|
||||
* row has words for rather than a time it invents.
|
||||
*/
|
||||
data class LimitNote(override val seq: Long, val resetsAt: Double?) : TranscriptItem()
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -458,6 +470,7 @@ fun foldEvent(items: List<TranscriptItem>, entry: SeqEvent): List<TranscriptItem
|
||||
} else {
|
||||
items + TranscriptItem.ImageItem(entry.seq, event.ref)
|
||||
}
|
||||
is SessionEvent.LimitReached -> items + TranscriptItem.LimitNote(entry.seq, event.resetsAt)
|
||||
is SessionEvent.Cleared -> items + TranscriptItem.ClearedNote(entry.seq)
|
||||
is SessionEvent.Compacted ->
|
||||
items + TranscriptItem.CompactedNote(entry.seq, event.preTokens, event.postTokens)
|
||||
|
||||
@@ -18,7 +18,7 @@ import java.util.concurrent.atomic.AtomicReference
|
||||
*/
|
||||
class TranscriptSource(
|
||||
private val settings: ServerSettings,
|
||||
private val sessionId: String,
|
||||
private val address: TranscriptAddress,
|
||||
val cache: SessionCache,
|
||||
) {
|
||||
private val stream = AtomicReference<EventStream?>(null)
|
||||
@@ -65,7 +65,7 @@ class TranscriptSource(
|
||||
val tail = cache.tail() ?: return false
|
||||
// `before = seq + 1` is the newest event with seq <= the cursor, which is the event *at*
|
||||
// the cursor when the server still has one there.
|
||||
val answer = fetchTranscript(settings, sessionId, before = tail.seq + 1, limit = 1)
|
||||
val answer = fetchTranscript(settings, address, before = tail.seq + 1, limit = 1)
|
||||
val matches =
|
||||
answer.size == 1 &&
|
||||
try {
|
||||
@@ -83,7 +83,7 @@ class TranscriptSource(
|
||||
*/
|
||||
suspend fun fetchOpening(): List<SeqEvent> {
|
||||
DebugStats.count("transcript page from server")
|
||||
val page = fetchTranscript(settings, sessionId, limit = OPENING_WINDOW)
|
||||
val page = fetchTranscript(settings, address, limit = OPENING_WINDOW)
|
||||
page.forEach { (line, entry) -> cache.append(line, entry.seq) }
|
||||
cache.flush()
|
||||
return page.map { it.second }
|
||||
@@ -108,7 +108,7 @@ class TranscriptSource(
|
||||
val page =
|
||||
fetchTranscript(
|
||||
settings,
|
||||
sessionId,
|
||||
address,
|
||||
before = before,
|
||||
limit = limit,
|
||||
coalesce = coalesce,
|
||||
@@ -131,7 +131,7 @@ class TranscriptSource(
|
||||
* well lose.
|
||||
*/
|
||||
fun follow(after: Long, onOpen: () -> Unit, onReset: () -> Unit, onEvent: (SeqEvent) -> Unit) {
|
||||
val opened = EventStream(settings, sessionId)
|
||||
val opened = EventStream(settings, address)
|
||||
stream.getAndSet(opened)?.close()
|
||||
try {
|
||||
opened.run(after, onOpen, onReset) { raw, entry ->
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<resources>
|
||||
<!-- Overridden by the `bench` build type's resValue (build.gradle.kts) to "AI Sessions bench",
|
||||
so the two are never mistaken for each other in the launcher or in Settings. -->
|
||||
<string name="app_name">AI Sessions</string>
|
||||
</resources>
|
||||
@@ -0,0 +1,32 @@
|
||||
package com.example.aiapp
|
||||
|
||||
import java.time.ZoneId
|
||||
import kotlin.test.Test
|
||||
import kotlin.test.assertEquals
|
||||
import kotlin.test.assertTrue
|
||||
|
||||
/**
|
||||
* What the transcript says where a session ran out of quota.
|
||||
*
|
||||
* The pair worth a test is the one that reads the same when it goes wrong: a reset time that
|
||||
* arrived and one that never did. The second must not turn into a plausible-looking time, because a
|
||||
* reader has no way of telling an invented one from a reported one.
|
||||
*/
|
||||
class LimitRowTest {
|
||||
private val utc = ZoneId.of("UTC")
|
||||
|
||||
@Test
|
||||
fun `a reported reset time is shown as a time`() {
|
||||
// 2026-09-05T12:00:00Z. Asserted as a prefix and the clock reading rather than as the
|
||||
// whole string: the platform's own short-time format is what this asks for, and it
|
||||
// differs by JDK and locale down to which space character separates the meridiem.
|
||||
val summary = limitSummary(1_788_609_600.0, utc)
|
||||
assertTrue(summary.startsWith("Usage limit reached • resets "), summary)
|
||||
assertTrue(summary.contains("12:00"), summary)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `a limit with no reset time says only what is known`() {
|
||||
assertEquals("Usage limit reached", limitSummary(null, utc))
|
||||
}
|
||||
}
|
||||
@@ -23,7 +23,7 @@ class TranscriptCacheTest {
|
||||
|
||||
private fun cache() = TranscriptCache(File(temp, "v1/host_8443")) { said += it }
|
||||
|
||||
private fun session(id: String = "s") = cache().session(id)
|
||||
private fun session(id: String = "s") = cache().session(TranscriptAddress(id))
|
||||
|
||||
private fun line(seq: Long, type: String = "toolStart") =
|
||||
"""{"seq":$seq,"ts":1.5,"type":"$type","id":"x"}"""
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
# The P0 benchmark fixture
|
||||
|
||||
`transcript.jsonl` is a synthetic transcript in the app's own event model (the JSON lines
|
||||
`GET /sessions/{id}/transcript` returns; see `Events.kt`'s `parseSeqEvent` and
|
||||
`server/src/session/driver.rs`) -- never a real one. It is what both the Compose `bench` build
|
||||
and iris's bench build open with no server, so the two apps draw exactly the same content and a
|
||||
frame-time comparison is measuring the renderer rather than the data.
|
||||
|
||||
Generated by `./generate.py` (Python stdlib only, seeded -- `SEED = 20260905` -- so re-running it
|
||||
reproduces the same file byte for byte). It writes into `assets/` -- a separate directory from this
|
||||
script and README, because the Compose `bench` build type points its own asset source set straight
|
||||
at `assets/` (`app/androidApp/build.gradle.kts`'s `sourceSets { getByName("bench") }`), and a Python
|
||||
script and a markdown file have no business inside an APK:
|
||||
|
||||
- `transcript.jsonl` -- 3,601 events. The first 3,200 (`BACKLOG_COUNT`) are the scrolled-back
|
||||
history the benchmark opens with: user turns, tool calls with kilobyte-scale input/output,
|
||||
assistant replies built from headings, bold/italic/inline code, a link, fenced code blocks that
|
||||
rotate through rust/kotlin/python/sh/json/toml, a markdown table, two embedded images, and
|
||||
periodic `usageDelta`/`compacted` events. The remaining 400 (`STREAM_COUNT`) are not part of the
|
||||
opening window -- both bench harnesses replay them at a fixed rate (20/s) through the same live
|
||||
fold path a real SSE reply arrives on, which is P0's "streaming phase."
|
||||
- `bench1.png`, `bench2.png` -- tiny (8x8) flat-colour PNGs, base64-free on disk but served the
|
||||
same way a real attachment is (`GET /sessions/{id}/files/{name}`), referenced by the two
|
||||
`"type":"image"` events in the transcript.
|
||||
|
||||
Regenerate after changing the shape (a new event type, a different backlog/stream split) with
|
||||
`./generate.py`, and commit the result -- it is checked in rather than generated at build time so
|
||||
both apps' bench builds embed the identical bytes without needing this script at build time.
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 74 B |
Binary file not shown.
|
After Width: | Height: | Size: 74 B |
File diff suppressed because it is too large.
Load diff
Executable
+186
@@ -0,0 +1,186 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Generates transcript.jsonl -- the synthetic fixture P0's benchmark opens in both apps.
|
||||
|
||||
Deterministic (fixed seed), so a Compose bench APK and an iris bench APK draw byte-identical
|
||||
content: the point of the fixture is a like-for-like comparison, not a realistic one.
|
||||
|
||||
Never a real transcript -- see AGENTS.md's ui-sandbox.sh, which this borrows its vocabulary
|
||||
style from (headings, code fences, a table, a link) rather than reusing its Claude-Code JSONL
|
||||
shape. This file's shape is the *app's own event model* instead: one JSON object per line,
|
||||
matching what GET /sessions/{id}/transcript returns and what Events.kt's parseSeqEvent reads
|
||||
(server/src/session/driver.rs is the source of truth for the field names).
|
||||
|
||||
./generate.py writes transcript.jsonl and bench1.png/bench2.png here
|
||||
|
||||
BACKLOG_COUNT events (seq 1..BACKLOG_COUNT) are the scrolled-back history the benchmark opens
|
||||
with. A further STREAM_COUNT events (seq BACKLOG_COUNT+1..) are not part of the opening window;
|
||||
both bench harnesses replay them at a fixed rate as the "streaming reply" phase, appended through
|
||||
the same live path a real SSE reply arrives on. Keeping both halves in one file means one
|
||||
generator and one seed to keep in sync, rather than two fixtures that can drift apart.
|
||||
"""
|
||||
import base64
|
||||
import json
|
||||
import random
|
||||
import struct
|
||||
import zlib
|
||||
from pathlib import Path
|
||||
|
||||
SEED = 20260905
|
||||
BACKLOG_COUNT = 3200
|
||||
STREAM_COUNT = 400
|
||||
HERE = Path(__file__).resolve().parent / "assets"
|
||||
|
||||
random.seed(SEED)
|
||||
|
||||
LANGUAGES = ["rust", "kotlin", "python", "sh", "json", "toml"]
|
||||
|
||||
CODE_SNIPPETS = {
|
||||
"rust": '''fn fold_event(items: Vec<Item>, seq: u64) -> Vec<Item> {
|
||||
// a comment worth keeping: this is the fold the app's own screen runs
|
||||
let mut out = items;
|
||||
out.push(Item::new(seq));
|
||||
out
|
||||
}''',
|
||||
"kotlin": '''fun foldEvent(items: List<TranscriptItem>, entry: SeqEvent): List<TranscriptItem> {
|
||||
// mirrors the server's own event model, one item per line
|
||||
return items + TranscriptItem.from(entry)
|
||||
}''',
|
||||
"python": '''def render_report(frames, cpu_ms, rss_kb):
|
||||
# printed for a human to paste back, so every number carries its unit
|
||||
return f"{frames} frames, {cpu_ms}ms cpu, {rss_kb}kb peak rss"''',
|
||||
"sh": '''#!/bin/sh
|
||||
# scripted scroll loop, the shape transcript-bench.sh drives on a phone
|
||||
for i in $(seq 1 24); do
|
||||
ui-trace record --do "swipe 540 700 540 1600 200"
|
||||
done''',
|
||||
"json": '{"seq": 1, "type": "status", "state": "running"}',
|
||||
"toml": '''[package]
|
||||
name = "bench-fixture"
|
||||
version = "0.1.0"''',
|
||||
}
|
||||
|
||||
HEADINGS = [
|
||||
"## Plan",
|
||||
"## What changed",
|
||||
"## Why this approach",
|
||||
"### Open questions",
|
||||
"## Results",
|
||||
]
|
||||
|
||||
WORDS = (
|
||||
"session render report frame budget scroll transcript fold event cache "
|
||||
"cursor probe stream backlog swipe fixture bench compose iris widget layout "
|
||||
"measure place draw tool call token context window anchor"
|
||||
).split()
|
||||
|
||||
|
||||
def paragraph(n=24):
|
||||
words = [random.choice(WORDS) for _ in range(n)]
|
||||
words[0] = words[0].capitalize()
|
||||
text = " ".join(words) + "."
|
||||
# Sprinkle markdown inline spans so the syntax highlighter/markdown parser sees a real mix.
|
||||
text = text.replace(" fold ", " **fold** ", 1)
|
||||
text = text.replace(" cursor ", " *cursor* ", 1)
|
||||
text = text.replace(" cache ", " `cache` ", 1)
|
||||
if "bench" in text:
|
||||
text = text.replace(
|
||||
" bench ", " [bench](https://example.com/bench) ", 1
|
||||
)
|
||||
return text
|
||||
|
||||
|
||||
def make_png(rgb, size=8):
|
||||
"""A tiny, valid PNG -- flat colour, no external dependency."""
|
||||
|
||||
def chunk(tag, data):
|
||||
c = tag + data
|
||||
return struct.pack(">I", len(data)) + c + struct.pack(">I", zlib.crc32(c))
|
||||
|
||||
sig = b"\x89PNG\r\n\x1a\n"
|
||||
ihdr = struct.pack(">IIBBBBB", size, size, 8, 2, 0, 0, 0)
|
||||
raw = b""
|
||||
for _ in range(size):
|
||||
raw += b"\x00" + bytes(rgb) * size
|
||||
idat = zlib.compress(raw)
|
||||
return sig + chunk(b"IHDR", ihdr) + chunk(b"IDAT", idat) + chunk(b"IEND", b"")
|
||||
|
||||
|
||||
def main():
|
||||
HERE.mkdir(exist_ok=True)
|
||||
lines = []
|
||||
seq = 1
|
||||
ts = 1_788_000_000.0
|
||||
|
||||
def emit(type_, **fields):
|
||||
nonlocal seq, ts
|
||||
obj = {"seq": seq, "ts": round(ts, 3), "type": type_}
|
||||
obj.update(fields)
|
||||
lines.append(json.dumps(obj, separators=(",", ":")))
|
||||
seq += 1
|
||||
ts += random.uniform(0.05, 2.0)
|
||||
|
||||
emit("status", state="running")
|
||||
emit("settings", model="bench-model", permissionMode="auto")
|
||||
|
||||
image_refs = []
|
||||
turn = 0
|
||||
while seq <= BACKLOG_COUNT:
|
||||
turn += 1
|
||||
emit("userMessage", text=f"Turn {turn}: {paragraph(12)}", id=None, attachments=[])
|
||||
|
||||
# A tool call with kilobyte-scale input/output every few turns.
|
||||
if turn % 3 == 0:
|
||||
tool_id = f"tool-{turn}"
|
||||
big_input = json.dumps({"path": f"/repo/file_{turn}.rs", "content": paragraph(400)})
|
||||
emit("toolStart", id=tool_id, tool="Edit", input=big_input)
|
||||
big_output = "\n".join(paragraph(60) for _ in range(20))
|
||||
emit("toolUpdate", id=tool_id, output=big_output[: len(big_output) // 2])
|
||||
emit("toolEnd", id=tool_id, output=big_output)
|
||||
|
||||
# A reply: a heading, prose, a fenced block in a rotating language, a table, then deltas.
|
||||
emit("assistantText", delta=f"{random.choice(HEADINGS)}\n\n")
|
||||
emit("assistantText", delta=paragraph(30) + "\n\n")
|
||||
lang = LANGUAGES[turn % len(LANGUAGES)]
|
||||
emit("assistantText", delta=f"```{lang}\n{CODE_SNIPPETS[lang]}\n```\n\n")
|
||||
if turn % 5 == 0:
|
||||
emit(
|
||||
"assistantText",
|
||||
delta="| column | value |\n|---|---|\n| a | " + paragraph(3) + " |\n\n",
|
||||
)
|
||||
# A run of small deltas -- the shape a live reply actually streams in.
|
||||
for _ in range(random.randint(3, 8)):
|
||||
emit("assistantText", delta=paragraph(6) + " ")
|
||||
|
||||
# A couple of images, base64 PNGs, the way a real transcript embeds a screenshot.
|
||||
if turn in (10, 40):
|
||||
ref = f"bench{len(image_refs) + 1}.png"
|
||||
image_refs.append(ref)
|
||||
emit("image", ref=ref, about=None)
|
||||
|
||||
emit("usageDelta", tokens=random.randint(200, 4000), context=random.randint(2000, 180000))
|
||||
|
||||
if turn % 15 == 0:
|
||||
emit(
|
||||
"compacted",
|
||||
preTokens=180000,
|
||||
postTokens=20000,
|
||||
trigger="auto",
|
||||
)
|
||||
|
||||
# The streaming-phase tail: one long reply, built entirely from text deltas, the shape a
|
||||
# bench harness replays at a fixed events/sec through the live fold path.
|
||||
emit("userMessage", text="One more, streamed live for the benchmark's timing phase.", id=None, attachments=[])
|
||||
while seq <= BACKLOG_COUNT + STREAM_COUNT:
|
||||
emit("assistantText", delta=paragraph(5) + " ")
|
||||
emit("status", state="idle")
|
||||
|
||||
(HERE / "transcript.jsonl").write_text("\n".join(lines) + "\n")
|
||||
|
||||
(HERE / "bench1.png").write_bytes(make_png((220, 90, 90)))
|
||||
(HERE / "bench2.png").write_bytes(make_png((90, 150, 220)))
|
||||
|
||||
print(f"wrote {len(lines)} events ({BACKLOG_COUNT} backlog + {STREAM_COUNT} stream) to transcript.jsonl")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+9
-3
@@ -4,6 +4,11 @@
|
||||
# ./build-apk.sh the release build, signed (what the phone runs)
|
||||
# ./build-apk.sh debug the debug build, for reproducing something the
|
||||
# emulator scripts would build anyway
|
||||
# ./build-apk.sh bench P0's benchmark build (own app id, "AI Sessions
|
||||
# bench" label, opens straight onto the fixture
|
||||
# session -- see docs/RUST.md's P0 box and
|
||||
# app/bench-fixture/README.md). Signed the same
|
||||
# as release; never touches the CA it pins.
|
||||
#
|
||||
# Dev Updater's `.dev-updater.ron` at the checkout root spells these out as
|
||||
# build modes, one command line each; it passes nothing else, so the word
|
||||
@@ -25,8 +30,9 @@ VARIANT=${1:-release}
|
||||
case "$VARIANT" in
|
||||
release) TASK=assembleRelease ;;
|
||||
debug) TASK=assembleDebug ;;
|
||||
bench) TASK=assembleBench ;;
|
||||
*)
|
||||
echo "build-apk.sh: unknown variant '$VARIANT' (release, debug)" >&2
|
||||
echo "build-apk.sh: unknown variant '$VARIANT' (release, debug, bench)" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
@@ -81,7 +87,7 @@ fi
|
||||
# uninstalling it first: the signatures differ, and Android refuses to
|
||||
# update across them.
|
||||
KEYSTORE="${AI_APP_KEYSTORE:-${XDG_CONFIG_HOME:-$HOME/.config}/ai-app/release.jks}"
|
||||
if [ "$VARIANT" = release ] && [ ! -f "$KEYSTORE" ]; then
|
||||
if { [ "$VARIANT" = release ] || [ "$VARIANT" = bench ]; } && [ ! -f "$KEYSTORE" ]; then
|
||||
KEYTOOL="${JAVA_HOME:+$JAVA_HOME/bin/keytool}"
|
||||
KEYTOOL="${KEYTOOL:-keytool}"
|
||||
if ! command -v "$KEYTOOL" >/dev/null 2>&1; then
|
||||
@@ -97,7 +103,7 @@ if [ "$VARIANT" = release ] && [ ! -f "$KEYSTORE" ]; then
|
||||
-keyalg RSA -keysize 2048 -validity 10000 \
|
||||
-storepass "$PASSWORD" -keypass "$PASSWORD" -dname "CN=ai-app" >/dev/null 2>&1)
|
||||
fi
|
||||
if [ "$VARIANT" = release ]; then
|
||||
if [ "$VARIANT" = release ] || [ "$VARIANT" = bench ]; then
|
||||
AI_APP_KEYSTORE="$KEYSTORE"
|
||||
AI_APP_KEYSTORE_PASSWORD=$(cat "$KEYSTORE.password")
|
||||
export AI_APP_KEYSTORE AI_APP_KEYSTORE_PASSWORD
|
||||
|
||||
Executable
+34
@@ -0,0 +1,34 @@
|
||||
#!/bin/sh
|
||||
# RUST.md's I5 "Where iris's frame time goes" pass (2026-09-05). The same
|
||||
# 24-swipe/6-cycle loop as transcript-bench.sh's, extracted for iris's own
|
||||
# demo app -- transcript-bench.sh itself is Compose-specific (opens by
|
||||
# session title through the Compose app's own UI) and cannot be called
|
||||
# directly against dev.iris.android.demo.
|
||||
#
|
||||
# MUST be run from inside this checkout (not /tmp): ui-trace/adb pick which
|
||||
# emulator to target from the current directory's basename (the
|
||||
# per-checkout-AVD rule), and a previous pass lost two attempts to a `cd`
|
||||
# into /tmp that made this resolve to a nonexistent "tmp" checkout.
|
||||
set -eu
|
||||
cd "$(dirname "$0")"
|
||||
. ./android-env.sh >/dev/null 2>&1
|
||||
|
||||
cycles=${1:-6}
|
||||
|
||||
ui-trace record -d 3000 --do "tap 'Reset frame report'" -o /tmp/iris-bench-reset.txt >/dev/null
|
||||
adb logcat -c
|
||||
|
||||
DO=""
|
||||
i=0
|
||||
while [ "$i" -lt "$cycles" ]; do
|
||||
DO="$DO --do 'swipe 540 700 540 1600 200' --do 'wait 500'"
|
||||
DO="$DO --do 'swipe 540 700 540 1600 200' --do 'wait 500'"
|
||||
DO="$DO --do 'swipe 540 1600 540 700 200' --do 'wait 500'"
|
||||
DO="$DO --do 'swipe 540 1600 540 700 200' --do 'wait 500'"
|
||||
i=$((i + 1))
|
||||
done
|
||||
eval ui-trace record -d $((cycles * 16000 + 20000)) $DO -o /tmp/iris-bench-scroll.txt >/dev/null
|
||||
|
||||
ui-trace record -d 3000 --do "tap 'Frame report'" -o /tmp/iris-bench-report.txt >/dev/null
|
||||
sleep 1
|
||||
adb logcat -d -s iris-android-app:I | grep "iris frame report:"
|
||||
@@ -17,6 +17,11 @@ dependencyResolutionManagement {
|
||||
|
||||
include(":androidApp")
|
||||
|
||||
// E3 (RUST.md): the Kotlin/Java shell over android-shell's JNI bridge, a
|
||||
// separate module from :androidApp so the ~13,000 lines of working Compose
|
||||
// UI there are untouched. See shellApp/build.gradle.kts's module comment.
|
||||
include(":shellApp")
|
||||
|
||||
// The app half of wg-app-link, resolved by path through the submodule so
|
||||
// this checkout and the crate it consumes move together -- the same
|
||||
// arrangement `server/` uses for the Rust half. See that repo's README.
|
||||
|
||||
@@ -0,0 +1,163 @@
|
||||
plugins { alias(libs.plugins.androidApplication) }
|
||||
|
||||
// E3 (RUST.md): the Kotlin/Java shell being replaced by a thin JNI bridge
|
||||
// into Rust (`../../android-shell`). Deliberately its own module rather
|
||||
// than a rewrite of `:androidApp` in place -- that module is ~13,000 lines
|
||||
// of working Compose UI this experiment does not touch, and the two can be
|
||||
// installed side by side on the same development device (see
|
||||
// `settings.SCHEME`'s doc in `android-shell` for why the deep-link scheme
|
||||
// and Keystore alias are not the production app's). No Compose plugin, no
|
||||
// Kotlin source of its own: `MainActivity`/`NotificationService` are plain
|
||||
// Java, and the CA constant below is generated as Java too.
|
||||
//
|
||||
// The CA this build pins is baked in the same way `androidApp`'s does --
|
||||
// see that module's `build.gradle.kts` comment for the reasoning (the
|
||||
// trust boundary follows the machine that builds, never a pasted copy).
|
||||
// `PinnedCa.java`'s package must match `android-shell`'s
|
||||
// `settings::load_pinned_ca` lookup (`com/example/aiapp/shell/PinnedCa`).
|
||||
val pinnedCaPath: String =
|
||||
System.getenv("AI_APP_CA")
|
||||
?: "${System.getenv("XDG_CONFIG_HOME") ?: "${System.getProperty("user.home")}/.config"}" +
|
||||
"/ai-app/certs/ca.pem"
|
||||
|
||||
abstract class GeneratePinnedCa : DefaultTask() {
|
||||
@get:Input abstract val caPath: Property<String>
|
||||
|
||||
@get:InputFile
|
||||
@get:Optional
|
||||
@get:PathSensitive(PathSensitivity.NONE)
|
||||
abstract val caCertificate: RegularFileProperty
|
||||
|
||||
@get:OutputDirectory abstract val outputDir: DirectoryProperty
|
||||
|
||||
@TaskAction
|
||||
fun generate() {
|
||||
val path = caPath.get()
|
||||
val ca = File(path)
|
||||
if (!ca.isFile) {
|
||||
throw GradleException(
|
||||
"No CA certificate at $path.\n" +
|
||||
"Start ai-server (or app/ui-sandbox.sh) once on this machine first -- it " +
|
||||
"generates the CA this build pins.\n" +
|
||||
"Set AI_APP_CA=/path/to/ca.pem to build against a different one."
|
||||
)
|
||||
}
|
||||
val pem = ca.readText().trim()
|
||||
if (!pem.startsWith("-----BEGIN CERTIFICATE-----")) {
|
||||
throw GradleException("$path is not a PEM certificate.")
|
||||
}
|
||||
val dir = outputDir.get().dir("com/example/aiapp/shell").asFile
|
||||
dir.mkdirs()
|
||||
// Same reasoning as androidApp's generatePinnedCert: the text block
|
||||
// must start immediately after the opening `"""`, or
|
||||
// CertificateFactory stops recognising the "-----BEGIN" preamble.
|
||||
File(dir, "PinnedCa.java")
|
||||
.writeText(
|
||||
"""
|
||||
|// Generated from $path by the generatePinnedCa task. Do not edit.
|
||||
|package com.example.aiapp.shell;
|
||||
|
|
||||
|public final class PinnedCa {
|
||||
| private PinnedCa() {}
|
||||
| public static final String PINNED_CA_PEM = ""${'"'}
|
||||
|$pem""${'"'};
|
||||
|}
|
||||
|"""
|
||||
.trimMargin()
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
val generatePinnedCa =
|
||||
tasks.register<GeneratePinnedCa>("generatePinnedCa") {
|
||||
val ca = file(pinnedCaPath)
|
||||
caPath.set(pinnedCaPath)
|
||||
if (ca.isFile) {
|
||||
caCertificate.set(ca)
|
||||
}
|
||||
}
|
||||
|
||||
android {
|
||||
namespace = "com.example.aiapp.shell"
|
||||
compileSdk = 37
|
||||
|
||||
defaultConfig {
|
||||
applicationId = "com.example.aiapp.shell"
|
||||
minSdk = 24
|
||||
targetSdk = 37
|
||||
versionCode = 1
|
||||
versionName = "1.0"
|
||||
}
|
||||
// Same reasoning and same key as androidApp's (see that module's comment): E5 (RUST.md)
|
||||
// signs its own, Gradle-free build with this same keystore, and the two can only
|
||||
// `adb install -r` over each other if they carry the same certificate.
|
||||
val keystore = System.getenv("AI_APP_KEYSTORE")
|
||||
signingConfigs {
|
||||
if (keystore != null) {
|
||||
create("release") {
|
||||
storeFile = file(keystore)
|
||||
storePassword = System.getenv("AI_APP_KEYSTORE_PASSWORD")
|
||||
keyAlias = "ai-app"
|
||||
keyPassword = storePassword
|
||||
}
|
||||
}
|
||||
}
|
||||
buildTypes {
|
||||
getByName("release") {
|
||||
isMinifyEnabled = false
|
||||
if (keystore != null) signingConfig = signingConfigs.getByName("release")
|
||||
}
|
||||
}
|
||||
compileOptions {
|
||||
sourceCompatibility = JavaVersion.VERSION_21
|
||||
targetCompatibility = JavaVersion.VERSION_21
|
||||
}
|
||||
}
|
||||
|
||||
// E5 (RUST.md): the xtask dexes and packages this module's Java sources itself, but it does
|
||||
// not resolve Maven dependencies -- reimplementing a dependency resolver was out of scope for a
|
||||
// packaging step, so this one task is the single place Gradle still runs in that pipeline. It
|
||||
// asks the dependency graph for the *post-transform* jars (AARs already unpacked to a classes
|
||||
// jar, the same artifact type AGP's own dexing task consumes) rather than the raw configuration,
|
||||
// which would hand back .aar files d8 cannot read directly.
|
||||
val artifactType = Attribute.of("artifactType", String::class.java)
|
||||
|
||||
tasks.register("printRuntimeClasspathJars") {
|
||||
description = "Writes the resolved release runtime classpath jars, one per line, for xtask."
|
||||
val outputFile = layout.buildDirectory.file("xtask/runtime-classpath.txt")
|
||||
outputs.file(outputFile)
|
||||
val jars =
|
||||
configurations
|
||||
.getByName("releaseRuntimeClasspath")
|
||||
.incoming
|
||||
.artifactView { attributes.attribute(artifactType, "android-classes-jar") }
|
||||
.files
|
||||
// Captured as a plain FileCollection (not the ArtifactView itself, which the
|
||||
// configuration cache cannot serialize) so this task is still cacheable.
|
||||
inputs.files(jars)
|
||||
doLast {
|
||||
val file = outputFile.get().asFile
|
||||
file.parentFile.mkdirs()
|
||||
file.writeText(jars.joinToString("\n") { it.absolutePath })
|
||||
}
|
||||
}
|
||||
|
||||
androidComponents {
|
||||
onVariants { variant ->
|
||||
variant.sources.java?.addGeneratedSourceDirectory(generatePinnedCa, GeneratePinnedCa::outputDir)
|
||||
}
|
||||
}
|
||||
|
||||
dependencies {
|
||||
// The Keystore-sealed enrollment (ServerStore/ServerSettings) --
|
||||
// android-shell's settings.rs calls into this Kotlin class directly
|
||||
// over JNI rather than re-sealing the token in Rust; see that file's
|
||||
// module doc.
|
||||
implementation(project(":link"))
|
||||
// NotificationCompat/NotificationManagerCompat/NotificationChannelCompat/
|
||||
// ServiceCompat -- android-shell's notify.rs calls these classes over
|
||||
// JNI so the pre-26 fallback behaviour (no channels) lives once, in
|
||||
// the library that already has it, rather than being re-derived as a
|
||||
// set of Build.VERSION.SDK_INT branches in Rust.
|
||||
implementation(libs.androidx.core.ktx)
|
||||
}
|
||||
@@ -0,0 +1,62 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<manifest xmlns:android="http://schemas.android.com/apk/res/android"
|
||||
xmlns:tools="http://schemas.android.com/tools">
|
||||
<!-- Mirrors androidApp's manifest (AGENTS.md: reuse it rather than
|
||||
re-deriving it) for the permissions and declarations E3 actually
|
||||
exercises. Not carried over: the QR scanner activity (this
|
||||
experiment enrolls via the aiappshell://enroll deep link directly,
|
||||
per AGENTS.md's ui-sandbox.sh banner) and the app icon warning
|
||||
suppression below, for the same reason androidApp's is there. -->
|
||||
<uses-permission android:name="android.permission.INTERNET" />
|
||||
<uses-permission android:name="android.permission.ACCESS_LOCAL_NETWORK" />
|
||||
<uses-permission android:name="android.permission.POST_NOTIFICATIONS" />
|
||||
<uses-permission android:name="android.permission.FOREGROUND_SERVICE" />
|
||||
<uses-permission android:name="android.permission.FOREGROUND_SERVICE_SPECIAL_USE" />
|
||||
|
||||
<application
|
||||
android:label="AI Sessions (shell)"
|
||||
android:allowBackup="true"
|
||||
android:theme="@android:style/Theme.Material.Light.NoActionBar"
|
||||
tools:ignore="MissingApplicationIcon">
|
||||
<activity
|
||||
android:name=".MainActivity"
|
||||
android:exported="true"
|
||||
android:launchMode="singleTop">
|
||||
<intent-filter>
|
||||
<action android:name="android.intent.action.MAIN" />
|
||||
<category android:name="android.intent.category.LAUNCHER" />
|
||||
</intent-filter>
|
||||
<!-- Enrollment: aiappshell://enroll?host=...&port=...&token=...,
|
||||
per AGENTS.md's ui-sandbox.sh banner (fed to this app with
|
||||
`adb shell am start -a android.intent.action.VIEW -d
|
||||
'aiappshell://enroll?...'`, or -n'd at this component
|
||||
directly if a second app also claims the aiapp scheme). -->
|
||||
<intent-filter>
|
||||
<action android:name="android.intent.action.VIEW" />
|
||||
<category android:name="android.intent.category.DEFAULT" />
|
||||
<category android:name="android.intent.category.BROWSABLE" />
|
||||
<data android:scheme="aiappshell" android:host="enroll" />
|
||||
</intent-filter>
|
||||
<!-- The share sheet - see android-shell's share.rs. -->
|
||||
<intent-filter>
|
||||
<action android:name="android.intent.action.SEND" />
|
||||
<action android:name="android.intent.action.SEND_MULTIPLE" />
|
||||
<category android:name="android.intent.category.DEFAULT" />
|
||||
<data android:mimeType="*/*" />
|
||||
</intent-filter>
|
||||
</activity>
|
||||
|
||||
<!-- specialUse, not dataSync, for the reason androidApp's manifest
|
||||
gives: a connection that has to keep listening overnight
|
||||
cannot accept dataSync's six-hour cap. -->
|
||||
<service
|
||||
android:name=".NotificationService"
|
||||
android:exported="false"
|
||||
android:foregroundServiceType="specialUse">
|
||||
<property
|
||||
android:name="android.app.PROPERTY_SPECIAL_USE_FGS_SUBTYPE"
|
||||
android:value="E3 experiment: holds one connection to the sandbox server so a
|
||||
session that needs an answer can be reported while the app is closed." />
|
||||
</service>
|
||||
</application>
|
||||
</manifest>
|
||||
@@ -0,0 +1,50 @@
|
||||
package com.example.aiapp.shell;
|
||||
|
||||
import android.app.Activity;
|
||||
import android.content.Context;
|
||||
import android.content.Intent;
|
||||
import android.os.Bundle;
|
||||
import android.os.Handler;
|
||||
import android.os.Looper;
|
||||
import android.widget.Toast;
|
||||
|
||||
/**
|
||||
* E3's floor, per RUST.md's "How much Java is unavoidable": a class the framework
|
||||
* constructs by name from the manifest, with its lifecycle methods handing straight to Rust
|
||||
* (android-shell's {@code share::handle_intent}). No Compose, no layout -- there is no screen to
|
||||
* draw yet (that is E4's job, on iris); {@link #toast} is this experiment's stand-in for showing
|
||||
* something happened.
|
||||
*/
|
||||
public class MainActivity extends Activity {
|
||||
static {
|
||||
System.loadLibrary("android_shell");
|
||||
}
|
||||
|
||||
@Override
|
||||
protected void onCreate(Bundle savedInstanceState) {
|
||||
super.onCreate(savedInstanceState);
|
||||
NotificationService.sync(this);
|
||||
nativeHandleIntent(this, getIntent());
|
||||
}
|
||||
|
||||
// launchMode="singleTop": a notification tap or a share while this activity is already on
|
||||
// top lands here rather than in a second instance -- same reasoning as MainActivity.kt's.
|
||||
@Override
|
||||
protected void onNewIntent(Intent intent) {
|
||||
super.onNewIntent(intent);
|
||||
setIntent(intent);
|
||||
nativeHandleIntent(this, intent);
|
||||
}
|
||||
|
||||
/**
|
||||
* Called from android-shell, sometimes from a background thread (a share's network call is
|
||||
* never made on the calling thread -- see share.rs). {@code Toast} itself is main-thread-only,
|
||||
* so this hops there with a {@link Handler} rather than assuming the caller already has.
|
||||
*/
|
||||
static void toast(Context context, String message) {
|
||||
new Handler(Looper.getMainLooper())
|
||||
.post(() -> Toast.makeText(context, message, Toast.LENGTH_LONG).show());
|
||||
}
|
||||
|
||||
private static native void nativeHandleIntent(Activity activity, Intent intent);
|
||||
}
|
||||
@@ -0,0 +1,45 @@
|
||||
package com.example.aiapp.shell;
|
||||
|
||||
import android.app.Service;
|
||||
import android.content.Context;
|
||||
import android.content.Intent;
|
||||
import android.os.IBinder;
|
||||
|
||||
/**
|
||||
* E3's second unavoidable Java class (RUST.md): a foreground service constructed by the framework
|
||||
* from the manifest, existing only to hand its lifecycle to android-shell's {@code notify} module
|
||||
* -- the SSE follow loop, deciding what a notification says, and posting it are all Rust reached
|
||||
* through these three native calls. See {@code Notifications.kt}'s {@code NotificationService} for
|
||||
* the Kotlin original this mirrors.
|
||||
*/
|
||||
public class NotificationService extends Service {
|
||||
static {
|
||||
System.loadLibrary("android_shell");
|
||||
}
|
||||
|
||||
@Override
|
||||
public IBinder onBind(Intent intent) {
|
||||
return null;
|
||||
}
|
||||
|
||||
@Override
|
||||
public int onStartCommand(Intent intent, int flags, int startId) {
|
||||
return nativeOnStartCommand(this);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDestroy() {
|
||||
nativeOnDestroy();
|
||||
}
|
||||
|
||||
/** Starts this service if there is a server to connect to, and stops it otherwise. */
|
||||
static void sync(Context context) {
|
||||
nativeSync(context);
|
||||
}
|
||||
|
||||
private static native void nativeSync(Context context);
|
||||
|
||||
private static native int nativeOnStartCommand(Service service);
|
||||
|
||||
private static native void nativeOnDestroy();
|
||||
}
|
||||
@@ -7,7 +7,7 @@ edition = "2024"
|
||||
# with `server/` via `event-model`), the REST + SSE clients for its HTTP
|
||||
# surface (see `server/src/routes.rs`'s module doc for the table), the
|
||||
# transcript fold and cache, the markdown block model, the syntax
|
||||
# highlighter and the ANSI parser. See CLIENT_CORE.md at the repo root for
|
||||
# highlighter and the ANSI parser. See `docs/CLIENT_CORE.md` for
|
||||
# what this holds today, what it does not yet, and how it corresponds to
|
||||
# the Kotlin it replaces.
|
||||
#
|
||||
|
||||
@@ -0,0 +1,152 @@
|
||||
//! What a Rust client needs to reach one enrolled server: host, port and
|
||||
//! bearer token. Mirrors the shape `ServerConfig.kt`/`Api.kt`'s
|
||||
//! `handleEnrollment` parses out of an `aiapp://enroll?host=H&port=P&token=T`
|
||||
//! deep link -- the exact link `wg-app-link`'s `enroll` module mints and
|
||||
//! `app/ui-sandbox.sh`'s banner prints, so any Rust client can enrol from
|
||||
//! the same text a phone would scan as a QR, with no second format
|
||||
//! invented for it (RUST.md's E4).
|
||||
//!
|
||||
//! What this type deliberately does not decide: where it is persisted, and
|
||||
//! under what file permissions. A phone seals its token in the Android
|
||||
//! Keystore; a desktop client has its own `$XDG_CONFIG_HOME/<app>/`
|
||||
//! directory and its own file-mode conventions (MACHINE.md: owner-only,
|
||||
//! never in the repo). Both are caller-specific, so they stay out of this
|
||||
//! crate per the code rules' "ask for the least you need" -- see
|
||||
//! `iris/desktop-app/src/config.rs` for the desktop instance.
|
||||
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
/// One enrolled server: reachable at `https://{host}:{port}`, authenticated
|
||||
/// with `token` as a bearer header. Does not carry the pinned CA -- that is
|
||||
/// a public certificate rather than a secret, and where to find it differs
|
||||
/// by caller (a phone pins the one its APK was built against; a desktop
|
||||
/// client is told a path).
|
||||
#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
|
||||
pub struct EnrolledServer {
|
||||
pub host: String,
|
||||
pub port: u16,
|
||||
pub token: String,
|
||||
}
|
||||
|
||||
impl EnrolledServer {
|
||||
/// Parses `aiapp://enroll?host=H&port=P&token=T` (query order does not
|
||||
/// matter; unrecognised keys are ignored). `token` is percent-decoded,
|
||||
/// since `ui-sandbox.sh` encodes it precisely because a raw token can
|
||||
/// contain `+`, which turns into a space if left to a naive splitter.
|
||||
pub fn parse_link(link: &str) -> Result<Self, String> {
|
||||
let query = link.split_once('?').map(|(_, q)| q).ok_or_else(|| {
|
||||
format!(
|
||||
"'{link}' has no query string (expected \
|
||||
aiapp://enroll?host=...&port=...&token=...)"
|
||||
)
|
||||
})?;
|
||||
|
||||
let mut host = None;
|
||||
let mut port = None;
|
||||
let mut token = None;
|
||||
for pair in query.split('&') {
|
||||
let Some((key, value)) = pair.split_once('=') else {
|
||||
continue;
|
||||
};
|
||||
let value = percent_decode(value);
|
||||
match key {
|
||||
"host" => host = Some(value),
|
||||
"port" => port = Some(value),
|
||||
"token" => token = Some(value),
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
|
||||
let host = host.ok_or_else(|| format!("'{link}' is missing 'host'"))?;
|
||||
let port_str = port.ok_or_else(|| format!("'{link}' is missing 'port'"))?;
|
||||
let port: u16 = port_str
|
||||
.parse()
|
||||
.map_err(|e| format!("'{link}''s port ('{port_str}') is not a number: {e}"))?;
|
||||
let token = token.ok_or_else(|| format!("'{link}' is missing 'token'"))?;
|
||||
|
||||
Ok(Self { host, port, token })
|
||||
}
|
||||
|
||||
/// Where a `client_core::api::UreqTransport` reaches this server.
|
||||
pub fn base_url(&self) -> String {
|
||||
format!("https://{}:{}", self.host, self.port)
|
||||
}
|
||||
}
|
||||
|
||||
fn percent_decode(s: &str) -> String {
|
||||
let bytes = s.as_bytes();
|
||||
let mut out = Vec::with_capacity(bytes.len());
|
||||
let mut i = 0;
|
||||
while i < bytes.len() {
|
||||
if bytes[i] == b'%'
|
||||
&& i + 2 < bytes.len()
|
||||
&& let Ok(byte) =
|
||||
u8::from_str_radix(std::str::from_utf8(&bytes[i + 1..i + 3]).unwrap_or(""), 16)
|
||||
{
|
||||
out.push(byte);
|
||||
i += 3;
|
||||
continue;
|
||||
}
|
||||
out.push(bytes[i]);
|
||||
i += 1;
|
||||
}
|
||||
String::from_utf8_lossy(&out).into_owned()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn parses_host_port_and_token() {
|
||||
let server =
|
||||
EnrolledServer::parse_link("aiapp://enroll?host=127.0.0.1&port=8547&token=abcDEF123")
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
server,
|
||||
EnrolledServer {
|
||||
host: "127.0.0.1".to_string(),
|
||||
port: 8547,
|
||||
token: "abcDEF123".to_string(),
|
||||
}
|
||||
);
|
||||
assert_eq!(server.base_url(), "https://127.0.0.1:8547");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn field_order_does_not_matter() {
|
||||
let server =
|
||||
EnrolledServer::parse_link("aiapp://enroll?token=tok&port=443&host=example.com")
|
||||
.unwrap();
|
||||
assert_eq!(server.host, "example.com");
|
||||
assert_eq!(server.port, 443);
|
||||
assert_eq!(server.token, "tok");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_percent_encoded_token_is_decoded() {
|
||||
// ui-sandbox.sh's own reason for encoding: a raw '+' would
|
||||
// otherwise arrive as a space.
|
||||
let server =
|
||||
EnrolledServer::parse_link("aiapp://enroll?host=h&port=1&token=a%2Bb%2Fc").unwrap();
|
||||
assert_eq!(server.token, "a+b/c");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_missing_field_is_named_in_the_error() {
|
||||
let err = EnrolledServer::parse_link("aiapp://enroll?host=h&port=1").unwrap_err();
|
||||
assert!(
|
||||
err.contains("token"),
|
||||
"error should name the missing field: {err}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_non_numeric_port_is_named_in_the_error() {
|
||||
let err = EnrolledServer::parse_link("aiapp://enroll?host=h&port=x&token=t").unwrap_err();
|
||||
assert!(
|
||||
err.contains("port"),
|
||||
"error should name the offending field: {err}"
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -1,11 +1,13 @@
|
||||
//! The app's pure logic, shared between the server and any Rust client --
|
||||
//! see `CLIENT_CORE.md` at the repo root for what lives here and what does
|
||||
//! see `docs/CLIENT_CORE.md` for what lives here and what does
|
||||
//! not yet.
|
||||
|
||||
pub mod ansi;
|
||||
pub mod api;
|
||||
pub mod config;
|
||||
pub mod event_stream;
|
||||
pub mod highlight;
|
||||
pub mod notifications;
|
||||
pub mod sse;
|
||||
pub mod transcript_cache;
|
||||
pub mod transcript_fold;
|
||||
|
||||
@@ -0,0 +1,162 @@
|
||||
//! `GET /notifications`, the attention stream PLAN.md's "Notifications: two
|
||||
//! places, never both" describes. Ported from the parsing half of
|
||||
//! `app/.../Notifications.kt`'s `NotificationService` -- the framing
|
||||
//! ([`crate::sse`]) and the wire shape ([`SessionNotification`],
|
||||
//! [`NotificationKind`], mirroring `server/src/session/mod.rs`'s
|
||||
//! `Notification`/`NotificationKind`).
|
||||
//!
|
||||
//! What is deliberately **not** here, because it is a decision rather than
|
||||
//! logic: whether a given notification is shown at all (the session on
|
||||
//! screen gets nothing), handed to the app as a banner, or posted to the
|
||||
//! platform's own notification drawer. That three-way choice reads
|
||||
//! process-wide state (what screen is open, whether the app is in front)
|
||||
//! that has no meaning to a pure crate with no UI and no Android in it --
|
||||
//! see `android-shell` for where it lives for this port.
|
||||
|
||||
use std::io::{BufRead, BufReader};
|
||||
|
||||
use serde::Deserialize;
|
||||
|
||||
use crate::api::{ApiError, Transport};
|
||||
use crate::sse::SseReader;
|
||||
|
||||
/// One frame of `GET /notifications`, matching `server/src/session/mod.rs`'s
|
||||
/// `Notification` field for field.
|
||||
#[derive(Debug, Clone, PartialEq, Deserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub struct SessionNotification {
|
||||
pub session_id: String,
|
||||
pub title: String,
|
||||
pub kind: NotificationKind,
|
||||
/// Epoch seconds, so a phone that was asleep can say how long ago.
|
||||
pub at: f64,
|
||||
}
|
||||
|
||||
/// Mirrors `server/src/session/mod.rs`'s `NotificationKind` -- serialized
|
||||
/// the same way, so this deserializes the wire's `"awaitingInput"` /
|
||||
/// `"finished"` directly rather than through a string match.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Deserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub enum NotificationKind {
|
||||
AwaitingInput,
|
||||
Finished,
|
||||
}
|
||||
|
||||
impl NotificationKind {
|
||||
/// What a notification asks of the reader, in the words they see --
|
||||
/// ported verbatim from `Notifications.kt`'s `attentionLine`. One
|
||||
/// function because the same fact is shown in two places (the
|
||||
/// platform's drawer and the app's own banner) and two mappings of one
|
||||
/// word drift.
|
||||
pub fn attention_line(self) -> &'static str {
|
||||
match self {
|
||||
NotificationKind::AwaitingInput => "Waiting for you",
|
||||
NotificationKind::Finished => "Finished",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Follows `/notifications`, calling `on_notification` for each frame until
|
||||
/// the connection drops or the callback asks to stop (by returning
|
||||
/// `false`). Reconnecting is the caller's job -- mirroring
|
||||
/// `NotificationService.follow`'s retry loop, which is a platform policy
|
||||
/// (how long to wait, whether to give up) rather than parsing logic.
|
||||
pub fn follow_notifications(
|
||||
transport: &dyn Transport,
|
||||
mut on_notification: impl FnMut(SessionNotification) -> bool,
|
||||
) -> Result<(), ApiError> {
|
||||
let body = transport.stream("/notifications")?;
|
||||
let mut lines = BufReader::new(body).lines();
|
||||
let mut reader = SseReader::new();
|
||||
while let Some(line) = lines.next().transpose().map_err(|e| ApiError {
|
||||
message: format!("Can't reach the server -- retrying. ({e})"),
|
||||
status: None,
|
||||
})? {
|
||||
let Some(frame) = reader.feed_line(&line) else {
|
||||
continue;
|
||||
};
|
||||
if frame.data.is_empty() {
|
||||
continue;
|
||||
}
|
||||
let notification: SessionNotification =
|
||||
serde_json::from_str(&frame.data).map_err(|e| ApiError {
|
||||
message: format!("The server sent a notification this build couldn't parse: {e}"),
|
||||
status: None,
|
||||
})?;
|
||||
if !on_notification(notification) {
|
||||
return Ok(());
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::api::{Body, RawResponse};
|
||||
use std::io::Cursor;
|
||||
|
||||
struct FixtureTransport {
|
||||
body: &'static str,
|
||||
}
|
||||
|
||||
impl Transport for FixtureTransport {
|
||||
fn request(
|
||||
&self,
|
||||
_method: &str,
|
||||
_path: &str,
|
||||
_body: Option<Body>,
|
||||
) -> Result<RawResponse, ApiError> {
|
||||
unimplemented!("this fixture only serves a stream")
|
||||
}
|
||||
|
||||
fn stream(&self, _path: &str) -> Result<Box<dyn std::io::Read + Send>, ApiError> {
|
||||
Ok(Box::new(Cursor::new(self.body.as_bytes().to_vec())))
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_notification_frame_parses_both_kinds() {
|
||||
let transport = FixtureTransport {
|
||||
body: "data:{\"sessionId\":\"s1\",\"title\":\"fix the bug\",\"kind\":\"awaitingInput\",\"at\":1.0}\n\n\
|
||||
data:{\"sessionId\":\"s2\",\"title\":\"add tests\",\"kind\":\"finished\",\"at\":2.0}\n\n",
|
||||
};
|
||||
let mut seen = Vec::new();
|
||||
follow_notifications(&transport, |n| {
|
||||
seen.push((n.session_id, n.kind));
|
||||
true
|
||||
})
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
seen,
|
||||
vec![
|
||||
("s1".to_string(), NotificationKind::AwaitingInput),
|
||||
("s2".to_string(), NotificationKind::Finished),
|
||||
]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_caller_can_stop_early() {
|
||||
let transport = FixtureTransport {
|
||||
body: "data:{\"sessionId\":\"s1\",\"title\":\"a\",\"kind\":\"finished\",\"at\":1.0}\n\n\
|
||||
data:{\"sessionId\":\"s2\",\"title\":\"b\",\"kind\":\"finished\",\"at\":2.0}\n\n",
|
||||
};
|
||||
let mut count = 0;
|
||||
follow_notifications(&transport, |_| {
|
||||
count += 1;
|
||||
count < 1
|
||||
})
|
||||
.unwrap();
|
||||
assert_eq!(count, 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn attention_line_matches_the_kotlin_original() {
|
||||
assert_eq!(
|
||||
NotificationKind::AwaitingInput.attention_line(),
|
||||
"Waiting for you"
|
||||
);
|
||||
assert_eq!(NotificationKind::Finished.attention_line(), "Finished");
|
||||
}
|
||||
}
|
||||
@@ -1,7 +1,7 @@
|
||||
//! This phone's copy of the transcripts it has already been sent, so
|
||||
//! reopening a session does not download it again. Ported from
|
||||
//! `app/.../TranscriptCache.kt`; see `TRANSCRIPT_CACHE.md` at the repo root
|
||||
//! for the design and `CLIENT_CORE.md` for how this file corresponds to it.
|
||||
//! `app/.../TranscriptCache.kt`; see `docs/TRANSCRIPT_CACHE.md`
|
||||
//! for the design and `docs/CLIENT_CORE.md` for how this file corresponds to it.
|
||||
//!
|
||||
//! What is stored is the server's own JSON for one event per line, in
|
||||
//! transcript order. Reading the cache means running the same [`seq_of`]
|
||||
|
||||
@@ -112,6 +112,13 @@ pub enum TranscriptItem {
|
||||
ClearedNote {
|
||||
seq: u64,
|
||||
},
|
||||
/// The account's usage limit stopped the turn; `resets_at` is epoch
|
||||
/// seconds when the dialect said when it lifts (`LimitNote` in
|
||||
/// `TranscriptItems.kt`).
|
||||
LimitNote {
|
||||
seq: u64,
|
||||
resets_at: Option<f64>,
|
||||
},
|
||||
CompactedNote {
|
||||
seq: u64,
|
||||
pre_tokens: Option<u64>,
|
||||
@@ -131,6 +138,7 @@ impl TranscriptItem {
|
||||
| Self::CommandRow { seq, .. }
|
||||
| Self::Note { seq, .. }
|
||||
| Self::ClearedNote { seq }
|
||||
| Self::LimitNote { seq, .. }
|
||||
| Self::CompactedNote { seq, .. } => *seq,
|
||||
Self::QuestionCard(card) => card.seq,
|
||||
}
|
||||
@@ -525,6 +533,14 @@ pub fn fold_event(items: &[TranscriptItem], entry: &SeqEvent) -> Vec<TranscriptI
|
||||
items.push(TranscriptItem::ClearedNote { seq });
|
||||
items
|
||||
}
|
||||
Event::LimitReached { resets_at } => {
|
||||
let mut items = items.to_vec();
|
||||
items.push(TranscriptItem::LimitNote {
|
||||
seq,
|
||||
resets_at: *resets_at,
|
||||
});
|
||||
items
|
||||
}
|
||||
Event::Compacted {
|
||||
pre_tokens,
|
||||
post_tokens,
|
||||
@@ -606,6 +622,37 @@ pub fn group_tool_runs(items: &[TranscriptItem]) -> Vec<TranscriptRow> {
|
||||
rows
|
||||
}
|
||||
|
||||
/// Folds a page of raw transcript lines (`ApiClient::fetch_transcript_page`'s
|
||||
/// `Vec<Value>`) into the flat item list this module works over. A line
|
||||
/// this build can't parse fails the whole page rather than being skipped --
|
||||
/// CODE_RULES's "an enumeration must be able to say 'it broke'" -- since
|
||||
/// silently dropping one event could hide, say, a user message that then
|
||||
/// looks like it was never sent. Moved here from `desktop-app`'s `app.rs`
|
||||
/// (RUST.md's E4) when the Android transcript client (I5) needed the same
|
||||
/// fold: "write the logic once" applies to any caller embedding
|
||||
/// `transcript-ui` against a live server, not just the first one.
|
||||
pub fn fold_page(values: &[serde_json::Value]) -> Result<Vec<TranscriptItem>, String> {
|
||||
let mut items = Vec::new();
|
||||
for value in values {
|
||||
let event: SeqEvent = serde_json::from_value(value.clone()).map_err(|e| {
|
||||
format!("the server sent a transcript line this build couldn't parse: {e}")
|
||||
})?;
|
||||
items = fold_event(&items, &event);
|
||||
}
|
||||
Ok(items)
|
||||
}
|
||||
|
||||
/// The wire `seq` a raw transcript line carries -- the live-stream resume
|
||||
/// cursor after loading a page must be this, not a folded item's `seq()`.
|
||||
/// A folded `AssistantMsg` keeps the seq of the *first* delta it
|
||||
/// accumulated (`fold_event`'s own doc), so resuming from that seq would
|
||||
/// re-deliver every delta already folded into it, duplicating the tail of
|
||||
/// a reply that was mid-stream when the page was fetched -- found via a
|
||||
/// real screenshot in E4 (RUST.md), where the assistant's line doubled.
|
||||
pub fn raw_seq(value: &serde_json::Value) -> Option<u64> {
|
||||
value.get("seq")?.as_u64()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
@@ -824,4 +871,88 @@ mod tests {
|
||||
other => panic!("expected a QuestionCard, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
fn line(seq: u64, json: serde_json::Value) -> serde_json::Value {
|
||||
let mut obj = json;
|
||||
obj["seq"] = serde_json::json!(seq);
|
||||
obj["ts"] = serde_json::json!(1.0);
|
||||
obj
|
||||
}
|
||||
|
||||
/// The regression for a bug a real `run-headless.sh` screenshot found
|
||||
/// in `desktop-app` (E4, RUST.md): resuming the live stream from the
|
||||
/// last *item's* seq re-delivers the deltas already folded into a
|
||||
/// still-open assistant message, doubling its tail. `raw_seq` of the
|
||||
/// last wire line must be the true high-water mark instead, which for a
|
||||
/// run of deltas is higher than every item's own `seq()`.
|
||||
#[test]
|
||||
fn the_resume_cursor_is_the_last_wire_seq_not_the_last_items_seq() {
|
||||
let values = vec![
|
||||
line(1, serde_json::json!({"type": "userMessage", "text": "hi"})),
|
||||
line(
|
||||
2,
|
||||
serde_json::json!({"type": "assistantText", "delta": "a"}),
|
||||
),
|
||||
line(
|
||||
3,
|
||||
serde_json::json!({"type": "assistantText", "delta": "b"}),
|
||||
),
|
||||
line(
|
||||
4,
|
||||
serde_json::json!({"type": "assistantText", "delta": "c"}),
|
||||
),
|
||||
];
|
||||
let after = raw_seq(values.last().unwrap()).unwrap();
|
||||
assert_eq!(after, 4);
|
||||
|
||||
let items = fold_page(&values).unwrap();
|
||||
let assistant_seq = items
|
||||
.iter()
|
||||
.find(|i| matches!(i, TranscriptItem::AssistantMsg { .. }))
|
||||
.unwrap()
|
||||
.seq();
|
||||
assert_eq!(assistant_seq, 2);
|
||||
assert_ne!(
|
||||
after, assistant_seq,
|
||||
"the fixed bug: these must differ here"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_page_folds_into_one_settled_assistant_message() {
|
||||
let values = vec![
|
||||
line(1, serde_json::json!({"type": "userMessage", "text": "hi"})),
|
||||
line(
|
||||
2,
|
||||
serde_json::json!({"type": "assistantText", "delta": "hel"}),
|
||||
),
|
||||
line(
|
||||
3,
|
||||
serde_json::json!({"type": "assistantText", "delta": "lo"}),
|
||||
),
|
||||
];
|
||||
let items = fold_page(&values).unwrap();
|
||||
assert_eq!(
|
||||
items,
|
||||
vec![
|
||||
TranscriptItem::UserMsg {
|
||||
seq: 1,
|
||||
text: "hi".to_string(),
|
||||
attachments: Vec::new(),
|
||||
},
|
||||
TranscriptItem::AssistantMsg {
|
||||
seq: 2,
|
||||
text: "hello".to_string(),
|
||||
settled: false,
|
||||
},
|
||||
]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_unparseable_line_fails_the_whole_page() {
|
||||
let values = vec![serde_json::json!({"seq": 1, "ts": 1.0, "type": "not-a-real-type"})];
|
||||
let err = fold_page(&values).unwrap_err();
|
||||
assert!(err.contains("couldn't parse"));
|
||||
}
|
||||
}
|
||||
@@ -26,6 +26,7 @@ next (a Masonry or iris transcript screen, most likely).
|
||||
| `api.rs` | `Api.kt` | Partial -- see below |
|
||||
| `event_stream.rs` | `EventStream.kt` | Done |
|
||||
| `transcript_fold.rs` | `TranscriptItems.kt`, `ToolRows.kt` | Partial -- see below |
|
||||
| `config.rs` | `ServerConfig.kt`'s `handleEnrollment` | New, desktop-only so far -- see below |
|
||||
| *(not started)* | `TranscriptSource.kt` | Not started |
|
||||
| *(not ported, and may never be)* | `TranscriptUnits.kt` | Out of scope -- see below |
|
||||
|
||||
@@ -103,6 +104,21 @@ deciding how `event_model` itself represents "a shape I don't recognise"
|
||||
-- a shared-model decision affecting `server/` too, not a `client-core`-only
|
||||
fix, so it is recorded here rather than silently worked around.
|
||||
|
||||
## `config.rs`: `EnrolledServer`
|
||||
|
||||
`EnrolledServer` (host, port, bearer token) plus `parse_link`, which reads
|
||||
the exact `aiapp://enroll?host=H&port=P&token=T` deep link
|
||||
`wg-app-link`'s `enroll` mints and `ServerConfig.kt`'s `handleEnrollment`
|
||||
parses on the phone -- so any Rust client enrols from the same text a
|
||||
phone would scan as a QR, with no second format invented for it (RUST.md's
|
||||
E4, DECISIONS.md 2026-09-05). Deliberately does not decide where it is
|
||||
persisted or under what file permissions -- a phone seals its token in the
|
||||
Android Keystore, `iris/desktop-app/src/config.rs` writes it to
|
||||
`$XDG_CONFIG_HOME/ai-app-desktop/enrollment.json` at 0600 -- since that is
|
||||
caller-specific (the code rules' "ask for the least you need"). Its only
|
||||
caller today is `desktop-app`; a future Android build of this crate would
|
||||
be a second one, not a reason to move the type.
|
||||
|
||||
## What is not started at all
|
||||
|
||||
- **`TranscriptSource.kt`** -- the layer that decides whether a page comes
|
||||
@@ -0,0 +1,293 @@
|
||||
# Decisions taken for Iris to review
|
||||
|
||||
Short list of design choices made by the design agent without asking, so
|
||||
they can be judged and reversed later. Detail lives in RUST.md (and IRIS.md
|
||||
for iris API changes); this file is only the summary. Newest first. Items
|
||||
marked **DEFERRED** are ones the agent chose not to decide alone.
|
||||
|
||||
## 2026-09-05
|
||||
|
||||
- **iris no longer asks every device for compute-shader limits it never
|
||||
uses.** `adapter.request_device` (both `iris/src/android/render.rs` and
|
||||
`iris/src/default/render.rs`) used `Limits::default()` plus an override
|
||||
for `max_buffer_size`, and `Limits::default()` unconditionally requests
|
||||
desktop-tier compute limits (`max_compute_workgroups_per_dimension:
|
||||
65535`, per `wgpu_types`) even though nothing in `iris`/`iris-core`
|
||||
creates a `ComputePipeline` or writes a `@compute` shader stage —
|
||||
confirmed by grepping the whole tree, not assumed. That crashed
|
||||
`request_device` outright on the Android emulator's software GL path
|
||||
(`EMU_GPU=software`, `--features force-gles`): SwiftShader's GL reports
|
||||
itself as OpenGL ES 3.0, which has no compute shaders at all, so the
|
||||
adapter's real limit is 0 against the unconditional request for 65535 —
|
||||
`RUST.md`'s "Software mode ... crashes for a third, different reason,"
|
||||
2026-09-05, earlier today. The same would happen on any real
|
||||
GLES-3.0-only Android device, not just the emulator. Fixed by a new
|
||||
`iris_core::device_limits()` (`iris/core/src/render/mod.rs`), shared by
|
||||
both platform backends so the two requests cannot drift, that zeros the
|
||||
six `max_compute_*` fields explicitly rather than switching to a
|
||||
downlevel `Limits` preset — `Limits::downlevel_webgl2_defaults()` was
|
||||
considered and rejected: it also zeros
|
||||
`max_storage_buffers_per_shader_stage`, and `shader.wgsl`'s vertex stage
|
||||
reads four `var<storage>` buffers (rects, glyphs, masks, move_offsets),
|
||||
so that preset would trade the compute crash for a bind-group-layout
|
||||
one on the same downlevel hardware this is meant to support. No
|
||||
capability check or fallback path was needed since nothing is being
|
||||
disabled — the request is simply narrowed to what the pipeline actually
|
||||
uses. `rigs/gpu-probe`'s own mirrored limits (it is deliberately its own
|
||||
crate, not a workspace member, so it cannot call `device_limits()`
|
||||
directly) were updated to match, and confirm `IRIS DEVICE: ok` against
|
||||
this VM's own Vulkan and GL adapters. **Not verified this pass**: the
|
||||
specific SwiftShader-ES-3.0 crash this fixes, on-device — the
|
||||
`EMU_GPU=software` cold boot this needs would have force-restarted this
|
||||
checkout's emulator while another session was actively running its own
|
||||
app on it (`com.example.aiapp` had window focus at the time), so it was
|
||||
left for a pass when the emulator is free rather than disrupting that
|
||||
session. Everything reachable without the emulator is clean: `cargo
|
||||
fmt`/`clippy --workspace --all-targets`/`test --workspace`, `cargo ndk
|
||||
build`/`clippy` for `iris-android-app` with `force-gles`, and
|
||||
`gpu-probe` against this VM's own Vulkan and GL(ES 3.2, which still has
|
||||
compute and so would not have reproduced the crash even before this
|
||||
fix — not a substitute for the real ES-3.0 test).
|
||||
|
||||
- **P0's Compose half is built and smoke-tested on the emulator** — the
|
||||
`bench` build type, the shared `app/bench-fixture/` transcript, and an
|
||||
in-process fake backend (`BenchFixture.kt`/`BenchNetwork.kt`) that
|
||||
answers `TranscriptSource`/`EventStream` from an in-memory event log
|
||||
instead of a real server, so the fold and paging under test are the real
|
||||
ones. Full account, the smoke run's report, and what is deliberately
|
||||
left (the iris half, the real on-phone runs) are in RUST.md's P0 box.
|
||||
Not a decision to review so much as the gate itself now being runnable —
|
||||
flagged here because it is the first half of something Iris explicitly
|
||||
asked to see before P1.
|
||||
|
||||
- **P0's iris half is also built and smoke-tested on the emulator,
|
||||
2026-09-05.** A new `bench` Cargo feature on `iris-android-app`, on top
|
||||
of `transcript-screen`: the same checked-in fixture (`include_str!`, no
|
||||
asset pipeline needed), the same 24-swipe scroll loop animated through
|
||||
`List::scroll` and the same 400-event/20s streaming phase through
|
||||
`fold_event`, "Run benchmark"/"Copy report" as named accessible
|
||||
controls, and the same three added report fields (process CPU time,
|
||||
peak RSS, battery current) via direct JNI calls
|
||||
(`bench_jni.rs::PlatformHandle`) since `android_view` has no
|
||||
`BatteryManager`/`ClipboardManager` wrapper of its own. One small public
|
||||
API addition to get there: `AndroidAppState::platform_ready` (`IRIS.md`),
|
||||
a default-no-op lifecycle hook handing an implementor a `JavaVM` +
|
||||
`GlobalRef` it can call Java through from any thread. Packaged with a
|
||||
new `release` build type on `iris-android-app`'s own Gradle project
|
||||
(there was previously only `debug`), signed with the same key
|
||||
`app/build-apk.sh` generates. Smoke run and the full report are in
|
||||
RUST.md's P0 box; not attempted this pass: the real on-phone runs and
|
||||
Iris's pass/fail call, which is the actual gate.
|
||||
|
||||
- **The intermittent touch-scroll dropout is root-caused and fixed: a
|
||||
missed `ACTION_DOWN` hit-test, not the previously-suspected coalesced
|
||||
first `ACTION_MOVE`.** Diagnosed by temporary logcat tracing of every
|
||||
touch event, `DragArbiter` state transition and `Selection::drag`
|
||||
dispatch (removed once confirmed), reproduced on this checkout's own
|
||||
emulator against a real sandbox session. The trace showed the actual
|
||||
mechanism: a gesture's `ACTION_DOWN` lands wherever the finger actually
|
||||
is, which is not guaranteed to fall inside the same row-local sensor
|
||||
region a later `ACTION_MOVE` in the same gesture lands in (a row's own
|
||||
padding/gap, or its non-selectable sender-name header, is
|
||||
pointer-transparent to `iris::sense::CursorSense`). When that happens,
|
||||
the widget that ends up handling the gesture never saw `PressStart`, so
|
||||
`DragArbiter` sits in `Idle` — which answers every subsequent frame with
|
||||
`Undecided` and has no way to tell "no press is happening" from "a press
|
||||
is happening but I missed its start," so it never recovers on its own
|
||||
for the rest of that gesture. One real trace showed exactly this: touch
|
||||
`Down`/`Move`/`Up` all delivered correctly, but zero `PressStart`
|
||||
reaching the arbiter, `state=Idle` unchanged from first frame to last.
|
||||
Fixed at the call site that has the context to recover
|
||||
(`iris::transcript_ui::selection::Selection::drag`,
|
||||
`iris/transcript-ui/src/selection.rs`): a new `DragArbiter::is_idle()`
|
||||
(`iris/src/sense.rs`) lets it notice a `Pressing` frame arriving with the
|
||||
arbiter still `Idle` — which can only mean a missed `PressStart`, since a
|
||||
`Pressing` sense requires the button to genuinely be down — and start the
|
||||
press there instead of where it was missed. Three new unit tests in
|
||||
`sense.rs`'s `drag_arbiter_tests` and one in `transcript-ui`'s
|
||||
`selection::tests` (the latter fails on the code before this fix).
|
||||
Commit follows. Not the same failure the earlier pass's `DECISIONS.md`
|
||||
DEFERRED item speculated about (a coalesced first `ACTION_MOVE` skipping
|
||||
slop detection) — that hypothesis is now ruled out; the arbiter's own
|
||||
slop/long-press logic was never wrong. RUST.md's I5 box,
|
||||
"Touch-scroll dropout root-caused, 2026-09-05" has the full trace.
|
||||
- **P0, a phone benchmark gate before any porting, asked for by Iris
|
||||
2026-09-05**: "before P1 I'd like to see benchmarks & also maybe stress
|
||||
test on my own phone ... If it doesn't match compose reasonably well then
|
||||
I don't think I'd wanna continue." Design (RUST.md's P0 box has the
|
||||
detail): the same embedded synthetic fixture in both apps with no server
|
||||
needed; the same scripted scroll loop then a streaming phase, run
|
||||
programmatically since the phone has no usable system tracing and no
|
||||
agent can drive it; the same report from both (frames, janky %, p50/p90/
|
||||
p99, process CPU time, peak RSS, battery current where readable) with a
|
||||
copy button; the iris app under its own id and the Compose one as a new
|
||||
`bench` build type with an id suffix, so neither replaces her production
|
||||
install; two arm64 APKs plus instructions delivered under `~/host/bench/`.
|
||||
The gate is hers: iris within a reasonable margin of Compose release on
|
||||
p50, p99 and CPU time, no crashes, no visible stutter. If it fails, the
|
||||
port stops.
|
||||
- **The rest of the port is one UI crate, `iris/app-ui`, grown out of
|
||||
`iris/transcript-ui` rather than started beside it.** It holds a
|
||||
`Screen` enum plus a back stack — the Rust equivalent of `AppRoot.kt`'s
|
||||
`when` — and `iris/desktop-app`/`iris/android-app` become thin entry
|
||||
points over it. Chosen over a fresh crate because `transcript-ui`
|
||||
already has the right generic shape (`Rsc: HasEvents` +
|
||||
`Rsc::State: FocusHost`) and the `client-core`/`event-model` path
|
||||
dependencies every later screen needs, so growing it in place is the
|
||||
smaller diff. Platform-only code (notification service, share target,
|
||||
QR scanner, Keystore token, deep-link enrolment) stays in the E3/E5
|
||||
Java shell (`android-shell/` + `app/shellApp`) rather than moving into
|
||||
this crate, since none of it is a screen. The Android APK is built by
|
||||
`cargo xtask apk` (E5), merging the app-ui cdylib into the E3 shell so
|
||||
there is one app rather than a demo shell plus a service shell.
|
||||
`app/androidApp` (the Compose app) stays untouched and is the baseline
|
||||
every step is measured against, until parity is reached (P7 decides
|
||||
the switch, and is itself a load-bearing decision left to Iris). Order
|
||||
is by risk to the daily-use path: session screen first (P1, where
|
||||
every hard behaviour already lives), then the shell merge and a real
|
||||
phone install (P2), then root tabs (P3), the explorer (P4),
|
||||
settings/enrolment (P5), desktop parity (P6), and the cutover itself
|
||||
(P7). Full plan: RUST.md's "The port, in order (decided 2026-09-05)".
|
||||
- **iris gets its own measured frame report, rather than waiting on a
|
||||
`dumpsys`/`gfxinfo` answer that cannot see a `SurfaceView`'s GPU-drawn
|
||||
frames.** `iris_core::FrameReport` (`iris/core/src/render/frame_report.rs`)
|
||||
times each frame's wall clock from the same point `render()`'s redraw
|
||||
starts to just after `queue.submit` + `present()` — the span Compose's
|
||||
own render report and `gfxinfo` both count — into a fixed 4096-entry
|
||||
ring (no allocation per frame; `report()` is the only place that
|
||||
allocates, and only on a button tap). The report gives total frames,
|
||||
janky % over the same 16.7ms budget `gfxinfo` uses, P50/P90/P99 and the
|
||||
worst, plus a reset. Exposed the way the Compose app's copy-button
|
||||
report already is: two named controls ("Frame report", "Reset frame
|
||||
report") on the transcript screen, tappable by accessibility name via
|
||||
`ui-trace`, logging under this crate's fixed `android_logger` tag
|
||||
(`iris-android-app`) so a script can grep `"iris frame report"` the way
|
||||
`transcript-bench.sh` greps `"ai-app render report"`. The report's own
|
||||
`Display` line says plainly that it measures up to the `present()` call
|
||||
returning, not GPU/compositor completion — wgpu's `present()` is not
|
||||
fenced against either, so presenting that span as "time to reach the
|
||||
screen" would be a measured-looking number that is actually inferred,
|
||||
which the standing UI rule forbids.
|
||||
- **`ui-trace` gains a hold-then-drag gesture, additive, in
|
||||
`emulator-tools`.** Neither of its two existing actions can produce
|
||||
"hold stationary for `LONG_PRESS`, then move without lifting" — `tap`
|
||||
has no hold and `swipe X1 Y1 X2 Y2 MS` interpolates motion across its
|
||||
whole duration from t=0. A new action presses, waits, then moves to a
|
||||
second point and releases as one continuous touch (raw
|
||||
`sendevent`/`MotionEvent` injection, extending whatever mechanism the
|
||||
existing `swipe` already uses), so `DragArbiter`'s pan-vs-select rule
|
||||
(`iris/src/sense.rs`, already covered by 8 unit tests against a
|
||||
synthetic clock) can finally be driven on a real device instead of only
|
||||
in a test harness.
|
||||
- **Touch drag on a transcript row follows Android's own rule**: a vertical
|
||||
drag pans the list immediately; a stationary press held 500 ms starts a
|
||||
text selection which further dragging extends; a horizontal drag while
|
||||
something is already selected extends that selection without the wait.
|
||||
One `DragArbiter` per list decides it (`iris/src/sense.rs`). Chosen over a
|
||||
"text layer always wins" or "list always wins" rule because either loses
|
||||
one of the two gestures a reader expects.
|
||||
- **E4's desktop shape is a new `iris/desktop-app` crate**: a winit window
|
||||
holding `transcript-ui`'s screen beside a session list, talking to a real
|
||||
`ai-server` through `client-core`. It enrols by pasting the same
|
||||
`aiapp://enroll?…` link a phone scans (`client-core::config::EnrolledServer`)
|
||||
and keeps it owner-only under `$XDG_CONFIG_HOME/ai-app-desktop/`. The
|
||||
pinned CA is a path given on the command line, not baked in. Chosen so
|
||||
the phone and desktop share one enrolment format and no second one is
|
||||
invented.
|
||||
- **I5's Android integration extends `iris-android-app` (I2's shell)
|
||||
behind a Cargo feature (`transcript-screen`), rather than a third
|
||||
shell crate.** That project already has the Gradle module, the
|
||||
`IrisView`/`MainActivity` Java, and the JNI registration; the only
|
||||
thing a second screen needs on top is a different `AndroidAppState`,
|
||||
the same axis `tabs_ui::build`/`transcript_ui::build` already vary
|
||||
along on the winit side. `tabs-screen`/`transcript-screen` are
|
||||
mutually exclusive and each pulls in only its own deps, so the plain
|
||||
tabs build (I2/I4) is untouched.
|
||||
- **Order of remaining work, updated 2026-09-05**: the two in-flight
|
||||
pieces and I5's Android integration are all done; next is giving iris
|
||||
its own frame-timing report so item 3 below can be decided by a number.
|
||||
- **DECIDED by Iris, 2026-09-05: iris is the app's framework; Masonry was
|
||||
the calibration.** Her words: "I think iris definitely makes more sense
|
||||
based on the limitations we've found." The limitations: Masonry has no
|
||||
touch scroll on Android (E2), no per-span rich text and no cross-row
|
||||
selection on the pinned commit (E2), and its keyboard bridge is a TODO
|
||||
(E1); iris carries the same screen under the Compose baseline on the
|
||||
host GPU (p50 15.0 ms against Compose's 20.0 ms, RUST.md's I5 box). What
|
||||
follows: the E-steps are closed as calibration, and the port proceeds
|
||||
on iris — screens, the shell (E3/E5), and `client-core` underneath.
|
||||
The item below is kept as the record of what she decided from.
|
||||
- **Was DEFERRED — whether to commit to iris over Masonry for `ai-app`.**
|
||||
Updated 2026-09-05 with the clean comparison the recommendation wanted:
|
||||
same sandbox session content, same emulator, `EMU_GPU=software`, one
|
||||
session. Headline numbers (RUST.md's I5 box, "Clean scroll comparison,
|
||||
2026-09-05," has the full table and every caveat):
|
||||
|
||||
| app | build | frames | janky % | p50 | p90 | p99 | worst |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| Compose (in-app report) | debug | 1102 | 99.0% late | 33.8ms | 50.6ms | 79.5ms | -- |
|
||||
| Compose (`dumpsys gfxinfo`) | debug | 1499 | 21.15% (95.66% legacy) | 32ms | 48ms | 150ms (p99) | -- |
|
||||
| iris (`FrameReport`) | **release** | 299 | 94.65% | 79.1ms | 98.6ms | 117.8ms | 212.6ms |
|
||||
| iris (`FrameReport`, repeat) | **release** | 233 | 94.42% | 109.3ms | 130.8ms | 147.1ms | 150.5ms |
|
||||
|
||||
**Not a clean apples-to-apples reading, stated plainly rather than
|
||||
smoothed over**: iris had to be built **release** (debug `SIGSEGV`s on
|
||||
this emulator's Vulkan loader, I4's finding) against Compose's mandated
|
||||
**debug** build, so this asymmetry likely *understates* iris's gap
|
||||
rather than the reverse; the three frame-time sources measure different
|
||||
things (Compose's own phase accounting vs. Android's HWUI deadline-miss
|
||||
definition vs. iris's redraw-start-to-present window, the last of which
|
||||
`dumpsys gfxinfo` cannot see at all for iris's `SurfaceView`); and both
|
||||
figures are emulator numbers under software rasterisation, which
|
||||
Compose's *own* in-app report shows already costs 20-34ms/frame in
|
||||
`swap`+`gpu` alone under this GPU mode, so a same-mode iris number well
|
||||
above 16.7ms was expected going in for either app. A second pair under
|
||||
`-gpu host` was not taken this pass. The earlier session's suspected
|
||||
intermittent touch-delivery dropout was **not reproduced** this pass —
|
||||
the zero-frame results this time traced to this pass's own script bug
|
||||
(a `cd` that changed which emulator `ui-trace` targeted), not the
|
||||
emulator; a CPU-load rise during the gesture was observed by a sampler
|
||||
running throughout, but did not correlate with any failure, so the
|
||||
original candidate is neither confirmed nor ruled out.
|
||||
The choice in front of Iris, updated: decide now on the
|
||||
structural-plus-functional case already made (iris works end-to-end
|
||||
where Masonry's scroll gesture doesn't exist at all on Android) plus
|
||||
this table — reading the two build profiles and three jank definitions
|
||||
with the caveats above rather than as a single number — or ask for a
|
||||
same-profile, same-GPU-mode rerun first. RUST.md's I5 box has the full
|
||||
account.
|
||||
|
||||
**Updated 2026-09-05, the `-gpu host` pair taken.** Real GPU rendering
|
||||
(`force-gles` -- the default Vulkan backend has no adapter at all under
|
||||
plain host-GPU boot, confirmed by the exact `wgpu` error) reverses the
|
||||
software-mode shape:
|
||||
|
||||
| app | build | GPU mode | frames | janky % | p50 | p90 | p99 | worst | cpu p50 | gpu-wait p50 |
|
||||
|---|---|---|---|---|---|---|---|---|---|---|
|
||||
| Compose (in-app report) | debug | host (virgl) | 1268 | 96.4% late | 20.0ms | 28.4ms | 37.7ms | -- | -- | -- |
|
||||
| iris (`FrameReport`), **best of three, 2026-09-05** | release, `force-gles` | host (virgl) | 439 | 46.24% | 15.7ms | 23.3ms | 31.2ms | 57.4ms | 1.2ms | 13.2ms |
|
||||
|
||||
Under real GPU rendering iris's median frame is *faster* than
|
||||
Compose's, not the 2-3x-slower shape the software-mode table shows. A
|
||||
new split inside `FrameReport` (redraw-to-submit vs. submit-to-present,
|
||||
commit `e2a1fad`) says why: iris's own CPU work per frame is a median
|
||||
~1ms -- almost the entire frame is time spent handing the frame to the
|
||||
driver, not in iris's layout/text/primitive code. This is consistent
|
||||
with the earlier software-mode gap being mostly SwiftShader's CPU
|
||||
rasterisation cost rather than an iris-specific slowness. **Still not
|
||||
proof, and now closed as unanswerable rather than merely untaken**: a
|
||||
same-mode software `force-gles` run to isolate the backend was retried
|
||||
2026-09-05 after fixing the compute-limit crash the first attempt hit,
|
||||
and hit a second, structural wall instead — SwiftShader's ES 3.0 GL
|
||||
path has no storage-buffer capacity at all, and `shader.wgsl` reads
|
||||
`var<storage>` buffers unconditionally, so reaching that path needs a
|
||||
shader rewrite, not a limits fix (RUST.md's I5 box, "The three
|
||||
remaining I5 verifications, closed 2026-09-05," item 2). The
|
||||
intermittent touch-scroll dropout this pass also reproduced is
|
||||
root-caused and fixed as of the same date (a missed `ACTION_DOWN` on a
|
||||
row's padding/header left `DragArbiter` stuck in `Idle`); three clean
|
||||
`iris-scroll.sh` runs post-fix each scrolled all 24/24 swipes, replacing
|
||||
the single-attempt 62-frame reading this table used to carry. RUST.md's
|
||||
I5 box, "Where iris's frame time goes, 2026-09-05, the `-gpu host`
|
||||
pass," and "The three remaining I5 verifications, closed 2026-09-05,"
|
||||
have the full account. The iris-vs-Masonry choice itself is still
|
||||
Iris's to make.
|
||||
File renamed without changes.
+653
@@ -0,0 +1,653 @@
|
||||
# iris: notable public API changes
|
||||
|
||||
For Iris to read on her own time. Each entry is a change to iris's public
|
||||
surface that a widget author or app author would notice: a trait method
|
||||
added, removed or re-shaped; a type that callers construct differently; a
|
||||
capability that moved. Small and trivial changes do not go here.
|
||||
|
||||
An entry gives the date, what changed, why, and a short before/after where
|
||||
it helps judge the change without the session that made it. Newest first.
|
||||
|
||||
## 2026-09-06: `List::anchor_position_display`, `FrameReport::mark_phase`/`phase_stats`/`late_at_hz` (RUST.md's "Benchmark v2")
|
||||
|
||||
`List` gained `anchor_position_display(&self) -> String`, reporting the
|
||||
anchor's own row index and pixel offset (`idx=N/off=Mpx`, or
|
||||
`idx=more-before`/`idx=more-after`/`idx=none`) -- what a scripted
|
||||
benchmark reads to report fling travel. Note the anchor does not
|
||||
necessarily change *slot* over a long scroll (this widget's own documented
|
||||
design: the anchor is a stable identity, not re-derived from what's on
|
||||
screen each frame), so this is not the same measurement as a Compose
|
||||
`LazyListState.firstVisibleItemIndex`, which does track the true topmost
|
||||
visible row -- the `off` half is what actually reflects how far a fling
|
||||
travelled.
|
||||
|
||||
`iris_core::render::frame_report::FrameReport` gained three methods for
|
||||
per-phase benchmark reporting: `mark_phase(name)` records a named phase
|
||||
boundary at the current frame/instant; `phase_stats(now, refresh_hz)`
|
||||
returns one `PhaseStats` (frames, wall duration, late count/percent,
|
||||
p50/p90/p99, worst) per marked phase, sliced from the existing ring by a
|
||||
new parallel `index_ring`; `late_at_hz(refresh_hz)` gives the whole run's
|
||||
late count/percent judged against an arbitrary refresh rate rather than
|
||||
the fixed 60Hz `JANK_THRESHOLD` every existing caller still uses (a
|
||||
separate method, not a parameter on `report()`, so nothing else changes
|
||||
behaviour). `RING_CAPACITY` grew 4096->16384 to hold a full multi-phase
|
||||
run without evicting earlier phases' samples.
|
||||
|
||||
## 2026-09-06: `List::fling`, `VelocityTracker`, `FlingCalculator` (IRIS_TODO.md's "swiping has no momentum")
|
||||
|
||||
`iris::widget::List` gained a real fling: `fling(velocity_px_per_s)` starts
|
||||
one (cancelled by the next touch-down via `cancel_fling`, or automatically
|
||||
once it settles or reaches loaded content's start/end), `is_scrolling()`
|
||||
reports whether one is running, and `tick_fling(now: Instant) -> bool`
|
||||
advances it and returns whether it is still going -- a caller that owns a
|
||||
`RequestRedraw` handle can hand it to the list once via the new
|
||||
`set_redraw_handle`, after which `List` re-arms its own next frame while
|
||||
flinging with no further polling needed; a caller driving a scripted
|
||||
benchmark instead calls `tick_fling` itself in a loop, same as it already
|
||||
drives `scroll`.
|
||||
|
||||
The physics is `iris::sense::FlingCalculator` + `VelocityTracker`
|
||||
(`sense.rs`, beside `DragArbiter`): a port of AOSP `SplineOverScroller`'s
|
||||
deceleration curve (the same one Compose's own `ScrollableDefaults.
|
||||
flingBehavior()` uses), cited at the definition, so a fling here travels
|
||||
the same distance a Compose `LazyColumn` would for the same initial
|
||||
velocity. `VelocityTracker` estimates that velocity from the drag's last
|
||||
~100ms of samples rather than one frame's last delta. Unit-tested:
|
||||
velocity from known samples, fling distance/duration against the closed-
|
||||
form spline result (within 1%), cancel-on-touch, and the start/end clamp
|
||||
(a fling stops rather than scrolling into content that was never loaded).
|
||||
|
||||
Before: a touch-drag panned exactly as far as the finger moved and stopped
|
||||
dead on release. After: releasing mid-drag continues scrolling and
|
||||
decelerates, matching the muscle memory every other Android scroll view
|
||||
already trained. `transcript_ui::selection::Selection::drag` wires this in
|
||||
-- a release only flings if the gesture had committed to panning
|
||||
(`DragArbiter::is_panning`, new), never a selection or an undecided tap.
|
||||
|
||||
## 2026-09-06: `UiRenderNode::new` returns `Result`, not `Self` (RUST.md's P0 box, phone-crash fix)
|
||||
|
||||
`iris_core::UiRenderNode::new(device, queue, config)` now returns
|
||||
`Result<Self, String>` instead of `Self`. Why: it used to let a bind-group-
|
||||
layout validation failure reach wgpu's default error handler, which panics
|
||||
with no way for a caller to intervene -- exactly what aborted the P0 bench
|
||||
APK on Iris's phone with the crash report truncated to "wgpu error:
|
||||
Validation Error" and nothing else recoverable. It now runs its creation
|
||||
calls inside wgpu error scopes and returns the full error text (wgpu's own
|
||||
"Caused by" chain) as `Err` instead.
|
||||
|
||||
Both callers changed to match: `android::render::AndroidRenderer::new`
|
||||
itself now returns `Result<Self, String>` too, building a fuller report
|
||||
(adapter identity, the limits/downlevel flags a layout validates against,
|
||||
then wgpu's text) on failure -- its caller,
|
||||
`android::view::IrisViewPeer::surface_changed`, logs that report as one
|
||||
logcat line and shows it on screen (a new `IrisView.showRendererError`,
|
||||
called via an ordinary JNI method call rather than a new `native fn`)
|
||||
instead of letting the process abort. `default::render::UiRenderer::new`
|
||||
(the winit/desktop backend) still panics on failure -- there is no
|
||||
on-screen fallback there -- but the panic message is now the same full
|
||||
text rather than whatever wgpu's own handler would have printed.
|
||||
|
||||
No change for an app that never constructs a `UiRenderNode` directly (every
|
||||
current one goes through `AndroidRenderer`/`UiRenderer`), but anyone who
|
||||
does needs an `?`/`.expect()`/`match` at the call site now. Full audit and
|
||||
the named hypothesis for what actually failed on the phone are in
|
||||
RUST.md's P0 box, "iris bench crash on the phone, 2026-09-06."
|
||||
|
||||
## 2026-09-05: `AndroidAppState::platform_ready` (RUST.md's P0 box, iris half)
|
||||
|
||||
Added a second, optional lifecycle method to `iris::android::AndroidAppState`
|
||||
(`iris/src/android/view.rs`), called once from `new_peer` right after `new`:
|
||||
|
||||
```rust
|
||||
fn platform_ready(&mut self, rsc: &mut AndroidRsc<Self>, vm: JavaVM, view: GlobalRef) {}
|
||||
```
|
||||
|
||||
Default does nothing, so every existing implementor (`Client`,
|
||||
`TranscriptClient`) is unaffected. It exists for a caller that needs to call
|
||||
into Java itself beyond what a `RequestRedraw` handle already covers --
|
||||
P0's bench build (`iris-android-app`'s new `bench` feature,
|
||||
`bench_client.rs`/`bench_jni.rs`) uses it to hold a `JavaVM` + `GlobalRef`
|
||||
to the view so its "Copy report" control and once-a-second battery sampler
|
||||
can call `BatteryManager`/`ClipboardManager` through the view's own
|
||||
`Context`, from a background tokio task as well as the UI thread. `new`
|
||||
itself was not extended with these two parameters: most implementors need
|
||||
nothing here, and `new`'s job is building the widget tree, not holding a
|
||||
platform handle. `vm`/`view` are independent handles from the ones
|
||||
`new_peer` keeps for its own `RequestRedraw` (a fresh `get_java_vm`/
|
||||
`new_global_ref` each), so storing them has no effect on that mechanism.
|
||||
|
||||
## 2026-09-05 (later still): `iris_core::device_limits()`, and iris no longer requests compute-shader limits
|
||||
|
||||
New public function, `iris_core::device_limits() -> wgpu::Limits`. Why:
|
||||
`adapter.request_device`'s `required_limits` was `Limits::default()` plus
|
||||
a `max_buffer_size` override in both platform backends, and
|
||||
`Limits::default()` requests desktop-tier compute-shader limits
|
||||
unconditionally (`max_compute_workgroups_per_dimension: 65535`) even
|
||||
though nothing in `iris`/`iris-core` uses a `ComputePipeline` — that
|
||||
crashed device creation outright on a downlevel GL adapter reporting
|
||||
OpenGL ES 3.0 (no compute shaders at all: the Android emulator's
|
||||
`EMU_GPU=software` path, and any real GLES-3.0-only Android device).
|
||||
`device_limits()` is what both `android::render::AndroidRenderer::new`
|
||||
and `default::render::UiRenderer::new` now build their `required_limits`
|
||||
from, so the request cannot drift between the two backends.
|
||||
|
||||
Before: `Limits { max_buffer_size: 1 << 30, ..Default::default() }`
|
||||
inlined in each backend. After: `iris_core::device_limits()`, which is
|
||||
the same thing with the six `max_compute_*` fields additionally zeroed.
|
||||
A caller building its own `DeviceDescriptor` outside these two backends
|
||||
(there are none today, but a third platform backend would want this)
|
||||
should call `device_limits()` rather than reaching for
|
||||
`Limits::default()` directly, unless it genuinely adds a compute pass —
|
||||
in which case it wants the specific compute limits that pass needs, not
|
||||
the desktop-tier default for everything.
|
||||
|
||||
## 2026-09-05 (later the same day): `iris_core::FrameReport` (RUST.md's I5 box)
|
||||
|
||||
New public type, `iris_core::FrameReport` (re-exported from `iris_core`'s
|
||||
`render` module alongside `FrameStats` and `JANK_THRESHOLD`). Why: `dumpsys
|
||||
gfxinfo` cannot see a `SurfaceView`'s own GPU-drawn frames at all, so a
|
||||
`wgpu`-rendered iris screen had no way to ask "was this smooth" the way
|
||||
Compose's own in-app render report already can -- item 3 of RUST.md's
|
||||
recommendation was stuck on a one-sided number for exactly this reason.
|
||||
|
||||
`FrameReport::record(elapsed: Duration)` is called once per frame (wired
|
||||
into `android/view.rs`'s `render()`, wrapping the same span from redraw
|
||||
start to after `queue.submit`+`present()` that Compose's report and
|
||||
`gfxinfo` both count) and writes into a fixed 4096-entry ring -- no
|
||||
allocation on the hot path. `FrameReport::report() -> Option<FrameStats>`
|
||||
gives total frames, janky % (over `JANK_THRESHOLD`, the same 16.7ms 60Hz
|
||||
budget `gfxinfo` uses), P50/P90/P99 and the worst; `None` if nothing has
|
||||
been recorded since the last `reset()`, not a zeroed report that would
|
||||
read as a real measurement. `FrameStats`'s `Display` line says plainly
|
||||
that it measures up to `present()` being called, not GPU/compositor
|
||||
completion, since wgpu's `present()` isn't fenced against either.
|
||||
|
||||
`AndroidUiState` gained a `pub frame_report: FrameReport` field --
|
||||
anything with `HasAndroidUiState` can now read or reset it. Before this,
|
||||
there was no way to ask iris's own render path how long a frame took at
|
||||
all, on any backend.
|
||||
|
||||
Before/after, for a caller that already has `ui_state: &AndroidUiState`:
|
||||
|
||||
```rust
|
||||
// before: no such question could be asked
|
||||
// after:
|
||||
match ui_state.frame_report.report() {
|
||||
Some(stats) => log::info!("iris frame report: {stats}"),
|
||||
None => log::info!("iris frame report: no frames recorded yet"),
|
||||
}
|
||||
ui_state.frame_report.reset(); // via android_state_mut()
|
||||
```
|
||||
|
||||
`iris-android-app`'s transcript screen exposes this as two named,
|
||||
tappable controls ("Frame report", "Reset frame report") rather than
|
||||
requiring a caller to wire its own UI -- see `transcript_client.rs`'s
|
||||
`frame_report_controls`.
|
||||
|
||||
## 2026-09-05: `Tasks::redraw_handle` (RUST.md's I5 Android integration)
|
||||
|
||||
New public method on `iris::task::Tasks`, `redraw_handle(&self) ->
|
||||
Arc<dyn RequestRedraw>`. Why: a caller running its own long-lived loop
|
||||
*inside* one spawned task (a live SSE follow, the Android transcript
|
||||
client's `select_session`) has no other way to ask for a frame after each
|
||||
`TaskCtx::update` -- `Tasks::spawn`'s own wrapper only requests one, after
|
||||
the whole async closure finishes, which fits a single request-then-update
|
||||
but not a stream that needs to be seen redrawing after *each* event. This
|
||||
is the same gap `iris/desktop-app`'s module doc names for why it uses
|
||||
winit's `Proxy<AppEvent>` instead of `Tasks` -- android-view has no
|
||||
`Proxy`, so this is what closes it there.
|
||||
|
||||
**A real bug this uncovered, not a hypothetical**: calling the returned
|
||||
handle's `request_redraw()` from the background thread crashed the process
|
||||
(`SIGABRT`, `Result::unwrap() on an Err value: JavaException`) the first
|
||||
time an Android transcript fetch called it a second time. `android/render.rs`'s
|
||||
`AndroidRedrawHandle` was already attaching the calling thread to the JVM
|
||||
correctly, but its `request_redraw` called `View::post_frame_callback`,
|
||||
whose Java side calls `Choreographer.getInstance()` -- which throws unless
|
||||
the *calling* thread already has a `Looper`, and a tokio worker thread,
|
||||
even freshly JNI-attached, has none. Fixed by routing through
|
||||
`View::post_delayed(0)` instead (Android's own thread-safe "queue work onto
|
||||
this View's UI thread" primitive, needing no caller-side `Looper`), landing
|
||||
on a new `IrisViewPeer::delayed_callback` override that drains tasks and
|
||||
renders -- same body as `do_frame`, on the UI thread where
|
||||
`post_frame_callback` is safe again. Any future caller of `redraw_handle()`
|
||||
from a background thread gets this for free; nothing about the fix is
|
||||
specific to the transcript screen.
|
||||
|
||||
## 2026-09-05: `transcript_ui::build_tree` (RUST.md's E4)
|
||||
|
||||
`transcript_ui::build` claimed the whole window (`ui_state.set_root(tree)`)
|
||||
as its last step, which is right for a window that *is* the transcript
|
||||
screen (the winit example, an eventual Android cdylib) and wrong for the
|
||||
desktop app, which puts a session list beside it. `build_tree` is `build`
|
||||
minus that last step: it returns `(TranscriptScreen, StrongWidget)` instead
|
||||
of just `TranscriptScreen`, and the caller decides where the tree goes —
|
||||
into `ui_state.set_root`, or into a `WidgetPtr` alongside something else
|
||||
(`iris/desktop-app`'s `rebuild_transcript`). `build` is now one line calling
|
||||
`build_tree` and doing the `set_root` itself, so existing callers are
|
||||
unaffected.
|
||||
|
||||
```rust
|
||||
// before, and still available, for a caller that wants to *be* the window:
|
||||
let screen = transcript_ui::build(rsc, &mut ui_state, rows);
|
||||
|
||||
// new, for a caller embedding the screen beside something else:
|
||||
let (screen, tree) = transcript_ui::build_tree(rsc, rows);
|
||||
some_widget_ptr(rsc).set(tree);
|
||||
```
|
||||
|
||||
|
||||
## 2026-09-05: `DragArbiter`, pan-vs-select for one shared touch gesture (RUST.md's I5)
|
||||
|
||||
New public type, `iris::sense::DragArbiter`. Why: a widget author who
|
||||
registers both a list-level pan and a row-level drag-to-select on the same
|
||||
touch gesture has no way to arbitrate between them — `core/src/sense.rs`'s
|
||||
`run_sensors` always gives the innermost layer first refusal, so the inner
|
||||
one wins every frame it is pressed, not just the frame the press started
|
||||
(this is exactly what left transcript-ui's touch-drag panning unreachable
|
||||
until now). `DragArbiter` is one small state machine, one instance per
|
||||
gesture surface (a whole list, not per row), that a caller drives with its
|
||||
own `press_start`/`update`/`release` calls and a caller-supplied `Instant`
|
||||
(so it is unit-testable without a real clock or a render harness). It
|
||||
decides the way Android itself does: an ordinary vertical drag pans
|
||||
immediately; a stationary press held `LONG_PRESS` (500ms) starts a
|
||||
selection, which any further drag then extends; a horizontal drag while
|
||||
something is already selected extends it immediately, skipping the wait.
|
||||
|
||||
```rust
|
||||
// One per list, held alongside whatever state coordinates the rows:
|
||||
let mut arbiter = DragArbiter::new();
|
||||
|
||||
// On press-down:
|
||||
arbiter.press_start(pos, Instant::now(), already_selected);
|
||||
// Every frame the button/finger stays down:
|
||||
match arbiter.update(pos, Instant::now()) {
|
||||
DragOutcome::Pan(dy) => list.scroll(-dy),
|
||||
DragOutcome::SelectStart => selection.begin(...),
|
||||
DragOutcome::SelectExtend => selection.extend(...),
|
||||
DragOutcome::Undecided => {}
|
||||
}
|
||||
// On release:
|
||||
arbiter.release();
|
||||
```
|
||||
|
||||
`transcript-ui`'s `Selection::drag` (`transcript-ui/src/selection.rs`) is
|
||||
the reference caller: every row's `CursorSense::click_or_drag() |
|
||||
CursorSense::unclick()` handler routes through one `Selection`-owned
|
||||
arbiter instead of calling `begin`/`extend` directly, so a drag that starts
|
||||
on a row's own rendered text now pans the list correctly instead of
|
||||
always starting a selection. 8 new unit tests in `iris/src/sense.rs`'s
|
||||
`drag_arbiter_tests` module.
|
||||
|
||||
### 2026-09-05, later: `DragArbiter::is_idle()`, recovering a missed `press_start`
|
||||
|
||||
Follow-up to the above, from a real touch-scroll dropout: a gesture's
|
||||
`ACTION_DOWN` can land on a caller's own dead space (a row's padding, a
|
||||
gap, a header with no handler) that never calls `press_start`, so the
|
||||
first frame the arbiter actually sees is a `Pressing`-shaped `update`
|
||||
with no matching start. Before this, `update`'s `Idle` arm had no way to
|
||||
tell that apart from "nothing is happening" and answered `Undecided`
|
||||
forever for the rest of that gesture. `is_idle(&self) -> bool` lets a
|
||||
caller notice the gap and recover: if `is_idle()` is true on a frame the
|
||||
caller knows a press is genuinely down (its own `Pressing`/equivalent
|
||||
sense fired), call `press_start` right there instead of assuming one
|
||||
already happened. `transcript-ui`'s `Selection::drag` is the reference
|
||||
caller — one new match arm, checked before the ordinary `update`-only
|
||||
case. Any other `DragArbiter` caller with the same "one sensor per
|
||||
sub-region, no fallback for dead space" shape has the same gap and wants
|
||||
the same recovery.
|
||||
|
||||
## 2026-09-05: `SpanStyle`, per-range text styling (RUST.md's I5)
|
||||
|
||||
A `TextBuffer` used to have exactly one style (`TextAttrs`: colour, size,
|
||||
family, ...) for its whole string, applied via `push_default` into parley's
|
||||
ranged builder. `SpanStyle` is a second, optional layer: a byte range plus
|
||||
whichever of colour/family/font size/bold/italic/underline it overrides,
|
||||
pushed with parley's own `push(property, range)` instead. Why: a transcript
|
||||
row's markdown (a heading, **bold**, `inline code`, a link) all inside one
|
||||
wrapped paragraph needs each to carry its own look while the paragraph
|
||||
still wraps and selects as a single buffer — the thing `masonry`'s
|
||||
`TextArea` cannot do (`StyleSet` is one style for the whole editor,
|
||||
`text_area.rs:43-44`'s `// TODO: RichTextInput`), and the reason this
|
||||
existed at all.
|
||||
|
||||
```rust
|
||||
let (text, spans) = transcript_ui::markdown::render_markdown(src, 16.0);
|
||||
wtext(text)
|
||||
.spans(spans) // new: TextBuilder::spans, on both Text and TextEdit
|
||||
.editable(EditMode::MultiLine)
|
||||
.add(rsc);
|
||||
```
|
||||
|
||||
Two things a widget author should know before reaching for it:
|
||||
|
||||
- **Call `.spans()` before or after `.editable()`, both work** — the field
|
||||
lives on `TextBuilder` itself, not either output type, and both
|
||||
`TextOutput::run` and `TextEditOutput::run` apply it to the buffer via
|
||||
`TextBuffer::set_spans`. **These two call sites are a pair**: adding a
|
||||
third `TextBuilderOutput` impl without also calling `set_spans` there
|
||||
reproduces the exact bug this box shipped once already (spans silently
|
||||
dropped for `TextEdit`, found only by screenshotting, not by any test —
|
||||
`markdown.rs`'s own unit tests check string/range logic, which is
|
||||
correct in isolation and proves nothing about whether the render path
|
||||
ever sees it).
|
||||
- **Colour is now per-glyph, not per-buffer.** `PlacedGlyph` gained a
|
||||
`color: UiColor` field (from parley's own per-run `Style::brush`), and
|
||||
`Painter::glyphs` draws each glyph in its own colour instead of
|
||||
`RenderedText::color` uniformly. `RenderedText::color` still exists (the
|
||||
buffer's *base* colour, for a caller that wants it as a whole, e.g. to
|
||||
tint a cursor) but no longer drives what a glyph actually renders as.
|
||||
|
||||
## 2026-09-05: accessibility names via AccessKit (RUST.md's I4)
|
||||
|
||||
`.label()` (already in `trait_fns.rs`, previously unused anywhere in-tree)
|
||||
is now load-bearing: it's the one thing that puts a widget in the AccessKit
|
||||
tree `iris_core::ui::access::AccessTree` builds and both backends push
|
||||
out. A widget author who wants a control to be findable by name (and
|
||||
tappable by name, through `ui-trace`/a real screen reader) calls `.label()`
|
||||
on it; nothing else is required, and a widget nobody labels is invisible
|
||||
to this system at zero cost, not just zero UI.
|
||||
|
||||
```rust
|
||||
let button = rect(Color::LIME)
|
||||
.on(CursorSense::click(), move |_, rsc| { ... })
|
||||
.label("Add task"); // now findable by uiautomator/AccessKit as "Add task"
|
||||
```
|
||||
|
||||
Two new things a widget author might touch directly:
|
||||
|
||||
- **`Widget::access_role(&self) -> accesskit::Role`**, default `Unknown`.
|
||||
Override it if your widget has a real platform equivalent —
|
||||
`TextEdit` now returns `TextInput`/`MultilineTextInput` by `EditMode`.
|
||||
Only consulted for a widget that also has a `.label()`; an unlabelled
|
||||
widget's `access_role` is never called.
|
||||
- **`Widgets::named() -> impl Iterator<Item = WidgetId>`** — every widget
|
||||
with an explicit label, for anything else that wants to walk the same
|
||||
set `AccessTree` does.
|
||||
|
||||
Nothing about `Painter`, `draw`, or the layout/move machinery changed —
|
||||
this sits entirely beside them, reading `resolved_region`'s output rather
|
||||
than participating in producing it.
|
||||
|
||||
## 2026-09-05: `List`, a virtualised bottom-anchored list (RUST.md's I3)
|
||||
|
||||
A new widget, `iris::widget::List` (`iris/src/widget/list.rs` -- read its
|
||||
module doc first), for the transcript's kind of screen: variable-height
|
||||
rows, keyed by a `u64`, composed only while visible, moved rather than
|
||||
re-laid-out on scroll, a scroll anchor that survives a row inserted above
|
||||
it, "more" sentinels at each end, and "hold the edge nearest the tap" when
|
||||
a row's height changes (`note_tap`, resolved in the layout pass).
|
||||
|
||||
```rust
|
||||
let mut list = List::new(Axis::Y);
|
||||
list.push_back(ListRow::new(key, row_widget)); // O(1)
|
||||
list.push_front(ListRow::new(older_key, row)); // O(1), anchor unaffected
|
||||
list.set_more_before(Some(spinner_widget)); // sentinel, drawn at the edge
|
||||
list.note_tap(viewport_y); // before mutating a row's height
|
||||
let (top, bottom) = list.extent(key).unwrap(); // last frame's on-screen box, if visible
|
||||
```
|
||||
|
||||
Built entirely out of existing primitives (`Painter::widget`/`widget_within`/
|
||||
`reposition`/`draw_twice`, and `draw_inner`'s own old-children diffing) --
|
||||
no new mechanism was added to the render core for it. One correctness
|
||||
lesson worth reading even for other widgets: a row that fills whatever
|
||||
region it is offered (`Rect`, `is_size_independent`) cannot be measured at
|
||||
a throwaway oversized region and then merely `reposition`ed into place --
|
||||
`reposition` only ever writes an offset, never a size, so the oversized
|
||||
primitive stays oversized. `List` fixes this by caching each row's real
|
||||
height once measured and placing an already-known row directly at its
|
||||
exact box; see `list.rs`'s `place` for the full reasoning and
|
||||
`a_fill_shaped_background_is_not_left_oversized` for the regression test.
|
||||
|
||||
## 2026-09-05: a second backend (android-view), and what moved to make room for it
|
||||
|
||||
RUST.md's I2. Three changes a widget or app author would notice, all in
|
||||
service of the same thing: `default` (winit) and the new `android`
|
||||
(android-view) backends sharing what does not depend on windowing.
|
||||
|
||||
- **`Selector`/`Selectable`'s bound changed from `Rsc::State:
|
||||
HasDefaultUiState` to `Rsc::State: FocusHost`** (new trait, `attr.rs`).
|
||||
`HasDefaultUiState` still exists and still works — `default/attr.rs` now
|
||||
implements `FocusHost` for anything that has it — so a winit app's
|
||||
existing code is unaffected. An Android app implements `FocusHost` via
|
||||
`HasAndroidUiState` instead. Affects only an app that referenced
|
||||
`HasDefaultUiState` directly at a `Selectable`/`Selector` call site
|
||||
rather than through `.attr::<Selectable>(())`, which nothing in-tree
|
||||
does.
|
||||
- **`Tasks::init` takes `Arc<dyn RequestRedraw>` instead of
|
||||
`Arc<winit::window::Window>`.** `RequestRedraw` (`task.rs`) is one method,
|
||||
`fn request_redraw(&self)`; `winit::window::Window` implements it
|
||||
(`default/render.rs`), so `Tasks::init(window)` at a call site is
|
||||
unchanged by inference. Only matters if something constructed a `Tasks`
|
||||
directly rather than through `DefaultRsc`/`AndroidRsc`.
|
||||
- **`TextEdit::apply_event`/`TextInputResult` are `#[cfg(not(target_os =
|
||||
"android"))]`** — they take a `winit::event::KeyEvent`, which does not
|
||||
exist on Android; `android/input.rs` drives the same primitives
|
||||
(`backspace`/`delete`/`motion`/`insert`, all still unconditional) from
|
||||
`ndk::event::Keycode` directly instead. New unconditional getters on the
|
||||
way: `TextEdit::text()`/`selection_range()`/`caret()`, and
|
||||
`TextEditCtx::delete_byte_range`/`set_cursor_byte` — the primitives
|
||||
`android/ime.rs`'s `InputConnection` bridge needed and that were not
|
||||
previously exposed publicly.
|
||||
|
||||
## 2026-09-04: `Widget::draw` reports the size it used; `desired_width`/`desired_height` are gone
|
||||
|
||||
A widget used to implement three methods (`draw`, `desired_width`,
|
||||
`desired_height`); it now implements one, `fn draw(&mut self, painter: &mut
|
||||
Painter) -> Size`, which draws into `painter.region()` and returns how much
|
||||
of it was used. Why: the two extra methods routinely re-simulated what
|
||||
`draw` was about to do anyway (`Span::desired_ortho` copied its own draw
|
||||
loop to get cross-axis sizing right) — one visit per widget per frame
|
||||
instead of up to three. A container that needs a child's size before
|
||||
placing it (alignment, centering) draws the child once at a provisional
|
||||
region, reads the returned `Size`, and calls the new `Painter::reposition`
|
||||
to move it into its final spot — an O(1) offset write, not a second draw. A
|
||||
widget whose drawn output never depends on the size it's given (a
|
||||
fixed-size `Rect`, a decoded `Image`) overrides the new `fn
|
||||
is_size_independent(&self) -> bool { false }` to `true`, which skips
|
||||
redrawing it when only its offered region changes shape.
|
||||
|
||||
```rust
|
||||
// before
|
||||
fn draw(&mut self, painter: &mut Painter) { /* ... */ }
|
||||
fn desired_width(&mut self, ctx: &mut SizeCtx) -> Len { /* ... */ }
|
||||
fn desired_height(&mut self, ctx: &mut SizeCtx) -> Len { /* ... */ }
|
||||
|
||||
// after
|
||||
fn draw(&mut self, painter: &mut Painter) -> Size { /* ... */ }
|
||||
```
|
||||
|
||||
`SizeCtx` and `Cache` are gone with it — see `LAYOUT.md` for the full
|
||||
design, the move-offset mechanism this shipped alongside, and the file
|
||||
list.
|
||||
|
||||
## 2026-09-04: texture pipeline rebuilt off the binding array
|
||||
|
||||
`Textures`/`TextureHandle`, `GlyphPrimitive`, and `UiRenderNode::new` all
|
||||
changed shape. Why: the old pipeline bound every texture ever drawn in one
|
||||
`binding_array<texture_2d<f32>>` and asked every device, unconditionally,
|
||||
for `VK_EXT_descriptor_indexing` — a real share of Android GPUs lack it,
|
||||
and it failed outright on the Android emulator's software Vulkan. See
|
||||
TEXTURES.md's "Recommended shape" and "Implemented, 2026-09-04".
|
||||
|
||||
- **`UiRenderNode::new` drops its `limits: UiLimits` parameter, and
|
||||
`UiLimits` is gone.** Before: `UiRenderNode::new(&device, &queue,
|
||||
&config, UiLimits::default())`. After: `UiRenderNode::new(&device,
|
||||
&queue, &config)`. Nothing replaces it — there are no more
|
||||
binding-array limits to size.
|
||||
- **`src/default/render.rs`'s device request asks for no features and no
|
||||
binding-array limits.** Before: `required_features:
|
||||
Features::TEXTURE_BINDING_ARRAY | Features::PARTIALLY_BOUND_BINDING_ARRAY
|
||||
| Features::SAMPLED_TEXTURE_AND_STORAGE_BUFFER_ARRAY_NON_UNIFORM_INDEXING`
|
||||
plus two `max_binding_array_*` limits. After: `Features::empty()` (the
|
||||
`DeviceDescriptor` default) and only `max_buffer_size` set, which was
|
||||
never about the binding array.
|
||||
- **`TextureHandle` has no `primitive()` method any more**; a caller
|
||||
outside `iris` shouldn't have been calling it (it fed the old renderer's
|
||||
internals), but if something did: use `image_index()` for a standalone
|
||||
image's bind-group index. There is no equivalent for a page — a page has
|
||||
no bind group of its own now, see below.
|
||||
- **`GlyphPrimitive` has no public constructor from a struct literal.**
|
||||
Before: `GlyphPrimitive { uv_min, uv_max, view_idx, sampler_idx, color,
|
||||
flags }`. After: `GlyphPrimitive::new(uv_min, uv_max, layer, color,
|
||||
flags)` — one `layer` (the shared atlas array's layer) instead of a
|
||||
`view_idx`/`sampler_idx` pair, since a page is now a layer of one array
|
||||
texture rather than its own bound texture.
|
||||
- **A widget author drawing images is unaffected**: `Painter::texture`/
|
||||
`texture_at`/`texture_within` and `Textures::add` keep their signatures.
|
||||
What changed underneath is that each standalone image now gets its own
|
||||
`wgpu::BindGroup` and draw call instead of a slot in the shared array —
|
||||
invisible from the widget API, visible only in `UiRenderNode`'s internals
|
||||
and in `iris`'s device requirements.
|
||||
|
||||
## 2026-09-05: `FrameReport` splits each frame at `queue.submit`
|
||||
|
||||
`FrameStats` gains two fields, and `FrameReport` gains a second recording
|
||||
method, to answer "is a slow frame iris's own CPU work or the driver/GPU"
|
||||
with a number instead of a guess (RUST.md's I5 box).
|
||||
|
||||
- **`FrameReport::record_split(total, submit_to_present)`** is a second way
|
||||
to record a frame, alongside the existing `record(total)` (unchanged,
|
||||
and still what a caller with no split should use — it now reads as
|
||||
`cpu_p50 == total`, `gpu_wait_p50 == 0`, rather than fabricating a
|
||||
number for a half it never measured).
|
||||
- **`FrameStats` gains `cpu_p50` and `gpu_wait_p50`**: medians of
|
||||
redraw-start-to-submit and submit-to-after-`present()` respectively,
|
||||
independent of each other and of the existing `p50`/`p90`/`p99`/`worst`
|
||||
(which are unchanged, and still over the whole frame). The Android
|
||||
renderer's `draw()` now returns the `submit_to_present` `Duration` it
|
||||
measured, which `android::view::render()` passes to `record_split`.
|
||||
- **Caveat carried in both doc comments**: `submit_to_present` is not
|
||||
fenced against the GPU actually finishing — it is "how long the CPU was
|
||||
blocked handing the frame to the driver," not a confirmed GPU-completion
|
||||
time. Enough to separate "iris is slow building the frame" from "iris is
|
||||
slow handing it off," not enough to claim an exact GPU budget.
|
||||
|
||||
## 2026-09-05: `List::replace_back`/`List::clear`, and `TranscriptScreen::apply`
|
||||
|
||||
Fixes the "every client refolds and rebuilds the whole widget tree per
|
||||
streamed event" cost RUST.md's P0 box measured (20 events/second against a
|
||||
~3,200-row transcript). Two small additions to `iris::widget::List`
|
||||
(`iris/src/widget/list.rs`), plus one new method on `transcript-ui`'s
|
||||
`TranscriptScreen`.
|
||||
|
||||
- **`List::replace_back(row: ListRow) -> Option<ListRow>`**: swaps the
|
||||
*last* row's widget for a new one without moving it — same slot index,
|
||||
so an anchor already pinned there (in particular a list flush with its
|
||||
own end) stays pinned, and a `List` scrolled elsewhere is untouched.
|
||||
`None` if the list is empty. `RowKey` may differ between the old and new
|
||||
row; only `heights`/`extents` care, and both are invalidated for the
|
||||
evicted key the same way `pop_back` already does.
|
||||
- **`List::clear()`**: drops every loaded row and resets to `List::new`'s
|
||||
state (`more_before`/`more_after` untouched — a caller that wants those
|
||||
cleared too calls `set_more_before(None)`/`set_more_after(None)` itself).
|
||||
The fallback path for a change that touches more than the tail.
|
||||
- **`transcript_ui::TranscriptScreen::apply(&self, rsc, old: &[TranscriptItem], new: &[TranscriptItem])`**:
|
||||
the incremental alternative to rebuilding the whole screen from
|
||||
`transcript_ui::build_tree` on every folded event. Diffs the two
|
||||
`group_tool_runs` outputs and picks the cheapest update: nothing changed
|
||||
(no-op), a pure append (`push_row`, unchanged cost), or — the common
|
||||
streaming case, a delta into a still-open assistant message — a rebuild
|
||||
of just the one changed row via `List::replace_back`, with any further
|
||||
new rows appended after it. A row changing *before* the tail (only
|
||||
`group_tool_runs` retroactively grouping tool calls into a run does
|
||||
this) falls back to `List::clear` plus a full rebuild, counted in
|
||||
`TranscriptScreen::take_rebuilds()`. `bench_client.rs`, `transcript_client.rs`
|
||||
and `desktop-app/app.rs` all call this now instead of rebuilding on every
|
||||
event; only the opening page (and `apply`'s own fallback) still calls
|
||||
`build_tree`.
|
||||
- **`TextEditCtx::set_with_spans(text, spans)`**: `set()` plus a fresh
|
||||
`Vec<SpanStyle>` in one call, needed because a streamed row's markdown
|
||||
re-renders to both a new string and a new span list on every delta and
|
||||
the two have to land together — a stale span list drawn against new
|
||||
text can point past its end. `set()` itself is unchanged (still clears
|
||||
spans to none, as before).
|
||||
|
||||
Measured on this checkout's emulator (`iris/android-app/run-bench.sh`,
|
||||
release, x86_64, `force-gles`): worst-frame and p99 during the streaming
|
||||
phase dropped from 369.3ms/284.5ms (full rebuild per event, prior pass) to
|
||||
~101–130ms/~76–103ms across three runs (this fix) — see RUST.md's P0 box
|
||||
for the full numbers and the comparison's caveats (different AVD
|
||||
instances, not a controlled A/B on identical hardware state).
|
||||
|
||||
## 2026-09-06: bundled fonts, `content_scale`, `AndroidAppState::on_insets_changed`
|
||||
|
||||
From RUST.md's P0 box, working Iris's first real-phone report (font/scale/
|
||||
inset bugs the emulator never showed).
|
||||
|
||||
- **`TextData` now bundles Noto Sans + Noto Sans Mono** (regular/bold/
|
||||
italic/bold-italic static faces, OFL) and registers them ahead of the
|
||||
platform's own fonts in the `SansSerif`/`Monospace` generic-family
|
||||
lists, rather than relying on the platform's font enumeration alone.
|
||||
`TextData::font_diagnostics() -> FontDiagnostics` reports what was found
|
||||
and what each style axis resolved to — logged once at startup and shown
|
||||
on a screen's Diagnostics page if it has one. Adds ~3.6 MB uncompressed
|
||||
to any binary linking `iris-core`; `build-apk.sh`'s own output says the
|
||||
delivered (compressed) number.
|
||||
- **`UiRenderNode::new`/`resize` now take the window size explicitly**
|
||||
(`window_size: impl Into<Vec2>`) instead of deriving it from the
|
||||
surface's physical `SurfaceConfiguration`. Existing callers pass a
|
||||
*logical* size (physical ÷ density/scale-factor) now; this is what makes
|
||||
a `font_size: 16.0` 16 dp instead of 16 raw device pixels on a
|
||||
high-density phone. Before this, `scale_factor` did not exist anywhere
|
||||
in the crate, on either platform.
|
||||
- **`AndroidUiState::content_scale: f32`** (`DisplayMetrics.density`, read
|
||||
once in `new_peer`) and the desktop equivalent (`window.scale_factor()`)
|
||||
now divide every physical-pixel number before it reaches layout or
|
||||
touch handling — see `content_scale`'s own field doc for the full list
|
||||
of what depends on it.
|
||||
- **New: `AndroidAppState::on_insets_changed(&mut self, rsc, LogicalInsets)`**,
|
||||
a default-no-op hook called from `render()` exactly when
|
||||
`AndroidUiState::insets()` changes. Nothing previously consumed
|
||||
`insets().top` at all; a screen with chrome under the status bar
|
||||
implements this to pad it, in the same logical units `content_scale`
|
||||
converts everything else to.
|
||||
- **New: `iris_core::WgpuErrorLog`**, installed via `Device::
|
||||
on_uncaptured_error` on the Android device (wgpu's default handler is an
|
||||
unconditional panic outside `UiRenderNode::new`'s own error scopes).
|
||||
Explicit `Arc`-backed value passed to the callback and kept on
|
||||
`AndroidRenderer`, not a global — a caller wanting one on desktop builds
|
||||
its own the same way.
|
||||
|
||||
## 2026-09-06: `Len::dp`, physical pixels throughout, the keyboard glyph wipe
|
||||
|
||||
Iris's phone report on build a9232ac (screenshots): text now the right
|
||||
size but blurry; the keyboard still wipes every glyph; the header buttons
|
||||
have nothing behind them. All three are fixed; this entry is the public
|
||||
API side. docs/LAYOUT.md has the layout-side writeup, docs/RUST.md's P0
|
||||
box has the full investigation and the phone verification still to do.
|
||||
|
||||
- **The keyboard wipe was `surface_changed` rebuilding the whole renderer
|
||||
on every resize**, including an IME-driven one — a fresh, empty glyph
|
||||
atlas while the CPU-side glyph cache kept UV coordinates from the old
|
||||
one. `surface_changed` now calls `AndroidRenderer::resize` (reconfigures
|
||||
the surface and window uniform only) when a renderer is already live,
|
||||
and only builds a new one when there genuinely isn't one yet.
|
||||
- **`Len` has a third field, `dp`** (Android's dp / CSS's reference pixel,
|
||||
1/160in), beside the existing `abs` (now explicitly *physical* pixels)
|
||||
and `rel`/`rest`. `len_fns::dp`/`Len::dp` construct one, used exactly
|
||||
like `abs`/`rel`/`rest` — `dp(16)` instead of a bare `16` wherever a
|
||||
size should look the same physical size on any density. This is the
|
||||
unit IRIS_TODO.md's "density-independent length unit" item asked for;
|
||||
it replaces the previous stopgap (the whole rendered scene divided by
|
||||
`content_scale` then implicitly stretched back up), which is also what
|
||||
made text blurry — a glyph rasterised at the small, pre-stretch size and
|
||||
then upscaled onto the real framebuffer.
|
||||
- **`UiRenderState`/`Painter` gained `density()`/`set_density()`** (physical
|
||||
pixels per dp). Every place a length resolves (`Len::apply_rest`,
|
||||
`Size::to_uivec2`) now takes it; `Span::gap` and `Padding`'s four sides
|
||||
moved from a bare `f32` to `Len` so they take `dp(...)` too. A bare
|
||||
number anywhere is unaffected — still `abs`, physical pixels.
|
||||
- **Text is rasterised at physical resolution now.** `TextBuffer::shape`
|
||||
takes `density` and multiplies `font_size`/`line_height` (and any span
|
||||
override) by it before handing them to parley, so the atlas holds a
|
||||
bitmap at the size it is actually shown at rather than a low-resolution
|
||||
one stretched afterward.
|
||||
- **Everything at the Android boundary is physical pixels now** — window
|
||||
size, touch coordinates, insets (`LogicalInsets` renamed
|
||||
`WindowInsets`). The previous "logical" division by `content_scale` is
|
||||
gone; `content_scale` now feeds `set_density` instead.
|
||||
- Not yet verified on Iris's actual phone (this pass had no device) —
|
||||
built and checked on this checkout's emulator only. RUST.md's P0 box
|
||||
says what she should check for: crisp text at two densities, the
|
||||
keyboard no longer wiping, and the header's background.
|
||||
@@ -0,0 +1,452 @@
|
||||
# iris: known problems and things still to build
|
||||
|
||||
Iris's own list for the library, recorded 2026-09-04 in her words where it
|
||||
matters, so the agents working through RUST.md pick these up in a sensible
|
||||
order rather than rediscovering them. Each item says where it sits in the
|
||||
order and what "done" looks like. Tick and date them in place.
|
||||
|
||||
## Fix
|
||||
|
||||
- [x] **`request_device` asked for compute-shader limits it never uses
|
||||
(2026-09-05).** `Limits::default()` (both `iris/src/android/render.rs`
|
||||
and `iris/src/default/render.rs`) requests desktop-tier compute limits
|
||||
unconditionally, even though nothing in `iris`/`iris-core` creates a
|
||||
`ComputePipeline` or writes a `@compute` shader stage — confirmed by
|
||||
grepping the whole tree, not assumed. That crashed device creation
|
||||
outright on the Android emulator's software GL path (`EMU_GPU=software`,
|
||||
`--features force-gles`): SwiftShader's GL reports itself as OpenGL ES
|
||||
3.0, which has no compute shaders, so the adapter's real limit is 0
|
||||
against the unconditional request for 65535 — the same would happen on
|
||||
any real GLES-3.0-only Android device. Fixed by a new, shared
|
||||
`iris_core::device_limits()` (`iris/core/src/render/mod.rs`) that zeros
|
||||
exactly the six `max_compute_*` fields rather than switching to a
|
||||
downlevel `Limits` preset — `downlevel_webgl2_defaults()` also zeros
|
||||
`max_storage_buffers_per_shader_stage`, which `shader.wgsl`'s vertex
|
||||
stage needs (four `var<storage>` buffers), so that preset would trade
|
||||
this crash for a bind-group-layout one on the same hardware.
|
||||
`rigs/gpu-probe`'s own hand-mirrored `Limits` (it is deliberately its
|
||||
own crate, not able to call `device_limits()` directly) was updated to
|
||||
match. See `DECISIONS.md` and RUST.md's I5 box for the account,
|
||||
including what could not be re-verified on-device this pass (the
|
||||
emulator was in concurrent use by another session).
|
||||
|
||||
- [x] **Input does not fall through by input type (2026-09-04).**
|
||||
`SensorUi::run_sensors` (`src/default/sense.rs`) used to set "consumed,
|
||||
stop checking lower layers" from mere hover — a widget registered for
|
||||
nothing but `click()` blocked a `Scroll` meant for whatever was behind
|
||||
it, since "the cursor is over this widget" and "this widget handled the
|
||||
event" were the same check. Fixed by judging consumption per input
|
||||
kind: with no button transition and no scroll happening this frame
|
||||
("momentary" activity), the topmost hovered widget still wins, same as
|
||||
before; when something momentary *is* happening, only a widget whose
|
||||
registered senses actually include a matching non-hover one (checked
|
||||
via a new `TypeEventManager::registered`, which lists what a widget
|
||||
registered without running anything) consumes it, so a widget with only
|
||||
`Hovering`/click handlers can no longer block a scroll from reaching a
|
||||
list underneath. `iris/src/sense_tests.rs` builds a button-over-a-list
|
||||
`Stack` with a plain `HasEvents` impl (no GPU or window) and checks both
|
||||
directions: a scroll over the button reaches the list, and a real click
|
||||
still reaches the button — confirmed to fail on the pre-fix code and
|
||||
pass after.
|
||||
|
||||
- [x] **Appending one image to an already-loaded list rebuilds every other
|
||||
image's bind group (2026-09-05, fixed 2026-09-05).** Found by the
|
||||
benchmark below: `GpuTextures::update` (`core/src/render/texture.rs`)
|
||||
triggered `rebuild_image_bind_groups` — a loop over *every live
|
||||
standalone image*, rebuilding its `BindGroup` — whenever the shared
|
||||
`masks` or `move_offsets` GPU buffer was resized (`masks_resized ||
|
||||
moves_resized` in `UiRenderNode::update`, `core/src/render/mod.rs`), and
|
||||
a widget getting its *first* move-offset slot (LAYOUT.md section 2 —
|
||||
every widget gets one on first draw) could be exactly what grows that
|
||||
buffer. So one new message with one new image, appended to a transcript
|
||||
that already has N images loaded, did not cost O(1): it cost one
|
||||
`create_image` for the new image plus one `make_image_bind_group` per
|
||||
*existing* image, because the new widget's own move slot pushed the
|
||||
arena past its capacity. Measured directly in
|
||||
`iris/examples/bench_images.rs`: appending a 1,001st image to 1,000
|
||||
already-settled ones reported **1,001** bind-group creates for that one
|
||||
frame, not 1 (`./run-bench.sh images`, frame 5 in the transcript below).
|
||||
|
||||
**Fix**: `masks`/`move_offsets` never belonged in a standalone image's own
|
||||
bind group (group 2) in the first place — the group also holds that
|
||||
image's own texture view, which is the only thing that is genuinely
|
||||
per-image, so a buffer shared by *everything* forced a rebuild of
|
||||
*every* group the moment it moved. Gave masks/move_offsets their own
|
||||
bind group (group 3 in `shader.wgsl` and `UiRenderNode`: `masks_layout`/
|
||||
`masks_group`), bound once per frame in `UiRenderNode::draw` rather than
|
||||
once per draw call, instead of duplicating them into every per-image
|
||||
group. `GpuTextures` and its image bind groups now know nothing about
|
||||
either buffer — `rebuild_image_bind_groups` is called only from
|
||||
`grow_array` (the atlas array texture growing, which genuinely does
|
||||
change what every image's own bind group must reference) — so a
|
||||
masks/move_offsets resize now touches exactly one bind group, ever,
|
||||
regardless of how many images are live. Numbers after the fix, same
|
||||
benchmark and command:
|
||||
|
||||
./run-bench.sh images
|
||||
frame=1 bind_group_creates=1000 (cold load, unchanged)
|
||||
frame=2 bind_group_creates=0 (was 1000 -- see the item below)
|
||||
frame=3 bind_group_creates=0
|
||||
frame=4 bind_group_creates=0
|
||||
(append one image here)
|
||||
frame=5 bind_group_creates=1 (was 1001)
|
||||
frame=6 bind_group_creates=0
|
||||
|
||||
`run-headless.sh tabs --shot` still 27266 bytes, byte-for-byte unchanged,
|
||||
confirming the bind-group restructuring changed nothing about what is
|
||||
drawn.
|
||||
- [x] **Bind-group creation takes two frames to reach the steady state, not
|
||||
one (2026-09-05, closed by the fix above, 2026-09-05).** Same benchmark:
|
||||
loading 1,000 images cold used to report 1,000 creates on frame 1
|
||||
(expected — `create_image`, one per new image) *and again* 1,000 on
|
||||
frame 2, before settling to 0 from frame 3. This was `rebuild_image_bind_groups`
|
||||
firing a second time for the same masks/move-offsets buffer-growth
|
||||
reason as the item above, confirming the guess recorded here — the two
|
||||
were exactly the same root cause measured two different ways. Frame 2
|
||||
now reports 0 (see the numbers above); not a separate fix.
|
||||
|
||||
- [ ] **A read-only text display has no widget of its own — P0's bench
|
||||
report area is a `TextEdit` standing in for one (2026-09-05).** The only
|
||||
way to get selectable text on screen today is `.editable(...)` plus
|
||||
`.attr::<Selectable>(())` (`Selectable` is only implemented for
|
||||
`TextEdit`, `iris/src/attr.rs`), which also makes the field focusable —
|
||||
tapping the bench report opens the soft keyboard over text nothing lets
|
||||
you type into. Harmless for a bench-only debug screen (not fixed this
|
||||
pass), but a real "selectable, not editable" text primitive would
|
||||
remove the keyboard side effect and is worth having before another
|
||||
screen wants the same thing (P1's own transcript rows already read
|
||||
their content from a `TextEdit` for the same reason).
|
||||
|
||||
## From the phone, 2026-09-06
|
||||
|
||||
Found on Iris's own phone while working RUST.md's P0 box's phone-report
|
||||
follow-ups. Recorded here rather than fixed in that pass, so a follow-up
|
||||
agent takes them without colliding with that pass's `bench_client.rs`/
|
||||
`android/view.rs`/`android/sense.rs` changes.
|
||||
|
||||
- [x] **Swiping has no momentum, fixed 2026-09-06.** `List::fling`/
|
||||
`VelocityTracker`/`FlingCalculator` (`iris/src/widget/list.rs`,
|
||||
`iris/src/sense.rs`) -- IRIS.md's 2026-09-06 entry has the full account.
|
||||
Wired through `Selection::drag`'s release path, cancelled by the next
|
||||
touch-down, clamped at the loaded content's start/end. Verified by unit
|
||||
test (fling distance against the closed-form spline result, cancel-on-
|
||||
touch, the clamp), not yet by an on-device or emulator feel-check --
|
||||
that is still open.
|
||||
- [x] **Scrolling down sometimes jitters the text, fixed 2026-09-06.**
|
||||
Root-caused by reading `DragArbiter::update`'s `Undecided`-to-`Panning`
|
||||
transition rather than by an on-device trace (no emulator was used this
|
||||
pass): it was the first named suspect, not the second. `self.last` stays
|
||||
at the press origin for every `Undecided` frame (nothing pans while the
|
||||
gesture might still be a selection), so the frame that finally crosses
|
||||
`DRAG_SLOP` returned `Pan(dy)` with `dy` measured from `press_start` --
|
||||
the *whole* pre-threshold drag, applied to the list in one step, however
|
||||
many frames it had taken to get there. Fixed by applying only the
|
||||
excess past `DRAG_SLOP` on that one frame (`dy - DRAG_SLOP.copysign
|
||||
(dy)`), the same "consume the slop, don't replay it" rule Android's own
|
||||
touch handling follows. New regression test,
|
||||
`crossing_the_slop_by_a_little_pans_by_a_little` (`iris/src/sense.rs`).
|
||||
**Not yet done**: an emulator trace of the real per-frame offset
|
||||
confirming this was the whole story on real touch input rather than
|
||||
only the arbiter's own unit tests -- worth a follow-up pass before
|
||||
calling it fully closed.
|
||||
|
||||
## Build
|
||||
|
||||
- [x] **Benchmarks**, not unit tests, run on demand (2026-09-05; a
|
||||
`benches/` or a script under `iris/`, never in `cargo test`). The
|
||||
scenario that matters most is a **message list** — chat apps and this
|
||||
app's transcript alike — stressed with many messages and many images.
|
||||
One case in particular: **resizing an input box** (typing enough text to
|
||||
grow it) that pushes a long list of messages above it must stay very
|
||||
fast and recalculate almost nothing — a move of everything above, not a
|
||||
re-layout. That is exactly the O(1) move chain in LAYOUT.md; the
|
||||
benchmark is what proves it. Done when the numbers are in this file with
|
||||
the command, and the input-box case reports draws re-run, not just frame
|
||||
time.
|
||||
|
||||
**Built as two rigs**, chosen per scenario by whether a real `wgpu`
|
||||
device is needed (`UiRenderState`/`Widgets` touch no GPU or window, so
|
||||
most of this runs as an ordinary binary — the same property
|
||||
`layout_tests.rs` relies on):
|
||||
|
||||
- `iris/benches/message_list.rs` — a plain `Instant`-timed binary
|
||||
(`[[bench]] harness = false` in `iris/Cargo.toml`), not criterion: see
|
||||
the file's own header for why (short version — every scenario here
|
||||
reduces to a *count* `UiRenderState::take_counters` already produces,
|
||||
which criterion's statistical machinery adds nothing to and which a
|
||||
new dependency is not worth pulling in for). Covers (a) first-frame
|
||||
cost of a message list of N wrapped-text rows (one in 20 also carrying
|
||||
a small in-memory image) for N = 100/1,000/10,000; (b) per-frame cost
|
||||
of scrolling that list, 200 ticks; (c) the input-box case — a
|
||||
fixed-height field at the bottom of the screen growing by a line 40
|
||||
times, with the message list above it filling the rest of the screen.
|
||||
Run: `cd iris && cargo bench --bench message_list` (always release —
|
||||
`cargo bench` builds the `bench` profile, which is optimized).
|
||||
- `iris/examples/bench_images.rs` — needs a real device, so it runs
|
||||
through `iris/run-headless.sh bench_images`, printing
|
||||
`UiRenderNode::take_image_bind_group_creates()` (a new counter, added
|
||||
in `core/src/render/texture.rs` and `core/src/render/mod.rs`,
|
||||
mirroring `UiRenderState::take_counters`) each frame. Covers (d): 1,000
|
||||
image rows, checked both cold (does bind-group creation reach zero
|
||||
once loaded) and after appending one more image once settled (does
|
||||
*that* stay cheap) — the second question is what actually matters for
|
||||
a live transcript and is what turned up the two Fix items above.
|
||||
- `iris/run-bench.sh [list|images]` runs either or both and is what to
|
||||
run before/after touching `Scroll`, `Span`, `Sized`, the move-offset
|
||||
chain, or `GpuTextures`.
|
||||
|
||||
**Numbers (2026-09-05, release, `cargo bench`/`run-headless.sh`, this
|
||||
VM: AMD Ryzen 7 3800X, 8 cores, rustc 1.98.0 nightly-2026-09-03):**
|
||||
|
||||
cd iris && cargo bench --bench message_list
|
||||
(a) first frame, N=100: 30.30ms draws=227 rewrites=15 moves=0
|
||||
(a) first frame, N=1000: 186.04ms draws=2252 rewrites=150 moves=0
|
||||
(a) first frame, N=10000:1770.36ms draws=22502 rewrites=1500 moves=0
|
||||
(b) scroll, N=100/1000/10000, 200 ticks each:
|
||||
draws=200 rewrites=0 moves=200 (identical at every N)
|
||||
per-tick average: 0.0002ms (identical at every N)
|
||||
(c) input grows 40 lines, N=100/1000/10000 rows above it:
|
||||
draws=320 rewrites=40 moves=160 (identical at every N)
|
||||
per-line average: 0.0012-0.0013ms (identical at every N)
|
||||
|
||||
cd iris && ./run-bench.sh images (2026-09-05, before the fix)
|
||||
frame=1 bind_group_creates=1000 (cold load)
|
||||
frame=2 bind_group_creates=1000 (see Fix item above)
|
||||
frame=3 bind_group_creates=0
|
||||
frame=4 bind_group_creates=0
|
||||
(append one image here)
|
||||
frame=5 bind_group_creates=1001 (see Fix item above)
|
||||
frame=6 bind_group_creates=0
|
||||
|
||||
cd iris && ./run-bench.sh images (2026-09-05, after the fix)
|
||||
frame=1 bind_group_creates=1000 (cold load, unchanged -- genuine work)
|
||||
frame=2 bind_group_creates=0
|
||||
frame=3 bind_group_creates=0
|
||||
frame=4 bind_group_creates=0
|
||||
(append one image here)
|
||||
frame=5 bind_group_creates=1 (one image's own create_image, O(1))
|
||||
frame=6 bind_group_creates=0
|
||||
|
||||
**Reading it**: (a) is real, necessary work — shaping and laying out N
|
||||
never-before-seen text rows — and scales with N as it must, ~10x cost
|
||||
per 10x N. (b) and (c) are the pass conditions that matter: both are
|
||||
**exactly flat across N = 100 to 10,000**, confirming LAYOUT.md's O(1)
|
||||
move chain holds for both scrolling and for a growing input box pushing
|
||||
the message list — draws/moves per tick or per line do not grow with
|
||||
list size, and the per-operation cost (a fraction of a microsecond) is
|
||||
nowhere near a frame budget. (d)'s cold-load and steady-state halves
|
||||
behave as designed; its *append* half did not, until the fix above moved
|
||||
masks/move_offsets out of the per-image bind group — now flat at O(1)
|
||||
the same way (b) and (c) are.
|
||||
|
||||
- **I5's transcript screen (`iris/transcript-ui/`, 2026-09-05) — what it
|
||||
left, each recorded at the point in the code it would go rather than
|
||||
silently dropped. See RUST.md's I5 box for the full account of what
|
||||
*was* built (the screen, `SpanStyle`, cross-row selection, the growing
|
||||
composer).**
|
||||
- [x] **Android integration for this screen — done, 2026-09-05.**
|
||||
`iris-android-app`'s `transcript-screen` Cargo feature
|
||||
(`transcript_client.rs`) runs this screen against a real `ai-server`
|
||||
through `client-core`, confirmed on-device: real scrolling, real
|
||||
touch-drag panning, tap-by-name on the composer. Two real bugs found
|
||||
and fixed along the way (a missing `INTERNET` permission; a
|
||||
background-thread redraw request that crashed via a `Looper`
|
||||
requirement, fixed by routing through `View::post_delayed` — see
|
||||
`IRIS.md`'s `Tasks::redraw_handle` entry). See RUST.md's I5 box,
|
||||
"The Android integration, done 2026-09-05" for the full account.
|
||||
- [x] **A render-time number for iris, comparable to Compose's
|
||||
`transcript-bench.sh` report — instrumentation done and a real number
|
||||
obtained, 2026-09-05 (later the same day); the clean comparable loop
|
||||
is not.** `iris_core::FrameReport` (`iris/core/src/render/
|
||||
frame_report.rs`, `IRIS.md`'s new entry) times every frame from
|
||||
`render()`'s redraw start to after `queue.submit`+`present()`, exposed
|
||||
as two named on-screen controls ("Frame report", "Reset frame
|
||||
report"). Driven against a real on-device touch-drag it read
|
||||
`frames=34 janky%=61.76 p50=26.5ms p90=48.0ms p99=98.1ms
|
||||
worst=98.1ms` — real, not inferred, but accumulated across several
|
||||
gestures rather than one clean 24-swipe loop, because of the new
|
||||
finding below. See RUST.md's I5 box, "Update, 2026-09-05, later the
|
||||
same day" for the full account.
|
||||
- [ ] **New, 2026-09-05: intermittent touch delivery to iris's
|
||||
`SurfaceView` under this checkout's `EMU_GPU=software` emulator.**
|
||||
The same swipe coordinates, confirmed (by scanning a screenshot
|
||||
column for the first non-black pixel) to sit over real row text,
|
||||
sometimes produced 30+ real frames and a screenshot diff and
|
||||
sometimes produced zero of either, across otherwise-identical
|
||||
`ui-trace` invocations. Not the already-understood "already at that
|
||||
scroll edge" case (reproduced with content confirmed taller than the
|
||||
viewport, in both directions). Leading candidate, not yet confirmed:
|
||||
this checkout's emulator was independently observed at ~78% of one
|
||||
CPU core, continuously, while idle on-screen — SwiftShader's software
|
||||
rasterisation is CPU-bound by design, and a synthetic touch competing
|
||||
with that load for delivery is plausible but unmeasured *during* a
|
||||
failing gesture (the standing rule against diagnosing from
|
||||
after-the-fact measurements applies here). Needs a sampler (load,
|
||||
`dumpsys input`, a `-i 0` `ui-trace` capture) running while a failing
|
||||
gesture is driven, and ideally a comparison under `-gpu host` (real
|
||||
Vulkan) to see whether it is specific to software rendering. This is
|
||||
what blocks the clean, comparable 24-swipe loop above.
|
||||
- [x] **Long-press-then-drag-to-select — confirmed on-device, 2026-09-05
|
||||
(later the same day).** `ui-trace` gained a `holddrag X1 Y1 X2 Y2
|
||||
HOLD_MS MOVE_MS` action (`emulator-tools`, additive, extends the same
|
||||
`MotionEvent`/`injectInputEvent` mechanism `swipe` already used):
|
||||
press, hold past `LONG_PRESS`, move, release, as one continuous touch.
|
||||
Driven against a real row (`holddrag 300 1850 300 2050 600 300`) it
|
||||
produced `iris selection: begin at row ...` then a sequence of
|
||||
`iris selection: extend to row ...` log lines
|
||||
(`transcript-ui/src/selection.rs`, a new small `log` dependency since
|
||||
selection has no accessibility label of its own yet — see the next
|
||||
item), and a screenshot taken right after shows the expected
|
||||
highlighted selection spanning multiple rows. `DragArbiter`'s own
|
||||
unit tests already covered this sequence against a synthetic clock;
|
||||
this is the first time it has been driven by a real device touch.
|
||||
- [x] **Touch-drag panning over a row's own rendered text — done,
|
||||
2026-09-05.** `row.rs` used to register `CursorSense::click_or_drag()`
|
||||
on each row's `TextEdit` for cross-row selection; `TextEdit::draw`'s
|
||||
`painter.child_layer()` (`iris/src/widget/text/edit.rs:87`) meant that
|
||||
registration won `core/src/sense.rs::run_sensors`'s per-layer
|
||||
arbitration on every frame it was pressed, not just the frame the
|
||||
press started, so a list pan gesture registered on `List` itself never
|
||||
got a turn while a row was under the finger. Fixed with
|
||||
`iris::sense::DragArbiter` (recorded in `IRIS.md`), one small state
|
||||
machine per list deciding pan vs. select the way Android does (a
|
||||
vertical drag pans immediately; a stationary press held `LONG_PRESS`
|
||||
(500ms) starts a selection which further drag extends; a horizontal
|
||||
drag while something is already selected extends immediately) —
|
||||
`transcript-ui/src/selection.rs`'s `Selection::drag` is the one place
|
||||
every row's drag now routes through. 8 new unit tests
|
||||
(`iris/src/sense.rs`'s `drag_arbiter_tests`); `cargo fmt/clippy/test
|
||||
--workspace` and `cargo ndk` (both `iris` and `transcript-ui`) all
|
||||
clean; `run-headless.sh` screenshot byte-identical to before the
|
||||
change (38578 bytes). See RUST.md's I5 box, "Gap closed, 2026-09-05".
|
||||
- [x] **Intermittent touch-scroll dropout — root-caused and fixed,
|
||||
2026-09-05.** Not the coalesced-`ACTION_MOVE` hypothesis the earlier
|
||||
pass suspected (ruled out): a gesture's `ACTION_DOWN` can land on a
|
||||
row's own padding/gap or its header, which no `CursorSense` covers,
|
||||
so `DragArbiter` never gets `press_start` and sits in `Idle`
|
||||
(answers `Undecided` forever) for that whole gesture. Fixed via a new
|
||||
`DragArbiter::is_idle()` that `Selection::drag`
|
||||
(`transcript-ui/src/selection.rs`) checks to recover a missed press
|
||||
on the next `Pressing` frame. Four new unit tests. See RUST.md's I5
|
||||
box, "Touch-scroll dropout root-caused, 2026-09-05", for the trace and
|
||||
what a peer session sharing this checkout's emulator mid-pass
|
||||
prevented from being re-verified end-to-end (the aggregate
|
||||
`iris-scroll.sh` three-run confirmation and a re-taken FrameReport
|
||||
row) — a future pass should finish that once the emulator is free.
|
||||
- [ ] **Row-level accessibility names.** The composer carries
|
||||
`.label("Message")`; transcript rows do not carry a `.label()` of
|
||||
their own yet, so `Widgets::named()` (I4) does not include them —
|
||||
`row.rs`'s `build_text_row` is where one would go, keyed to something
|
||||
stable per row (its sender + a short excerpt, matching what a screen
|
||||
reader announcing a chat message would say).
|
||||
- [ ] **A tappable link and a background chip behind inline code.**
|
||||
Both need per-range glyph geometry that `TextEditCtx` does not expose
|
||||
outside `iris::widget::text` (`edit.rs`'s `layout()` helper is
|
||||
private) — see `markdown.rs`'s module doc for the exact shape the fix
|
||||
would take (the same primitive `TextEdit::draw`'s own selection
|
||||
highlight already uses internally,
|
||||
`iris/src/widget/text/edit.rs:99`).
|
||||
- [ ] **`Selection`'s anchor-row shortcut.** The row a drag started in
|
||||
is selected in full (`select_all`) the moment the drag leaves it,
|
||||
rather than "from the click point to whichever edge points away from
|
||||
the drag" — needs the same private `layout()` access as the item
|
||||
above. `selection.rs`'s module doc has the exact reasoning.
|
||||
- [ ] **No syntax highlighting inside a fenced code block.**
|
||||
`client_core::highlight` exists (built for the file explorer) and
|
||||
could feed per-token `SpanStyle`s into a code block's span; wiring it
|
||||
in was not attempted this pass.
|
||||
|
||||
- [ ] **Masks defined relative to each other.** Wanted: mask A multiplies
|
||||
by something *and also* applies mask B — a mask can reference a parent
|
||||
mask, the way the move chain references a parent offset. Today masks
|
||||
are independent regions. Design it beside the move chain (same shape:
|
||||
a parent index and a bounded walk in the shader); do it when a real
|
||||
widget needs it, not before.
|
||||
- [ ] **Positions as a single float per scroll.** Iris raised, and half
|
||||
rejected, letting a scroll update one float rather than positions:
|
||||
input handling cares about most elements in a list, so absolute
|
||||
positions must be computed on the CPU anyway. LAYOUT.md's design
|
||||
already lands here (GPU walks the chain, CPU resolves on demand for
|
||||
hit tests). Keep the CPU resolution lazy and per query; do not
|
||||
materialise every row's absolute position per frame.
|
||||
- [ ] **Animations, last.** Cosmetic, so after everything above. Must be
|
||||
**modular — a piece of the library rather than a core part forced into
|
||||
everything, the same way input is**. Whatever the mechanism, a widget
|
||||
that does not animate must pay nothing and import nothing for it.
|
||||
|
||||
## Build (for the port)
|
||||
|
||||
Widgets `RUST.md`'s "The port, in order (decided 2026-09-05)" needs and
|
||||
iris does not have yet, one entry per gap, named against the P-step that
|
||||
first needs it. Move an entry up to "Fix" or tick it in place once built;
|
||||
do not duplicate it there.
|
||||
|
||||
- [ ] **A history-paging cushion measured in on-screen viewports, not a
|
||||
row count.** (**P1**.) `iris::widget::List` has no equivalent of the
|
||||
Compose app's `HISTORY_SCREENS` — AGENTS.md's "Things that have
|
||||
bitten" is explicit that a fixed row count under-fills a screen on a
|
||||
tool-heavy transcript and over-fills one on a text-heavy one, so
|
||||
whatever loads the next page has to ask the list how many viewports
|
||||
are actually on screen, not assume a constant.
|
||||
- [ ] **A scaled thumbnail/image widget for an in-transcript image.**
|
||||
(**P1**.) `SessionImage.kt`'s bitmap decode-and-downscale has no iris
|
||||
counterpart; iris's own image widget (used by `bench_images.rs`) draws
|
||||
a loaded texture but does nothing about sourcing or scaling one from a
|
||||
server-produced attachment.
|
||||
- [ ] **A modal/dialog primitive.** (**P1**, reused by **P3** and
|
||||
**P5**.) Needed for the session settings dialog, `UsageDialog`'s
|
||||
equivalent, and the delete-with-`deleteForeign` confirmation with its
|
||||
toggle switch. Build once, wherever it is first needed, rather than
|
||||
once per screen that wants one.
|
||||
- [ ] **A horizontal gauge/bar widget.** (**P1**.) For
|
||||
`SessionUsageBar`'s equivalent — a bounded fill reflecting a fraction,
|
||||
nothing fancier.
|
||||
- [ ] **A `BusyItem` equivalent: a dimmed row carrying an operation
|
||||
label that does not block its list's own scroll/drag.** (**P3**.) The
|
||||
Compose version tried an overlay first and it swallowed the drag along
|
||||
with the tap (AGENTS.md's "Shared appearance") — worth not repeating
|
||||
that attempt in iris before building the row-level version directly.
|
||||
- [ ] **A toggle switch.** (**P3**.) For the delete dialog's
|
||||
`deleteForeign` control; iris has no switch/checkbox widget yet as far
|
||||
as this pass found.
|
||||
|
||||
## Reconsider
|
||||
|
||||
- [ ] **`WidgetView`.** Iris is unsure of it: what she wants is an easy way
|
||||
to compose a widget from others (a button is the main case). With
|
||||
sizing folded into `draw`, composing may be easy enough that `View` is
|
||||
redundant. Decide after the layout change lands, by writing a button
|
||||
both ways and keeping the one that is shorter to explain; delete the
|
||||
other rather than keeping two ways.
|
||||
|
||||
## Build (asked for by Iris, 2026-09-06): a density-independent length unit
|
||||
|
||||
- [x] **A third length kind beside relative and pixels, so display scales
|
||||
"just work".** Done 2026-09-06 — `Len::dp`/`len_fns::dp`, resolved
|
||||
against `UiRenderState`/`Painter::density()` at `apply_rest` time; text
|
||||
additionally rasterises at the resolved (physical) size instead of
|
||||
scaling a low-resolution bitmap afterward, which was making text blurry.
|
||||
`Span::gap`/`Padding` moved from `f32` to `Len` so they take `dp(...)`
|
||||
too; transcript-ui's row/composer padding and one example migrated.
|
||||
`em` was not added — nothing in this pass needed a text-relative unit,
|
||||
and `dp`'s own doc says why it and physical pixels are kept as separate
|
||||
fields rather than one the caller pre-multiplies. Not yet verified on
|
||||
Iris's own phone at two densities (this pass had no device) — see
|
||||
docs/RUST.md's P0 box and docs/IRIS.md's 2026-09-06 entry for what to
|
||||
check. Iris's words: "another length type similar to absolute &
|
||||
relative, so instead there would be relative, pixels, and another unit
|
||||
like em or whatever is standard. That way different display scales
|
||||
should just work." Today a length is either a fraction of the parent
|
||||
(`rest`/relative) or physical pixels, and the phone drew 16 px text at
|
||||
roughly a third of its intended size until the P0 fixes applied the
|
||||
display's scale factor globally. That global scale is a stopgap for the
|
||||
benchmark; the real shape is a unit resolved against the display's
|
||||
density at layout time — Android's `dp` / CSS's reference pixel is the
|
||||
standard (1 unit = 1/160 in), with `em` as the text-relative option —
|
||||
so a widget author writes `16.dp()` once and never sees the scale.
|
||||
Done when: `Length` (or whatever the enum is called) has the third
|
||||
variant; every place that resolves a length takes the density; the
|
||||
examples and `transcript-ui` use the new unit for text sizes, padding
|
||||
and control sizes; the emulator at two densities and the phone draw the
|
||||
same layout at the same physical size. After the bench setup is
|
||||
finished, before P1 draws any new screen.
|
||||
@@ -861,6 +861,58 @@ unspecified rather than getting them wrong:
|
||||
conditions, so the remaining slack was accepted rather than chased
|
||||
further.
|
||||
|
||||
## Density: `Len::dp`, resolved at `apply_rest` time (2026-09-06)
|
||||
|
||||
Iris asked for a third length kind beside `abs` (physical pixels) and
|
||||
`rel`/`rest` (a fraction of the parent) — IRIS_TODO.md's "density-
|
||||
independent length unit" — after the P0 phone pass found 16px text
|
||||
drawing at roughly a third size on a real phone. The fix that shipped
|
||||
first (RUST.md's P0 box) was a global stopgap: divide the whole window
|
||||
into a "logical" coordinate space (physical ÷ `content_scale`) and let
|
||||
the shader's NDC mapping stretch it back up onto the real framebuffer.
|
||||
That fixed the *size* but not the *sharpness* — a glyph rasterised at the
|
||||
small, pre-stretch size and then stretched onto more physical pixels than
|
||||
it has texels for is blurry, which is exactly what Iris's next report
|
||||
said.
|
||||
|
||||
**The fix**: `Len` gained a `dp` field, resolved against a `density: f32`
|
||||
(physical pixels per dp) at the one place a `Len` becomes a `UiScalar`
|
||||
(`Len::apply_rest`) — `abs + dp * density`. `density` lives on
|
||||
`UiRenderState` (`set_density`/`density()`) and `Painter` (`density()`),
|
||||
set once from `DisplayMetrics.density` in `android::view::new_peer`; the
|
||||
desktop backend has no per-monitor density wired up yet and stays at
|
||||
`1.0`. Every layout call site that used to call `.apply_rest()`/
|
||||
`.to_uivec2()` now passes `painter.density()` (nine call sites — `Span`,
|
||||
`Sized`, `MaxSize`, `Aligned`, `Scroll`, `List::place`, and
|
||||
`UiRenderState::reposition` itself). This also meant the Android
|
||||
boundary's global logical-space stopgap could come out entirely: window
|
||||
size, touch coordinates and insets are physical pixels again, matching
|
||||
`AndroidRenderer`'s own swapchain resolution, with `dp` doing the
|
||||
per-length work the global divide used to do for everything at once.
|
||||
|
||||
**Text is the case that needed more than the `Len` plumbing.** A widget's
|
||||
`font_size`/`line_height` are plain `f32`, not routed through `Len` at
|
||||
all (there is no sensible `rel`/`rest` for a font size). `TextBuffer::
|
||||
shape` now takes `density` directly and multiplies `font_size`/
|
||||
`line_height` (and any span override) by it before handing them to
|
||||
parley — so the size that reaches both the line-breaker and the
|
||||
rasteriser (`TextData::place`, which reads back whatever `shape` set) is
|
||||
the display's *physical* size, and the glyph atlas holds a bitmap at the
|
||||
resolution it is actually shown at. `GlyphKey.size` already keys on the
|
||||
resolved size, so a cache entry is naturally per-physical-size with no
|
||||
further change. The one caller with no `Painter` to read density from
|
||||
(`TextEditCtx::layout`, cursor movement and hit-testing) reads a second
|
||||
copy kept directly on `TextData` (`TextData::density`) instead — an
|
||||
accepted duplication rather than threading a `Painter` into every input
|
||||
handler for one field, the same tradeoff `AndroidRenderer::content_scale`
|
||||
already makes for the Diagnostics page.
|
||||
|
||||
**What did not change**: `rel`/`rest` are unaffected (already
|
||||
resolution-independent, a fraction of the parent). `Span::gap` and
|
||||
`Padding`'s four sides moved from bare `f32` to `Len` so `dp(...)` works
|
||||
on them the same as any other size; a bare number is still `abs`,
|
||||
physical pixels, unchanged.
|
||||
|
||||
## For IRIS.md
|
||||
|
||||
When this lands, copy this entry into `IRIS.md` (newest first):
|
||||
+147
-10
@@ -147,12 +147,37 @@ turn.
|
||||
|
||||
Spawn: `claude -p --verbose --input-format stream-json --output-format
|
||||
stream-json --permission-mode <mode>` in the chosen working directory, plus
|
||||
`--model`. Wire-format notes are pinned against CLI 2.1.237 in
|
||||
`session/claude.rs`'s module doc: permissions need the hidden
|
||||
`--permission-prompt-tool stdio` flag, AskUserQuestion answers ride
|
||||
`updatedInput.answers` keyed by question text, and `set_model`/`interrupt`
|
||||
`--model` and, where one has been chosen, `--effort`. Wire-format notes are
|
||||
pinned against CLI 2.1.237 in `session/claude.rs`'s module doc: permissions
|
||||
need the hidden `--permission-prompt-tool stdio` flag, AskUserQuestion answers
|
||||
ride `updatedInput.answers` keyed by question text, and `set_model`/`interrupt`
|
||||
are control requests.
|
||||
|
||||
**The thinking level is settled at launch** (added 2026-09-04, because it is
|
||||
the largest saving available on a long session: output is about an eighth of
|
||||
what a session costs and thinking is the bulk of output, against the ~1.5% that
|
||||
is prose). The CLI's only two setting control requests are `set_model` and
|
||||
`set_permission_mode` -- checked against the 2.1.258 binary -- so there is no
|
||||
way to ask a running process to think differently. `set_session_effort` is
|
||||
therefore shaped like `set_session_cwd` rather than like `set_session_model`:
|
||||
it records the level and **stops the process**, and the next message or Start
|
||||
launches one that has it. It lives in the session settings dialog beside the
|
||||
working directory for that reason, not on the session bar beside the model and
|
||||
the mode, which do take effect mid-turn. `None` is a level in its own right --
|
||||
the CLI's own default -- so the picker can return to it; a level this app named
|
||||
as the default instead would be this app choosing one.
|
||||
|
||||
**What a new session starts at is `Config::default_effort`**, applied in
|
||||
`spawn_session` rather than filled in by the spawn screen, so it holds for an
|
||||
import and a bare API call as well. It is set by the spawn screen's own
|
||||
picker, whose label says so: one control, where new sessions are made, rather
|
||||
than a settings page for a single value. It is not on a provider, because
|
||||
providers are discovered and the next rediscovery would erase it, and not on
|
||||
the phone, because a second device would then spawn at a level nobody there
|
||||
chose. `GET`/`POST /defaults` carry it, as a struct rather than a bare value
|
||||
so the permission mode -- still hardcoded to `auto` on the spawn screen -- can
|
||||
move there without a second route.
|
||||
|
||||
**`--resume` only ever runs when nothing else has that session open.** That
|
||||
is the rule behind the import refusal, the single `ClaudeDriver::launch`
|
||||
entry point, and the `Exited` correction below; two CLIs on one session file
|
||||
@@ -169,10 +194,33 @@ deliberate and easy to undo by accident:
|
||||
when the process restarts. That leaves the Claude driver as the odd one
|
||||
out rather than this one — the CLI's memory is a cache in front of the same
|
||||
transcript. Resolve any inconsistency in this direction.
|
||||
- **A llama session on an ssh host is refused.** The model is reached over
|
||||
HTTP and forwarding that port is not built, so refusing beats silently
|
||||
talking to the wrong machine. A transport is "run this" plus "reach this
|
||||
port", and only the first half exists.
|
||||
- **A llama session runs on whatever machine its setup names** (2026-09-04,
|
||||
the last of phase 5). A transport is "run this" plus "reach this port", and
|
||||
the second half is `Transport::reserve_port` — the port the server binds
|
||||
*there* and the port that reaches it *here*, the same number locally —
|
||||
carried by `Launch::reaching` onto the connection that already runs the
|
||||
command. `llama-server` binds loopback on the far machine, so nothing is
|
||||
served to its network. The far port is a guess from a range below the
|
||||
ephemeral one, because no portable way to ask a machine for a free port
|
||||
avoids racing the bind anyway; a collision is not silent, since the server
|
||||
fails to bind and the readiness poll reports what its log said.
|
||||
- **The model file lives on the machine that serves it** (2026-09-04). Each
|
||||
setup names its own models directory (`SshConfig::models_dir`, default
|
||||
`~/.local/share/ai-app/models` expanded *there*), and a spawn resolves the
|
||||
key on that machine — one round trip answering "at /abs/path" or "missing",
|
||||
so a model that is not there is refused at the spawn rather than becoming a
|
||||
server that never becomes ready. The spawn screen offers
|
||||
`GET /setups/{id}/models`, that machine's list, rather than `GET /models`,
|
||||
which is this backend's downloads. Downloading *to* another machine is
|
||||
deliberately not built: a multi-gigabyte transfer with no progress
|
||||
anywhere, and the file gets there however anything else on that machine
|
||||
did.
|
||||
- **The readiness poll watches the process, not only the port.** A model that
|
||||
will not load, a port already taken, a flag an older build does not know:
|
||||
all exit within a second and none will ever answer `/health`, so waiting
|
||||
out the 300s timeout turned the server's own account of the problem into
|
||||
"gave up". The failure carries the tail of `llama-server.log`, which on a
|
||||
remote session is the only copy anybody reading the phone can see.
|
||||
|
||||
### Models (2026-08-28)
|
||||
|
||||
@@ -203,6 +251,13 @@ deliberate and easy to undo by accident:
|
||||
A driver says what to run; something above it turns that into a process.
|
||||
Otherwise transport knowledge sits inside a translator whose job is a wire
|
||||
format, and every future driver has to remember to do the same.
|
||||
- **A forwarded launch gets a pty and every other one does not** (measured
|
||||
2026-09-04). Killing the ssh client ends a CLI because it closes the stdin
|
||||
that CLI is reading; `llama-server` never reads its stdin, so the same kill
|
||||
left it running on the far machine with the model loaded — one orphan per
|
||||
stopped session. With `-tt` the far side takes SIGHUP when the connection
|
||||
goes. Its log then arrives through a line discipline, which nothing parses.
|
||||
`-T` stays everywhere else, where a pty would rewrite the JSONL.
|
||||
- **`command -v` follows ssh's non-login PATH**, which is narrower than an
|
||||
interactive shell's, so a binary somewhere unusual is invisible to
|
||||
discovery. Point `command` at an absolute path.
|
||||
@@ -499,6 +554,25 @@ rate-limited bucket). Poll at ≥180 s, only while a Claude session exists or
|
||||
the usage screen is open, and cache the last answer. It is undocumented, so
|
||||
`usage.rs` treats every field as optional and degrades rather than erroring.
|
||||
|
||||
**Per provider, not per machine (2026-09-04).** A machine is not what is
|
||||
metered; the provider a session runs is. One machine offers echo, the Claude
|
||||
CLI and a local model side by side, and only the second spends anything — so
|
||||
pairing a session with a snapshot by machine alone drew the CLI's five-hour
|
||||
window under every echo session on it, a quota that session cannot spend. A
|
||||
session now names its meter (`usageProvider`, from
|
||||
`DriverKind::usage_provider`, which `usage::providers_for` reads too, so the
|
||||
two lists cannot disagree) and `GET /usage` is matched on machine *and*
|
||||
provider. `None` is a session that meters nothing, and the phone draws
|
||||
nothing at all for it — not a zero, and not "unknown".
|
||||
|
||||
`DriverKind::Echo` names a meter of its own that exists only when a test has
|
||||
asked for one: `/usage` in an echo session sets an invented answer
|
||||
(`usage::Fixture`), and with none set there is no snapshot and no bar. That
|
||||
is what makes those screens' states reachable — a number near the top, a
|
||||
window between blocks with no reset time, a machine nobody logged into, one
|
||||
that could not be reached — without spending real quota to arrange them,
|
||||
which is why none of them had ever been looked at.
|
||||
|
||||
**Per machine, not per backend (2026-08-29).** The credential store that
|
||||
matters is the one on the machine the session runs on, because that is the
|
||||
account being billed — and in the layout this aims at, `ai-server` is on the
|
||||
@@ -522,6 +596,68 @@ always running. So absent means **not running**, and only a timestamp that
|
||||
arrives and cannot be parsed is unknown. `WindowEnd` in `ResetCountdown.kt`
|
||||
is the one rule both readers go through.
|
||||
|
||||
### Auto-resume (2026-09-05)
|
||||
|
||||
**A session may pick itself back up when the account's usage limit lifts.**
|
||||
Off unless somebody switched that session to it, because it spends quota the
|
||||
moment quota exists and does so with nobody looking — that is not a thing a
|
||||
default may decide. It sends one message, `continue` unless another was
|
||||
typed, and then it is done; there is no retry loop around the conversation
|
||||
itself.
|
||||
|
||||
**Running out of quota is a state, not an error.** `Event::LimitReached`
|
||||
carries the dialect's reset time where it gave one, and recognising it
|
||||
belongs to the driver — the Claude CLI ends the turn with `is_error` and
|
||||
`Claude AI usage limit reached|1788546972`, and nothing above the driver
|
||||
matches on a string. The transcript draws it as a divider, like a clear or a
|
||||
compaction: what a reader scrolling back wants from it is why the
|
||||
conversation stops at that line.
|
||||
|
||||
**The schedule is a plan to ask, never a plan to send.** Every reset time
|
||||
available here is untrustworthy in the direction that matters: the dialect's
|
||||
is written when the turn fails, and the endpoint's moves when the window
|
||||
does. So the wait ends in a question to `usage.rs`, and only `ok` with no
|
||||
window at 100% sends anything. A window still spent reschedules to *its own*
|
||||
reset time — which is what makes a limit that lifts later than promised wait
|
||||
longer, and one that lifts sooner resume sooner. A meter that cannot be
|
||||
asked at all is a longer wait too, never a send: "we could not find out"
|
||||
must not be able to produce the same action as "there is room".
|
||||
|
||||
Bounded, because something has to be: a day after the limit was hit the wait
|
||||
stops and says so in the session's own transcript. A machine that can never
|
||||
be asked would otherwise be retried for ever with nothing on screen saying
|
||||
so.
|
||||
|
||||
The schedule is persisted on the session (`resume: Some(ScheduledResume)`),
|
||||
not held in memory: a five-hour window routinely outlasts a backend restart,
|
||||
and a wait forgotten across one is a session that silently never comes back.
|
||||
`resume.rs` is the top layer — it holds the manager and the monitor and
|
||||
neither holds it — which is what lets the decision be a pure function of a
|
||||
snapshot and a clock. The pump reports limits downward on a broadcast, for
|
||||
the reason `Shared` exists: the pump runs underneath the manager.
|
||||
|
||||
**Exercised with echo, never with a real account.** `/limit [minutes]` in an
|
||||
echo session reports the same event a real driver does, and `/usage` sets
|
||||
what the meter answers — deliberately two commands, because the two
|
||||
disagreeing is the state the whole design is about. The loop was driven end
|
||||
to end that way on 2026-09-05: the wait moved from the dialect's two minutes
|
||||
to the meter's seven when the meter changed its mind, and the message went
|
||||
out on the first check after the meter came back under the limit.
|
||||
|
||||
### Subagents (2026-09-05)
|
||||
|
||||
**A subagent is a second transcript owned by a session, in the same event
|
||||
model, with no process and no controls of its own.** Full design and wire
|
||||
shape in `SUBAGENTS.md`, kept separate because the app half is being built
|
||||
against it in parallel and it is the shared contract between the two. The
|
||||
one-paragraph reason: a session's Task-tool helpers already speak the common
|
||||
event model on the parent's own stdout (each line carrying
|
||||
`parent_tool_use_id`), so giving each one its own small transcript — same
|
||||
file format, same paging routes, same SSE stream, reused by addressing rather
|
||||
than by copying — costs a routing step in the translator and a registry
|
||||
(`session/subagent.rs`) rather than a second session type with a driver, a
|
||||
process and a config entry it does not need.
|
||||
|
||||
### HTTP surface
|
||||
|
||||
**`routes.rs`'s module doc comment is the table.** REST for actions, one SSE
|
||||
@@ -800,8 +936,9 @@ Noticed and deliberately not fixed, so they are not re-found from scratch.
|
||||
|
||||
Phases 1–3 (the skeleton pipe, the full Claude driver, the usage screen) done
|
||||
2026-08-24. Phase 4 (llama.cpp: model browsing, downloads, and `llama-server`
|
||||
through its OpenAI-compatible endpoint) and phase 5 (ssh) done 2026-08-28.
|
||||
The file explorer and the transcript cache followed in September. What is
|
||||
through its OpenAI-compatible endpoint) and phase 5 (ssh) done 2026-08-28,
|
||||
except for the remote `llama-server` and its port forward, which landed
|
||||
2026-09-04. The file explorer and the transcript cache followed in September. What is
|
||||
left is real-phone/WireGuard bring-up, which is operational rather than code.
|
||||
|
||||
Each phase ended runnable and verified against the real thing. The backend
|
||||
+5069
File diff suppressed because it is too large.
Load diff
File renamed without changes.
File renamed without changes.
File renamed without changes.
@@ -0,0 +1,73 @@
|
||||
# Compose bench report from Iris's phone, 2026-09-06
|
||||
|
||||
The Compose half of P0 (RUST.md), run by Iris on her own phone and pasted
|
||||
back verbatim. The iris half's report goes beside it in this directory
|
||||
when it exists. Her caveat, worth keeping with the numbers: "I don't think
|
||||
this is entirely fair because the UI for iris is more minimal" -- the
|
||||
Compose screen also draws the usage bar, the status row and tool cards,
|
||||
which the iris bench screen does not yet. Her impression of the iris build
|
||||
before its first-touch bug: "it already feels very smooth so far".
|
||||
|
||||
What to read first: the phone runs at 120 Hz, so the budget is 8.3 ms;
|
||||
`late` is measured against that. Compose's tail is the streaming phase --
|
||||
`markdown reparsed while streaming: 396, 8.5ms mean, 25.8ms worst` and
|
||||
`record: one block: 398, 6.3ms mean, 19.9ms worst` -- which is exactly the
|
||||
path iris's `TranscriptScreen::apply` (replace the last row only) is meant
|
||||
to beat. Process CPU over the run is 20.9 s of a 38.5 s run; peak RSS
|
||||
587 MB; battery current mean 419 mA.
|
||||
|
||||
```
|
||||
ai-app render report
|
||||
device: Pixel 9 Pro XL (Google), Android 17
|
||||
build: release
|
||||
|
||||
transcript:
|
||||
43 events, 41 rows, 93 units loaded
|
||||
viewport 1333px, 2 units visible
|
||||
on screen: the list's own 0px, AssistantMsg 24520px
|
||||
0 tool calls and 0 groups open
|
||||
|
||||
frames:
|
||||
1613 frames over 38.5s at 120Hz (8.3ms budget)
|
||||
late: 742 (46.0%)
|
||||
total p50 7.7ms p90 29.2ms p99 41.1ms
|
||||
waited p50 0.5ms p90 12.9ms p99 27.2ms
|
||||
input p50 0.0ms p90 0.0ms p99 0.0ms
|
||||
anim p50 1.1ms p90 5.8ms p99 9.5ms
|
||||
layout p50 0.0ms p90 0.1ms p99 0.2ms
|
||||
draw p50 0.4ms p90 15.9ms p99 27.9ms
|
||||
sync p50 0.1ms p90 0.5ms p99 1.0ms
|
||||
issue p50 1.1ms p90 1.7ms p99 3.0ms
|
||||
swap p50 0.4ms p90 0.5ms p99 0.7ms
|
||||
gpu p50 1.8ms p90 2.1ms p99 6.6ms
|
||||
|
||||
where the draw phase went:
|
||||
draw phase 3.83ms per frame, of which:
|
||||
the transcript: 0.25ms (measure 0.15, place 0.10, record 0.00)
|
||||
everything else: 3.58ms (93%)
|
||||
|
||||
work since this was last copied:
|
||||
draw: the whole transcript: 12, 0.2ms total, 0.0ms mean, 0.0ms worst
|
||||
grouped tool runs: 398, 10.3ms total, 0.0ms mean, 0.1ms worst
|
||||
markdown cut into pieces: 1, 0.0ms total, 0.0ms mean, 0.0ms worst
|
||||
markdown parsed while composing: 7, 3.0ms total, 0.4ms mean, 0.5ms worst
|
||||
markdown ready: 46
|
||||
markdown reparsed while streaming: 396, 3384.6ms total, 8.5ms mean, 25.8ms worst
|
||||
markdown warmed: 1, 1.4ms total, 1.4ms mean, 1.4ms worst
|
||||
measure: the whole transcript: 957, 248.7ms total, 0.3ms mean, 15.7ms worst
|
||||
message composed: 403
|
||||
message cut into parts: 1, 0.1ms total, 0.1ms mean, 0.1ms worst
|
||||
place: the whole transcript: 1319, 162.4ms total, 0.1ms mean, 1.3ms worst
|
||||
record: one block: 398, 2498.7ms total, 6.3ms mean, 19.9ms worst
|
||||
session screen recomposed: 413
|
||||
status row recomposed: 1
|
||||
unit composed: 538
|
||||
units flattened: 399, 44.3ms total, 0.1ms mean, 0.5ms worst
|
||||
usage bar recomposed: 413
|
||||
|
||||
bench:
|
||||
scroll: 6 cycles (24 swipes), streamed 400/400 fixture events
|
||||
process CPU time over this run: 20907ms
|
||||
peak RSS: 587356kB
|
||||
battery current: mean -418509µA over 39 samples (min -1988281, max -107812)
|
||||
```
|
||||
@@ -0,0 +1,93 @@
|
||||
# Compose bench v2 report from Iris's phone, 2026-09-06
|
||||
|
||||
Bench v2 (fling / stream / type / keyboard, RUST.md's P0 box) on the
|
||||
Compose `bench` build, run by Iris on her Pixel 9 Pro XL, verbatim. Note
|
||||
the display was at **60 Hz** for this run (16.7 ms budget) where the v1
|
||||
run was at 120 Hz -- the phone's adaptive refresh rate decides, and
|
||||
`late` is judged against whichever it was, so compare a run with a run at
|
||||
the same rate. The iris v2 report goes beside this when it exists.
|
||||
|
||||
What it says: fling, type and keyboard are all essentially clean on
|
||||
Compose (0.1%, 0.9% and 0% late; fling p50 5.5 ms, p99 11.6 ms). The
|
||||
whole tail is the streaming phase again -- 41.9% late, p99 42.5 ms,
|
||||
driven by `markdown reparsed while streaming` (8.6 ms mean, 30.3 ms
|
||||
worst) and `record: one block` (6.3 ms mean, 25.5 ms worst). Process CPU
|
||||
69.6 s over the 125.5 s run; peak RSS 577 MB; battery current mean
|
||||
571 mA over 126 samples.
|
||||
|
||||
```
|
||||
ai-app render report
|
||||
device: Pixel 9 Pro XL (Google), Android 17
|
||||
build: release
|
||||
|
||||
transcript:
|
||||
108 events, 26 rows, 58 units loaded
|
||||
viewport 1531px, 2 units visible
|
||||
on screen: the list's own 0px, AssistantMsg 24520px
|
||||
0 tool calls and 0 groups open
|
||||
|
||||
per phase:
|
||||
fling: 3278 frames over 32.7s
|
||||
late: 4 (0.1%)
|
||||
total p50 5.5ms p90 8.7ms p99 11.6ms
|
||||
worst 49.0ms
|
||||
stream: 1041 frames over 21.3s
|
||||
late: 436 (41.9%)
|
||||
total p50 13.4ms p90 31.7ms p99 42.5ms
|
||||
worst 52.5ms
|
||||
type: 2446 frames over 61.5s
|
||||
late: 23 (0.9%)
|
||||
total p50 7.3ms p90 13.2ms p99 16.5ms
|
||||
worst 38.9ms
|
||||
keyboard: 358 frames over 10.0s
|
||||
late: 0 (0.0%)
|
||||
total p50 6.3ms p90 8.6ms p99 11.1ms
|
||||
worst 12.0ms
|
||||
|
||||
frames:
|
||||
7122 frames over 125.5s at 60Hz (16.7ms budget)
|
||||
late: 463 (6.5%)
|
||||
total p50 6.0ms p90 13.8ms p99 34.0ms
|
||||
waited p50 0.5ms p90 1.1ms p99 19.9ms
|
||||
input p50 0.0ms p90 0.0ms p99 0.0ms
|
||||
anim p50 0.7ms p90 4.5ms p99 7.6ms
|
||||
layout p50 0.1ms p90 0.1ms p99 0.2ms
|
||||
draw p50 0.7ms p90 2.9ms p99 21.9ms
|
||||
sync p50 0.1ms p90 0.2ms p99 0.6ms
|
||||
issue p50 1.4ms p90 2.4ms p99 3.2ms
|
||||
swap p50 0.4ms p90 0.8ms p99 1.2ms
|
||||
gpu p50 1.5ms p90 2.1ms p99 6.6ms
|
||||
|
||||
where the draw phase went:
|
||||
draw phase 1.74ms per frame, of which:
|
||||
the transcript: 0.24ms (measure 0.10, place 0.14, record 0.00)
|
||||
everything else: 1.51ms (86%)
|
||||
|
||||
work since this was last copied:
|
||||
draw: the whole transcript: 280, 1.9ms total, 0.0ms mean, 0.0ms worst
|
||||
grouped tool runs: 407, 18.6ms total, 0.0ms mean, 0.2ms worst
|
||||
markdown cut into pieces: 40, 0.4ms total, 0.0ms mean, 0.0ms worst
|
||||
markdown parsed while composing: 2, 1.6ms total, 0.8ms mean, 1.1ms worst
|
||||
markdown ready: 323
|
||||
markdown reparsed while streaming: 395, 3406.9ms total, 8.6ms mean, 30.3ms worst
|
||||
markdown warmed: 40, 38.2ms total, 1.0ms mean, 4.3ms worst
|
||||
measure: the whole transcript: 1978, 683.9ms total, 0.3ms mean, 15.9ms worst
|
||||
message composed: 397
|
||||
message cut into parts: 40, 2.9ms total, 0.1ms mean, 0.2ms worst
|
||||
place: the whole transcript: 4308, 996.0ms total, 0.2ms mean, 2.4ms worst
|
||||
record: one block: 394, 2472.7ms total, 6.3ms mean, 25.5ms worst
|
||||
session screen recomposed: 1630
|
||||
status row recomposed: 1
|
||||
transcript page from server: 10
|
||||
unit composed: 927
|
||||
units flattened: 408, 85.5ms total, 0.2ms mean, 1.8ms worst
|
||||
|
||||
bench:
|
||||
fling: 8 flings out + 8 back at 12000px/s, travel start=idx=0/off=0px outward=idx=188/off=182px end=idx=0/off=0px
|
||||
scroll: 6 cycles (24 swipes, legacy tween), streamed 400/400 fixture events
|
||||
type: 600 characters inserted then deleted, one per 50ms
|
||||
keyboard: shown 5/5, hidden 5/5 (confirmed via isImeVisible)
|
||||
process CPU time over this run: 69564ms
|
||||
peak RSS: 577452kB
|
||||
battery current: mean -571483µA over 126 samples (min -2361718, max -99218)
|
||||
```
|
||||
@@ -0,0 +1,31 @@
|
||||
# iris bench report from Iris's phone, 2026-09-06, before the phone fixes
|
||||
|
||||
Build 46246ea (Vulkan, bench v1: 24-swipe scroll loop then 400 streamed
|
||||
events), run by Iris on her Pixel 9 Pro XL before the first-touch wipe,
|
||||
the missing bold faces, the density scale and the status-bar inset were
|
||||
fixed -- so the rows were drawn at roughly a third of their intended size
|
||||
and the run may have included frames after the wipe. Preliminary, kept
|
||||
because it is the first iris number from real hardware. Compare with
|
||||
`compose-phone-2026-09-06.md`, taken on the same phone with the same
|
||||
fixture and gesture loop.
|
||||
|
||||
Reading it: the phone is 120 Hz (8.3 ms budget). `janky%` here counts
|
||||
frames over 16.7 ms, so it is not Compose's `late` (over 8.3 ms). Like for
|
||||
like: iris p50 6.2 ms vs Compose 7.7 ms; p90 32.0 vs 29.2; p99 42.1 vs
|
||||
41.1. `cpu_p50=4.7ms` is iris's own per-frame CPU work on the phone,
|
||||
against 0.2-0.4 ms on the emulator's x86 cores. Process CPU 15.6 s vs
|
||||
20.9 s, but over a shorter run (692 frames vs 1613 -- iris only renders on
|
||||
change and had no fling settle time), so per-second CPU is not directly
|
||||
comparable; peak RSS 365 MB vs 587 MB. Battery current mean 563 mA vs
|
||||
419 mA is the one figure that reads worse, and it is the least
|
||||
comparable: 22 samples vs 39, over runs of different length and different
|
||||
idle share. Bench v2's per-phase accounting is what makes these comparable.
|
||||
|
||||
```
|
||||
iris bench report
|
||||
frames=692 janky%=32.37 p50=6.2ms p90=32.0ms p99=42.1ms worst=52.6ms (measures redraw-start to after present() is called, not GPU/compositor completion) cpu_p50=4.7ms gpu_wait_p50=1.3ms (redraw-start-to-submit vs. submit-to-after-present)
|
||||
scroll: 6 cycles (24 swipes), streamed 400/400 fixture events
|
||||
process CPU time over this run: 15554ms
|
||||
peak RSS: 365328kB
|
||||
battery current: mean -563493µA over 22 samples (min -1807812, max -132812)
|
||||
```
|
||||
@@ -305,6 +305,25 @@ pub enum Event {
|
||||
/// it, which is why this is written down rather than left to be inferred
|
||||
/// from a second example that does not exist.
|
||||
Cleared,
|
||||
/// The account behind this session has no quota left, so the turn stopped
|
||||
/// without finishing.
|
||||
///
|
||||
/// Its own event rather than an [`Event::Error`] carrying the dialect's
|
||||
/// sentence, because two things act on it that cannot read English: the
|
||||
/// transcript draws it as a state the session is in rather than as a
|
||||
/// failure of something it did, and `crate::resume` schedules the message
|
||||
/// that picks the work back up. Recognising it belongs to the driver, which
|
||||
/// is the only layer that knows its dialect's wording -- above here nothing
|
||||
/// matches on strings.
|
||||
///
|
||||
/// `resets_at` is epoch seconds, and `None` is a real state: the dialect
|
||||
/// said the limit was hit without saying when it lifts. Nothing here
|
||||
/// invents one -- what the wait is actually decided against is the usage
|
||||
/// endpoint, and this is the hint that starts the waiting.
|
||||
LimitReached {
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
resets_at: Option<f64>,
|
||||
},
|
||||
Error {
|
||||
message: String,
|
||||
},
|
||||
|
||||
Generated
+1110
-5
File diff suppressed because it is too large.
Load diff
+31
-1
@@ -13,6 +13,7 @@ swash = { workspace = true }
|
||||
pollster = { workspace = true }
|
||||
wgpu = { workspace = true }
|
||||
image = { workspace = true }
|
||||
accesskit = { workspace = true }
|
||||
tokio = { workspace = true, features = ["sync", "rt", "rt-multi-thread"] }
|
||||
|
||||
# winit everywhere except Android; android-view (below) is what stands in
|
||||
@@ -27,6 +28,11 @@ tokio = { workspace = true, features = ["sync", "rt", "rt-multi-thread"] }
|
||||
[target.'cfg(not(target_os = "android"))'.dependencies]
|
||||
winit = { workspace = true }
|
||||
arboard = { workspace = true, features = ["wayland-data-control"] }
|
||||
# I4 (RUST.md): the desktop half of the AccessKit push, `winit`'s own
|
||||
# adapter over `accesskit`. No pin needed the way android-view's rev is
|
||||
# pinned -- this is an ordinary crates.io release with no local abort to
|
||||
# track (that finding is Android-only, see below).
|
||||
accesskit_winit = "0.34.0"
|
||||
|
||||
# Pinned to the exact commit RUST.md's E1 (2026-09-04) measured on this
|
||||
# emulator -- real Vulkan rendering, a working `InputConnection`, and the
|
||||
@@ -35,6 +41,14 @@ arboard = { workspace = true, features = ["wayland-data-control"] }
|
||||
# is dated rather than floating.
|
||||
[target.'cfg(target_os = "android")'.dependencies]
|
||||
android-view = { git = "https://github.com/rust-mobile/android-view.git", rev = "bec6c62a96cef8239b0fd7fedeef9b184d02e3a1" }
|
||||
# I4 (RUST.md): the Android half of the AccessKit push, over android-view's
|
||||
# `AccessibilityNodeProvider`. **0.8.0 carries the same detach-abort E1
|
||||
# found on 0.4.0** (the `State` enum still never returns to `Inactive`,
|
||||
# and `send_completed_event` still unwraps a Java exception) -- advancing
|
||||
# the version is not the fix, so pinning to a specific rev buys nothing
|
||||
# here the way it does for android-view itself. `android/view.rs`'s
|
||||
# `raise_if_enabled` is the mitigation, carried from E1.
|
||||
accesskit_android = "0.8.0"
|
||||
# Not re-exported by android-view (only `jni` and `ndk` are), and needed
|
||||
# for `android/insets.rs`'s own id -> state map -- the same reason
|
||||
# android-view's own `PEER_MAP` carries one.
|
||||
@@ -43,6 +57,16 @@ send_wrapper = "0.6.0"
|
||||
# installs it -- this crate never installs a logger itself.
|
||||
log = "0.4.28"
|
||||
|
||||
[features]
|
||||
# RUST.md's I5 "Where iris's frame time goes" diagnosis: forces the Android
|
||||
# `wgpu::Instance` to `Backends::GL` instead of `Backends::PRIMARY`, so the
|
||||
# same build can be measured against SwiftShader's software Vulkan ICD (the
|
||||
# default) or virgl's GLES path, without a second env-var plumbing path that
|
||||
# nothing on this machine can hand to an already-launched Android process
|
||||
# (there is no `am start` environment and no system-property reader here to
|
||||
# add one). Android-only; `android/render.rs` is the only reader.
|
||||
force-gles = []
|
||||
|
||||
[dev-dependencies]
|
||||
tokio = { workspace = true, features = ["sync", "rt", "rt-multi-thread", "time"] }
|
||||
# The tabs example's widget tree. A dev-dependency cycle back to this
|
||||
@@ -59,7 +83,7 @@ name = "message_list"
|
||||
harness = false
|
||||
|
||||
[workspace]
|
||||
members = ["core", "macro", "tabs-ui"]
|
||||
members = ["core", "macro", "tabs-ui", "transcript-ui", "desktop-app"]
|
||||
# android-app pulls in android-view, which needs the NDK sysroot to link
|
||||
# -- excluded so `cargo build --workspace --all-targets` on the host stays
|
||||
# buildable. Cross-compile it from its own directory (its own single-crate
|
||||
@@ -82,6 +106,12 @@ parley = "0.11.1"
|
||||
swash = "0.2.10"
|
||||
fxhash = "0.2.1"
|
||||
arboard = "3.6.1"
|
||||
accesskit = "0.25.0"
|
||||
iris-core = { path = "core" }
|
||||
iris-macro = { path = "macro" }
|
||||
tokio = "1.49.0"
|
||||
# Current stable as of 2026-09-05 (`cargo search`) -- I5's markdown block
|
||||
# model, the same crate E2's uncommitted `e2-transcript` experiment used for
|
||||
# the identical job (RUST.md), rather than reimplementing a CommonMark
|
||||
# parser.
|
||||
pulldown-cmark = "0.13.4"
|
||||
Generated
+1065
File diff suppressed because it is too large.
Load diff
@@ -17,10 +17,55 @@ crate-type = ["cdylib"]
|
||||
|
||||
[dependencies]
|
||||
iris = { path = "../" }
|
||||
tabs-ui = { path = "../tabs-ui" }
|
||||
android-view = { git = "https://github.com/rust-mobile/android-view.git", rev = "bec6c62a96cef8239b0fd7fedeef9b184d02e3a1" }
|
||||
android_logger = "0.15.0"
|
||||
log = "0.4.28"
|
||||
# `tabs-screen` (default, I2/I4's demo) and `transcript-screen` (I5's
|
||||
# Android integration) are mutually exclusive -- one `ActiveClient` type is
|
||||
# compiled in, never both (`lib.rs`'s doc comment) -- so both sets of deps
|
||||
# are optional and each screen's feature pulls in only its own. Without
|
||||
# this, building `--features transcript-screen` alone (default features
|
||||
# still on) left `tabs-ui` linked but never referenced under that cfg,
|
||||
# which Cargo's `unused_dependencies` lint (on by default) correctly flags.
|
||||
tabs-ui = { path = "../tabs-ui", optional = true }
|
||||
transcript-ui = { path = "../transcript-ui", optional = true }
|
||||
client-core = { path = "../../client-core", optional = true }
|
||||
event-model = { path = "../../event-model", optional = true }
|
||||
serde_json = { version = "1", features = ["float_roundtrip"], optional = true }
|
||||
# P0's bench build only (docs/RUST.md): `getrusage(RUSAGE_SELF)` for
|
||||
# process CPU time, matching `libc::getrusage`'s mention in that box over
|
||||
# parsing `/proc/self/stat` by hand and assuming `USER_HZ`. Already in the
|
||||
# workspace's own dependency tree transitively (`iris/Cargo.lock`, pinned
|
||||
# at 0.2.179) -- this makes it a direct dependency at the same version
|
||||
# rather than a second, possibly-drifting resolution.
|
||||
libc = { version = "0.2.179", optional = true }
|
||||
# P0's bench build only: the scroll animation and the streaming phase are
|
||||
# both a sequence of `sleep`s inside the async task `rsc.spawn_task` already
|
||||
# runs on iris's own tokio runtime (`iris/src/task.rs`'s `Tasks::init`), and
|
||||
# the battery sampler is a second, concurrent task on that same runtime
|
||||
# (`tokio::spawn`) -- so this crate needs `tokio` directly rather than only
|
||||
# through `iris`. `rt`+`time` only: no I/O, no macros, nothing this crate
|
||||
# doesn't call. Version matches the one `iris`'s own dependency tree already
|
||||
# resolves to (`iris/Cargo.lock`), so there is one copy of the runtime, not
|
||||
# two.
|
||||
tokio = { version = "1.53.1", features = ["rt", "time"], optional = true }
|
||||
|
||||
[features]
|
||||
default = ["tabs-screen"]
|
||||
tabs-screen = ["dep:tabs-ui"]
|
||||
transcript-screen = ["dep:transcript-ui", "dep:client-core", "dep:event-model", "dep:serde_json"]
|
||||
# RUST.md's I5 "Where iris's frame time goes": forces the GLES backend
|
||||
# instead of SwiftShader's software Vulkan. See `iris/Cargo.toml`'s own doc
|
||||
# on the feature this forwards to.
|
||||
force-gles = ["iris/force-gles"]
|
||||
# P0's iris half (docs/RUST.md, docs/AGENTS.md's "The rigs"): the same
|
||||
# checked-in fixture, scroll loop and streaming phase the Compose `bench`
|
||||
# build type drives, run here against `transcript-ui`'s real screen with no
|
||||
# server. Depends on `transcript-screen` for `transcript-ui`/`client-core`/
|
||||
# `event-model` -- `lib.rs`'s `ActiveClient` selection gives this feature
|
||||
# priority over `transcript-screen`'s own `TranscriptClient` when both are
|
||||
# listed, which is how this crate's build command names both explicitly.
|
||||
bench = ["transcript-screen", "dep:libc", "dep:tokio"]
|
||||
|
||||
[profile.release]
|
||||
panic = "abort"
|
||||
|
||||
@@ -19,9 +19,39 @@ android {
|
||||
versionName = "1.0"
|
||||
}
|
||||
|
||||
// A release build must be signed, and the key is per machine rather than per repo -- same
|
||||
// reasoning and the same key as `app/build-apk.sh` (the Compose app): it is what a phone
|
||||
// recognises the app by, and a secret never lives in a checkout (the mount is shared with an
|
||||
// untrusted VM). `build-apk.sh` generates this key once and points at it through the
|
||||
// environment; without it a release build here is unsigned, which is fine for everything
|
||||
// except installing.
|
||||
def keystore = System.getenv("AI_APP_KEYSTORE")
|
||||
signingConfigs {
|
||||
if (keystore != null) {
|
||||
release {
|
||||
storeFile = file(keystore)
|
||||
storePassword = System.getenv("AI_APP_KEYSTORE_PASSWORD")
|
||||
keyAlias = "ai-app"
|
||||
keyPassword = storePassword
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
buildTypes {
|
||||
debug {
|
||||
}
|
||||
// P0's iris half (docs/RUST.md's P0 box): the build a phone actually runs. The `.so`
|
||||
// itself is built separately with `cargo ndk --release --features "transcript-screen
|
||||
// force-gles bench"` straight into src/main/jniLibs/ (this crate's own Cargo.toml) --
|
||||
// Gradle here only packages and signs whatever is already there, the same division as the
|
||||
// debug/tabs-screen build this project started with. `applicationIdSuffix` keeps it
|
||||
// installable beside a debug build of the tabs demo rather than replacing it.
|
||||
release {
|
||||
applicationIdSuffix ".bench"
|
||||
if (keystore != null) {
|
||||
signingConfig = signingConfigs.release
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
compileOptions {
|
||||
|
||||
@@ -1,6 +1,15 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<manifest xmlns:android="http://schemas.android.com/apk/res/android">
|
||||
|
||||
<!-- Only needed by the transcript-screen feature (RUST.md's I5),
|
||||
which talks to a real ai-server; the plain tabs demo (I2/I4) makes
|
||||
no network call and never noticed this was missing. Absent,
|
||||
UreqTransport::new's connect failed with EPERM (Operation not
|
||||
permitted), not the ECONNREFUSED/ENETUNREACH a firewall or a dead
|
||||
server would give: a seccomp-level socket denial reads nothing
|
||||
like a network problem, which is what made it worth a comment. -->
|
||||
<uses-permission android:name="android.permission.INTERNET" />
|
||||
|
||||
<application
|
||||
android:allowBackup="true"
|
||||
android:label="iris android-view demo"
|
||||
|
||||
@@ -1,6 +1,17 @@
|
||||
package dev.iris.android.demo;
|
||||
|
||||
import android.app.Activity;
|
||||
import android.content.ClipData;
|
||||
import android.content.ClipboardManager;
|
||||
import android.content.Context;
|
||||
import android.view.Gravity;
|
||||
import android.view.View;
|
||||
import android.view.ViewGroup;
|
||||
import android.widget.Button;
|
||||
import android.widget.FrameLayout;
|
||||
import android.widget.LinearLayout;
|
||||
import android.widget.ScrollView;
|
||||
import android.widget.TextView;
|
||||
|
||||
import org.linebender.android.rustview.RustView;
|
||||
|
||||
@@ -33,4 +44,112 @@ public final class IrisView extends RustView {
|
||||
unregisterInsetsNative(mViewPeer);
|
||||
super.onDetachedFromWindow();
|
||||
}
|
||||
|
||||
/**
|
||||
* Called from the Rust side (iris/src/android/view.rs's
|
||||
* `show_renderer_error`) when `AndroidRenderer::new` fails instead of
|
||||
* drawing -- an ordinary instance method rather than a `native` one,
|
||||
* since this call is Rust reaching into Java rather than the other
|
||||
* direction. Replaces the whole activity content with plain,
|
||||
* selectable, scrollable text rather than leaving the last frame (or a
|
||||
* blank surface) on screen with no way to report what happened:
|
||||
* UI_RULES.md's "a failure is reported where it happened, and says
|
||||
* what to do next." No dialog and no styling beyond what is needed to
|
||||
* read and copy the text -- this path exists for exactly the crash it
|
||||
* replaces, so it must not depend on anything that could itself fail
|
||||
* to render.
|
||||
*/
|
||||
void showRendererError(String report) {
|
||||
Context context = getContext();
|
||||
if (!(context instanceof Activity)) {
|
||||
return;
|
||||
}
|
||||
Activity activity = (Activity) context;
|
||||
TextView text = new TextView(activity);
|
||||
text.setText(report);
|
||||
text.setTextIsSelectable(true);
|
||||
text.setGravity(Gravity.TOP | Gravity.START);
|
||||
int pad = (int) (16 * activity.getResources().getDisplayMetrics().density);
|
||||
text.setPadding(pad, pad, pad, pad);
|
||||
ScrollView scroll = new ScrollView(activity);
|
||||
scroll.addView(text);
|
||||
activity.setContentView(scroll);
|
||||
}
|
||||
|
||||
private static final String DIAGNOSTICS_OVERLAY_TAG = "iris-diagnostics-overlay";
|
||||
|
||||
/**
|
||||
* The bench build's keyboard diagnostics capture
|
||||
* (`bench_client.rs`'s `on_insets_changed` /
|
||||
* `capture_keyboard_diagnostics`, via `bench_jni.rs`'s
|
||||
* `PlatformHandle::show_diagnostics_overlay`): unlike
|
||||
* `showRendererError` above, this adds a panel *over* this view
|
||||
* (`MainActivity`'s `FrameLayout` still holds `IrisView` underneath,
|
||||
* running) rather than replacing the activity's content, and gives it
|
||||
* a Copy button and a Close that removes the panel -- so it draws
|
||||
* (and can be read) whether or not iris itself is still putting
|
||||
* anything on screen, without abandoning the session that produced
|
||||
* it. Runs on the UI thread regardless of which thread calls it,
|
||||
* since the call comes from a background task (a delayed capture
|
||||
* after the keyboard opens), and touching the view tree off the UI
|
||||
* thread is undefined.
|
||||
*/
|
||||
void showDiagnosticsOverlay(String report) {
|
||||
Context context = getContext();
|
||||
if (!(context instanceof Activity)) {
|
||||
return;
|
||||
}
|
||||
Activity activity = (Activity) context;
|
||||
activity.runOnUiThread(() -> {
|
||||
ViewGroup parent = (ViewGroup) getParent();
|
||||
if (parent == null) {
|
||||
return;
|
||||
}
|
||||
View existing = parent.findViewWithTag(DIAGNOSTICS_OVERLAY_TAG);
|
||||
if (existing != null) {
|
||||
parent.removeView(existing);
|
||||
}
|
||||
|
||||
float density = activity.getResources().getDisplayMetrics().density;
|
||||
int pad = (int) (16 * density);
|
||||
|
||||
LinearLayout overlay = new LinearLayout(activity);
|
||||
overlay.setTag(DIAGNOSTICS_OVERLAY_TAG);
|
||||
overlay.setOrientation(LinearLayout.VERTICAL);
|
||||
overlay.setBackgroundColor(0xEE000000);
|
||||
overlay.setPadding(pad, pad, pad, pad);
|
||||
|
||||
TextView text = new TextView(activity);
|
||||
text.setText(report);
|
||||
text.setTextIsSelectable(true);
|
||||
text.setTextColor(0xFFFFFFFF);
|
||||
ScrollView scroll = new ScrollView(activity);
|
||||
scroll.addView(text);
|
||||
overlay.addView(scroll, new LinearLayout.LayoutParams(
|
||||
LinearLayout.LayoutParams.MATCH_PARENT, 0, 1f));
|
||||
|
||||
LinearLayout buttonRow = new LinearLayout(activity);
|
||||
buttonRow.setOrientation(LinearLayout.HORIZONTAL);
|
||||
buttonRow.setPadding(0, pad, 0, 0);
|
||||
|
||||
Button copy = new Button(activity);
|
||||
copy.setText("Copy");
|
||||
copy.setOnClickListener(v -> {
|
||||
ClipboardManager clipboard =
|
||||
(ClipboardManager) activity.getSystemService(Context.CLIPBOARD_SERVICE);
|
||||
if (clipboard != null) {
|
||||
clipboard.setPrimaryClip(ClipData.newPlainText("iris diagnostics", report));
|
||||
}
|
||||
});
|
||||
Button close = new Button(activity);
|
||||
close.setText("Close");
|
||||
close.setOnClickListener(v -> parent.removeView(overlay));
|
||||
buttonRow.addView(copy);
|
||||
buttonRow.addView(close);
|
||||
overlay.addView(buttonRow);
|
||||
|
||||
parent.addView(overlay, new FrameLayout.LayoutParams(
|
||||
FrameLayout.LayoutParams.MATCH_PARENT, FrameLayout.LayoutParams.MATCH_PARENT));
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -36,9 +36,28 @@ public final class MainActivity extends Activity {
|
||||
int top = insets.getSystemWindowInsetTop();
|
||||
int right = insets.getSystemWindowInsetRight();
|
||||
int bottom = insets.getSystemWindowInsetBottom();
|
||||
// The manifest declares adjustResize (AGENTS.md: without it the
|
||||
// keyboard pans the whole window instead of resizing it), and
|
||||
// under adjustResize the window itself shrinks to make room for
|
||||
// the keyboard -- which is exactly the condition under which
|
||||
// WindowInsets.Type.ime()'s own *inset amount* reports zero: it
|
||||
// measures how much of the window the keyboard overlaps, and
|
||||
// resize already made that overlap zero by construction. That
|
||||
// numeric inset is not a usable "is the keyboard open" signal
|
||||
// here (found while root-causing why bench_client.rs's keyboard
|
||||
// phase and auto-diagnostics never fired on the emulator despite
|
||||
// the keyboard visibly opening -- RUST.md's P0 box). What does
|
||||
// survive adjustResize is the boolean isVisible() answer, set
|
||||
// from the platform's own start/end of the transition over a
|
||||
// different path than the inset amount -- the same fact
|
||||
// AGENTS.md's "Things that have bitten" already names for the
|
||||
// Compose side's identical trap. Passed through as a 0/1 stand-
|
||||
// in for the ime_bottom pixel amount, since nothing on the Rust
|
||||
// side reads it as a real pixel value -- only `> 0.0`.
|
||||
int imeBottom = 0;
|
||||
if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.R) {
|
||||
imeBottom = insets.getInsets(WindowInsets.Type.ime()).bottom;
|
||||
if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.R
|
||||
&& insets.isVisible(WindowInsets.Type.ime())) {
|
||||
imeBottom = 1;
|
||||
}
|
||||
((IrisView) v).applyWindowInsets(left, top, right, bottom, imeBottom);
|
||||
return insets;
|
||||
|
||||
Executable
+90
@@ -0,0 +1,90 @@
|
||||
#!/bin/sh
|
||||
# Builds iris-android-app end to end: the cdylib (cargo ndk, straight into
|
||||
# app/src/main/jniLibs/) then the APK (Gradle). Written to stop re-typing
|
||||
# the same incantation by hand every time (ANDROID_HOME/NDK exports, the
|
||||
# cargo ndk invocation, the keystore env for a release build, apksigner/
|
||||
# aapt2 verification) -- see docs/RUST.md's P0 box. Same shape as `app/
|
||||
# build-apk.sh` (the Compose app's own build script) and `app/
|
||||
# iris-scroll.sh` (no coordinates, set -eu, exit 0 on success).
|
||||
#
|
||||
# Usage: ./build-apk.sh [debug|release] [--abi arm64-v8a|x86_64] [--features "a b c"]
|
||||
# debug/release default to debug (matches this-machine-android's "the
|
||||
# emulator stays on debug" rule -- pass `release` explicitly for a phone
|
||||
# build). --abi defaults to arm64-v8a (a phone/real device); pass
|
||||
# x86_64 for this checkout's own AVD. --features defaults to
|
||||
# "transcript-screen bench" -- deliberately *without* `force-gles`, unlike
|
||||
# an earlier version of this default. `force-gles` (`iris/Cargo.toml`'s
|
||||
# own doc) exists only to force the emulator off its default software
|
||||
# Vulkan and onto GLES for one specific measurement (RUST.md's I5, "Where
|
||||
# iris's frame time goes") -- it was never meant to reach a real device,
|
||||
# but this script's old default put it in every arm64 build regardless,
|
||||
# so the P0 bench APK delivered to Iris's phone forced GLES there too.
|
||||
# That is the named hypothesis in RUST.md's P0 box ("iris bench crash on
|
||||
# the phone, 2026-09-06"): a real Vulkan driver is what a phone should
|
||||
# run, and GLES is the backend the same box's own SwiftShader finding
|
||||
# already flagged as the fragile one for this shader's storage buffers.
|
||||
# Pass `--features "transcript-screen force-gles bench"` explicitly for
|
||||
# an emulator backend-isolation run; never for a build meant for a phone.
|
||||
set -eu
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
BUILD_TYPE="debug"
|
||||
ABI="arm64-v8a"
|
||||
FEATURES="transcript-screen bench"
|
||||
case "${1:-}" in
|
||||
debug|release) BUILD_TYPE="$1"; shift ;;
|
||||
esac
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--abi) ABI="$2"; shift 2 ;;
|
||||
--features) FEATURES="$2"; shift 2 ;;
|
||||
*) echo "build-apk.sh: unknown argument: $1" >&2; exit 1 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
SDK_ROOT="$HOME/Android/Sdk"
|
||||
export ANDROID_HOME="$SDK_ROOT"
|
||||
export ANDROID_SDK_ROOT="$SDK_ROOT"
|
||||
NDK_DIR=$(ls -d "$SDK_ROOT"/ndk/*/ 2>/dev/null | sort -V | tail -1)
|
||||
if [ -z "$NDK_DIR" ]; then
|
||||
echo "build-apk.sh: no NDK found under $SDK_ROOT/ndk" >&2
|
||||
exit 1
|
||||
fi
|
||||
export ANDROID_NDK_HOME="$NDK_DIR"
|
||||
|
||||
echo "build-apk.sh: cargo ndk -t $ABI build ${BUILD_TYPE:+(${BUILD_TYPE})} --features \"$FEATURES\""
|
||||
if [ "$BUILD_TYPE" = "release" ]; then
|
||||
cargo ndk -t "$ABI" -P 26 -o app/src/main/jniLibs/ build --release --features "$FEATURES"
|
||||
else
|
||||
cargo ndk -t "$ABI" -P 26 -o app/src/main/jniLibs/ build --features "$FEATURES"
|
||||
fi
|
||||
|
||||
GRADLE_TASK="assembleDebug"
|
||||
APK_DIR="app/build/outputs/apk/debug"
|
||||
APK_NAME="app-debug.apk"
|
||||
if [ "$BUILD_TYPE" = "release" ]; then
|
||||
GRADLE_TASK="assembleRelease"
|
||||
APK_DIR="app/build/outputs/apk/release"
|
||||
APK_NAME="app-release.apk"
|
||||
# Same key `app/build-apk.sh` (the Compose app) generates once under
|
||||
# ~/.config/ai-app/release.jks -- see AGENTS.md's "Checking your work".
|
||||
export AI_APP_KEYSTORE="$HOME/.config/ai-app/release.jks"
|
||||
if [ ! -f "$AI_APP_KEYSTORE" ]; then
|
||||
echo "build-apk.sh: no release key at $AI_APP_KEYSTORE -- run app/build-apk.sh once first" >&2
|
||||
exit 1
|
||||
fi
|
||||
export AI_APP_KEYSTORE_PASSWORD
|
||||
AI_APP_KEYSTORE_PASSWORD=$(cat "$AI_APP_KEYSTORE.password")
|
||||
fi
|
||||
|
||||
gradle ":app:$GRADLE_TASK" --console=plain
|
||||
|
||||
APK_PATH="$(pwd)/$APK_DIR/$APK_NAME"
|
||||
BUILD_TOOLS=$(ls -d "$SDK_ROOT"/build-tools/*/ | sort -V | tail -1)
|
||||
echo "--- aapt2 dump badging ---"
|
||||
"${BUILD_TOOLS}aapt2" dump badging "$APK_PATH" | head -5
|
||||
if [ "$BUILD_TYPE" = "release" ]; then
|
||||
echo "--- apksigner verify ---"
|
||||
"${BUILD_TOOLS}apksigner" verify --print-certs "$APK_PATH"
|
||||
fi
|
||||
echo "$APK_PATH"
|
||||
@@ -0,0 +1,100 @@
|
||||
// Only does anything under the `transcript-screen` feature (RUST.md's I5
|
||||
// Android integration) -- the plain tabs build (I2/I4) needs none of this
|
||||
// and stays untouched, same reasoning as the feature gate in Cargo.toml.
|
||||
//
|
||||
// Bakes the sandbox server's host, port, token and pinned CA in at build
|
||||
// time, the same way `app/androidApp/build.gradle.kts`'s
|
||||
// `GeneratePinnedCert` task bakes the CA for the Compose app -- see that
|
||||
// file's comment for why reading the machine's own certificate at build
|
||||
// time is the right trust boundary. This build additionally bakes the
|
||||
// host/port/token, which the Compose app does not: that app enrolls at
|
||||
// runtime from a scanned QR/deep link, and a from-scratch enrollment UI
|
||||
// (Keystore-sealed token storage, a QR/link scanner) is real, separate
|
||||
// scope this integration does not need to build to answer RUST.md's
|
||||
// question -- there is nothing here yet resembling `ServerConfig.kt`. So
|
||||
// this is a **deliberate simplification for this rig only**: an APK built
|
||||
// this way is good for exactly the emulator/server pair that built it, and
|
||||
// must never be treated as a template for a real enrollment flow. Recorded
|
||||
// in RUST.md's I5 box rather than left to be rediscovered.
|
||||
use std::path::PathBuf;
|
||||
|
||||
fn main() {
|
||||
if std::env::var_os("CARGO_FEATURE_TRANSCRIPT_SCREEN").is_none() {
|
||||
return;
|
||||
}
|
||||
// P0's bench build (docs/RUST.md) opens the checked-in fixture with no
|
||||
// server at all -- `bench_client.rs` never references the `pinned`
|
||||
// module this generates, so requiring a live server's host/port/token/
|
||||
// CA to build it (as plain `transcript-screen` does, below) would be a
|
||||
// pointless requirement for a build that talks to nothing.
|
||||
if std::env::var_os("CARGO_FEATURE_BENCH").is_some() {
|
||||
return;
|
||||
}
|
||||
println!("cargo:rerun-if-env-changed=AI_APP_TRANSCRIPT_HOST");
|
||||
println!("cargo:rerun-if-env-changed=AI_APP_TRANSCRIPT_PORT");
|
||||
println!("cargo:rerun-if-env-changed=AI_APP_TRANSCRIPT_TOKEN");
|
||||
println!("cargo:rerun-if-env-changed=AI_APP_CA");
|
||||
println!("cargo:rerun-if-env-changed=XDG_CONFIG_HOME");
|
||||
|
||||
let host = require_env(
|
||||
"AI_APP_TRANSCRIPT_HOST",
|
||||
"the sandbox server's host as the emulator reaches it, e.g. 10.0.2.2",
|
||||
);
|
||||
let port = require_env(
|
||||
"AI_APP_TRANSCRIPT_PORT",
|
||||
"the sandbox server's port -- app/ui-sandbox.sh's start banner prints it",
|
||||
);
|
||||
let token = require_env(
|
||||
"AI_APP_TRANSCRIPT_TOKEN",
|
||||
"the bearer token -- ~/.config/ai-app/sandbox-token, or the start banner's enrollment link",
|
||||
);
|
||||
|
||||
let ca_path = std::env::var_os("AI_APP_CA")
|
||||
.map(PathBuf::from)
|
||||
.unwrap_or_else(|| {
|
||||
let base = std::env::var_os("XDG_CONFIG_HOME")
|
||||
.map(PathBuf::from)
|
||||
.unwrap_or_else(|| {
|
||||
let home = std::env::var_os("HOME").expect("HOME must be set");
|
||||
PathBuf::from(home).join(".config")
|
||||
});
|
||||
base.join("ai-app").join("certs").join("ca.pem")
|
||||
});
|
||||
let ca_pem = std::fs::read_to_string(&ca_path).unwrap_or_else(|e| {
|
||||
panic!(
|
||||
"no CA certificate at {} ({e}).\n\
|
||||
Start ai-server (or app/ui-sandbox.sh) once on this machine first -- it \
|
||||
generates the CA this build pins. Set AI_APP_CA=/path/to/ca.pem to build \
|
||||
against a different one.",
|
||||
ca_path.display()
|
||||
)
|
||||
});
|
||||
let ca_pem = ca_pem.trim();
|
||||
if !ca_pem.starts_with("-----BEGIN CERTIFICATE-----") {
|
||||
panic!("{} is not a PEM certificate.", ca_path.display());
|
||||
}
|
||||
|
||||
let out_dir = PathBuf::from(std::env::var_os("OUT_DIR").unwrap());
|
||||
let generated = format!(
|
||||
"// Generated by build.rs from {host}:{port} and {ca}. Do not edit.\n\
|
||||
pub const HOST: &str = {host_lit:?};\n\
|
||||
pub const PORT: u16 = {port};\n\
|
||||
pub const TOKEN: &str = {token_lit:?};\n\
|
||||
pub const CA_PEM: &str = {ca_lit:?};\n",
|
||||
host = host,
|
||||
port = port
|
||||
.parse::<u16>()
|
||||
.unwrap_or_else(|e| panic!("AI_APP_TRANSCRIPT_PORT={port:?} is not a u16: {e}")),
|
||||
ca = ca_path.display(),
|
||||
host_lit = host,
|
||||
token_lit = token,
|
||||
ca_lit = ca_pem,
|
||||
);
|
||||
std::fs::write(out_dir.join("pinned_config.rs"), generated).unwrap();
|
||||
}
|
||||
|
||||
fn require_env(name: &str, what: &str) -> String {
|
||||
std::env::var(name).unwrap_or_else(|_| {
|
||||
panic!("{name} must be set to build the transcript-screen feature -- {what}")
|
||||
})
|
||||
}
|
||||
Executable
+69
@@ -0,0 +1,69 @@
|
||||
#!/bin/sh
|
||||
# Installs and runs the iris `bench` build on this checkout's own emulator
|
||||
# (per this-machine-android's per-checkout-AVD rule; `emu serial` picks it)
|
||||
# and prints the report -- the iris half of `app/transcript-bench.sh`'s
|
||||
# job. No coordinates: the button is found by its accessibility label
|
||||
# through `ui-trace`, per AGENTS.md's "Driving the UI".
|
||||
#
|
||||
# Usage: ./run-bench.sh [--apk PATH]
|
||||
# Defaults to this checkout's own release APK
|
||||
# (app/build/outputs/apk/release/app-release.apk) if it exists, else the
|
||||
# debug one -- build one first with ./build-apk.sh.
|
||||
set -eu
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
APK=""
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--apk) APK="$2"; shift 2 ;;
|
||||
*) echo "run-bench.sh: unknown argument: $1" >&2; exit 1 ;;
|
||||
esac
|
||||
done
|
||||
if [ -z "$APK" ]; then
|
||||
if [ -f app/build/outputs/apk/release/app-release.apk ]; then
|
||||
APK=app/build/outputs/apk/release/app-release.apk
|
||||
else
|
||||
APK=app/build/outputs/apk/debug/app-debug.apk
|
||||
fi
|
||||
fi
|
||||
if [ ! -f "$APK" ]; then
|
||||
echo "run-bench.sh: no APK at $APK -- run ./build-apk.sh first" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
SERIAL=$(emu serial)
|
||||
PKG=$(aapt2 dump badging "$APK" 2>/dev/null | sed -n "s/^package: name='\\([^']*\\)'.*/\\1/p")
|
||||
if [ -z "$PKG" ]; then
|
||||
BUILD_TOOLS=$(ls -d "$HOME"/Android/Sdk/build-tools/*/ | sort -V | tail -1)
|
||||
PKG=$("${BUILD_TOOLS}aapt2" dump badging "$APK" | sed -n "s/^package: name='\\([^']*\\)'.*/\\1/p")
|
||||
fi
|
||||
|
||||
echo "run-bench.sh: installing $APK ($PKG) on $SERIAL"
|
||||
adb -s "$SERIAL" install -r "$APK" >/dev/null
|
||||
adb -s "$SERIAL" shell am force-stop "$PKG"
|
||||
adb -s "$SERIAL" logcat -c
|
||||
adb -s "$SERIAL" shell am start -n "$PKG/dev.iris.android.demo.MainActivity" >/dev/null
|
||||
|
||||
ui-trace record -s "$SERIAL" -d 3000 --do "tap 'Run benchmark'" -o /tmp/run-bench-tap.txt >/dev/null
|
||||
|
||||
# Poll for the report line rather than a fixed sleep -- the run itself is
|
||||
# a fixed script (RUST.md's "Benchmark v2": 16 flings, a 20s streaming
|
||||
# phase, ~61s of typing, 10s of keyboard toggles, roughly 2.5 minutes end
|
||||
# to end) but device speed varies. 260s cap rather than v1's 90s -- v2 is
|
||||
# a longer script than v1's swipe-loop-only run.
|
||||
i=0
|
||||
while [ "$i" -lt 260 ]; do
|
||||
LINE=$(adb -s "$SERIAL" logcat -d -s iris-android-app:I 2>/dev/null | grep "iris bench report:" || true)
|
||||
if [ -n "$LINE" ]; then
|
||||
break
|
||||
fi
|
||||
i=$((i + 1))
|
||||
sleep 1
|
||||
done
|
||||
if [ -z "$LINE" ]; then
|
||||
echo "run-bench.sh: no report after 260s -- check logcat by hand" >&2
|
||||
exit 1
|
||||
fi
|
||||
# -A 60 rather than v1's -A 6 -- v2's report has a per-phase block (four
|
||||
# phases, four lines each) on top of the frames/bench sections v1 had.
|
||||
adb -s "$SERIAL" logcat -d -s iris-android-app:I | grep -A 60 "iris bench report:"
|
||||
@@ -0,0 +1,988 @@
|
||||
//! P0's iris half (docs/RUST.md's P0 box, docs/AGENTS.md's "The rigs"):
|
||||
//! the same fixture, scroll loop and streaming phase the Compose `bench`
|
||||
//! build type's `BenchRun.kt`/`BenchFixture.kt` drive, run here against
|
||||
//! `transcript-ui`'s real screen with no server -- a frame-time comparison
|
||||
//! that measures the renderer rather than the data or the network.
|
||||
//!
|
||||
//! **Reuses `transcript_client.rs`'s shape** (folded items, the same
|
||||
//! `TranscriptScreen::apply` incremental update on every event) with the
|
||||
//! network half replaced by the checked-in fixture, embedded with
|
||||
//! `include_str!` -- `app/bench-fixture/assets/transcript.jsonl`,
|
||||
//! 1,915,760 bytes, generated by `app/bench-fixture/generate.py` and never
|
||||
//! a real transcript (that file's own README). The first 3,200 lines are
|
||||
//! the opening backlog, folded once through
|
||||
//! `client_core::transcript_fold::fold_page` exactly as a real
|
||||
//! `/transcript` page would be (then a full `transcript_ui::build_tree`,
|
||||
//! same as any first load); the remaining ~400 are the streaming tail,
|
||||
//! replayed one at a time through `fold_event` -- the same fold path a
|
||||
//! live SSE reply arrives on -- by the "Run benchmark" control below.
|
||||
//! Streaming through `apply` rather than a full rebuild per event is what
|
||||
//! this file exists to measure -- see docs/RUST.md's P0 box for the
|
||||
//! before/after report.
|
||||
|
||||
use crate::bench_jni::PlatformHandle;
|
||||
use android_view::jni::{JavaVM, objects::GlobalRef};
|
||||
use client_core::transcript_fold::{TranscriptItem, fold_event, fold_page, group_tool_runs};
|
||||
use event_model::SeqEvent;
|
||||
use iris::android::{AndroidAppState, AndroidRsc, AndroidUiState, HasAndroidUiState};
|
||||
use iris::prelude::*;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::sync::{Arc, Mutex};
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
/// bench-fixture/README.md: the first `BACKLOG_COUNT` non-blank lines are
|
||||
/// the opening window; the rest are the streaming tail. Kept in sync with
|
||||
/// `BenchFixture.kt`'s identical constant by hand -- both read the same
|
||||
/// checked-in file, so a mismatch would only mean the two apps' bench
|
||||
/// builds open a different split of it, not a wrong-vs-right answer.
|
||||
const BACKLOG_COUNT: usize = 3200;
|
||||
|
||||
/// RUST.md's "Benchmark v2" spec, written once so both apps' bench clients
|
||||
/// implement the identical four phases -- see that box before changing any
|
||||
/// constant here, since a mismatch would make the two reports stop
|
||||
/// measuring the same thing while still looking like they do.
|
||||
const STREAM_EVENTS_PER_SEC: u64 = 20;
|
||||
const STREAM_SECONDS: u64 = 20;
|
||||
|
||||
/// Kept only so this phase's own label text still reads "scroll: 6 cycles
|
||||
/// (24 swipes, legacy tween)" the way `BenchRun.kt`'s v2 report does --
|
||||
/// `docs/bench/compose-phone-v2-2026-09-06.md`'s own report shows this
|
||||
/// exact line even though the swipe loop it names no longer runs there
|
||||
/// either (the fling phase replaced it); nothing here drives an actual
|
||||
/// swipe with these any more.
|
||||
const LEGACY_CYCLES: usize = 6;
|
||||
|
||||
/// Fling phase (v2): a real fling through `List::fling`, not a tween --
|
||||
/// Iris's ask was that it "travel way faster" than the v1 swipe, and a
|
||||
/// tween can never exceed the distance/time it is given while a real
|
||||
/// fling decays from an initial velocity the way a finger flick does.
|
||||
/// 12,000 px/s matches `BenchRun.kt`'s own constant exactly.
|
||||
const FLING_VELOCITY_PX_S: f32 = 12_000.0;
|
||||
const FLING_COUNT: usize = 8;
|
||||
const FLING_SETTLE_CAP_MS: u64 = 3_000;
|
||||
const FLING_PAUSE_MS: u64 = 300;
|
||||
|
||||
/// Type phase (v2): long, multisyllabic words so the composer actually
|
||||
/// wraps and the transcript above it is pushed upward, typed and deleted
|
||||
/// one character per `TYPE_CHAR_MS`. Exactly `BenchRun.TYPE_TEXT` --
|
||||
/// verified 600 characters by `type_text_is_exactly_600_characters` below.
|
||||
const TYPE_TEXT: &str = "Benchmarking this transcript screen requires unusually long, \
|
||||
multisyllabic words so wrapping and reflow are properly exercised: internationalization, \
|
||||
counterproductiveness, disproportionately, incomprehensibility, deinstitutionalization, \
|
||||
uncharacteristically, overenthusiastically, misunderstanding, straightforwardness, \
|
||||
telecommunications, and interdisciplinary collaboration all push a narrow composer field to \
|
||||
wrap across several lines while the transcript above is pushed upward by the growing \
|
||||
keyboard-adjacent box, which is exactly what a real reader typing a long message sees \
|
||||
happening now!!!";
|
||||
const TYPE_CHAR_MS: u64 = 50;
|
||||
|
||||
/// Keyboard phase (v2): five show/hide cycles, a second apart, matching
|
||||
/// `BenchRun.kt`'s `KEYBOARD_CYCLES`/`KEYBOARD_SHOW_WAIT_MS`/
|
||||
/// `KEYBOARD_HIDE_WAIT_MS`.
|
||||
const KEYBOARD_CYCLES: usize = 5;
|
||||
const KEYBOARD_WAIT_MS: u64 = 1_000;
|
||||
|
||||
/// One animation step's target cadence -- close enough to 60Hz that a
|
||||
/// fling/scroll is many small moves rather than one jump, so frames are
|
||||
/// actually rendered along the way, and close enough that a `ctx.update`
|
||||
/// closure's effect (only applied once the next frame callback drains the
|
||||
/// task channel -- `IrisViewPeer::drain_tasks`) is visible again quickly
|
||||
/// when a later step in the same phase needs to read state back.
|
||||
const ANIM_STEP_MS: u64 = 16;
|
||||
|
||||
const FIXTURE_JSONL: &str = include_str!("../../../app/bench-fixture/assets/transcript.jsonl");
|
||||
|
||||
pub struct BenchClient {
|
||||
ui_state: AndroidUiState,
|
||||
content: WeakWidget<WidgetPtr>,
|
||||
report_display: WeakWidget<TextEdit>,
|
||||
/// The top button row, in a `WidgetPtr` slot rather than added
|
||||
/// directly (like `content`) so `on_insets_changed` can swap in a
|
||||
/// version padded for the status bar once insets are known -- RUST.md's
|
||||
/// P0 box, "the status-bar inset is not applied," found the row sitting
|
||||
/// directly under it because nothing here read `insets().top` at all.
|
||||
top_bar: WeakWidget<WidgetPtr>,
|
||||
screen: Option<transcript_ui::TranscriptScreen>,
|
||||
items: Vec<TranscriptItem>,
|
||||
/// The events not yet streamed -- consumed by `start_benchmark`'s own
|
||||
/// clone, kept here only as the source a second run would need (the
|
||||
/// button can be pressed more than once; `running` just stops overlap,
|
||||
/// not repeat).
|
||||
stream_tail: Vec<SeqEvent>,
|
||||
platform: Option<Arc<PlatformHandle>>,
|
||||
last_report: Option<String>,
|
||||
running: bool,
|
||||
/// The keyboard phase's own confirmation channel -- updated from
|
||||
/// `on_insets_changed` (the platform's own answer for whether the IME
|
||||
/// is actually visible, per `WindowInsets::ime_bottom`), read from the
|
||||
/// benchmark's spawned task via the shared `Arc<Mutex<_>>` rather than
|
||||
/// `ctx.update`, since neither side needs the widget tree for this.
|
||||
ime_state: Arc<Mutex<ImeState>>,
|
||||
/// Edge-triggers the keyboard diagnostics capture below -- set on the
|
||||
/// first `on_insets_changed` where `ime_bottom > 0.0`, cleared on the
|
||||
/// first where it is not, so opening the keyboard fires this once
|
||||
/// rather than on every insets update while it stays open (a rotation
|
||||
/// or a status-bar change with the keyboard already up would otherwise
|
||||
/// re-fire it).
|
||||
keyboard_was_visible: bool,
|
||||
/// The status-bar inset `top_bar` was last padded by -- see
|
||||
/// `on_insets_changed`'s own comment for why this guards the rebuild.
|
||||
last_top_pad: f32,
|
||||
}
|
||||
|
||||
/// See `BenchClient::ime_state`'s doc. `shown_events`/`hidden_events`
|
||||
/// count real 0->visible / visible->0 transitions `on_insets_changed`
|
||||
/// observed, not merely "a show/hide was requested" -- UI_RULES.md: never
|
||||
/// present an inferred value as a measured one. `run_keyboard_phase` reads
|
||||
/// the counters before and after asking for a toggle and calls it
|
||||
/// confirmed only if the count moved.
|
||||
#[derive(Default)]
|
||||
struct ImeState {
|
||||
visible: bool,
|
||||
shown_events: u32,
|
||||
hidden_events: u32,
|
||||
}
|
||||
|
||||
impl HasAndroidUiState for BenchClient {
|
||||
fn android_state(&self) -> &AndroidUiState {
|
||||
&self.ui_state
|
||||
}
|
||||
fn android_state_mut(&mut self) -> &mut AndroidUiState {
|
||||
&mut self.ui_state
|
||||
}
|
||||
}
|
||||
|
||||
/// Parses the fixture once: `serde_json::Value`s for the backlog
|
||||
/// (`fold_page` takes a page of raw wire JSON, same as a real
|
||||
/// `/transcript` response) and folded `SeqEvent`s for the tail (`fold_event`
|
||||
/// takes one live wire event at a time, same as a real SSE frame).
|
||||
fn parse_fixture() -> (Vec<serde_json::Value>, Vec<SeqEvent>) {
|
||||
let lines: Vec<&str> = FIXTURE_JSONL
|
||||
.lines()
|
||||
.filter(|line| !line.trim().is_empty())
|
||||
.collect();
|
||||
let mut backlog = Vec::with_capacity(BACKLOG_COUNT.min(lines.len()));
|
||||
let mut stream_tail = Vec::new();
|
||||
for (i, line) in lines.iter().enumerate() {
|
||||
let value: serde_json::Value =
|
||||
serde_json::from_str(line).expect("bench fixture is generated JSON, always valid");
|
||||
if i < BACKLOG_COUNT {
|
||||
backlog.push(value);
|
||||
} else {
|
||||
let event: SeqEvent = serde_json::from_value(value)
|
||||
.expect("bench fixture event matches event-model's SeqEvent");
|
||||
stream_tail.push(event);
|
||||
}
|
||||
}
|
||||
(backlog, stream_tail)
|
||||
}
|
||||
|
||||
fn placeholder<Rsc: HasEvents>(rsc: &mut Rsc, message: &str) -> StrongWidget {
|
||||
wtext(message.to_string())
|
||||
.color(Color::WHITE)
|
||||
.wrap(true)
|
||||
.pad(16)
|
||||
.add_strong(rsc)
|
||||
.any()
|
||||
}
|
||||
|
||||
/// `getrusage(RUSAGE_SELF)`'s user+system time, in ms -- `None` only if
|
||||
/// the syscall itself fails, which UI_RULES.md's "never present an
|
||||
/// inferred value as a measured one" says to keep apart from a real (and
|
||||
/// here, impossible) zero.
|
||||
fn process_cpu_ms() -> Option<u64> {
|
||||
// SAFETY: `rusage` is a plain-old-data struct `getrusage` fully
|
||||
// initialises on success; on failure it is never read.
|
||||
unsafe {
|
||||
let mut usage: libc::rusage = std::mem::zeroed();
|
||||
if libc::getrusage(libc::RUSAGE_SELF, &mut usage) != 0 {
|
||||
return None;
|
||||
}
|
||||
let user_ms = usage.ru_utime.tv_sec as u64 * 1000 + usage.ru_utime.tv_usec as u64 / 1000;
|
||||
let sys_ms = usage.ru_stime.tv_sec as u64 * 1000 + usage.ru_stime.tv_usec as u64 / 1000;
|
||||
Some(user_ms + sys_ms)
|
||||
}
|
||||
}
|
||||
|
||||
/// `VmHWM` from `/proc/self/status` -- the process's peak RSS since it
|
||||
/// started, in kB. Same source `BenchRun.kt`'s `peakRssLine` reads, so the
|
||||
/// two reports' numbers mean the same thing.
|
||||
fn peak_rss_kb() -> Option<u64> {
|
||||
std::fs::read_to_string("/proc/self/status")
|
||||
.ok()?
|
||||
.lines()
|
||||
.find_map(|line| line.strip_prefix("VmHWM:"))
|
||||
.and_then(|rest| rest.trim().strip_suffix("kB"))
|
||||
.and_then(|n| n.trim().parse().ok())
|
||||
}
|
||||
|
||||
fn battery_line(samples: &[i32]) -> String {
|
||||
if samples.is_empty() {
|
||||
return " battery current: unavailable on this device".to_string();
|
||||
}
|
||||
let mean = samples.iter().map(|&v| v as i64).sum::<i64>() / samples.len() as i64;
|
||||
let min = samples.iter().min().unwrap();
|
||||
let max = samples.iter().max().unwrap();
|
||||
format!(
|
||||
" battery current: mean {mean}\u{b5}A over {} samples (min {min}, max {max})",
|
||||
samples.len()
|
||||
)
|
||||
}
|
||||
|
||||
impl AndroidAppState for BenchClient {
|
||||
fn new(mut ui_state: AndroidUiState, rsc: &mut AndroidRsc<Self>) -> Self {
|
||||
let content = WidgetPtr::new().add(rsc);
|
||||
let loading = placeholder(rsc, "Loading fixture...");
|
||||
content(rsc).set(loading);
|
||||
|
||||
let report_display = wtext("")
|
||||
.editable(EditMode::MultiLine)
|
||||
.text_align(Align::LEFT)
|
||||
.wrap(true)
|
||||
.size(14)
|
||||
.color(Color::WHITE)
|
||||
.attr::<Selectable>(())
|
||||
.label("Benchmark report")
|
||||
.add(rsc);
|
||||
|
||||
let top_bar = WidgetPtr::new().add(rsc);
|
||||
let controls = bench_controls(rsc, 0.0);
|
||||
top_bar(rsc).set(controls);
|
||||
let tree = (
|
||||
top_bar,
|
||||
content.height(rest(2)),
|
||||
report_display.height(rest(1)).pad(dp(8)),
|
||||
)
|
||||
.span(Dir::DOWN)
|
||||
.add_strong(rsc)
|
||||
.any();
|
||||
ui_state.set_root(tree);
|
||||
|
||||
// Startup log line (RUST.md's P0 box, "log once at startup ... the
|
||||
// number of font families found, the default family resolved"):
|
||||
// what font discovery actually found on this device, before
|
||||
// anything is drawn.
|
||||
let font = rsc.ui.text.font_diagnostics();
|
||||
log::info!(
|
||||
"iris fonts: {} families found, default={:?} mono={:?}, resolved regular={:?} \
|
||||
bold={:?} italic={:?} mono={:?}",
|
||||
font.families_found,
|
||||
font.default_family,
|
||||
font.default_mono_family,
|
||||
font.regular_resolved,
|
||||
font.bold_resolved,
|
||||
font.italic_resolved,
|
||||
font.mono_resolved,
|
||||
);
|
||||
|
||||
let mut client = Self {
|
||||
ui_state,
|
||||
content,
|
||||
report_display,
|
||||
top_bar,
|
||||
screen: None,
|
||||
items: Vec::new(),
|
||||
stream_tail: Vec::new(),
|
||||
platform: None,
|
||||
last_report: None,
|
||||
running: false,
|
||||
ime_state: Arc::new(Mutex::new(ImeState::default())),
|
||||
keyboard_was_visible: false,
|
||||
last_top_pad: 0.0,
|
||||
};
|
||||
|
||||
let (backlog, stream_tail) = parse_fixture();
|
||||
client.stream_tail = stream_tail;
|
||||
match fold_page(&backlog) {
|
||||
Ok(items) => {
|
||||
client.items = items;
|
||||
client.rebuild_transcript(rsc);
|
||||
}
|
||||
Err(message) => {
|
||||
client.show_message(rsc, &format!("Couldn't fold the bench fixture: {message}"))
|
||||
}
|
||||
}
|
||||
client
|
||||
}
|
||||
|
||||
fn platform_ready(&mut self, _rsc: &mut AndroidRsc<Self>, vm: JavaVM, view: GlobalRef) {
|
||||
self.platform = Some(Arc::new(PlatformHandle::new(vm, view)));
|
||||
}
|
||||
|
||||
fn back_pressed(&mut self, _rsc: &mut AndroidRsc<Self>, _render: &mut UiRenderState) -> bool {
|
||||
false
|
||||
}
|
||||
|
||||
/// Pads the top button row by the status-bar inset -- see `top_bar`'s
|
||||
/// field comment. Rebuilds the row rather than mutating a stored
|
||||
/// `Padding` in place, since nothing here holds a handle to one --
|
||||
/// but **only when `insets.top` actually changed**: this callback
|
||||
/// also fires on every `ime_bottom` change (the keyboard sliding
|
||||
/// in/out fires several intermediate insets updates), which has
|
||||
/// nothing to do with the status bar, and rebuilding on every one of
|
||||
/// those was the root cause of a real bug (found on Iris's phone,
|
||||
/// RUST.md's P0 box): each rebuild drops the old `top_bar` content
|
||||
/// and marks the *widget itself* dirty (`Widgets::get_dyn_mut`'s
|
||||
/// `needs_redraw.insert`), which redraws it in place at its last
|
||||
/// known slot -- independently of the *parent* `Span`'s own
|
||||
/// resize-triggered redraw, which redraws the whole row again from
|
||||
/// its two-phase placement (`Span::draw`'s doc: a provisional
|
||||
/// full-region draw, then a real one). A `.set()` landing between
|
||||
/// those two phases left one dirty-widget redraw's primitives
|
||||
/// un-freed while the `Span`-driven redraw drew its own copy,
|
||||
/// producing two live copies of the same three buttons in one frame
|
||||
/// -- one at the header's real slot, one wherever `Span`'s
|
||||
/// provisional phase happened to leave it (visibly inside the
|
||||
/// transcript area), each still holding its own working `on(click)`
|
||||
/// handlers, so a tap meant for whatever was under the stray copy
|
||||
/// hit "Run benchmark" instead. Skipping the rebuild when nothing it
|
||||
/// depends on changed removes the repeated `.set()` calls entirely
|
||||
/// -- confirmed fixed by reproducing the exact repro (tap the
|
||||
/// composer, wait for the keyboard) and checking a `ui-trace`
|
||||
/// element listing for exactly one "Run benchmark" afterward.
|
||||
///
|
||||
/// Also two things downstream of the same `ime_bottom` transition:
|
||||
/// **the keyboard phase's own confirmation signal** (`ime_state`'s
|
||||
/// doc -- the platform's own answer for whether the IME actually
|
||||
/// opened or closed, rather than assumed from having called
|
||||
/// `show_ime`/`hide_ime`), and **the trigger for the keyboard
|
||||
/// diagnostics capture** (RUST.md's P0 box): the IME resizing the
|
||||
/// surface is exactly the case a previous commit found wiped text,
|
||||
/// and Iris needs a way to get a report off the phone even if that
|
||||
/// (or some other keyboard-triggered regression) is still happening
|
||||
/// on the build she is holding -- `capture_keyboard_diagnostics`
|
||||
/// below fires ~500ms after the keyboard becomes visible, once per
|
||||
/// keyboard opening, and shows its report in a plain overlay view
|
||||
/// that draws independently of whatever iris itself is doing.
|
||||
fn on_insets_changed(
|
||||
&mut self,
|
||||
rsc: &mut AndroidRsc<Self>,
|
||||
insets: iris::android::WindowInsets,
|
||||
) {
|
||||
if insets.top != self.last_top_pad {
|
||||
self.last_top_pad = insets.top;
|
||||
let controls = bench_controls(rsc, insets.top);
|
||||
(self.top_bar)(rsc).set(controls);
|
||||
}
|
||||
|
||||
let ime_visible = insets.ime_bottom > 0.0;
|
||||
|
||||
let mut ime = self.ime_state.lock().unwrap();
|
||||
if ime_visible && !ime.visible {
|
||||
ime.shown_events += 1;
|
||||
}
|
||||
if !ime_visible && ime.visible {
|
||||
ime.hidden_events += 1;
|
||||
}
|
||||
ime.visible = ime_visible;
|
||||
drop(ime);
|
||||
|
||||
if ime_visible && !self.keyboard_was_visible {
|
||||
self.keyboard_was_visible = true;
|
||||
let redraw = rsc.tasks.redraw_handle();
|
||||
rsc.spawn_task(async move |mut ctx| {
|
||||
tokio::time::sleep(Duration::from_millis(KEYBOARD_DIAGNOSTICS_DELAY_MS)).await;
|
||||
ctx.update(|state: &mut BenchClient, rsc| {
|
||||
state.capture_keyboard_diagnostics(rsc);
|
||||
});
|
||||
redraw.request_redraw();
|
||||
});
|
||||
} else if !ime_visible {
|
||||
self.keyboard_was_visible = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// How long to wait after the keyboard becomes visible before capturing
|
||||
/// diagnostics -- long enough that the resize, the reported wipe (if it is
|
||||
/// still happening) and a couple of frames have all had time to land, per
|
||||
/// AGENTS.md's "so that operations that finish in milliseconds have states
|
||||
/// on the way that nothing can observe" reasoning applied the other way:
|
||||
/// this wants to observe the state *after* the transition settles, not
|
||||
/// mid-flight.
|
||||
const KEYBOARD_DIAGNOSTICS_DELAY_MS: u64 = 500;
|
||||
|
||||
type Rsc = AndroidRsc<BenchClient>;
|
||||
|
||||
/// The header row's own backdrop -- see `bench_controls`'s doc comment on
|
||||
/// why it needs one at all. A dark neutral rather than pure black
|
||||
/// (`android::render::CLEAR_COLOR`) so the row reads as a distinct panel
|
||||
/// instead of a hole in the background the buttons happen to float in.
|
||||
const HEADER_SURFACE: UiColor = UiColor::new(28, 28, 34, 255);
|
||||
|
||||
/// `top_pad` is the status-bar inset in physical pixels (0.0 until
|
||||
/// `on_insets_changed` has run once) -- folded in here, rather than
|
||||
/// exposing the unadded builder for a caller to `.pad()` itself, because
|
||||
/// naming that builder's type at each call site is more machinery than a
|
||||
/// top-of-screen padding number is worth.
|
||||
///
|
||||
/// **Backed by an opaque rect the full size of the row, not just the three
|
||||
/// buttons.** Iris's phone report (docs/RUST.md's P0 box, screenshots on
|
||||
/// build a9232ac): "the header buttons have nothing behind them and
|
||||
/// overlap the transcript text" -- before this, only each button's own
|
||||
/// `rect(...)` painted anything, so the gaps between and around them (and
|
||||
/// the status-bar strip above them) showed whatever was one layer back
|
||||
/// (`CLEAR_COLOR`, black), and the row's true height was three
|
||||
/// physical-pixel-sized (`abs`, not `dp`) button boxes rather than the
|
||||
/// density-correct size the transcript below was already using post-P0 --
|
||||
/// exactly what reads as "overlap" once the two disagree. Fixed two ways
|
||||
/// together: a `HEADER_SURFACE` rect stacked behind the whole row (this
|
||||
/// function), and every size below moved from a bare number (physical
|
||||
/// pixels) to `dp(...)` (IRIS_TODO.md's density-independent length unit),
|
||||
/// so the row's reserved height in the outer `Span::DOWN`
|
||||
/// (`AndroidAppState::new`) matches what is actually painted.
|
||||
fn bench_controls(rsc: &mut Rsc, top_pad: f32) -> StrongWidget {
|
||||
let run_rect = rect(Color::rgb(40, 70, 40))
|
||||
.on(
|
||||
CursorSense::click(),
|
||||
|ctx: EventIdCtx<'_, Rsc, _, _>, rsc: &mut Rsc| {
|
||||
ctx.state.start_benchmark(rsc);
|
||||
},
|
||||
)
|
||||
.label("Run benchmark");
|
||||
let run = (
|
||||
run_rect,
|
||||
wtext("Run benchmark").size(18).text_align(Align::CENTER),
|
||||
)
|
||||
.stack()
|
||||
.pad(dp(8))
|
||||
.add(rsc);
|
||||
|
||||
let copy_rect = rect(Color::rgb(50, 50, 60))
|
||||
.on(
|
||||
CursorSense::click(),
|
||||
|ctx: EventIdCtx<'_, Rsc, _, _>, _rsc: &mut Rsc| {
|
||||
ctx.state.copy_report();
|
||||
},
|
||||
)
|
||||
.label("Copy report");
|
||||
let copy = (
|
||||
copy_rect,
|
||||
wtext("Copy report").size(18).text_align(Align::CENTER),
|
||||
)
|
||||
.stack()
|
||||
.pad(dp(8))
|
||||
.add(rsc);
|
||||
|
||||
let diag_rect = rect(Color::rgb(60, 45, 70))
|
||||
.on(
|
||||
CursorSense::click(),
|
||||
|ctx: EventIdCtx<'_, Rsc, _, _>, rsc: &mut Rsc| {
|
||||
ctx.state.show_diagnostics(rsc);
|
||||
},
|
||||
)
|
||||
.label("Diagnostics");
|
||||
let diagnostics = (
|
||||
diag_rect,
|
||||
wtext("Diagnostics").size(18).text_align(Align::CENTER),
|
||||
)
|
||||
.stack()
|
||||
.pad(dp(8))
|
||||
.add(rsc);
|
||||
|
||||
let buttons = (run, copy, diagnostics).span(Dir::RIGHT).add(rsc);
|
||||
|
||||
(rect(HEADER_SURFACE), buttons)
|
||||
.stack()
|
||||
.height(dp(56))
|
||||
.pad(Padding::top(top_pad))
|
||||
.add_strong(rsc)
|
||||
.any()
|
||||
}
|
||||
|
||||
impl BenchClient {
|
||||
fn show_message(&mut self, rsc: &mut Rsc, message: &str) {
|
||||
let widget = placeholder(rsc, message);
|
||||
(self.content)(rsc).set(widget);
|
||||
self.screen = None;
|
||||
}
|
||||
|
||||
fn rebuild_transcript(&mut self, rsc: &mut Rsc) {
|
||||
let rows = group_tool_runs(&self.items);
|
||||
let (screen, tree) = transcript_ui::build_tree(rsc, rows);
|
||||
(self.content)(rsc).set(tree);
|
||||
self.screen = Some(screen);
|
||||
}
|
||||
|
||||
/// RUST.md's P0 box: "a named `Diagnostics` control ... with 'copy this
|
||||
/// and send it to Iris'." Fills `report_display` (the same TextEdit the
|
||||
/// benchmark report uses) rather than a separate widget, so the
|
||||
/// existing "Copy report" button and clipboard path work on whichever
|
||||
/// text is currently shown -- `last_report` is what `copy_report` reads,
|
||||
/// so it's set here too rather than adding a second copy path.
|
||||
fn show_diagnostics(&mut self, rsc: &mut Rsc) {
|
||||
let font = rsc.ui.text.font_diagnostics();
|
||||
let frame_report = match self.android_state().frame_report.report() {
|
||||
Some(stats) => format!("{stats}"),
|
||||
None => "no frames recorded yet".to_string(),
|
||||
};
|
||||
let report = match &self.android_state().renderer {
|
||||
Some(renderer) => renderer.diagnostics_report(&font, &frame_report),
|
||||
None => "iris diagnostics: no renderer yet (no surface)".to_string(),
|
||||
};
|
||||
self.report_display.edit(rsc).set(&report);
|
||||
self.last_report = Some(report);
|
||||
}
|
||||
|
||||
/// The keyboard's own diagnostics capture -- see `on_insets_changed`'s
|
||||
/// doc comment. Reuses `show_diagnostics`'s exact report (so it is the
|
||||
/// same text the on-screen `Diagnostics` button produces, plus the
|
||||
/// per-frame log `FrameReport` already keeps around the resize --
|
||||
/// `frame_report.report()` above covers "the frames around the
|
||||
/// resize" without a second accounting mechanism), then does three
|
||||
/// things the button does not: logs it (so a `logcat` pull gets it
|
||||
/// even if nothing on screen does), copies it to the clipboard
|
||||
/// unprompted, and shows it in the shell's plain overlay view, which
|
||||
/// draws independently of iris's own renderer -- the whole point,
|
||||
/// since the renderer is exactly what might be in the wiped state
|
||||
/// this exists to report on.
|
||||
fn capture_keyboard_diagnostics(&mut self, rsc: &mut Rsc) {
|
||||
self.show_diagnostics(rsc);
|
||||
let Some(report) = self.last_report.clone() else {
|
||||
return;
|
||||
};
|
||||
log::info!("iris keyboard diagnostics:\n{report}");
|
||||
let Some(platform) = &self.platform else {
|
||||
log::info!("iris keyboard diagnostics: no platform handle, can't reach the shell");
|
||||
return;
|
||||
};
|
||||
if platform.copy_to_clipboard("iris keyboard diagnostics", &report) {
|
||||
log::info!("iris keyboard diagnostics: copied to clipboard");
|
||||
} else {
|
||||
log::info!("iris keyboard diagnostics: clipboard copy failed");
|
||||
}
|
||||
platform.show_diagnostics_overlay(&report);
|
||||
}
|
||||
|
||||
fn copy_report(&mut self) {
|
||||
let Some(report) = &self.last_report else {
|
||||
log::info!("iris bench report: nothing to copy -- run the benchmark first");
|
||||
return;
|
||||
};
|
||||
let Some(platform) = &self.platform else {
|
||||
log::info!("iris bench report: no platform handle, can't reach the clipboard");
|
||||
return;
|
||||
};
|
||||
if platform.copy_to_clipboard("iris bench report", report) {
|
||||
log::info!("iris bench report: copied to clipboard");
|
||||
} else {
|
||||
log::info!("iris bench report: clipboard copy failed");
|
||||
}
|
||||
}
|
||||
|
||||
/// RUST.md's "Benchmark v2": fling, then stream (unchanged from v1),
|
||||
/// then type, then keyboard, then the report -- run in-process for the
|
||||
/// same reason `BenchRun.kt`'s own doc gives (no usable system tracing
|
||||
/// on a real phone, no agent that can drive one).
|
||||
fn start_benchmark(&mut self, rsc: &mut Rsc) {
|
||||
if self.running {
|
||||
log::info!("iris bench report: already running");
|
||||
return;
|
||||
}
|
||||
self.running = true;
|
||||
self.android_state_mut().frame_report.reset();
|
||||
self.report_display.edit(rsc).set("Running benchmark...");
|
||||
|
||||
let redraw = rsc.tasks.redraw_handle();
|
||||
let platform = self.platform.clone();
|
||||
let stream_tail = self.stream_tail.clone();
|
||||
let ime_state = self.ime_state.clone();
|
||||
let refresh_hz = platform
|
||||
.as_ref()
|
||||
.and_then(|p| p.refresh_rate_hz())
|
||||
.unwrap_or(60.0);
|
||||
let cpu_start = process_cpu_ms();
|
||||
let run_started_at = Instant::now();
|
||||
|
||||
rsc.spawn_task(async move |mut ctx| {
|
||||
// The battery sampler runs for the whole run, once a second,
|
||||
// the same cadence `BatterySampler` uses on the Compose side
|
||||
// -- via its own JNI-attached thread, not `ctx.update`, since
|
||||
// a sample needs no widget-tree access.
|
||||
let sampler_done = Arc::new(AtomicBool::new(false));
|
||||
let samples = Arc::new(Mutex::new(Vec::<i32>::new()));
|
||||
let sampler = platform.clone().map(|platform| {
|
||||
let done = sampler_done.clone();
|
||||
let samples = samples.clone();
|
||||
tokio::spawn(async move {
|
||||
while !done.load(Ordering::Relaxed) {
|
||||
if let Some(value) = platform.battery_current_ua() {
|
||||
samples.lock().unwrap().push(value);
|
||||
}
|
||||
tokio::time::sleep(Duration::from_secs(1)).await;
|
||||
}
|
||||
})
|
||||
});
|
||||
|
||||
let travel = run_fling_phase(&mut ctx, &redraw).await;
|
||||
let (sent, total) = run_stream_phase(&mut ctx, &redraw, stream_tail).await;
|
||||
run_type_phase(&mut ctx, &redraw, &platform).await;
|
||||
let keyboard = run_keyboard_phase(&mut ctx, &platform, &ime_state).await;
|
||||
|
||||
sampler_done.store(true, Ordering::Relaxed);
|
||||
if let Some(sampler) = sampler {
|
||||
let _ = sampler.await;
|
||||
}
|
||||
let battery = battery_line(&samples.lock().unwrap());
|
||||
let cpu_line = match (cpu_start, process_cpu_ms()) {
|
||||
(Some(start), Some(end)) => {
|
||||
format!(
|
||||
" process CPU time over this run: {}ms",
|
||||
end.saturating_sub(start)
|
||||
)
|
||||
}
|
||||
_ => " process CPU time over this run: unavailable".to_string(),
|
||||
};
|
||||
let rss_line = match peak_rss_kb() {
|
||||
Some(kb) => format!(" peak RSS: {kb}kB"),
|
||||
None => " peak RSS: unavailable (/proc/self/status unreadable)".to_string(),
|
||||
};
|
||||
let total_seconds = run_started_at.elapsed().as_secs_f64();
|
||||
|
||||
ctx.update(move |state: &mut BenchClient, rsc| {
|
||||
state.running = false;
|
||||
let now = Instant::now();
|
||||
let phase_lines: String = state
|
||||
.android_state()
|
||||
.frame_report
|
||||
.phase_stats(now, refresh_hz)
|
||||
.iter()
|
||||
.map(|p| format!("{p}\n"))
|
||||
.collect();
|
||||
let per_phase = if phase_lines.is_empty() {
|
||||
String::new()
|
||||
} else {
|
||||
format!("per phase:\n{phase_lines}\n")
|
||||
};
|
||||
let frames_block = match state.android_state().frame_report.report() {
|
||||
Some(stats) => {
|
||||
let (late, late_pct) =
|
||||
state.android_state().frame_report.late_at_hz(refresh_hz);
|
||||
format!(
|
||||
"frames:\n {} frames over {:.1}s at {:.0}Hz ({:.1}ms budget)\n \
|
||||
late: {late} ({late_pct:.1}%)\n total p50 {:.1}ms p90 {:.1}ms \
|
||||
p99 {:.1}ms\n worst {:.1}ms\n cpu_p50 {:.1}ms gpu_wait_p50 {:.1}ms",
|
||||
stats.total_frames,
|
||||
total_seconds,
|
||||
refresh_hz,
|
||||
1000.0 / refresh_hz as f64,
|
||||
stats.p50.as_secs_f64() * 1000.0,
|
||||
stats.p90.as_secs_f64() * 1000.0,
|
||||
stats.p99.as_secs_f64() * 1000.0,
|
||||
stats.worst.as_secs_f64() * 1000.0,
|
||||
stats.cpu_p50.as_secs_f64() * 1000.0,
|
||||
stats.gpu_wait_p50.as_secs_f64() * 1000.0,
|
||||
)
|
||||
}
|
||||
None => "frames:\n no frames recorded".to_string(),
|
||||
};
|
||||
let scroll_line = format!(
|
||||
" scroll: {LEGACY_CYCLES} cycles ({} swipes, legacy tween), streamed \
|
||||
{sent}/{total} fixture events",
|
||||
LEGACY_CYCLES * 4
|
||||
);
|
||||
let fling_line = format!(
|
||||
" fling: {FLING_COUNT} flings out + {FLING_COUNT} back at \
|
||||
{FLING_VELOCITY_PX_S}px/s, travel {travel}"
|
||||
);
|
||||
let type_line = format!(
|
||||
" type: {} characters inserted then deleted, one per {TYPE_CHAR_MS}ms",
|
||||
TYPE_TEXT.chars().count()
|
||||
);
|
||||
let report = format!(
|
||||
"iris bench report\n{per_phase}{frames_block}\n\nbench:\n{fling_line}\n\
|
||||
{scroll_line}\n{type_line}\n{keyboard}\n{cpu_line}\n{rss_line}\n{battery}"
|
||||
);
|
||||
log::info!("iris bench report: {report}");
|
||||
state.report_display.edit(rsc).set(&report);
|
||||
state.last_report = Some(report);
|
||||
});
|
||||
redraw.request_redraw();
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
/// Runs `f` against the real `BenchClient`/`Rsc` on the main thread (the
|
||||
/// same `ctx.update` every other mutation here goes through) and returns
|
||||
/// its result to the caller's async task -- `ctx.update` alone has no way
|
||||
/// to hand a value back, since the closure only actually runs once the
|
||||
/// next frame callback drains `IrisViewPeer`'s task channel
|
||||
/// (`drain_tasks`). **Must call `redraw.request_redraw()` itself, right
|
||||
/// after enqueueing** -- `ctx.update` only ever pushes onto a channel;
|
||||
/// nothing drains it until something schedules the frame callback that
|
||||
/// calls `drain_tasks`, and a caller relying on some *earlier*,
|
||||
/// already-in-flight `request_redraw()` to cover a *later* `ctx.update`
|
||||
/// deadlocks the moment that earlier callback has already fired and
|
||||
/// drained everything queued before this call existed. Cost a real hang
|
||||
/// in this file's first version of the fling phase: every loop iteration
|
||||
/// after the first sat forever with nothing scheduled to drain it.
|
||||
/// Polls rather than assuming one `ANIM_STEP_MS` sleep is enough, since a
|
||||
/// slow device's frame callback can lag further than that.
|
||||
async fn read_from_state<T, F>(
|
||||
ctx: &mut iris::task::TaskCtx<Rsc>,
|
||||
redraw: &Arc<dyn RequestRedraw>,
|
||||
f: F,
|
||||
) -> T
|
||||
where
|
||||
T: Send + 'static,
|
||||
F: FnOnce(&mut BenchClient, &mut Rsc) -> T + Send + 'static,
|
||||
{
|
||||
let (tx, rx) = std::sync::mpsc::channel();
|
||||
ctx.update(move |state: &mut BenchClient, rsc| {
|
||||
let _ = tx.send(f(state, rsc));
|
||||
});
|
||||
redraw.request_redraw();
|
||||
loop {
|
||||
if let Ok(value) = rx.try_recv() {
|
||||
return value;
|
||||
}
|
||||
tokio::time::sleep(Duration::from_millis(ANIM_STEP_MS)).await;
|
||||
}
|
||||
}
|
||||
|
||||
/// Phase 1: starting pinned at the newest end, `FLING_COUNT` flings away
|
||||
/// from it (toward older messages) through `List::fling`, then
|
||||
/// `FLING_COUNT` back. Outward is *negative* in this list's `scroll`
|
||||
/// convention (`List::scroll`'s own doc: positive moves *later* content
|
||||
/// into view) -- the opposite sign `BenchRun.kt`'s `runFlingPhase` uses,
|
||||
/// since `TranscriptList`'s `LazyColumn` and this list define "positive"
|
||||
/// the other way around; the two apps' *travel* is still directly
|
||||
/// comparable because both report it as a row index + pixel offset, not a
|
||||
/// signed distance.
|
||||
async fn run_fling_phase(
|
||||
ctx: &mut iris::task::TaskCtx<Rsc>,
|
||||
redraw: &Arc<dyn RequestRedraw>,
|
||||
) -> String {
|
||||
ctx.update(|state: &mut BenchClient, _rsc| {
|
||||
state.android_state_mut().frame_report.mark_phase("fling");
|
||||
});
|
||||
ctx.update(|state: &mut BenchClient, rsc| {
|
||||
if let Some(screen) = &state.screen {
|
||||
(screen.list)(rsc).jump_to_end();
|
||||
}
|
||||
});
|
||||
redraw.request_redraw();
|
||||
// Lets the next frame's `repair_anchor` resolve `jump_to_end`'s
|
||||
// `anchor = None` into a real slot before `start` is read.
|
||||
tokio::time::sleep(Duration::from_millis(ANIM_STEP_MS * 2)).await;
|
||||
let start = read_anchor_position(ctx, redraw).await;
|
||||
|
||||
for _ in 0..FLING_COUNT {
|
||||
ctx.update(|state: &mut BenchClient, rsc| {
|
||||
if let Some(screen) = &state.screen {
|
||||
(screen.list)(rsc).fling(-FLING_VELOCITY_PX_S);
|
||||
}
|
||||
});
|
||||
redraw.request_redraw();
|
||||
wait_for_fling_settle(ctx, redraw).await;
|
||||
tokio::time::sleep(Duration::from_millis(FLING_PAUSE_MS)).await;
|
||||
}
|
||||
let outward = read_anchor_position(ctx, redraw).await;
|
||||
|
||||
for _ in 0..FLING_COUNT {
|
||||
ctx.update(|state: &mut BenchClient, rsc| {
|
||||
if let Some(screen) = &state.screen {
|
||||
(screen.list)(rsc).fling(FLING_VELOCITY_PX_S);
|
||||
}
|
||||
});
|
||||
redraw.request_redraw();
|
||||
wait_for_fling_settle(ctx, redraw).await;
|
||||
tokio::time::sleep(Duration::from_millis(FLING_PAUSE_MS)).await;
|
||||
}
|
||||
let end = read_anchor_position(ctx, redraw).await;
|
||||
|
||||
format!("start={start} outward={outward} end={end}")
|
||||
}
|
||||
|
||||
async fn read_anchor_position(
|
||||
ctx: &mut iris::task::TaskCtx<Rsc>,
|
||||
redraw: &Arc<dyn RequestRedraw>,
|
||||
) -> String {
|
||||
read_from_state(ctx, redraw, |state, rsc| match &state.screen {
|
||||
Some(screen) => (screen.list)(rsc).anchor_position_display(),
|
||||
None => "idx=none".to_string(),
|
||||
})
|
||||
.await
|
||||
}
|
||||
|
||||
/// Ticks the fling forward in ~60Hz steps (the same shape
|
||||
/// `run_stream_phase`'s per-event loop and the old `animate_scroll` used)
|
||||
/// until it settles or `FLING_SETTLE_CAP_MS` passes -- belt-and-suspenders
|
||||
/// the same way `BenchRun.kt`'s own `waitForSettle` is, since a fling's
|
||||
/// own spline-decided `duration()` already caps how long it can run.
|
||||
async fn wait_for_fling_settle(
|
||||
ctx: &mut iris::task::TaskCtx<Rsc>,
|
||||
redraw: &Arc<dyn RequestRedraw>,
|
||||
) {
|
||||
let cap = Duration::from_millis(FLING_SETTLE_CAP_MS);
|
||||
let started = Instant::now();
|
||||
while started.elapsed() < cap {
|
||||
let still_scrolling = read_from_state(ctx, redraw, |state, rsc| match &state.screen {
|
||||
Some(screen) => (screen.list)(rsc).tick_fling(Instant::now()),
|
||||
None => false,
|
||||
})
|
||||
.await;
|
||||
if !still_scrolling {
|
||||
return;
|
||||
}
|
||||
tokio::time::sleep(Duration::from_millis(ANIM_STEP_MS)).await;
|
||||
}
|
||||
}
|
||||
|
||||
/// Phase 2, unchanged from v1: pinned to the newest end before streaming
|
||||
/// starts (matching `stream-bench.sh`'s "Jump to latest" tap), then
|
||||
/// `STREAM_EVENTS_PER_SEC * STREAM_SECONDS` fixture events replayed
|
||||
/// through the real `fold_event`/`TranscriptScreen::apply` path. Returns
|
||||
/// `(sent, total)`.
|
||||
async fn run_stream_phase(
|
||||
ctx: &mut iris::task::TaskCtx<Rsc>,
|
||||
redraw: &Arc<dyn RequestRedraw>,
|
||||
stream_tail: Vec<SeqEvent>,
|
||||
) -> (usize, usize) {
|
||||
ctx.update(|state: &mut BenchClient, _rsc| {
|
||||
state.android_state_mut().frame_report.mark_phase("stream");
|
||||
});
|
||||
ctx.update(|state: &mut BenchClient, rsc| {
|
||||
if let Some(screen) = &state.screen {
|
||||
(screen.list)(rsc).jump_to_end();
|
||||
}
|
||||
});
|
||||
redraw.request_redraw();
|
||||
|
||||
let total = (STREAM_EVENTS_PER_SEC * STREAM_SECONDS) as usize;
|
||||
let mut sent = 0usize;
|
||||
for event in stream_tail.into_iter().take(total) {
|
||||
ctx.update(move |state: &mut BenchClient, rsc| {
|
||||
let old_items = state.items.clone();
|
||||
state.items = fold_event(&state.items, &event);
|
||||
match &state.screen {
|
||||
Some(screen) => screen.apply(rsc, &old_items, &state.items),
|
||||
None => state.rebuild_transcript(rsc),
|
||||
}
|
||||
});
|
||||
redraw.request_redraw();
|
||||
sent += 1;
|
||||
tokio::time::sleep(Duration::from_millis(1000 / STREAM_EVENTS_PER_SEC)).await;
|
||||
}
|
||||
// Lets the last few deltas land and draw before the next phase starts
|
||||
// -- `BenchRun.kt`'s own closing delay.
|
||||
tokio::time::sleep(Duration::from_millis(300)).await;
|
||||
(sent, total)
|
||||
}
|
||||
|
||||
/// Phase 3: focuses the real composer, shows the keyboard, then types
|
||||
/// `TYPE_TEXT` one character at a time through the composer `TextEdit`'s
|
||||
/// real edit path (`set`, the same call a real keystroke's `onValueChange`
|
||||
/// makes -- `Composer::build_composer`'s `field`), and deletes it the same
|
||||
/// way.
|
||||
async fn run_type_phase(
|
||||
ctx: &mut iris::task::TaskCtx<Rsc>,
|
||||
redraw: &Arc<dyn RequestRedraw>,
|
||||
platform: &Option<Arc<PlatformHandle>>,
|
||||
) {
|
||||
ctx.update(|state: &mut BenchClient, _rsc| {
|
||||
state.android_state_mut().frame_report.mark_phase("type");
|
||||
});
|
||||
ctx.update(|state: &mut BenchClient, rsc| {
|
||||
if let Some(screen) = &state.screen {
|
||||
(screen.list)(rsc).jump_to_end();
|
||||
state.set_focus(Some(screen.composer.field));
|
||||
}
|
||||
});
|
||||
redraw.request_redraw();
|
||||
if let Some(p) = platform {
|
||||
p.show_ime();
|
||||
}
|
||||
// Lets focus and the keyboard's opening animation land before typing
|
||||
// starts, so the frames this phase records are the wrap/reflow it is
|
||||
// measuring, not the keyboard opening -- `BenchRun.kt`'s own delay.
|
||||
tokio::time::sleep(Duration::from_millis(300)).await;
|
||||
|
||||
let mut typed = String::new();
|
||||
for ch in TYPE_TEXT.chars() {
|
||||
typed.push(ch);
|
||||
let text = typed.clone();
|
||||
ctx.update(move |state: &mut BenchClient, rsc| {
|
||||
if let Some(screen) = &state.screen {
|
||||
screen.composer.field.edit(rsc).set(&text);
|
||||
}
|
||||
});
|
||||
redraw.request_redraw();
|
||||
tokio::time::sleep(Duration::from_millis(TYPE_CHAR_MS)).await;
|
||||
}
|
||||
tokio::time::sleep(Duration::from_millis(200)).await;
|
||||
while !typed.is_empty() {
|
||||
typed.pop();
|
||||
let text = typed.clone();
|
||||
ctx.update(move |state: &mut BenchClient, rsc| {
|
||||
if let Some(screen) = &state.screen {
|
||||
screen.composer.field.edit(rsc).set(&text);
|
||||
}
|
||||
});
|
||||
redraw.request_redraw();
|
||||
tokio::time::sleep(Duration::from_millis(TYPE_CHAR_MS)).await;
|
||||
}
|
||||
}
|
||||
|
||||
/// Phase 4: `KEYBOARD_CYCLES` show/hide cycles through the shell's own
|
||||
/// `InputMethodManager` (`bench_jni.rs`'s `show_ime`/`hide_ime`), each
|
||||
/// confirmed by `on_insets_changed`'s real `ime_bottom` transition rather
|
||||
/// than assumed from the JNI call having returned -- `ImeState`'s doc.
|
||||
/// "keyboard: could not be shown" if the platform never confirms it even
|
||||
/// once, per UI_RULES.md ("design the unknown/failed state before the
|
||||
/// answer's").
|
||||
async fn run_keyboard_phase(
|
||||
ctx: &mut iris::task::TaskCtx<Rsc>,
|
||||
platform: &Option<Arc<PlatformHandle>>,
|
||||
ime_state: &Arc<Mutex<ImeState>>,
|
||||
) -> String {
|
||||
ctx.update(|state: &mut BenchClient, _rsc| {
|
||||
state
|
||||
.android_state_mut()
|
||||
.frame_report
|
||||
.mark_phase("keyboard");
|
||||
});
|
||||
let mut shown = 0;
|
||||
let mut hidden = 0;
|
||||
for _ in 0..KEYBOARD_CYCLES {
|
||||
let before_shown = ime_state.lock().unwrap().shown_events;
|
||||
if let Some(p) = platform {
|
||||
p.show_ime();
|
||||
}
|
||||
tokio::time::sleep(Duration::from_millis(KEYBOARD_WAIT_MS)).await;
|
||||
if ime_state.lock().unwrap().shown_events > before_shown {
|
||||
shown += 1;
|
||||
}
|
||||
|
||||
let before_hidden = ime_state.lock().unwrap().hidden_events;
|
||||
if let Some(p) = platform {
|
||||
p.hide_ime();
|
||||
}
|
||||
tokio::time::sleep(Duration::from_millis(KEYBOARD_WAIT_MS)).await;
|
||||
if ime_state.lock().unwrap().hidden_events > before_hidden {
|
||||
hidden += 1;
|
||||
}
|
||||
}
|
||||
if shown == 0 {
|
||||
format!(" keyboard: could not be shown ({KEYBOARD_CYCLES} attempts, 0 confirmed visible)")
|
||||
} else {
|
||||
format!(
|
||||
" keyboard: shown {shown}/{KEYBOARD_CYCLES}, hidden {hidden}/{KEYBOARD_CYCLES} \
|
||||
(confirmed via on_insets_changed)"
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::TYPE_TEXT;
|
||||
|
||||
/// `BenchRun.kt`'s own `TYPE_TEXT` is verified `.length == 600`; this
|
||||
/// is the same string, so it has to match exactly or the two apps'
|
||||
/// type phases stop typing the same content -- RUST.md's "Benchmark
|
||||
/// v2" spec is one shared string for both.
|
||||
#[test]
|
||||
fn type_text_is_exactly_600_characters() {
|
||||
assert_eq!(TYPE_TEXT.chars().count(), 600);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,252 @@
|
||||
//! JNI calls the `bench` feature needs that go through the shell's own
|
||||
//! Java side rather than anything `iris`/`android-view` already wraps:
|
||||
//! `BatteryManager.getIntProperty(BATTERY_PROPERTY_CURRENT_NOW)` for the
|
||||
//! per-second battery sample, `ClipboardManager.setPrimaryClip` for the
|
||||
//! "Copy report" control (P0's iris half, docs/RUST.md), and -- added for
|
||||
//! RUST.md's "Benchmark v2" -- `Display.getRefreshRate()` for the phase
|
||||
//! report's real late-frame budget and `InputMethodManager.
|
||||
//! showSoftInput`/`hideSoftInputFromWindow` for the keyboard phase. None
|
||||
//! of these are part of `android_view::context`'s own `Context`/
|
||||
//! `Resources` wrappers (that file's own `// TODO: more methods?`), so
|
||||
//! this calls them directly rather than growing that crate's wrapper for
|
||||
//! calls this crate alone needs.
|
||||
//!
|
||||
//! Holds its own `JavaVM` + `GlobalRef` to the view (handed in through
|
||||
//! [`iris::android::AndroidAppState::platform_ready`]) so it can attach
|
||||
//! whichever thread calls it -- the battery sampler runs on a background
|
||||
//! tokio task, not the UI thread the rest of `IrisViewPeer`'s JNI calls
|
||||
//! run on. `JavaVM::attach_current_thread` is safe to call from a thread
|
||||
//! already attached (the `jni` crate detects it and does not double
|
||||
//! attach), so no caller here needs to know or care which thread it is.
|
||||
|
||||
use android_view::jni::{
|
||||
JNIEnv, JavaVM,
|
||||
objects::{GlobalRef, JObject, JValue},
|
||||
};
|
||||
|
||||
/// `android.os.BatteryManager.BATTERY_PROPERTY_CURRENT_NOW` -- not exposed
|
||||
/// as a constant anywhere reachable without the Android SDK jar, so named
|
||||
/// here with its source rather than left as a bare `2`.
|
||||
const BATTERY_PROPERTY_CURRENT_NOW: i32 = 2;
|
||||
|
||||
pub struct PlatformHandle {
|
||||
vm: JavaVM,
|
||||
view: GlobalRef,
|
||||
}
|
||||
|
||||
impl PlatformHandle {
|
||||
pub fn new(vm: JavaVM, view: GlobalRef) -> Self {
|
||||
Self { vm, view }
|
||||
}
|
||||
|
||||
fn context<'e>(&self, env: &mut JNIEnv<'e>) -> Option<JObject<'e>> {
|
||||
env.call_method(
|
||||
self.view.as_obj(),
|
||||
"getContext",
|
||||
"()Landroid/content/Context;",
|
||||
&[],
|
||||
)
|
||||
.ok()?
|
||||
.l()
|
||||
.ok()
|
||||
}
|
||||
|
||||
fn system_service<'e>(
|
||||
&self,
|
||||
env: &mut JNIEnv<'e>,
|
||||
context: &JObject<'e>,
|
||||
name: &str,
|
||||
) -> Option<JObject<'e>> {
|
||||
let jname = env.new_string(name).ok()?;
|
||||
env.call_method(
|
||||
context,
|
||||
"getSystemService",
|
||||
"(Ljava/lang/String;)Ljava/lang/Object;",
|
||||
&[JValue::Object(jname.as_ref())],
|
||||
)
|
||||
.ok()?
|
||||
.l()
|
||||
.ok()
|
||||
}
|
||||
|
||||
/// One sample of `BATTERY_PROPERTY_CURRENT_NOW`, in microamps. `None`
|
||||
/// on any JNI failure, on a device with no `BatteryManager` service,
|
||||
/// or when the platform itself answers "not supported" -- `0` or
|
||||
/// `Integer.MIN_VALUE` are both documented SDK answers for that, and
|
||||
/// both would read as a real (and wrong) measurement if folded into an
|
||||
/// average rather than named apart. UI_RULES.md: never present an
|
||||
/// inferred value as a measured one.
|
||||
pub fn battery_current_ua(&self) -> Option<i32> {
|
||||
let mut guard = self.vm.attach_current_thread().ok()?;
|
||||
let env: &mut JNIEnv = &mut guard;
|
||||
let context = self.context(env)?;
|
||||
let battery_manager = self.system_service(env, &context, "batterymanager")?;
|
||||
let value = env
|
||||
.call_method(
|
||||
&battery_manager,
|
||||
"getIntProperty",
|
||||
"(I)I",
|
||||
&[JValue::Int(BATTERY_PROPERTY_CURRENT_NOW)],
|
||||
)
|
||||
.ok()?
|
||||
.i()
|
||||
.ok()?;
|
||||
if value == 0 || value == i32::MIN {
|
||||
None
|
||||
} else {
|
||||
Some(value)
|
||||
}
|
||||
}
|
||||
|
||||
/// Puts `text` on the system clipboard through `ClipboardManager` --
|
||||
/// `true` only if the whole JNI chain (service lookup, `ClipData`,
|
||||
/// `setPrimaryClip`) succeeded.
|
||||
pub fn copy_to_clipboard(&self, label: &str, text: &str) -> bool {
|
||||
self.try_copy_to_clipboard(label, text).is_some()
|
||||
}
|
||||
|
||||
fn try_copy_to_clipboard(&self, label: &str, text: &str) -> Option<()> {
|
||||
let mut guard = self.vm.attach_current_thread().ok()?;
|
||||
let env: &mut JNIEnv = &mut guard;
|
||||
let context = self.context(env)?;
|
||||
let clipboard = self.system_service(env, &context, "clipboard")?;
|
||||
let jlabel = env.new_string(label).ok()?;
|
||||
let jtext = env.new_string(text).ok()?;
|
||||
let clip = env
|
||||
.call_static_method(
|
||||
"android/content/ClipData",
|
||||
"newPlainText",
|
||||
"(Ljava/lang/CharSequence;Ljava/lang/CharSequence;)Landroid/content/ClipData;",
|
||||
&[
|
||||
JValue::Object(jlabel.as_ref()),
|
||||
JValue::Object(jtext.as_ref()),
|
||||
],
|
||||
)
|
||||
.ok()?
|
||||
.l()
|
||||
.ok()?;
|
||||
env.call_method(
|
||||
&clipboard,
|
||||
"setPrimaryClip",
|
||||
"(Landroid/content/ClipData;)V",
|
||||
&[JValue::Object(&clip)],
|
||||
)
|
||||
.ok()?;
|
||||
Some(())
|
||||
}
|
||||
|
||||
/// The display's own refresh rate in Hz (`View::getDisplay()` ->
|
||||
/// `Display::getRefreshRate()`), for RUST.md's "Benchmark v2": late
|
||||
/// frames are judged against *this* device's real budget, not an
|
||||
/// assumed 60Hz -- a 90Hz or 120Hz phone would otherwise call frames
|
||||
/// "late" that met their own faster deadline. `None` if the view is
|
||||
/// not yet attached to a window (`getDisplay` returns `null`) or the
|
||||
/// platform reports a non-positive rate, which is not a real answer
|
||||
/// either.
|
||||
pub fn refresh_rate_hz(&self) -> Option<f32> {
|
||||
let mut guard = self.vm.attach_current_thread().ok()?;
|
||||
let env: &mut JNIEnv = &mut guard;
|
||||
let display = env
|
||||
.call_method(
|
||||
self.view.as_obj(),
|
||||
"getDisplay",
|
||||
"()Landroid/view/Display;",
|
||||
&[],
|
||||
)
|
||||
.ok()?
|
||||
.l()
|
||||
.ok()?;
|
||||
if display.is_null() {
|
||||
return None;
|
||||
}
|
||||
let rate = env
|
||||
.call_method(&display, "getRefreshRate", "()F", &[])
|
||||
.ok()?
|
||||
.f()
|
||||
.ok()?;
|
||||
if rate > 0.0 { Some(rate) } else { None }
|
||||
}
|
||||
|
||||
/// `InputMethodManager.showSoftInput(view, 0)` -- the keyboard phase's
|
||||
/// own show, called directly rather than through the focus-driven
|
||||
/// `pending_show_keyboard` path `android/view.rs` uses for a real tap,
|
||||
/// since RUST.md's "Benchmark v2" spec asks for this "through the
|
||||
/// shell's InputMethodManager" independent of focus state. `true` only
|
||||
/// if the platform itself reports the request succeeded -- whether the
|
||||
/// IME actually became visible is confirmed separately, from
|
||||
/// `on_insets_changed`, per UI_RULES.md ("never present an inferred
|
||||
/// value as a measured one").
|
||||
pub fn show_ime(&self) -> bool {
|
||||
self.try_toggle_ime(true).unwrap_or(false)
|
||||
}
|
||||
|
||||
/// `InputMethodManager.hideSoftInputFromWindow(windowToken, 0)`.
|
||||
pub fn hide_ime(&self) -> bool {
|
||||
self.try_toggle_ime(false).unwrap_or(false)
|
||||
}
|
||||
|
||||
fn try_toggle_ime(&self, show: bool) -> Option<bool> {
|
||||
let mut guard = self.vm.attach_current_thread().ok()?;
|
||||
let env: &mut JNIEnv = &mut guard;
|
||||
let context = self.context(env)?;
|
||||
let imm = self.system_service(env, &context, "input_method")?;
|
||||
if show {
|
||||
env.call_method(
|
||||
&imm,
|
||||
"showSoftInput",
|
||||
"(Landroid/view/View;I)Z",
|
||||
&[JValue::Object(self.view.as_obj()), JValue::Int(0)],
|
||||
)
|
||||
.ok()?
|
||||
.z()
|
||||
.ok()
|
||||
} else {
|
||||
let token = env
|
||||
.call_method(
|
||||
self.view.as_obj(),
|
||||
"getWindowToken",
|
||||
"()Landroid/os/IBinder;",
|
||||
&[],
|
||||
)
|
||||
.ok()?
|
||||
.l()
|
||||
.ok()?;
|
||||
env.call_method(
|
||||
&imm,
|
||||
"hideSoftInputFromWindow",
|
||||
"(Landroid/os/IBinder;I)Z",
|
||||
&[JValue::Object(&token), JValue::Int(0)],
|
||||
)
|
||||
.ok()?
|
||||
.z()
|
||||
.ok()
|
||||
}
|
||||
}
|
||||
|
||||
/// Shows `report` in the shell's plain-view diagnostics overlay
|
||||
/// (`IrisView.showDiagnosticsOverlay`) -- a real `TextView` plus Copy
|
||||
/// and Close controls, added over whatever iris itself is drawing
|
||||
/// rather than replacing it (unlike `android::view::show_renderer_error`,
|
||||
/// which exists for the case the renderer can never recover from and
|
||||
/// intentionally never returns). Called from a background task after
|
||||
/// the keyboard-open delay (`bench_client.rs`'s `on_insets_changed`),
|
||||
/// so the Java side hops onto the UI thread itself before touching the
|
||||
/// view tree -- see that method's own comment.
|
||||
pub fn show_diagnostics_overlay(&self, report: &str) -> bool {
|
||||
self.try_show_diagnostics_overlay(report).is_some()
|
||||
}
|
||||
|
||||
fn try_show_diagnostics_overlay(&self, report: &str) -> Option<()> {
|
||||
let mut guard = self.vm.attach_current_thread().ok()?;
|
||||
let env: &mut JNIEnv = &mut guard;
|
||||
let jreport = env.new_string(report).ok()?;
|
||||
env.call_method(
|
||||
self.view.as_obj(),
|
||||
"showDiagnosticsOverlay",
|
||||
"(Ljava/lang/String;)V",
|
||||
&[JValue::Object(jreport.as_ref())],
|
||||
)
|
||||
.ok()?;
|
||||
Some(())
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//! The android-view demo app: iris's `tabs` widget tree (`tabs_ui::build`,
|
||||
//! shared with the winit example) running through
|
||||
//! The android-view demo app: by default, iris's `tabs` widget tree
|
||||
//! (`tabs_ui::build`, shared with the winit example) running through
|
||||
//! `iris::android`'s `ViewPeer`. This is RUST.md's I2 pass condition made
|
||||
//! concrete -- there is no UI here beyond what `tabs-ui` already draws.
|
||||
//!
|
||||
@@ -9,6 +9,32 @@
|
||||
//! wrapping `iris::android::new_peer`'s generic function in a concrete
|
||||
//! `extern "system" fn`, since `register_view_class` wants a plain
|
||||
//! function pointer.
|
||||
//!
|
||||
//! **`transcript-screen` feature (RUST.md's I5 Android integration):** with
|
||||
//! `--features transcript-screen`, `new_view_peer` instantiates
|
||||
//! `transcript_client::TranscriptClient` instead of the tabs `Client`
|
||||
//! below, against a real `ai-server` (see that module's doc). Chosen over a
|
||||
//! third shell crate: this one already has the Gradle project, the
|
||||
//! `IrisView`/`MainActivity` Java, and the JNI registration I2 built and
|
||||
//! measured against, and the only thing a transcript screen needs on top
|
||||
//! is a different `AndroidAppState` -- the same axis `tabs_ui::build` vs.
|
||||
//! `transcript_ui::build` already varies along on the winit side (compare
|
||||
//! `iris/examples/tabs.rs` and `iris/transcript-ui/examples/transcript.rs`).
|
||||
//! A build picks one screen or the other, never both, so `Client` and
|
||||
//! `TranscriptClient` are cfg-gated apart rather than switched at runtime --
|
||||
//! there is no in-app navigation to switch *to* on either side yet.
|
||||
//!
|
||||
//! **`bench` feature (P0's iris half, docs/RUST.md):** a third
|
||||
//! `AndroidAppState`, `bench_client::BenchClient`, on the same axis --
|
||||
//! `transcript_ui::build_tree` again, this time against the checked-in
|
||||
//! fixture (`app/bench-fixture/assets/transcript.jsonl`) instead of a real
|
||||
//! server, with a "Run benchmark" control that drives the same scroll loop
|
||||
//! and streaming phase the Compose `bench` build type's `BenchRun.kt`
|
||||
//! does. `bench` depends on `transcript-screen` (Cargo.toml) for
|
||||
//! `transcript-ui`/`client-core`/`event-model`, so both features end up
|
||||
//! enabled together -- `ActiveClient` below gives `bench` priority in that
|
||||
//! case, the same way `transcript-screen` already takes priority over the
|
||||
//! default `tabs-screen`.
|
||||
|
||||
use android_view::{
|
||||
Context, View,
|
||||
@@ -18,19 +44,30 @@ use android_view::{
|
||||
},
|
||||
register_view_class,
|
||||
};
|
||||
#[cfg(not(feature = "transcript-screen"))]
|
||||
use iris::android::{AndroidAppState, AndroidRsc, AndroidUiState, HasAndroidUiState};
|
||||
#[cfg(not(feature = "transcript-screen"))]
|
||||
use iris::prelude::*;
|
||||
use log::LevelFilter;
|
||||
use std::ffi::c_void;
|
||||
|
||||
#[cfg(feature = "bench")]
|
||||
mod bench_client;
|
||||
#[cfg(feature = "bench")]
|
||||
mod bench_jni;
|
||||
#[cfg(all(feature = "transcript-screen", not(feature = "bench")))]
|
||||
mod transcript_client;
|
||||
|
||||
/// The app's `View` subclass, matching the Java side's package --
|
||||
/// `app/src/main/java/dev/iris/android/demo/IrisView.java`.
|
||||
const VIEW_CLASS: &str = "dev/iris/android/demo/IrisView";
|
||||
|
||||
#[cfg(not(feature = "transcript-screen"))]
|
||||
pub struct Client {
|
||||
ui_state: AndroidUiState,
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "transcript-screen"))]
|
||||
impl HasAndroidUiState for Client {
|
||||
fn android_state(&self) -> &AndroidUiState {
|
||||
&self.ui_state
|
||||
@@ -40,6 +77,7 @@ impl HasAndroidUiState for Client {
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "transcript-screen"))]
|
||||
impl AndroidAppState for Client {
|
||||
fn new(mut ui_state: AndroidUiState, rsc: &mut AndroidRsc<Self>) -> Self {
|
||||
// `widgets.info` is the winit example's frame-debug readout, kept
|
||||
@@ -61,12 +99,19 @@ impl AndroidAppState for Client {
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "transcript-screen"))]
|
||||
type ActiveClient = Client;
|
||||
#[cfg(all(feature = "transcript-screen", not(feature = "bench")))]
|
||||
type ActiveClient = transcript_client::TranscriptClient;
|
||||
#[cfg(feature = "bench")]
|
||||
type ActiveClient = bench_client::BenchClient;
|
||||
|
||||
extern "system" fn new_view_peer<'local>(
|
||||
env: JNIEnv<'local>,
|
||||
view: View<'local>,
|
||||
context: Context<'local>,
|
||||
) -> jlong {
|
||||
iris::android::new_peer::<Client>(env, view, context)
|
||||
iris::android::new_peer::<ActiveClient>(env, view, context)
|
||||
}
|
||||
|
||||
/// # Safety
|
||||
|
||||
@@ -0,0 +1,392 @@
|
||||
//! RUST.md's I5 Android integration: `transcript-ui`'s screen filling the
|
||||
//! whole window on android-view, against a real `ai-server` through
|
||||
//! `client-core` -- the missing half `iris-android-app` (I2) only had for
|
||||
//! `tabs-ui` until now. Behind the `transcript-screen` Cargo feature so the
|
||||
//! plain build (`cargo ndk build`, no `--features`) stays exactly the tabs
|
||||
//! demo I2/I4 already measured against.
|
||||
//!
|
||||
//! **Deliberate simplification, recorded rather than left to be
|
||||
//! rediscovered (RUST.md's I5 box has the full account)**: there is no
|
||||
//! session list and no enrollment UI here. The server, port, token and
|
||||
//! pinned CA are baked in at build time (`build.rs`'s
|
||||
//! `AI_APP_TRANSCRIPT_HOST`/`_PORT`/`_TOKEN`/`AI_APP_CA`), and the first
|
||||
//! session `ApiClient::fetch_sessions` returns is opened automatically --
|
||||
//! there is nothing to tap to get there, which is what `transcript-bench.sh`
|
||||
//! and `ui-trace` need to land straight on the screen under test. A real
|
||||
//! app needs `desktop-app`'s `EnrolledServer`/QR-link flow or E3's
|
||||
//! Keystore-sealed `ServerConfig.kt`; building a second one of those was
|
||||
//! not this pass's job.
|
||||
//!
|
||||
//! **Reuses `iris/desktop-app`'s `app.rs` shape almost exactly** --
|
||||
//! `fold_event`/`group_tool_runs`/`fold_page`/`raw_seq` from
|
||||
//! `client_core::transcript_fold`, a `generation` counter guarding against
|
||||
//! a stale background response. What differs is only the redraw
|
||||
//! mechanism: android-view has no `winit::EventLoopProxy`, so this uses
|
||||
//! `iris::task::Tasks::redraw_handle` (new, added alongside this box) to
|
||||
//! request a frame after each `TaskCtx::update` instead of relying on
|
||||
//! `Tasks::spawn`'s single end-of-future redraw -- see that method's own
|
||||
//! doc for why.
|
||||
//!
|
||||
//! **Streaming no longer costs a full rebuild** (fixed after the P0 gate
|
||||
//! showed why it mattered -- 20 events/second means 20 rebuilds/second of
|
||||
//! a ~3,200-row transcript otherwise): `apply_event` calls
|
||||
//! `transcript_ui::TranscriptScreen::apply` with the item list before and
|
||||
//! after `fold_event`, which updates only the row(s) that actually
|
||||
//! changed (almost always just the one open assistant message) instead of
|
||||
//! refolding and rebuilding every row. `rebuild_transcript` still runs
|
||||
//! the whole widget tree once, for the opening page and for `apply`'s own
|
||||
//! rare regroup fallback.
|
||||
|
||||
use client_core::api::{ApiClient, UreqTransport};
|
||||
use client_core::event_stream::{StreamItem, follow_session_events};
|
||||
use client_core::transcript_fold::{TranscriptItem, fold_event, fold_page, group_tool_runs};
|
||||
use event_model::SeqEvent;
|
||||
use iris::android::{AndroidAppState, AndroidRsc, AndroidUiState, HasAndroidUiState};
|
||||
use iris::prelude::*;
|
||||
use std::sync::Arc;
|
||||
use std::sync::atomic::{AtomicU64, Ordering};
|
||||
|
||||
mod pinned {
|
||||
include!(concat!(env!("OUT_DIR"), "/pinned_config.rs"));
|
||||
}
|
||||
|
||||
pub struct TranscriptClient {
|
||||
ui_state: AndroidUiState,
|
||||
/// The screen's own content -- everything under the fixed
|
||||
/// [`frame_report_controls`] bar, which is built once (`new`, below)
|
||||
/// and never touched by `show_message`/`rebuild_transcript`'s own
|
||||
/// `set` calls the way `desktop-app`'s `transcript_ptr` isn't touched
|
||||
/// by rebuilding the session list beside it.
|
||||
content: WeakWidget<WidgetPtr>,
|
||||
screen: Option<transcript_ui::TranscriptScreen>,
|
||||
/// The folded transcript as of the last rebuild -- kept here (not
|
||||
/// re-derived) for the same reason `desktop-app`'s `Client::items`
|
||||
/// exists: a live `StreamEvent` only carries one new wire event, and
|
||||
/// `fold_event` needs everything folded so far to fold it in.
|
||||
items: Vec<TranscriptItem>,
|
||||
/// The session currently open -- `None` only before the first fetch
|
||||
/// resolves. Read back by `apply_event`'s rebuild, which has no session
|
||||
/// id of its own (a live `SeqEvent` doesn't carry one).
|
||||
session_id: Option<String>,
|
||||
/// Bumped every time a new session load starts; a background response
|
||||
/// checks it before touching state, so a slow reply for a session this
|
||||
/// screen has moved on from can't overwrite what replaced it. There is
|
||||
/// only ever one session here (no list to switch away to), but the
|
||||
/// guard still matters for the *first* fetch racing a `stop`/`start`.
|
||||
generation: Arc<AtomicU64>,
|
||||
}
|
||||
|
||||
impl HasAndroidUiState for TranscriptClient {
|
||||
fn android_state(&self) -> &AndroidUiState {
|
||||
&self.ui_state
|
||||
}
|
||||
fn android_state_mut(&mut self) -> &mut AndroidUiState {
|
||||
&mut self.ui_state
|
||||
}
|
||||
}
|
||||
|
||||
/// Builds one `UreqTransport` from the config `build.rs` baked in. Called
|
||||
/// twice per session load, same as `desktop-app`'s `build_transport`
|
||||
/// closure -- `ApiClient` and the live-stream follow each need their own,
|
||||
/// since `UreqTransport` holds its own `ureq::Agent`.
|
||||
fn build_transport() -> Result<UreqTransport, String> {
|
||||
let base_url = format!("https://{}:{}", pinned::HOST, pinned::PORT);
|
||||
UreqTransport::new(
|
||||
base_url,
|
||||
pinned::TOKEN.to_string(),
|
||||
pinned::CA_PEM.as_bytes(),
|
||||
)
|
||||
.map_err(|e| e.to_string())
|
||||
}
|
||||
|
||||
fn placeholder<Rsc: HasEvents>(rsc: &mut Rsc, message: &str) -> StrongWidget {
|
||||
wtext(message.to_string())
|
||||
.color(Color::WHITE)
|
||||
.wrap(true)
|
||||
.pad(16)
|
||||
.add_strong(rsc)
|
||||
.any()
|
||||
}
|
||||
|
||||
/// The two named controls RUST.md's I5 box ("Measurements taken" (b))
|
||||
/// drives by name over `ui-trace`, e.g. `ui-trace record --do "tap 'Frame
|
||||
/// report'"`. `dumpsys gfxinfo` cannot see this screen's own GPU-drawn
|
||||
/// frames at all -- this is the screen's own equivalent of the Compose
|
||||
/// app's "Copy render timings" control, logged rather than clipboarded
|
||||
/// (no clipboard wiring exists here) under this crate's own fixed
|
||||
/// `android_logger` tag (`iris-android-app`, `lib.rs`'s `JNI_OnLoad`),
|
||||
/// grep-able on the fixed string `"iris frame report"` the way
|
||||
/// `transcript-bench.sh` greps `"ai-app render report"`.
|
||||
fn frame_report_controls(rsc: &mut AndroidRsc<TranscriptClient>) -> WeakWidget {
|
||||
type Rsc = AndroidRsc<TranscriptClient>;
|
||||
let report_rect = rect(Color::rgb(50, 50, 60))
|
||||
.on(
|
||||
CursorSense::click(),
|
||||
|ctx: EventIdCtx<'_, Rsc, _, _>, _rsc: &mut Rsc| match ctx
|
||||
.state
|
||||
.android_state()
|
||||
.frame_report
|
||||
.report()
|
||||
{
|
||||
Some(stats) => log::info!("iris frame report: {stats}"),
|
||||
None => log::info!(
|
||||
"iris frame report: no frames recorded -- scroll first, then press this"
|
||||
),
|
||||
},
|
||||
)
|
||||
.label("Frame report");
|
||||
let report = (
|
||||
report_rect,
|
||||
wtext("Frame report").size(18).text_align(Align::CENTER),
|
||||
)
|
||||
.stack()
|
||||
.pad(8)
|
||||
.add(rsc);
|
||||
|
||||
let reset_rect = rect(Color::rgb(70, 40, 40))
|
||||
.on(
|
||||
CursorSense::click(),
|
||||
|ctx: EventIdCtx<'_, Rsc, _, _>, _rsc: &mut Rsc| {
|
||||
ctx.state.android_state_mut().frame_report.reset();
|
||||
log::info!("iris frame report: reset");
|
||||
},
|
||||
)
|
||||
.label("Reset frame report");
|
||||
let reset = (
|
||||
reset_rect,
|
||||
wtext("Reset").size(18).text_align(Align::CENTER),
|
||||
)
|
||||
.stack()
|
||||
.pad(8)
|
||||
.add(rsc);
|
||||
|
||||
(report, reset).span(Dir::RIGHT).height(56).add(rsc)
|
||||
}
|
||||
|
||||
impl AndroidAppState for TranscriptClient {
|
||||
fn new(mut ui_state: AndroidUiState, rsc: &mut AndroidRsc<Self>) -> Self {
|
||||
let content = WidgetPtr::new().add(rsc);
|
||||
let loading = placeholder(rsc, "Loading sessions...");
|
||||
content(rsc).set(loading);
|
||||
|
||||
let tree = (frame_report_controls(rsc), content.height(rest(1)))
|
||||
.span(Dir::DOWN)
|
||||
.add_strong(rsc)
|
||||
.any();
|
||||
ui_state.set_root(tree);
|
||||
|
||||
let mut client = Self {
|
||||
ui_state,
|
||||
content,
|
||||
screen: None,
|
||||
items: Vec::new(),
|
||||
session_id: None,
|
||||
generation: Arc::new(AtomicU64::new(0)),
|
||||
};
|
||||
client.spawn_fetch_sessions(rsc);
|
||||
client
|
||||
}
|
||||
|
||||
fn back_pressed(&mut self, _rsc: &mut AndroidRsc<Self>, _render: &mut UiRenderState) -> bool {
|
||||
// No screen stack of its own -- same "let the activity finish"
|
||||
// answer `iris-android-app`'s tabs `Client` already gives.
|
||||
false
|
||||
}
|
||||
}
|
||||
|
||||
impl TranscriptClient {
|
||||
fn show_message(&mut self, rsc: &mut AndroidRsc<Self>, message: &str) {
|
||||
let widget = placeholder(rsc, message);
|
||||
(self.content)(rsc).set(widget);
|
||||
self.screen = None;
|
||||
}
|
||||
|
||||
fn spawn_fetch_sessions(&mut self, rsc: &mut AndroidRsc<Self>) {
|
||||
let redraw = rsc.tasks.redraw_handle();
|
||||
let my_generation = self.generation.load(Ordering::SeqCst);
|
||||
let generation = self.generation.clone();
|
||||
rsc.spawn_task(async move |mut ctx| {
|
||||
let outcome = match build_transport() {
|
||||
Ok(transport) => ApiClient::new(transport)
|
||||
.fetch_sessions()
|
||||
.map_err(|e| e.to_string()),
|
||||
Err(e) => Err(format!("couldn't set up TLS: {e}")),
|
||||
};
|
||||
ctx.update(move |state: &mut TranscriptClient, rsc| {
|
||||
if generation.load(Ordering::SeqCst) != my_generation {
|
||||
return;
|
||||
}
|
||||
match outcome {
|
||||
Ok(sessions) => match sessions.into_iter().next() {
|
||||
Some(session) => state.select_session(rsc, session.id),
|
||||
None => state.show_message(rsc, "No sessions on the sandbox server."),
|
||||
},
|
||||
Err(message) => {
|
||||
state.show_message(rsc, &format!("Couldn't list sessions: {message}"))
|
||||
}
|
||||
}
|
||||
});
|
||||
redraw.request_redraw();
|
||||
});
|
||||
}
|
||||
|
||||
/// Loads the opening page, then follows the live SSE stream for the
|
||||
/// rest of this session's life -- `desktop-app`'s `select_session`
|
||||
/// almost verbatim, with `Proxy::send_event` replaced by `ctx.update` +
|
||||
/// `redraw.request_redraw()` (see this module's doc).
|
||||
fn select_session(&mut self, rsc: &mut AndroidRsc<Self>, session_id: String) {
|
||||
let my_generation = self.generation.fetch_add(1, Ordering::SeqCst) + 1;
|
||||
self.items.clear();
|
||||
self.session_id = Some(session_id.clone());
|
||||
self.show_message(rsc, "Loading transcript...");
|
||||
|
||||
let redraw = rsc.tasks.redraw_handle();
|
||||
let live_generation = self.generation.clone();
|
||||
rsc.spawn_task(async move |mut ctx| {
|
||||
let transports =
|
||||
build_transport().and_then(|rest| build_transport().map(|stream| (rest, stream)));
|
||||
let (rest, stream_transport) = match transports {
|
||||
Ok(pair) => pair,
|
||||
Err(e) => {
|
||||
let message = format!("couldn't set up TLS: {e}");
|
||||
ctx.update(move |state: &mut TranscriptClient, rsc| {
|
||||
if live_generation.load(Ordering::SeqCst) == my_generation {
|
||||
state.show_message(rsc, &message);
|
||||
}
|
||||
});
|
||||
redraw.request_redraw();
|
||||
return;
|
||||
}
|
||||
};
|
||||
let api = ApiClient::new(rest);
|
||||
|
||||
// The most recent 200 events, coalesced -- the same page size
|
||||
// `desktop-app` uses; RUST.md's I3/history-paging work is what
|
||||
// a real scrollback would reuse (out of scope here, same as
|
||||
// E4).
|
||||
let page: Result<Vec<serde_json::Value>, String> = api
|
||||
.fetch_transcript_page(&session_id, None, 200, true)
|
||||
.map_err(|e| e.to_string());
|
||||
// The wire `seq` of the last line, not a folded item's `seq()`
|
||||
// -- see `client_core::transcript_fold::raw_seq`'s doc for why
|
||||
// resuming from the latter re-delivers deltas already folded
|
||||
// into an in-progress reply.
|
||||
let after = page
|
||||
.as_ref()
|
||||
.ok()
|
||||
.and_then(|values| values.last())
|
||||
.and_then(client_core::transcript_fold::raw_seq)
|
||||
.unwrap_or(0);
|
||||
let result = page.and_then(|values| fold_page(&values));
|
||||
|
||||
{
|
||||
let live_generation = live_generation.clone();
|
||||
ctx.update(move |state: &mut TranscriptClient, rsc| {
|
||||
if live_generation.load(Ordering::SeqCst) != my_generation {
|
||||
return;
|
||||
}
|
||||
match result {
|
||||
Ok(items) => {
|
||||
state.items = items;
|
||||
state.rebuild_transcript(rsc);
|
||||
}
|
||||
Err(message) => {
|
||||
state.show_message(rsc, &format!("Couldn't load transcript: {message}"))
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
redraw.request_redraw();
|
||||
|
||||
if live_generation.load(Ordering::SeqCst) != my_generation {
|
||||
return;
|
||||
}
|
||||
// The outer closure here is an `FnMut` -- `follow_session_events`
|
||||
// calls it once per line -- so it captures `live_generation` by
|
||||
// move and re-clones it for each inner `ctx.update` closure
|
||||
// rather than moving a shared `stop`-style helper into itself:
|
||||
// a value moved out of an `FnMut`'s captures on one call leaves
|
||||
// nothing there for the next.
|
||||
let _ =
|
||||
follow_session_events(
|
||||
&stream_transport,
|
||||
&session_id,
|
||||
after,
|
||||
move |item| match item {
|
||||
StreamItem::Open | StreamItem::Reset => {
|
||||
live_generation.load(Ordering::SeqCst) == my_generation
|
||||
}
|
||||
StreamItem::Event { event, .. } => {
|
||||
if live_generation.load(Ordering::SeqCst) != my_generation {
|
||||
return false;
|
||||
}
|
||||
let live_generation = live_generation.clone();
|
||||
ctx.update(move |state: &mut TranscriptClient, rsc| {
|
||||
if live_generation.load(Ordering::SeqCst) != my_generation {
|
||||
return;
|
||||
}
|
||||
state.apply_event(rsc, &event);
|
||||
});
|
||||
redraw.request_redraw();
|
||||
true
|
||||
}
|
||||
},
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
/// Rebuilds the whole widget tree from `self.items` -- same tradeoff as
|
||||
/// `desktop-app`'s `rebuild_transcript` (this module's doc comment).
|
||||
/// Reads `self.session_id` rather than taking one, since every caller
|
||||
/// (the opening page, and every live event) already has it set there.
|
||||
fn rebuild_transcript(&mut self, rsc: &mut AndroidRsc<Self>) {
|
||||
let in_progress = self
|
||||
.screen
|
||||
.as_ref()
|
||||
.map(|screen| screen.composer.field.edit(rsc).text.text().to_string())
|
||||
.filter(|t| !t.is_empty());
|
||||
|
||||
let rows = group_tool_runs(&self.items);
|
||||
let (screen, tree) = transcript_ui::build_tree(rsc, rows);
|
||||
|
||||
if let Some(text) = in_progress {
|
||||
screen.composer.field.edit(rsc).set(&text);
|
||||
}
|
||||
if let Some(session_id) = self.session_id.clone() {
|
||||
let field = screen.composer.field;
|
||||
rsc.register_event(field, Submit, move |ctx, rsc| {
|
||||
let text = field.edit(rsc).take();
|
||||
let text = text.trim().to_string();
|
||||
if !text.is_empty() {
|
||||
ctx.state.send_message(session_id.clone(), text);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
(self.content)(rsc).set(tree);
|
||||
self.screen = Some(screen);
|
||||
}
|
||||
|
||||
fn apply_event(&mut self, rsc: &mut AndroidRsc<Self>, event: &SeqEvent) {
|
||||
let old_items = self.items.clone();
|
||||
self.items = fold_event(&self.items, event);
|
||||
match &self.screen {
|
||||
// The common path: update only the row(s) that actually
|
||||
// changed instead of refolding and rebuilding all ~3,200 of
|
||||
// them per event (RUST.md's P0 streaming-phase fix).
|
||||
Some(screen) => screen.apply(rsc, &old_items, &self.items),
|
||||
// No screen yet (the opening page hasn't landed) -- build one
|
||||
// the ordinary way once it has.
|
||||
None => self.rebuild_transcript(rsc),
|
||||
}
|
||||
}
|
||||
|
||||
fn send_message(&mut self, session_id: String, text: String) {
|
||||
std::thread::spawn(move || {
|
||||
if let Ok(transport) = build_transport() {
|
||||
let api = ApiClient::new(transport);
|
||||
let _ = api.send_message(&session_id, &text, &[]);
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
+200
-33
@@ -1,6 +1,7 @@
|
||||
//! On-demand benchmarks for iris's message-list scenario -- IRIS_TODO.md's
|
||||
//! "Benchmarks" item. Never run by `cargo test`; run explicitly with
|
||||
//! `cargo bench --bench message_list --release` or `./run-bench.sh`.
|
||||
//! "Benchmarks" item, and RUST.md's I3. Never run by `cargo test`; run
|
||||
//! explicitly with `cargo bench --bench message_list --release` or
|
||||
//! `./run-bench.sh`.
|
||||
//!
|
||||
//! **Why a plain `Instant`-timed binary, not criterion.** Every scenario
|
||||
//! here is really "how many `Widget::draw` calls and primitive rewrites did
|
||||
@@ -13,26 +14,49 @@
|
||||
//! -- and it avoids a new dependency this crate does not otherwise need.
|
||||
//! Per the code rules, the plain option is also the one shorter to explain.
|
||||
//!
|
||||
//! Scenarios (LAYOUT.md's O(1) move chain, and IRIS_TODO.md's "Benchmarks"
|
||||
//! wording):
|
||||
//! **The list under test is `iris::widget::List` (RUST.md's I3), not a
|
||||
//! `Scroll` over a `Span` of pre-built rows.** Earlier versions of this
|
||||
//! file built their own giant `Span` and wrapped it in `Scroll`, which
|
||||
//! meant (a)/(b)/(c) below were measuring "move one big child," never the
|
||||
//! virtualised widget the app's transcript screen actually needs. `List`
|
||||
//! still needs every row's *widget* built up front by the caller (its
|
||||
//! module doc explains why: it only ever sees `&dyn Widget` through
|
||||
//! `Painter`, so it cannot construct a row lazily on its own) -- what
|
||||
//! virtualisation buys is that only the rows currently on screen are ever
|
||||
//! *drawn*, which is what the draw/rewrite/move counters below are
|
||||
//! measuring, not construction time.
|
||||
//!
|
||||
//! Scenarios (LAYOUT.md's O(1) move chain, list.rs's module doc, and
|
||||
//! IRIS_TODO.md's "Benchmarks" wording):
|
||||
//!
|
||||
//! - (a) first-frame cost of a message list of N wrapped-text rows, some
|
||||
//! with an image, for N = 100 / 1,000 / 10,000.
|
||||
//! with an image, for N = 100 / 1,000 / 10,000. With a virtualised list
|
||||
//! this is expected to stop scaling with N once N exceeds a screenful --
|
||||
//! the draw/rewrite counters below are the number that used to grow 10x
|
||||
//! per 10x N and should not any more.
|
||||
//! - (b) per-frame cost of scrolling that list -- must be O(1) moves, not
|
||||
//! re-layout.
|
||||
//! - (c) the input-box case: growing a fixed-height field at the bottom of
|
||||
//! the screen must move the message list above it, not re-lay its rows.
|
||||
//! Reports frame time *and* the draw/rewrite/move counters LAYOUT.md
|
||||
//! section 8 defines.
|
||||
//! - (d) insert-above-anchor: paging older history onto the front of an
|
||||
//! already-scrolled list. `List::push_front` is an O(1) index update
|
||||
//! (list.rs's module doc); this measures that none of the rows already
|
||||
//! on screen are touched by it.
|
||||
//! - (e) expand-a-row-holding-its-edge: growing one row's height with a
|
||||
//! tap recorded near one of its edges (list.rs's `note_tap`) must move
|
||||
//! only the rows on the far side of it, never redraw the ones already
|
||||
//! correctly placed.
|
||||
//!
|
||||
//! (d), many images with zero steady-state bind-group creation, needs a
|
||||
//! (f), many images with zero steady-state bind-group creation, needs a
|
||||
//! real `wgpu` device and lives in `iris/examples/bench_images.rs` instead,
|
||||
//! driven through `run-headless.sh` -- see that file's header.
|
||||
//!
|
||||
//! `UiRenderState`/`Widgets` touch no GPU or window (as `layout_tests.rs`
|
||||
//! notes), so everything here runs as an ordinary `--release` binary with
|
||||
//! no compositor. Numbers are recorded in IRIS_TODO.md, not here -- this
|
||||
//! file is the rig, not the result.
|
||||
//! no compositor. Numbers are recorded in RUST.md's I3 box, not here --
|
||||
//! this file is the rig, not the result.
|
||||
|
||||
use iris::prelude::*;
|
||||
use std::time::Instant;
|
||||
@@ -82,21 +106,21 @@ fn build_row(rsc: &mut BenchRsc, i: usize, image_every: usize) -> StrongWidget {
|
||||
}
|
||||
}
|
||||
|
||||
/// A `Scroll` over `n` message rows, one in `image_every` of them carrying
|
||||
/// an image (0 disables images entirely). Returns the scroll widget (weak,
|
||||
/// so the caller can drive it) and the erased root to render.
|
||||
fn build_list(
|
||||
/// A virtualised `List` of `n` message rows, one in `image_every` of them
|
||||
/// carrying an image (0 disables images entirely). Returns the list widget
|
||||
/// (weak, so the caller can drive it) and the erased root to render.
|
||||
fn build_message_list(
|
||||
rsc: &mut BenchRsc,
|
||||
n: usize,
|
||||
image_every: usize,
|
||||
) -> (WeakWidget<Scroll>, StrongWidget) {
|
||||
let mut span = Span::empty(Dir::DOWN);
|
||||
) -> (WeakWidget<List>, StrongWidget) {
|
||||
let mut list = List::new(Axis::Y);
|
||||
for i in 0..n {
|
||||
span.push(build_row(rsc, i, image_every));
|
||||
let row = build_row(rsc, i, image_every);
|
||||
list.push_back(ListRow::new(i as u64, row));
|
||||
}
|
||||
let span = rsc.ui.widgets.add_strong(span);
|
||||
let scroll = rsc.ui.widgets.add_strong(Scroll::new(span.any(), Axis::Y));
|
||||
(scroll.weak(), scroll.any())
|
||||
let list = rsc.ui.widgets.add_strong(list);
|
||||
(list.weak(), list.any())
|
||||
}
|
||||
|
||||
fn report(label: &str, elapsed: std::time::Duration, draws: u64, rewrites: u64, moves: u64) {
|
||||
@@ -111,7 +135,7 @@ fn bench_first_frame(n: usize) {
|
||||
let mut rsc = BenchRsc {
|
||||
ui: UiData::default(),
|
||||
};
|
||||
let (_scroll, root) = build_list(&mut rsc, n, 20);
|
||||
let (_list, root) = build_message_list(&mut rsc, n, 20);
|
||||
let mut render = UiRenderState::new();
|
||||
render.resize((1080.0, 2000.0));
|
||||
|
||||
@@ -129,18 +153,18 @@ fn bench_first_frame(n: usize) {
|
||||
}
|
||||
|
||||
/// (b) Per-frame cost of scrolling an already-laid-out list of N rows.
|
||||
/// Warms up (as `layout_tests.rs`'s scrolling test documents: `Scroll`
|
||||
/// needs one no-op tick before a real scroll becomes a same-size move
|
||||
/// Warms up (one no-op tick, matching `Scroll`'s own need for it before an
|
||||
/// ordinary Rust `layout_tests.rs` scrolling test becomes a same-size move
|
||||
/// rather than a resize), then times a run of individual scroll ticks.
|
||||
fn bench_scroll(n: usize, ticks: usize) {
|
||||
let mut rsc = BenchRsc {
|
||||
ui: UiData::default(),
|
||||
};
|
||||
let (scroll, root) = build_list(&mut rsc, n, 20);
|
||||
let (list, root) = build_message_list(&mut rsc, n, 20);
|
||||
let mut render = UiRenderState::new();
|
||||
render.resize((1080.0, 2000.0));
|
||||
render.update(&root, &mut rsc);
|
||||
rsc.ui.widgets.get_mut(&scroll).unwrap().scroll(0.0);
|
||||
rsc.ui.widgets.get_mut(&list).unwrap().scroll(0.0);
|
||||
render.update(&root, &mut rsc);
|
||||
render.take_counters();
|
||||
|
||||
@@ -149,7 +173,7 @@ fn bench_scroll(n: usize, ticks: usize) {
|
||||
let mut total_rewrites = 0u64;
|
||||
let mut total_moves = 0u64;
|
||||
for _ in 0..ticks {
|
||||
rsc.ui.widgets.get_mut(&scroll).unwrap().scroll(-8.0);
|
||||
rsc.ui.widgets.get_mut(&list).unwrap().scroll(-8.0);
|
||||
let start = Instant::now();
|
||||
render.update(&root, &mut rsc);
|
||||
total += start.elapsed();
|
||||
@@ -174,18 +198,16 @@ fn bench_scroll(n: usize, ticks: usize) {
|
||||
/// (c) The input-box case: a fixed-height field at the bottom of the screen
|
||||
/// growing by a line at a time, with a message list of N rows filling the
|
||||
/// rest of the screen above it. Growing the input shrinks the *offered*
|
||||
/// height of the scroll container (a single widget, from the outer
|
||||
/// `Span`'s point of view) without changing the width it offers its
|
||||
/// content -- so the rows underneath, which only care about width, must
|
||||
/// not redraw; the scroll's own re-registration of where its content sits
|
||||
/// is the one O(1) move this is checking for. See LAYOUT.md's `Scroll`
|
||||
/// design note on offering the child last frame's content length, which is
|
||||
/// exactly what keeps this a move instead of a reflow.
|
||||
/// height of the list container (a single widget, from the outer `Span`'s
|
||||
/// point of view) without changing the width it offers its content -- so
|
||||
/// the rows underneath, which only care about width, must not redraw; the
|
||||
/// list's own re-registration of where its content sits is the one O(1)
|
||||
/// move this is checking for.
|
||||
fn bench_input_grows(n: usize, lines: usize) {
|
||||
let mut rsc = BenchRsc {
|
||||
ui: UiData::default(),
|
||||
};
|
||||
let (scroll, list_root) = build_list(&mut rsc, n, 20);
|
||||
let (list, list_root) = build_message_list(&mut rsc, n, 20);
|
||||
let list_area = rsc.ui.widgets.add_strong(Sized {
|
||||
inner: list_root,
|
||||
x: None,
|
||||
@@ -209,7 +231,7 @@ fn bench_input_grows(n: usize, lines: usize) {
|
||||
let mut render = UiRenderState::new();
|
||||
render.resize((1080.0, 2000.0));
|
||||
render.update(&root, &mut rsc);
|
||||
rsc.ui.widgets.get_mut(&scroll).unwrap().scroll(0.0);
|
||||
rsc.ui.widgets.get_mut(&list).unwrap().scroll(0.0);
|
||||
render.update(&root, &mut rsc);
|
||||
render.take_counters();
|
||||
|
||||
@@ -244,6 +266,145 @@ fn bench_input_grows(n: usize, lines: usize) {
|
||||
);
|
||||
}
|
||||
|
||||
/// (d) Insert-above-anchor: the list is scrolled to its very first loaded
|
||||
/// row (`jump_to_start`, an O(1) re-anchor) rather than left at the default
|
||||
/// bottom, so a row prepended above it is genuinely "inserted above the
|
||||
/// anchor" rather than merely far off-screen at the far end. Each
|
||||
/// `push_front` is O(1) (list.rs's module doc: the anchor's slot is an
|
||||
/// index, bumped by one) and, since the prepended rows never enter the
|
||||
/// viewport, none of them should cost a draw either.
|
||||
fn bench_insert_above_anchor(n: usize, inserts: usize) {
|
||||
let mut rsc = BenchRsc {
|
||||
ui: UiData::default(),
|
||||
};
|
||||
let (list, root) = build_message_list(&mut rsc, n, 20);
|
||||
let mut render = UiRenderState::new();
|
||||
render.resize((1080.0, 2000.0));
|
||||
render.update(&root, &mut rsc);
|
||||
rsc.ui.widgets.get_mut(&list).unwrap().jump_to_start();
|
||||
render.update(&root, &mut rsc);
|
||||
render.take_counters();
|
||||
|
||||
let mut total = std::time::Duration::ZERO;
|
||||
let mut total_draws = 0u64;
|
||||
let mut total_rewrites = 0u64;
|
||||
let mut total_moves = 0u64;
|
||||
for i in 0..inserts {
|
||||
// Older-history rows: distinct keys below every existing one, so a
|
||||
// real caller's paging code (prepending an older page) is exactly
|
||||
// what this loop does.
|
||||
let row = build_row(&mut rsc, usize::MAX - i, 20);
|
||||
rsc.ui
|
||||
.widgets
|
||||
.get_mut(&list)
|
||||
.unwrap()
|
||||
.push_front(ListRow::new(i as u64, row));
|
||||
let start = Instant::now();
|
||||
render.update(&root, &mut rsc);
|
||||
total += start.elapsed();
|
||||
let (draws, rewrites, moves) = render.take_counters();
|
||||
total_draws += draws;
|
||||
total_rewrites += rewrites;
|
||||
total_moves += moves;
|
||||
}
|
||||
report(
|
||||
&format!(
|
||||
"(d) insert-above-anchor, N={n}, {inserts} pushes (totals; \
|
||||
must not scale with N)"
|
||||
),
|
||||
total,
|
||||
total_draws,
|
||||
total_rewrites,
|
||||
total_moves,
|
||||
);
|
||||
println!(
|
||||
" per-push average: {:.4}ms",
|
||||
total.as_secs_f64() * 1000.0 / inserts as f64
|
||||
);
|
||||
}
|
||||
|
||||
/// (e) Expand-a-row-holding-its-edge: one row (fixed-height, so its size is
|
||||
/// directly controllable) is grown a little at a time, each time preceded
|
||||
/// by `note_tap` aimed at its own top edge -- the exact mechanism list.rs's
|
||||
/// module doc describes and its unit tests check for correctness. This
|
||||
/// measures its *cost*: only the rows on the far side of the grown one
|
||||
/// (below it, since the top edge is held) should ever move, and nothing
|
||||
/// should be redrawn purely because the list overall got taller.
|
||||
fn bench_expand_holds_edge(n: usize, growths: usize) {
|
||||
let mut rsc = BenchRsc {
|
||||
ui: UiData::default(),
|
||||
};
|
||||
let mut list = List::new(Axis::Y);
|
||||
// Near the end (not the very last row) so it is already on screen
|
||||
// under the list's default bottom-anchored placement, for every N --
|
||||
// no scrolling needed to bring it into view before measuring.
|
||||
let growable_index = n.saturating_sub(3);
|
||||
let mut growable = None;
|
||||
for i in 0..n {
|
||||
if i == growable_index {
|
||||
let rect = rsc.ui.widgets.add_strong(Rect::new(UiColor::WHITE));
|
||||
let sized = rsc.ui.widgets.add_strong(Sized {
|
||||
inner: rect.any(),
|
||||
x: None,
|
||||
y: Some(abs(40.0)),
|
||||
});
|
||||
growable = Some(sized.weak());
|
||||
list.push_back(ListRow::new(i as u64, sized.any()));
|
||||
} else {
|
||||
let row = build_row(&mut rsc, i, 20);
|
||||
list.push_back(ListRow::new(i as u64, row));
|
||||
}
|
||||
}
|
||||
let list = rsc.ui.widgets.add_strong(list);
|
||||
let list_weak = list.weak();
|
||||
let root = list.any();
|
||||
let growable = growable.unwrap();
|
||||
|
||||
let mut render = UiRenderState::new();
|
||||
render.resize((1080.0, 2000.0));
|
||||
render.update(&root, &mut rsc);
|
||||
render.take_counters();
|
||||
|
||||
let mut total = std::time::Duration::ZERO;
|
||||
let mut total_draws = 0u64;
|
||||
let mut total_rewrites = 0u64;
|
||||
let mut total_moves = 0u64;
|
||||
let mut height = 40.0f32;
|
||||
let key = growable_index as u64;
|
||||
for _ in 0..growths {
|
||||
height += 10.0;
|
||||
if let Some((top, _bottom)) = rsc.ui.widgets.get(&list_weak).unwrap().extent(key) {
|
||||
rsc.ui
|
||||
.widgets
|
||||
.get_mut(&list_weak)
|
||||
.unwrap()
|
||||
.note_tap(top + 1.0);
|
||||
}
|
||||
rsc.ui.widgets.get_mut(&growable).unwrap().y = Some(abs(height));
|
||||
let start = Instant::now();
|
||||
render.update(&root, &mut rsc);
|
||||
total += start.elapsed();
|
||||
let (draws, rewrites, moves) = render.take_counters();
|
||||
total_draws += draws;
|
||||
total_rewrites += rewrites;
|
||||
total_moves += moves;
|
||||
}
|
||||
report(
|
||||
&format!(
|
||||
"(e) expand-hold, N={n}, {growths} growths (totals; \
|
||||
must not scale with N)"
|
||||
),
|
||||
total,
|
||||
total_draws,
|
||||
total_rewrites,
|
||||
total_moves,
|
||||
);
|
||||
println!(
|
||||
" per-growth average: {:.4}ms",
|
||||
total.as_secs_f64() * 1000.0 / growths as f64
|
||||
);
|
||||
}
|
||||
|
||||
fn main() {
|
||||
println!("iris message-list benchmark -- release build, this machine's CPU");
|
||||
for &n in &[100usize, 1_000, 10_000] {
|
||||
@@ -255,4 +416,10 @@ fn main() {
|
||||
for &n in &[100usize, 1_000, 10_000] {
|
||||
bench_input_grows(n, 40);
|
||||
}
|
||||
for &n in &[100usize, 1_000, 10_000] {
|
||||
bench_insert_above_anchor(n, 200);
|
||||
}
|
||||
for &n in &[100usize, 1_000, 10_000] {
|
||||
bench_expand_holds_edge(n, 40);
|
||||
}
|
||||
}
|
||||
@@ -5,8 +5,15 @@ edition.workspace = true
|
||||
|
||||
[dependencies]
|
||||
wgpu = { workspace = true }
|
||||
# Only for `UiRenderNode::new`'s `push_error_scope`/`pop_error_scope` pair
|
||||
# (renderer-creation error reporting, RUST.md's P0 phone-crash box) --
|
||||
# `block_on` turns that one async pop into the same synchronous call shape
|
||||
# `device_limits()`'s two callers already use for `request_adapter`/
|
||||
# `request_device`, rather than making this crate's one entry point async.
|
||||
pollster = { workspace = true }
|
||||
bytemuck ={ workspace = true }
|
||||
image = { workspace = true }
|
||||
parley = { workspace = true }
|
||||
swash = { workspace = true }
|
||||
fxhash = { workspace = true }
|
||||
accesskit = { workspace = true }
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Loaded 100 of 178 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user