Compare commits
76
Commits
@@ -10,12 +10,18 @@
|
|||||||
# releases don't trigger each other's pipelines.
|
# releases don't trigger each other's pipelines.
|
||||||
#
|
#
|
||||||
# Required secrets:
|
# Required secrets:
|
||||||
# REGISTRY_TOKEN personal access token with read+write package scope —
|
# REGISTRY_TOKEN personal access token with read+write package scope —
|
||||||
# the built-in Actions token is NOT accepted by the
|
# the built-in Actions token is NOT accepted by the
|
||||||
# container registry (docker login → unauthorized).
|
# container registry (docker login → unauthorized).
|
||||||
# Create: user Settings → Applications → Generate token.
|
# Create: user Settings → Applications → Generate token.
|
||||||
# REGISTRY_USER optional; defaults to the pushing actor's username.
|
# REGISTRY_USER optional; defaults to the pushing actor's username.
|
||||||
# The release job needs only the built-in GITHUB_TOKEN.
|
# RELEASE_SIGNING_KEY base64 ed25519 seed that signs SHA256SUMS. Self-updating
|
||||||
|
# servers verify the signature against the public key baked
|
||||||
|
# into the binary (selfupdate.DefaultPublicKeyB64) and REFUSE
|
||||||
|
# unsigned releases, so this job hard-fails without it —
|
||||||
|
# a release nobody can install is better failed loudly here.
|
||||||
|
# Mint a pair with: go run ./cmd/release-sign -gen
|
||||||
|
# The release job otherwise needs only the built-in GITHUB_TOKEN.
|
||||||
|
|
||||||
name: server-release
|
name: server-release
|
||||||
on:
|
on:
|
||||||
@@ -44,6 +50,18 @@ jobs:
|
|||||||
done
|
done
|
||||||
(cd ../dist && sha256sum * > SHA256SUMS)
|
(cd ../dist && sha256sum * > SHA256SUMS)
|
||||||
|
|
||||||
|
- name: Sign SHA256SUMS
|
||||||
|
working-directory: server
|
||||||
|
env:
|
||||||
|
RELEASE_SIGNING_KEY: ${{ secrets.RELEASE_SIGNING_KEY }}
|
||||||
|
run: |
|
||||||
|
[ -n "$RELEASE_SIGNING_KEY" ] || { echo "::error::secret RELEASE_SIGNING_KEY is missing — self-updating servers refuse unsigned releases, so publishing one would strand the fleet. Add it under Settings → Actions → Secrets."; exit 1; }
|
||||||
|
go run ./cmd/release-sign ../dist/SHA256SUMS
|
||||||
|
# Verify with the key baked into the binary we just built — catches a
|
||||||
|
# secret that does not match DefaultPublicKeyB64 before it ships.
|
||||||
|
PUB=$(grep -o 'DefaultPublicKeyB64 = "[^"]*"' internal/selfupdate/selfupdate.go | cut -d'"' -f2)
|
||||||
|
go run ./cmd/release-sign -verify -pub "$PUB" ../dist/SHA256SUMS
|
||||||
|
|
||||||
- name: Create release + attach binaries
|
- name: Create release + attach binaries
|
||||||
env:
|
env:
|
||||||
TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
|||||||
@@ -43,3 +43,6 @@ web/.wrangler/
|
|||||||
|
|
||||||
# Eclipse/JDT output from the VSCodium Java extension — not a build artifact we own
|
# Eclipse/JDT output from the VSCodium Java extension — not a build artifact we own
|
||||||
echolot-app/*/bin/
|
echolot-app/*/bin/
|
||||||
|
|
||||||
|
# Kotlin compiler scratch/error logs
|
||||||
|
echolot-app/.kotlin/
|
||||||
|
|||||||
@@ -145,6 +145,17 @@ First build downloads AGP/Compose/Shizuku from Google Maven + Maven Central.
|
|||||||
Shizuku, **toggle Wireless debugging off/on** — Shizuku keeps running (separate process), a fresh
|
Shizuku, **toggle Wireless debugging off/on** — Shizuku keeps running (separate process), a fresh
|
||||||
port + mDNS record appear, and the beacon/connector recover. Plan the Shizuku-tier dev loop
|
port + mDNS record appear, and the beacon/connector recover. Plan the Shizuku-tier dev loop
|
||||||
around this (or USB, if ever available).
|
around this (or USB, if ever available).
|
||||||
|
- **A poisoned Gradle *build cache* entry can silently drop a whole module from the APK.**
|
||||||
|
Symptom: the app dies with `ClassNotFoundException` for a class that plainly exists, while the
|
||||||
|
build is green and `./gradlew :app:dependencies` lists the module on `debugRuntimeClasspath`.
|
||||||
|
The module's own jar is correct; its code simply never reaches AGP's intermediates. `clean`,
|
||||||
|
`rm -rf */build` and `--rerun-tasks` all fail to fix it, because **none of them touch the build
|
||||||
|
cache** — look for `compileKotlin FROM-CACHE` in the log. Fix: rebuild with `--no-build-cache`.
|
||||||
|
Verify by grepping the APK's dex for a string literal that only that module defines; grepping for
|
||||||
|
a *class name* proves nothing, because callers carry the name as a reference whether or not the
|
||||||
|
class is packaged:
|
||||||
|
`unzip -o -q app-debug.apk "classes*.dex" && grep -a "pin-sha256:" *.dex`
|
||||||
|
Suspect this whenever a runtime failure contradicts a successful build.
|
||||||
- **Empty-jar race with the IDE.** VSCodium's Java/Kotlin extension runs its own Gradle daemon on
|
- **Empty-jar race with the IDE.** VSCodium's Java/Kotlin extension runs its own Gradle daemon on
|
||||||
the same project; when it overlaps a CLI build, a module's `build/libs/*.jar` can end up
|
the same project; when it overlaps a CLI build, a module's `build/libs/*.jar` can end up
|
||||||
containing only a manifest, and Gradle then considers `jar` up-to-date. Dependent modules fail
|
containing only a manifest, and Gradle then considers `jar` up-to-date. Dependent modules fail
|
||||||
@@ -162,3 +173,11 @@ First build downloads AGP/Compose/Shizuku from Google Maven + Maven Central.
|
|||||||
are SUPPORTED on both known devices via `Os.recvmsg` + `StructMsghdr` reflection.
|
are SUPPORTED on both known devices via `Os.recvmsg` + `StructMsghdr` reflection.
|
||||||
3. Fold the confirmed capabilities + Shizuku dump-format samples back into the production
|
3. Fold the confirmed capabilities + Shizuku dump-format samples back into the production
|
||||||
`core-probe` / `core-shizuku` modules.
|
`core-probe` / `core-shizuku` modules.
|
||||||
|
|
||||||
|
## Enrolling a device with a server
|
||||||
|
|
||||||
|
`echolot-app/scripts/enroll-link.sh [note]` mints a §2.1 bootstrap link on fmr over SSH and prints
|
||||||
|
it (plus a QR if `qrencode` is installed, plus the `adb shell am start -a …VIEW -d '<uri>'` command
|
||||||
|
when a device is attached). The link carries a single-use token — treat it as a secret until spent.
|
||||||
|
Never hand-assemble one: the base64 pin needs percent-encoding, and a pin wrong by one character
|
||||||
|
fails as an inscrutable TLS error rather than as a bad pin.
|
||||||
|
|||||||
+749
-3
@@ -211,8 +211,8 @@ Two collection-loop gotchas found while driving the phone over USB:
|
|||||||
2. ~~If `trace.errqueue_reachable` = PARTIAL, add a C-over-JNI errqueue shim.~~ **Retired** —
|
2. ~~If `trace.errqueue_reachable` = PARTIAL, add a C-over-JNI errqueue shim.~~ **Retired** —
|
||||||
SUPPORTED on both known devices; `traceroute.udp4` reads real hops via `Os.recvmsg` +
|
SUPPORTED on both known devices; `traceroute.udp4` reads real hops via `Os.recvmsg` +
|
||||||
`StructMsghdr` reflection, so no `:native` module is needed.
|
`StructMsghdr` reflection, so no `:native` module is needed.
|
||||||
3. Start the Go server skeleton (enrollment + profile + sessions + UDP echo with observation
|
3. ~~Start the Go server skeleton per probe-protocol.md.~~ **Shipped** — live on fmr since
|
||||||
blocks + canary-DNS reference records) per probe-protocol.md.
|
v0.2.0 (2026-07-31); see the server sections below.
|
||||||
4. Fold confirmed capabilities into the production `core-probe` / `core-shizuku` modules.
|
4. Fold confirmed capabilities into the production `core-probe` / `core-shizuku` modules.
|
||||||
|
|
||||||
## Production probe server — LIVE on dedicated VM "fmr" (2026-07-31)
|
## Production probe server — LIVE on dedicated VM "fmr" (2026-07-31)
|
||||||
@@ -222,7 +222,7 @@ SSH only — verified untouched by the daemon (explicit multi-address binds, no
|
|||||||
Control: fmr-1:8443 (SPKI pin `zRV9qkiLnRexAeh4RrSfJzbPWO+U/2Oj2/NVM/KfXlg=`, verified
|
Control: fmr-1:8443 (SPKI pin `zRV9qkiLnRexAeh4RrSfJzbPWO+U/2Oj2/NVM/KfXlg=`, verified
|
||||||
externally over v4+v6). UDP data plane on all four service addresses :8442 — the second IP is
|
externally over v4+v6). UDP data plane on all four service addresses :8442 — the second IP is
|
||||||
the stun-5780 substrate. Daily randomized self-update timer installed (checksum-verified
|
the stun-5780 substrate. Daily randomized self-update timer installed (checksum-verified
|
||||||
against SHA256SUMS; signature verification still TODO before treating the source as untrusted).
|
against SHA256SUMS; signature verification landed 2026-08-02 — see "Release signing" below).
|
||||||
Host config in `/etc/echolot-server.env`. SSH access for sessions: `ssh claude-echolot`.
|
Host config in `/etc/echolot-server.env`. SSH access for sessions: `ssh claude-echolot`.
|
||||||
|
|
||||||
## Server v0.3.0 — STUN + TCP echo + observations + actions (2026-07-31)
|
## Server v0.3.0 — STUN + TCP echo + observations + actions (2026-07-31)
|
||||||
@@ -621,3 +621,749 @@ finding code and the metrics survive, then deleted it.
|
|||||||
- Accounts/OIDC on the server, which is what `uploads=account` is waiting for.
|
- Accounts/OIDC on the server, which is what `uploads=account` is waiting for.
|
||||||
- Nothing in this entry has been exercised on a phone yet — all of it was verified from the PC
|
- Nothing in this entry has been exercised on a phone yet — all of it was verified from the PC
|
||||||
against the live server. On-device verification is the next step.
|
against the live server. On-device verification is the next step.
|
||||||
|
|
||||||
|
### SemVer compatibility windows between app and server (server-v0.5.0 … v0.5.2, 2026-08-01)
|
||||||
|
Both artifacts are SemVer, and each now declares — and enforces — which peer versions it will talk
|
||||||
|
to. Spec: `docs/probe-protocol.md` §8.
|
||||||
|
|
||||||
|
**Two axes, deliberately not conflated.** Release versions are a *proxy* for what actually has to
|
||||||
|
match, so the real thing is checked first:
|
||||||
|
- `protocol_version` — **can** these builds talk. Advertised in the profile; a peer in a different
|
||||||
|
breaking series is refused whatever its release version says. Below 1.0.0 the **minor** is the
|
||||||
|
breaking axis (SemVer §4).
|
||||||
|
- release-version window — **may** they, per policy. `[min, max)`, min inclusive, max exclusive,
|
||||||
|
because the useful bound is always "the version that broke it".
|
||||||
|
|
||||||
|
Bounds sit at breaking boundaries, not at releases, so shipping a patch never requires editing a
|
||||||
|
range. The app requires server `>= 0.4.2` for a stated reason, not caution: earlier multi-homed
|
||||||
|
servers mis-addressed granted sends and the client measured 100 % downstream loss that never
|
||||||
|
happened. Operators override the server side with `ECHOLOT_MIN_APP_VERSION` /
|
||||||
|
`ECHOLOT_MAX_APP_VERSION`; a malformed bound is fatal at startup rather than ignored, so a typo
|
||||||
|
cannot silently disable a restriction.
|
||||||
|
|
||||||
|
Three rules that shaped the implementation:
|
||||||
|
1. **`GET /v1/profile` is never gated.** It is where a refused client learns which version it needs;
|
||||||
|
gating it leaves the user with a network error instead of an answer.
|
||||||
|
2. **An unparseable or absent version is `unknown`, and is allowed.** Dev builds report `dev`, and a
|
||||||
|
client too old to send the header cannot be identified anyway.
|
||||||
|
3. **Refusal is 426 with a body naming both versions and the window**, surfaced client-side as a
|
||||||
|
distinct `VersionRefused` rather than folded into "network error".
|
||||||
|
|
||||||
|
The app's `versionCode` is now derived from its SemVer (`major*1e6 + minor*1e4 + patch*10`) instead
|
||||||
|
of being a second number to remember.
|
||||||
|
|
||||||
|
Verified live against fmr (`LiveCompatTest`): profile advertises the window and stays readable for a
|
||||||
|
refused version; 0.1.0 and 99.0.0 are both refused with actionable messages; 0.2.0 and a missing
|
||||||
|
header are both served.
|
||||||
|
|
||||||
|
One user-visible bug caught in the process: Go's JSON encoder HTML-escapes `<`, `>` and `&` by
|
||||||
|
default, so the refusal reached the client as `needs \u003e= 0.2.0`. Disabled at the encoder (this
|
||||||
|
is an API, not a page), and the client now *parses* the error field instead of pattern-matching it,
|
||||||
|
so it survives whatever a future encoder decides to escape.
|
||||||
|
|
||||||
|
### Enrollment: the server mints the bootstrap link (server-v0.5.3 … v0.5.4, 2026-08-01)
|
||||||
|
Until now a device was configured by hand-typing a control URL, a base64 SPKI pin and a
|
||||||
|
credential. That is the step that goes wrong, and it goes wrong quietly: a pin off by one
|
||||||
|
character does not fail loudly, it just never matches, and surfaces days later as an inscrutable
|
||||||
|
TLS error.
|
||||||
|
|
||||||
|
`POST /admin/enroll-tokens` now returns the whole §2.1 bootstrap link alongside the token, because
|
||||||
|
the server is the only party holding all three parts at once. The app takes it from a paste or an
|
||||||
|
`echolot://enroll` deep link (so a QR scan configures a server in one action) and writes URL, pin
|
||||||
|
and credential **together or not at all** — a half-applied server fails later, somewhere else,
|
||||||
|
with an error pointing at the wrong thing.
|
||||||
|
|
||||||
|
The control URL comes from `ECHOLOT_PUBLIC_URL` (set on fmr to `https://fmr-1.echo-lot.app:8443`),
|
||||||
|
falling back to the first control listen address; a wildcard bind warns rather than emitting a
|
||||||
|
link to `0.0.0.0`.
|
||||||
|
|
||||||
|
**The encoding trap, which is the whole reason this is tested across both languages.** The pin is
|
||||||
|
base64, so it contains `+`, `/` and `=` — each of which means something else in a query string. An
|
||||||
|
unencoded `+` decodes to a space, leaving the pin wrong by exactly one character. Base64 has no
|
||||||
|
spaces, so the parser restores them; that cannot damage a correctly-encoded pin and it rescues
|
||||||
|
every hand-assembled link. `LiveEnrollmentTest` redeems a link the *server* produced, which is the
|
||||||
|
only way to catch a disagreement between the Go assembler and the Kotlin parser — a unit test on
|
||||||
|
either side alone cannot see it. It also asserts the token is refused the second time.
|
||||||
|
|
||||||
|
Also fixed a spec divergence found while reading §2.1: the spec names the field
|
||||||
|
`device_credential`, the first implementation shipped `credential`. The server now sends both and
|
||||||
|
the client prefers the spec's; the alias goes once nothing reads it.
|
||||||
|
|
||||||
|
Two process notes from this round:
|
||||||
|
- An edit to the admin handler silently failed to apply and the endpoint kept returning just the
|
||||||
|
token. Caught by deploying and *looking at the response*, not by trusting a green build.
|
||||||
|
- The live suite is now six tests (`LiveServerTest`, `LiveMeasurement`, `LiveGranted`,
|
||||||
|
`LiveUpload`, `LiveCompat`, `LiveEnrollment`), all green against fmr from the PC with no device.
|
||||||
|
|
||||||
|
### Directional loss: which way is the packet loss? (2026-08-01)
|
||||||
|
A round trip can only report that *something* was lost somewhere, which is the least useful form
|
||||||
|
of the answer — "3 % loss" sends an engineer looking in both directions at once. The server
|
||||||
|
already records every packet it received per sequence number (§6), so the two cases are actually
|
||||||
|
distinguishable, and `train.udp_updown` now reports them separately:
|
||||||
|
|
||||||
|
- sent, never seen by the server → **upstream** loss
|
||||||
|
- seen by the server, reply never arrived → **downstream** loss
|
||||||
|
|
||||||
|
Findings name the direction and say what is *not* implicated, which is half the value:
|
||||||
|
`connectivity.loss_upstream` ("the return path is not implicated: replies came back for everything
|
||||||
|
that arrived"), `connectivity.loss_downstream`, `nat.udp_unreachable_upstream`.
|
||||||
|
|
||||||
|
Two things the implementation gets deliberately right:
|
||||||
|
- **Downstream loss is measured against what reached the server**, not against what was sent.
|
||||||
|
Using "sent" as the denominator counts every upstream loss a second time and overstates the
|
||||||
|
return path. Pinned by a test with loss in both directions at once.
|
||||||
|
- **Per-direction jitter without synchronised clocks.** Absolute one-way delay would need clock
|
||||||
|
sync and we deliberately have none (the two-clock rule). But `server_rx − client_tx` carries a
|
||||||
|
constant unknown offset, and differencing successive samples cancels it — so RFC 3393 one-way
|
||||||
|
delay variation *is* honestly attributable to a direction even though latency is not. A test
|
||||||
|
pins that a 10-second clock offset changes nothing.
|
||||||
|
|
||||||
|
Correlation is by **wire sequence number**, which is not the loop index: the counter is shared
|
||||||
|
with every other packet type on the session, so "the nth echo" is not "sequence n". `ProbeSession`
|
||||||
|
now exposes `lastSeq`, including for a probe that was lost — a lost packet still has a sequence
|
||||||
|
number, and that number is exactly what tells you which way it was lost.
|
||||||
|
|
||||||
|
Live against fmr: 20/20 both ways, and jitter of **0.08 ms upstream vs 0.85 ms downstream** — a
|
||||||
|
tenfold asymmetry that a round-trip measurement cannot see at all.
|
||||||
|
|
||||||
|
10 unit tests on the arithmetic (a wrong denominator here does not crash, it produces a plausible
|
||||||
|
number pointing at the wrong half of the network) plus the live correlation check.
|
||||||
|
|
||||||
|
### frag_send: crafted IP fragments, so *ordering* is testable (server-v0.6.0, 2026-08-01)
|
||||||
|
`big_send` with `df=false` answers one question — do fragments get through. It cannot answer the
|
||||||
|
more interesting one, because the kernel always emits fragments in order, first one first.
|
||||||
|
|
||||||
|
The classic middlebox fault is exactly about that ordering. Only the **first** fragment carries the
|
||||||
|
UDP header, and therefore the ports; a stateful firewall or NAT that has not seen it has no flow to
|
||||||
|
match the rest against, and many simply drop them. That is invisible to every in-order test, and in
|
||||||
|
the field it looks like "large DNS answers fail on this network" or "the tunnel breaks when the MTU
|
||||||
|
drops" — it works until the network reorders, then fails intermittently, which is the hardest kind
|
||||||
|
of fault to chase.
|
||||||
|
|
||||||
|
So the server builds the fragments itself (raw socket, `IP_HDRINCL`) and controls their order:
|
||||||
|
`in_order` (baseline), `reversed` (last fragment first), `first_last` (first fragment held back
|
||||||
|
250 ms). The datagram is assembled and **signed whole** before being cut up, so what the client
|
||||||
|
reassembles is indistinguishable from an ordinary packet — otherwise the test would be measuring
|
||||||
|
our sender rather than the path. New test type `mtu.frag_ordering`; findings
|
||||||
|
`mtu.fragments_blocked` and `mtu.fragment_reorder_sensitive`.
|
||||||
|
|
||||||
|
Two details that would otherwise produce confidently wrong answers:
|
||||||
|
- **The UDP checksum is computed, not left zero.** Zero is legal in IPv4 and would be less code,
|
||||||
|
but zero-checksum datagrams are dropped by some middleboxes — and that drop would be recorded as
|
||||||
|
a fragmentation failure, which is the wrong conclusion entirely.
|
||||||
|
- **Fragment offsets are in 8-byte units**, so non-final fragments are rounded down to a multiple
|
||||||
|
of 8. A 100-byte fragment is not an error; it is a datagram no host will ever reassemble.
|
||||||
|
|
||||||
|
`frag-send` is advertised only when a raw socket can actually be opened — checked by opening one,
|
||||||
|
because a permission model has more ways to say no (userns, seccomp, LSM) than a capability bit has
|
||||||
|
to say yes. fmr runs as root with `cap_net_raw` in its bounding set, so it is available there.
|
||||||
|
|
||||||
|
Fragment ordering runs only after `mtu.frag_delivery` shows fragments arrive at all; otherwise the
|
||||||
|
three orderings would each report "not delivered" and read as three faults instead of one.
|
||||||
|
|
||||||
|
The header arithmetic is unit-tested (reassembly coverage with no gaps or double-delivery, MF
|
||||||
|
flags, shared IP ID, 8-byte offsets, checksum verification over odd and even lengths). Because the
|
||||||
|
code is `//go:build linux`, the tests are **cross-compiled and run on fmr** — there is no Go
|
||||||
|
toolchain there, so `go test -c` plus scp is the loop.
|
||||||
|
|
||||||
|
Live against fmr: 4 fragments per burst, and all three orderings reassembled — a healthy path, and
|
||||||
|
the baseline against which a mobile network will be interesting.
|
||||||
|
|
||||||
|
### Testing state (2026-08-01)
|
||||||
|
Six live tests against fmr, all green, no device involved: `LiveServerTest`, `LiveMeasurement`,
|
||||||
|
`LiveGranted`, `LiveDownstream`, `LiveUpload`, `LiveCompat`, `LiveEnrollment`. Plus 74 client unit
|
||||||
|
tests and the full Go suite. Everything in the last several entries is verified from the PC; the
|
||||||
|
app's UI (settings, history, deep-link enrollment) and `mtu.pmtud_up` remain device-only.
|
||||||
|
|
||||||
|
### throughput: a rate, plus the qualifier that makes it a measurement (server-v0.6.1 … v0.6.2)
|
||||||
|
A throughput test reports the *smallest* limit on the path — and the sender's own ceiling is one of
|
||||||
|
the candidates. If the server is asked for 50 Mbps and 50 Mbps arrives, the network was never the
|
||||||
|
constraint and "50 Mbps" says nothing about it. So `perf.throughput_udp` always carries
|
||||||
|
`limited_by` (duration | budget | rate | send_error) and `measures_network`, and a finding is
|
||||||
|
raised only when the path is actually implicated. The live run against fmr reports 20 Mbit/s with
|
||||||
|
`measures_network: false`, which is the correct and useful answer.
|
||||||
|
|
||||||
|
Loss is computed against the **sender's own count**, fetched from the observations API, not against
|
||||||
|
the requested rate. A receiver alone cannot tell "the network dropped it" from "the sender never
|
||||||
|
sent it", and guessing turns a healthy server-side limit into a phantom network fault. The server
|
||||||
|
keeps one summary per action rather than per-packet records — a ten-second run at 50 Mbps is half a
|
||||||
|
million packets, and a struct each would turn a measurement into memory exhaustion.
|
||||||
|
|
||||||
|
Sending is **paced**, on an absolute schedule. Unpaced would measure the server's NIC and the first
|
||||||
|
queue it meets, then collapse into loss that reads as a network fault; sleep-per-packet would
|
||||||
|
accumulate scheduler error and drift the rate down over a ten-second run.
|
||||||
|
|
||||||
|
Throughput gets its own grant budget sized from the request, so every *other* action stays bounded
|
||||||
|
at 8 MiB. When the byte cap binds before the clock does, the **duration is shortened and reported**
|
||||||
|
rather than the run being truncated: promising thirty seconds and delivering twenty-one is the same
|
||||||
|
information with a surprise attached, and it keeps "the clock ended the run" as the normal case —
|
||||||
|
the only case where the rate is a clean property of the path. That behaviour came out of a test
|
||||||
|
that failed honestly (30 s at 100 Mbps needs 375 MB against a 256 MB cap).
|
||||||
|
|
||||||
|
It is **opt-in** in the run config, default off. A 5-second run at 50 Mbps moves ~30 MB; on a
|
||||||
|
metered mobile connection that is the user's money, and a tool that spends it without being asked
|
||||||
|
is not one people keep installed.
|
||||||
|
|
||||||
|
#### The bug the live test found
|
||||||
|
The first live run delivered 104 packets and stopped after 50 ms. The grant's rate check exempted
|
||||||
|
the first 50 ms entirely, meaning to be lenient at startup — the effect was the opposite. A sender
|
||||||
|
could dump an unbounded burst into that free window, and the instant the check switched on it
|
||||||
|
compared those bytes against 50 ms worth of allowance and refused everything until real time caught
|
||||||
|
up. **Every short test passed** (downtrain sends 50 packets, big_send seven); every sustained send
|
||||||
|
died fifty milliseconds in.
|
||||||
|
|
||||||
|
Replaced with a token bucket (`allowance = burst + rate × elapsed`), which is smooth from t=0.
|
||||||
|
The burst is 100 ms of the allowed rate, floored at one ordinary datagram — deliberately one, since
|
||||||
|
at 8 kbps a 64 KB floor is sixty-four seconds' worth, exactly the instant dump the ceiling exists to
|
||||||
|
prevent. The pre-existing rate test caught that when I first tried the generous floor, and it was
|
||||||
|
right to. Second half of the same bug: callers treated *any* refusal as terminal, so `TryAllow` now
|
||||||
|
says why — a sender paces through a transient "too fast just now" and still stops dead on a spent
|
||||||
|
budget or an expired grant. Both halves are pinned by regression tests.
|
||||||
|
|
||||||
|
### Findings registry (2026-08-01)
|
||||||
|
Closes open item 1 of measurement-schema.md §9. A finding code is the stable, machine-readable half
|
||||||
|
of a result — what a dashboard groups by and what someone greps a year of archived runs for — and
|
||||||
|
that only holds if a code means exactly one thing forever. Ad-hoc string literals at fifteen call
|
||||||
|
sites cannot promise that, and by the time the registry was written the failure had already
|
||||||
|
happened.
|
||||||
|
|
||||||
|
**Two emitters had independently produced `connectivity.downstream_loss` and
|
||||||
|
`connectivity.loss_downstream` for the same claim**, and nothing anywhere objected. Anyone
|
||||||
|
aggregating either one would have silently seen half their data. Merged into
|
||||||
|
`connectivity.loss_downstream`, paired with `loss_upstream` so the two directions read as a set.
|
||||||
|
|
||||||
|
**Two codes were also renamed out of `nat.*`.** `nat.udp_unreachable` is not about NAT — it means
|
||||||
|
no replies came back — but the prefix determines the category, and the category determines which
|
||||||
|
verdict light the finding rolls up into (§7.3). A `nat.*` code landing under *connectivity* is not
|
||||||
|
a naming quibble; it changes which light turns red. Cheap to fix now, a breaking change later.
|
||||||
|
|
||||||
|
Codes are now declared as typed `FindingSpec`s carrying their category and default severity, and
|
||||||
|
emitters reference the spec instead of retyping the string — so a typo is a compile error and two
|
||||||
|
call sites cannot disagree about a finding's category.
|
||||||
|
|
||||||
|
`docs/findings-registry.md` is the contract, and a test reads it: it fails when the document and
|
||||||
|
the registry have codes the other lacks, or when a severity differs. Documentation that drifts from
|
||||||
|
its implementation is worse than none, because it still looks authoritative. The check scopes
|
||||||
|
itself to table rows, so the prose can keep explaining which codes were retired and why.
|
||||||
|
|
||||||
|
Six tests: uniqueness, declared-vs-listed, prefix↔category agreement, naming convention, a
|
||||||
|
word-order-anagram check (the shape the duplication actually took), and the document agreement.
|
||||||
|
|
||||||
|
### A real privacy leak, found by starting on the machine-readable schema (2026-08-01)
|
||||||
|
The intent was `measurement.schema.json` (§8's promised companion). The first step — checking
|
||||||
|
whether the anonymizer actually covers the fields the schema declares as sensitive — found that it
|
||||||
|
did not, so that became the work.
|
||||||
|
|
||||||
|
**At the `balanced` level, five identifying values were being uploaded verbatim:**
|
||||||
|
|
||||||
|
| value | field | why it matters |
|
||||||
|
|---|---|---|
|
||||||
|
| `2001:…::150` | `networks[].link.addresses[].addr` | the device's own global IPv6 address — a strong, geolocatable device identifier |
|
||||||
|
| `2a02:…::1` | `networks[].link.routes[].gateway` | identifies the ISP allocation |
|
||||||
|
| `203.0.113.77` | `networks[].link.dns.servers[]` | the configured resolver |
|
||||||
|
| `nas.example.lan` | `private_dns_hostname` | an internal hostname |
|
||||||
|
| `example.lan` | `search_domains[]` | the internal domain |
|
||||||
|
|
||||||
|
The settings screen describes that level as pseudonymizing addresses. It was not.
|
||||||
|
|
||||||
|
**Root cause:** classification keyed on field *names*, and the schema's actual names (`addr`,
|
||||||
|
`gateway`, `dst`, `servers`, `search_domains`, `private_dns_hostname`) had never been added to the
|
||||||
|
table. Not a subtle bug — just an unfalsifiable design. The existing tests all passed, because each
|
||||||
|
one checked a field somebody had remembered to write a case for.
|
||||||
|
|
||||||
|
**Two fixes, one of them structural:**
|
||||||
|
1. The missing names were added.
|
||||||
|
2. More importantly, a **shape-based backstop**: when a field name is unrecognised, the *value* is
|
||||||
|
inspected, and anything shaped like an IPv4/IPv6 address or a MAC is treated as one. A name
|
||||||
|
table can only protect fields someone thought of, which is precisely the wrong property for a
|
||||||
|
privacy control. Hostnames are deliberately *not* inferred by shape — `train.udp_updown` is
|
||||||
|
indistinguishable from a domain, and mangling a test type would corrupt the document to protect
|
||||||
|
nothing.
|
||||||
|
|
||||||
|
`LeakTest` is the new guard and is written to fail for fields nobody has considered: it plants
|
||||||
|
identifying values wherever one can actually occur and asserts none survive, rather than checking
|
||||||
|
a list of known cases. It also pins that RFC1918 addresses still come through readable, so the
|
||||||
|
test cannot pass by over-redacting everything.
|
||||||
|
|
||||||
|
Route prefixes and the unspecified address needed care in the transform: `0.0.0.0/0` and `::/0`
|
||||||
|
must stay themselves, or a routing table becomes unreadable for no privacy gain.
|
||||||
|
|
||||||
|
**Still outstanding:** `measurement.schema.json` itself. Worth noting what this episode implies for
|
||||||
|
it — much of a document's payload lives in `evidence`/`metrics`/`params`, which are per-test-type
|
||||||
|
`JsonObject` by design and therefore *outside* any schema. A schema-driven anonymizer would have
|
||||||
|
less coverage there than the name-plus-shape one now does, so the schema should be built for
|
||||||
|
validation and external tooling, not as a replacement for the classifier.
|
||||||
|
|
||||||
|
### ULA prefixes are pseudonymized whole (2026-08-01)
|
||||||
|
Spotted in a real uploaded run from the phone: the server had
|
||||||
|
`fda1:3fb1:ff92:6696::2662` for a DNS server. The general IPv6 path preserves the leading two
|
||||||
|
groups (deliberately — for a global address that keeps the ISP allocation, which is the
|
||||||
|
diagnostically useful part), and for a ULA that passed through **32 of the 40 random bits** of the
|
||||||
|
global ID.
|
||||||
|
|
||||||
|
ULA looks like the v6 equivalent of RFC1918 and the instinct is to treat it the same. That
|
||||||
|
reasoning does not carry over, and the difference is the whole point: an RFC1918 prefix is shared
|
||||||
|
by millions of networks and identifies none of them, while a ULA global ID is random and unique to
|
||||||
|
one network by construction (RFC 4193). The prefix *is* the identifier — it is a network
|
||||||
|
fingerprint that was surviving redaction.
|
||||||
|
|
||||||
|
Now pseudonymized as a unit, so two addresses on the same ULA subnet still land on the same
|
||||||
|
pseudonymous prefix: "these hosts are on one network" survives, "this is *that* network" does not.
|
||||||
|
Three tests, one of which uses the exact value observed on the wire.
|
||||||
|
|
||||||
|
Worth recording as a reasoning trap: I had originally raised this as "ULA should probably be kept
|
||||||
|
verbatim, like RFC1918, for consistency". The surface analogy pointed the wrong way, and the
|
||||||
|
correct answer was the opposite.
|
||||||
|
|
||||||
|
### Registry adopted everywhere; v6 findings renamed; Back works (2026-08-01)
|
||||||
|
The findings registry was only adopted in `core-engine`. The app module still emitted seven codes
|
||||||
|
as raw strings, so the registry test passed while codes existed outside it — including
|
||||||
|
`ipv6.broken`, which fired on a real network and was in no registry at all.
|
||||||
|
|
||||||
|
All seven now reference registry entries for code, category and severity, so those three cannot
|
||||||
|
disagree at a call site. A grep for `code = "…"` across the app, engine and probe modules returns
|
||||||
|
nothing.
|
||||||
|
|
||||||
|
**`ipv6.*` → `v6.*`.** The third instance of rule 1: they declared `Category.IPV6` while the prefix
|
||||||
|
map only knows `v6`, so `TestType.category("ipv6.broken")` fell through to *connectivity* and the
|
||||||
|
finding rolled up under the wrong verdict light. The test-type registry already used `v6.`.
|
||||||
|
|
||||||
|
Two severities reconciled while merging:
|
||||||
|
- `connectivity.captive_portal` is **medium**, not high. The registry had guessed high; the probe
|
||||||
|
that emits it had always said medium, and the probe was the considered value — a captive portal
|
||||||
|
on hotel wifi is what should be there, and logging in clears it. `connectivity.no_internet` is
|
||||||
|
the high one, because nothing the user does locally fixes that.
|
||||||
|
- `v6.not_offered` is **info, and the registry says it must stay info**. Most networks still do not
|
||||||
|
offer IPv6 and that is not a fault; a warning here lights a yellow verdict on a healthy network,
|
||||||
|
which teaches people to ignore the light.
|
||||||
|
|
||||||
|
Also: a `BackHandler` now returns from Settings/History to the run screen. The screen was a plain
|
||||||
|
state variable with nothing connecting it to the back stack, so the system Back gesture left the
|
||||||
|
app entirely. Enabled only when there is somewhere to go back to, so Back still exits from the run
|
||||||
|
screen.
|
||||||
|
|
||||||
|
### Upstream throughput (server-v0.6.3, 2026-08-01)
|
||||||
|
The mirror of the downstream case: the client generates the traffic and the server counts it. No
|
||||||
|
grant is involved — the client is sending its own packets, so there is nothing to amplify — but it
|
||||||
|
does need the server's tally, because **only the far end knows how much arrived**. Without that
|
||||||
|
number a sender measures how fast it can *transmit*, which is usually just the speed of the local
|
||||||
|
NIC and is a different question from the one being asked.
|
||||||
|
|
||||||
|
`TYPE_THROUGHPUT_UP` (0x0F) is counted and deliberately **never answered**: a reply would double
|
||||||
|
the traffic and drag the return path into a measurement that is specifically about the outbound
|
||||||
|
one.
|
||||||
|
|
||||||
|
The tally is a counter, not a list, and short-circuits **before** the observation log. A
|
||||||
|
five-second run at 20 Mbps is around ten thousand packets; one struct each would turn a
|
||||||
|
measurement into an allocation storm on a shared server, and nothing needs the per-packet detail
|
||||||
|
since the client holds the send-side record. The gap between the two counts is the loss.
|
||||||
|
|
||||||
|
`direction=up` on the throughput action sends nothing — it zeroes the counter, so a second run in
|
||||||
|
one session measures itself rather than inheriting the first one's packets. The live test asserts
|
||||||
|
`received <= sent`, which is what catches a counter that was never reset.
|
||||||
|
|
||||||
|
Live against fmr: **3125 sent, 3125 counted, 0 % loss, 10.0 Mbit/s** at a 10 Mbit/s request, with
|
||||||
|
`measures_network: false` — correct, since what arrived matched what was offered, so the path was
|
||||||
|
never the constraint.
|
||||||
|
|
||||||
|
### Raw shell dumps leaked the whole LAN (2026-08-01)
|
||||||
|
Found by running the Shizuku shell tier for the first time. The tier works — `tiers.shizuku: true`,
|
||||||
|
`exec_path: UserService` (so the UserService binds on the OnePlus, as recorded), `runs_as
|
||||||
|
shell(2000)`, 7/7 commands — and the run promptly uploaded **every MAC address on the local
|
||||||
|
network** to fmr at the `balanced` level: router, phones, whatever else was on the wifi. Fourteen
|
||||||
|
of them.
|
||||||
|
|
||||||
|
The probes embed raw command output verbatim (`ip neigh`, `ip route`, `id`), which is genuinely
|
||||||
|
good evidence and also a complete household device inventory. The anonymizer could not see it:
|
||||||
|
classification is by field name and by whole-value shape, and `ip_neigh` is one long string that is
|
||||||
|
itself neither a MAC nor an address. measurement-schema.md §9 item 2 had flagged raw dumps as "hard
|
||||||
|
to anonymize" and proposed dropping them from exports; nothing enforced either.
|
||||||
|
|
||||||
|
**Scrubbing beats dropping.** Identifiers inside any unclassified string are now replaced in place,
|
||||||
|
using the same pseudonyms as everywhere else — so a MAC that appears both in a parsed field and in
|
||||||
|
a raw dump still reads as one device. The dump stays readable and auditable: you can still see the
|
||||||
|
neighbour table's shape, the host count, RFC1918 addresses and vendor prefixes. Dropping the
|
||||||
|
evidence would have protected the same data while destroying the reason for collecting it.
|
||||||
|
|
||||||
|
Two implementation notes worth keeping:
|
||||||
|
- **One pass, not three.** Sequential passes re-process their own output: once a MAC became
|
||||||
|
`78:9a:18:xx:yy:zz`, the IPv6 pattern matched it — six hex groups separated by colons *is* an
|
||||||
|
address — and destroyed the vendor prefix the MAC rule had just preserved. Ordered alternation
|
||||||
|
resolves each position once, MAC first.
|
||||||
|
- The patterns are conservative on purpose. A missed address gets caught by another rule or not at
|
||||||
|
all; an over-eager one mangles timestamps and version strings, corrupting evidence to protect
|
||||||
|
nothing.
|
||||||
|
|
||||||
|
`RealDocumentTest` runs the anonymizer over a captured run when `ECHOLOT_REAL_RUN` points at one,
|
||||||
|
and fails on any MAC that survives. It self-skips otherwise, so no one's network is committed to the
|
||||||
|
repo. Against the actual leaked document: **14 MACs in, 0 surviving.**
|
||||||
|
|
||||||
|
Also fixed: the Settings *Preview what an upload would send* button did nothing. It read
|
||||||
|
`UiState.history`, which is empty until the History screen has been opened — the same root cause as
|
||||||
|
the "0 run(s)" count. It now reads the archive directly, and says so when there is nothing to
|
||||||
|
preview rather than silently ignoring the tap.
|
||||||
|
|
||||||
|
### Security: the admin listener was publicly exposed for ~15 minutes (2026-08-01)
|
||||||
|
Moving the admin listener to `[::2]:443` for the UI exposed `/admin/enroll-tokens` and
|
||||||
|
`/admin/selftest` to the internet **with no authentication**. Anyone who could reach
|
||||||
|
`fmr.echo-lot.app` could mint enrolment tokens.
|
||||||
|
|
||||||
|
The listener was designed localhost-only — its own flag help says *"keep localhost"* — and that
|
||||||
|
assumption travelled with it when the address changed. The compounding error: `checkAdminExposure`,
|
||||||
|
added the same day, verifies **encryption** and says nothing about **authentication**. It passed,
|
||||||
|
and a green light on an adjacent property is worse than no check, because it invites you to stop
|
||||||
|
looking.
|
||||||
|
|
||||||
|
Closed by returning to loopback (the TLS and ACME work is retained, just not exposed). All 68 device
|
||||||
|
enrolments matched the timestamps of test runs, so there is no evidence of abuse — but the window
|
||||||
|
existed on a freshly published hostname and absence cannot be proven. 39 unused enrolment tokens
|
||||||
|
were purged, since any could have been minted by someone else and they cost nothing to replace, and
|
||||||
|
63 test devices removed.
|
||||||
|
|
||||||
|
**The admin listener does not become reachable again until it authenticates.** That reorders the UI
|
||||||
|
work: auth on the listener first, everything else after.
|
||||||
|
|
||||||
|
### Open: encrypted uploads, where the operator cannot read the data
|
||||||
|
Not built. Recorded because the shape is decided by a few early choices, and the current design
|
||||||
|
happens to leave the door open.
|
||||||
|
|
||||||
|
The goal: hand someone an account, let them upload, and be unable to read what they uploaded.
|
||||||
|
|
||||||
|
Sketch: a random per-account **master key**, generated on the first device and wrapped under a
|
||||||
|
key derived from a passphrase (PBKDF2-HMAC-SHA256 — stdlib on both sides). The wrapped key is
|
||||||
|
stored server-side as an opaque blob, so a new device signs in, fetches it, and unwraps locally;
|
||||||
|
the server never sees either key. Runs are encrypted client-side with AES-256-GCM, fresh nonce per
|
||||||
|
run. All of this is stdlib in Go and `javax.crypto` in Kotlin — no dependency either side.
|
||||||
|
|
||||||
|
Four consequences that decide whether it is worth it:
|
||||||
|
|
||||||
|
1. **What stays readable determines what the UI can do.** The server builds its index by *parsing*
|
||||||
|
the document — verdict, finding count, started_at. An opaque payload means the client supplies
|
||||||
|
that metadata or the index disappears, and with it retention-by-verdict and any "runs with
|
||||||
|
findings" view. The honest version supplies only run id, timestamp and size, and moves the rest
|
||||||
|
client-side.
|
||||||
|
2. **Lose the passphrase, lose the data.** That is the feature working, and also the support
|
||||||
|
burden. It needs a recovery code printed at setup, not a reset flow — there is nothing to reset.
|
||||||
|
3. **Metadata is not hidden.** The operator still sees which account uploaded, when, how often and
|
||||||
|
how large. "Cannot see it" is about content, not existence, and saying otherwise would oversell.
|
||||||
|
4. **It makes `min_anonymization` unenforceable** — a server cannot check a level it cannot read.
|
||||||
|
That is not a conflict so much as a redundancy: the anonymization floor exists to protect the
|
||||||
|
user from the operator, and encryption does that better. The two should not both be demanded of
|
||||||
|
one upload.
|
||||||
|
|
||||||
|
What keeps this possible: uploads are already stored byte-for-byte as received, and every index
|
||||||
|
field is derived in one function (`runs.Put`). The thing to avoid is admin features that *require*
|
||||||
|
reading content — those would have to be unbuilt later.
|
||||||
|
|
||||||
|
### App sign-in, and an undisclosed dependency it surfaced (2026-08-01)
|
||||||
|
The app can now sign in to the server's identity provider: authorization code with PKCE, a
|
||||||
|
`Sign in` card in settings, and the `echolot://auth` redirect handled alongside the enrolment one
|
||||||
|
(told apart by host, since one spends a token and the other completes an authorization).
|
||||||
|
|
||||||
|
The detail that decides whether this works on a real phone: **the PKCE verifier is written to
|
||||||
|
storage before the browser opens**, not held in memory. Handing control to a browser backgrounds
|
||||||
|
the process and Android may kill it; the callback then arrives at a fresh process. An in-memory
|
||||||
|
verifier works on a developer's device and fails under memory pressure, which is the worst way for
|
||||||
|
a sign-in to break.
|
||||||
|
|
||||||
|
Nothing from the IdP is retained. The ID token proves who is signing in, once, and the device
|
||||||
|
credential authenticates everything after — no access tokens stored, no refresh tokens rotated.
|
||||||
|
|
||||||
|
**A server remains entirely optional.** All eight probes are device-tier; `serverConfigured` gates
|
||||||
|
only upload and the account. But answering that question exposed something worth fixing: two probes
|
||||||
|
hardcode the reference deployment —
|
||||||
|
|
||||||
|
```kotlin
|
||||||
|
DnsCanaryProbe(canaryZone = "c.echo-lot.app", ...) // "Hardcoded to the reference deployment"
|
||||||
|
StunProbe(serverHost = "fmr-1.echo-lot.app")
|
||||||
|
```
|
||||||
|
|
||||||
|
so a user with no server of their own still sends DNS and STUN traffic to fmr without being told.
|
||||||
|
For a tool that goes to this much trouble over what leaves the device, an undisclosed dependency on
|
||||||
|
a third party's infrastructure is the wrong default. It should prefer the configured server, and be
|
||||||
|
explicit when there is none. **Closed** — both probes take the enrolled server from settings
|
||||||
|
(canary zone learned from the profile, cleared on re-enroll) and report themselves SKIPPED with
|
||||||
|
the reason when none is configured.
|
||||||
|
|
||||||
|
### v6.broken was a false positive waiting to happen (2026-08-01)
|
||||||
|
A phone could not open `https://fmr.echo-lot.app` while loading the same server by IP literal
|
||||||
|
perfectly well. Two things came out of chasing it.
|
||||||
|
|
||||||
|
**The admin UI is IPv6-only, by consequence rather than intent.** `fmr.echo-lot.app` has an AAAA
|
||||||
|
and no A record — verified identical at Cloudflare, Google and Quad9, so DNS itself is healthy.
|
||||||
|
That follows from reserving all four measurement addresses for testing, which left only `::2` for
|
||||||
|
management, and `::2` has no IPv4 counterpart. Any client without working IPv6 sees an unreachable
|
||||||
|
admin interface — a poor property for the interface you reach *from the networks you are debugging*.
|
||||||
|
|
||||||
|
**And the app's own `v6.broken` finding was unsound.** It fired on exactly one signal — ICMPv6 echo
|
||||||
|
getting no reply — with `Confidence.HIGH`. ICMPv6 echo is widely filtered on networks where IPv6
|
||||||
|
works fine, which is precisely what that phone demonstrated: no ICMPv6 replies, working IPv6 TCP.
|
||||||
|
The finding asserted a cause it had no evidence for, which is the same class of error as the
|
||||||
|
multi-homed `100 % downstream loss` earlier: a confident measurement of something that was not
|
||||||
|
happening.
|
||||||
|
|
||||||
|
Now `v6.no_icmp_reply`, severity low, confidence medium, and the text names *both* explanations
|
||||||
|
instead of choosing one. It is still worth reporting, because filtered ICMPv6 breaks Path MTU
|
||||||
|
Discovery — large packets vanish rather than being reported as too big — which is a real fault even
|
||||||
|
when IPv6 works.
|
||||||
|
|
||||||
|
The proper fix is corroboration: attempt a real IPv6 connection and only call it broken when that
|
||||||
|
fails too. That needs a target, which runs into the hardcoded-reference-deployment issue already
|
||||||
|
open above. **Both closed 2026-08-02** — see "Corroborated IPv6 findings" below.
|
||||||
|
|
||||||
|
## Per-network probing is blocked while a VPN is up (2026-08-01)
|
||||||
|
|
||||||
|
`Network.bindSocket()` fails with `EPERM` for every underlying network when a VPN holds the
|
||||||
|
default route — verified on the OnePlus 15 with Netbird active: `Binding socket to network 101
|
||||||
|
failed: EPERM` for both cellular and wifi. This is Android preventing VPN leaks, not a bug to work
|
||||||
|
around, and it means the whole per-network measurement approach is unavailable to any user with a
|
||||||
|
VPN connected. Worth deciding deliberately rather than discovering per report:
|
||||||
|
|
||||||
|
- The run currently succeeds and simply measures nothing per network. Honest, but silent — the
|
||||||
|
document records `attempted: false` and the UI says green.
|
||||||
|
- A user with a corporate VPN permanently on would get a green run that measured almost nothing.
|
||||||
|
|
||||||
|
Options are to detect the VPN and say so plainly ("this network cannot be measured while a VPN is
|
||||||
|
active"), to measure the tunnel itself as the network under test, or both. **Decided and built
|
||||||
|
2026-08-02**: say so plainly, everywhere the run is read — see "Constrained runs" below.
|
||||||
|
|
||||||
|
Related: `icmp.ping6` now records `attempted` alongside `ok` per network, because collapsing them
|
||||||
|
made the app report "IPv6 is configured, but ICMPv6 gets no reply" about an interface it had never
|
||||||
|
succeeded in sending on — a claim about the user's carrier with no evidence behind it.
|
||||||
|
|
||||||
|
## Reserved measurement addresses, and the web UI on both families (2026-08-01)
|
||||||
|
|
||||||
|
fmr has two IPv4 (.150/.151) and three IPv6 (::150/::151/::2) addresses. `.150`/`::150` now carry
|
||||||
|
the services; `.151`/`::151` are reserved for measurement, declared in `ECHOLOT_RESERVED_ADDRS`.
|
||||||
|
|
||||||
|
Reserved does **not** mean silent. The UDP data plane, the canary DNS and STUN's RFC 5780 alternate
|
||||||
|
all belong there — reserving an address and then forbidding the measurements that need it would
|
||||||
|
defeat the purpose. What must never appear is a service, and above all not ports 80 or 443: a
|
||||||
|
handshake completing on a port known not to be listening is what proves interception, and that
|
||||||
|
proof survives exactly as long as nothing binds those ports. `config.CheckReserved` enforces it at
|
||||||
|
startup, refusing wildcard binds outright (every listener defaults to `:port`, so the next one added
|
||||||
|
will claim reserved addresses without anyone deciding to).
|
||||||
|
|
||||||
|
The first version of the guard was too strict and the live config caught it: it would have refused
|
||||||
|
the existing UDP and DNS binds on `.151`. The rule is about services and web ports, not about
|
||||||
|
listening at all.
|
||||||
|
|
||||||
|
**The adb-beacon receiver was wildcard-bound to `0.0.0.0:443`**, occupying port 443 on every IPv4
|
||||||
|
address including the reserved one — so the IPv4 interception test had been compromised for as long
|
||||||
|
as it had been running, silently. It is now `systemctl disable --now echolot-adb-beacon`; restore
|
||||||
|
with `systemctl enable --now`. Note what this implies: the guard covers this server's own listeners,
|
||||||
|
and a stray process outside its config can still pollute a reserved address. A startup probe that
|
||||||
|
*verifies* 80/443 are actually free on the reserved addresses would be a stronger guarantee than
|
||||||
|
checking our own configuration — built 2026-08-02 (`selftest.ReservedWebPortsFree`, fatal at
|
||||||
|
startup when anything is listening there).
|
||||||
|
|
||||||
|
The admin UI and the ACME responder now take comma-separated addresses like every other listener;
|
||||||
|
they were single-address, which is why the UI could only ever live on `::2`. It serves on `.150:443` and
|
||||||
|
`[::150]:443`; sshd on `.150:2322` and `[::150]:2322`.
|
||||||
|
|
||||||
|
`::2` is gone entirely — unbound, then removed from `/etc/systemd/network/ext.network`. The
|
||||||
|
transition kept it bound throughout and dropped it only after the CNAME landed, because removing it
|
||||||
|
first would have broken both the UI and ACME renewal for the very name the certificate is issued
|
||||||
|
to. Listeners came off before the address did, in that order, or the services would have failed to
|
||||||
|
bind on restart.
|
||||||
|
|
||||||
|
Verified after a full reboot: `fmr.echo-lot.app` answers 200 over both families, `.151`/`::151` are
|
||||||
|
closed on 80 and 443, canary DNS is still up on `.151`, and neither `::2` nor the beacon returns.
|
||||||
|
(`echolot-server` is `After=network-online.target` with `Restart=on-failure`, which is what makes
|
||||||
|
binding specific addresses safe across a boot — a wildcard bind would not have needed it, and that
|
||||||
|
is the trade for the reserved addresses being meaningful.)
|
||||||
|
|
||||||
|
The point of all this: `fmr.echo-lot.app` gained an A record, so the server stopped being reachable
|
||||||
|
only over IPv6 — which is what made it unreachable from a phone with no working IPv6, presenting as
|
||||||
|
"this host does not exist" in two different browsers.
|
||||||
|
|
||||||
|
### If the beacon comes back, it belongs in the web UI
|
||||||
|
|
||||||
|
Not as a separate listener. The receiver being its own Python service on `0.0.0.0:443` is exactly
|
||||||
|
what silently compromised the reserved address, and a second process racing for a port is a
|
||||||
|
recurring problem rather than a one-off: whoever loses the race simply fails to start, and on a
|
||||||
|
reboot which one that is comes down to unit ordering.
|
||||||
|
|
||||||
|
Folding it in costs little and settles several things at once. It would be two routes on the admin
|
||||||
|
UI (`POST` the observed adb port, `GET /apk` for the staged build), behind the TLS the UI already
|
||||||
|
terminates and the certificate it already renews, with no extra port and no wildcard. It also gets
|
||||||
|
authentication for free — the current receiver accepts a port report from anyone who can reach it,
|
||||||
|
which is tolerable for a dev tool on a trusted network and not something to keep once it lives
|
||||||
|
beside an admin session.
|
||||||
|
|
||||||
|
The one thing that changes on the device side is that the POST becomes HTTPS. That is a real
|
||||||
|
certificate rather than a self-signed one, so it costs a URL scheme rather than any trust plumbing.
|
||||||
|
|
||||||
|
## The control plane shares port 443 (2026-08-01)
|
||||||
|
|
||||||
|
`fmr-1.echo-lot.app:443` is the control plane, `fmr.echo-lot.app:443` the admin UI, both on
|
||||||
|
`.150`/`::150`, one listener, selected by SNI for the certificate and by `Host` for the handler.
|
||||||
|
|
||||||
|
The reason is not tidiness, it is reachability. Captive portals, hotel wifi and corporate firewalls
|
||||||
|
routinely permit only 80 and 443 — which is exactly the population of networks this tool exists to
|
||||||
|
diagnose. A control plane on 8443 is unreachable precisely when it matters most, and it fails as
|
||||||
|
"cannot reach server", which tells the user nothing.
|
||||||
|
|
||||||
|
They cannot share a certificate, which is why this needs two names. The control plane is trusted by
|
||||||
|
SPKI pin and so uses a long-lived self-signed certificate; a browser needs one a CA vouches for.
|
||||||
|
One name on one port is one certificate, so the port can only be shared by splitting the names.
|
||||||
|
Pinning the Let's Encrypt key instead was considered and rejected: it survives renewal only while
|
||||||
|
key reuse holds, so a routine key rotation would brick the whole fleet.
|
||||||
|
|
||||||
|
Verified per SNI on 443: `fmr.echo-lot.app` serves `issuer=Let's Encrypt`, `fmr-1.echo-lot.app`
|
||||||
|
serves the self-signed cert whose pin is unchanged (`zRV9…Xlg=`), `/v1/profile` answers 401 on the
|
||||||
|
control name and 303 to the login page on the UI name.
|
||||||
|
|
||||||
|
**8443 stays open.** Devices enrolled before this carry that URL in their settings, and closing it
|
||||||
|
for the sake of a port number would strand every one of them. It can go once no enrolled device
|
||||||
|
still points at it — not before.
|
||||||
|
|
||||||
|
The rule from the naming change still binds: `fmr` may be a CNAME to exactly one host and never a
|
||||||
|
multi-address record, because a pinned client that reaches a different key does not fail over.
|
||||||
|
|
||||||
|
## Constrained runs: a VPN'd run now says so, everywhere (2026-08-02, app 0.2.1)
|
||||||
|
|
||||||
|
The measurement schema gained a top-level `constraints` block (§3) and the app now fills it.
|
||||||
|
`ConstraintDetector` (core-probe) runs before any probe: one throwaway `Network.bindSocket()` per
|
||||||
|
non-VPN network, plus a transport check for an active VPN. The result lands in three places, and
|
||||||
|
all three are deliberate:
|
||||||
|
|
||||||
|
- **`run.constraints`** — for machines. A server aggregating thousands of runs can now separate
|
||||||
|
"measured a healthy network" from "measured almost nothing through a tunnel"; the shapes were
|
||||||
|
identical before.
|
||||||
|
- **A `measurement.vpn_constrained` finding** — for the person reading this run, naming the
|
||||||
|
interfaces that went unmeasured. A constrained run with a quiet findings list still reads as
|
||||||
|
"nothing wrong here".
|
||||||
|
- **The §7.3 verdict** — `Verdicts.derive` takes the constraints and returns INCONCLUSIVE
|
||||||
|
outright for a per-network-blocked run, whatever the category lights say; the run screen shows
|
||||||
|
an amber "Measured through a VPN" banner above the verdict so INCONCLUSIVE reads as the OS
|
||||||
|
refusing, not the app failing.
|
||||||
|
|
||||||
|
Detection is one bind per network rather than parsing per-test `attempted:false` breadcrumbs, so
|
||||||
|
it cannot drift when probe evidence formats change.
|
||||||
|
|
||||||
|
## Corroborated IPv6 findings: v6.broken is back, with evidence (2026-08-02, app 0.2.1)
|
||||||
|
|
||||||
|
The new `V6ConnectProbe` (test type `v6.brokenness`) attempts a real TCP connection over IPv6 to
|
||||||
|
the configured server's :443, per network that *claims* IPv6 (global address or v6 default
|
||||||
|
route) — IPv4-only networks are not attempted, since their failure is by design and would
|
||||||
|
manufacture the exact false positive this exists to kill. The finding derivation is now three-way:
|
||||||
|
|
||||||
|
- ICMPv6 silent, TCP works → `v6.no_icmp_reply` at **high** confidence, retitled "ICMPv6 is
|
||||||
|
filtered here — IPv6 itself works" (still reported: filtered ICMPv6 breaks PMTUD).
|
||||||
|
- ICMPv6 silent, TCP fails too → **`v6.broken`** (high severity, reinstated in the registry +
|
||||||
|
findings-registry.md): two independent transports silent on a network advertising IPv6.
|
||||||
|
- No corroboration (no server configured, or the connect never got as far as sending) → the
|
||||||
|
two-explanation `v6.no_icmp_reply` at medium confidence, unchanged.
|
||||||
|
|
||||||
|
Like STUN and the canary, the probe SKIPs honestly when no server is configured — corroboration
|
||||||
|
is a benefit of enrollment, not a reason to borrow fmr.
|
||||||
|
|
||||||
|
## Server: reserved 80/443 verified against the OS, and signed releases (2026-08-02)
|
||||||
|
|
||||||
|
**Reserved-address startup probe.** `serve()` now proves 80/443 are actually free on every
|
||||||
|
`ECHOLOT_RESERVED_ADDRS` address before starting: a throwaway bind per port
|
||||||
|
(`selftest.ReservedWebPortsFree`), fatal on EADDRINUSE with the offending address named — the
|
||||||
|
check `CheckReserved` cannot do, because a stray process outside our config (the adb-beacon
|
||||||
|
receiver on `0.0.0.0:443` was exactly that) is invisible to configuration checks. Bind errors
|
||||||
|
that are not "in use" (typo'd address, address not on this host) warn instead of refusing —
|
||||||
|
they are config problems, not pollution.
|
||||||
|
|
||||||
|
**Release signing.** Self-update now trusts a signature, not a host. CI signs `SHA256SUMS` with
|
||||||
|
an ed25519 key (`relsign` package, `cmd/release-sign`) and the updater refuses any release whose
|
||||||
|
`SHA256SUMS.sig` is missing or does not verify against the public key baked into the binary
|
||||||
|
(`selfupdate.DefaultPublicKeyB64`; operators with their own pipeline override via
|
||||||
|
`ECHOLOT_SELF_UPDATE_PUBKEY`). The private key exists in exactly two places: the Gitea Actions
|
||||||
|
secret `RELEASE_SIGNING_KEY`, and the offline original on the dev PC at
|
||||||
|
`~/.echolot/release-signing-key`. It is deliberately NOT on fmr and NOT in the repo — a
|
||||||
|
compromised release host can withhold updates but no longer inject one. CI hard-fails when the
|
||||||
|
secret is missing (an unsigned release would strand every verifying server) and cross-checks the
|
||||||
|
signature against the key in the source it just built.
|
||||||
|
|
||||||
|
**ACTION REQUIRED before the next `server-v*` tag:** add the Gitea repo secret
|
||||||
|
`RELEASE_SIGNING_KEY` (Settings → Actions → Secrets) with the contents of
|
||||||
|
`~/.echolot/release-signing-key` from the dev PC. Ordering is safe: the currently deployed
|
||||||
|
v0.3.x updater does not verify, so it will happily install the first signed release; every
|
||||||
|
release after that is verified. **Done 2026-08-02** — the secret is in place. (The "v0.3.x"
|
||||||
|
above should read "the currently deployed release": deployments had moved on to v0.9.x by the
|
||||||
|
time signing landed; the point — the deployed updater predates verification and will accept the
|
||||||
|
first signed release — is unchanged.)
|
||||||
|
|
||||||
|
## Prober fold: traceroute.udp4 and the mDNS inventory go production (2026-08-02, app 0.2.2)
|
||||||
|
|
||||||
|
The two highest-value validated capabilities moved from the prober into `core-probe`:
|
||||||
|
|
||||||
|
- **`traceroute.udp4`** (`TracerouteProbe`): UDP traceroute reading ICMP time-exceeded off the
|
||||||
|
socket error queue via `Os.recvmsg(MSG_ERRQUEUE)` through the reflection facade — no root, no
|
||||||
|
raw socket, no JNI, ~250 ms for six hops. Emits the schema's `TracerouteEvidence` (rtt in ns).
|
||||||
|
`OsAbi` came with it, including the measured fact that `Os.getsockoptInt` exists on neither
|
||||||
|
known device, so PMTU must always be read from the errqueue (`ee_info`), never
|
||||||
|
`getsockopt(IP_MTU)`. The load-bearing line survived the port: EAGAIN out of the reflected
|
||||||
|
`recvmsg` means "queue empty", not failure.
|
||||||
|
- **`local.mdns_inventory`** (`MdnsInventoryProbe`): MulticastLock + NSD discovery, the service
|
||||||
|
inventory that doubles as the VLAN-leakage detector. Both hardware lessons kept: the
|
||||||
|
`_services._dns-sd._udp.` meta-query returns 0 beside live services on both devices (so the
|
||||||
|
concrete types are the measurement and the meta-query result is itself evidence), and the
|
||||||
|
listen window is 10 s because 4 s missed services.
|
||||||
|
|
||||||
|
Still to fold, in order: the Shizuku dump *parsers* (the raw `link.ip_monitor` captures already
|
||||||
|
hold two divergent vendor formats that could feed `link.ra_source` and `sec.arp_watch`);
|
||||||
|
`multinetwork.request_and_bind` (extend ConstraintDetector to *request* transports rather than
|
||||||
|
only probing present ones); `peer.ble_advertise` (needs three new permissions and a peer mode to
|
||||||
|
exist first).
|
||||||
|
|
||||||
|
## Server v0.9.2: trains, real TTL/DSCP/ECN, rate limits (2026-08-02)
|
||||||
|
|
||||||
|
The spec-vs-implementation gap audit closed its top items; protocol_version 1.0.0 → 1.0.1
|
||||||
|
(additive — below 1.0.0 the minor is the breaking axis, and nothing here breaks an old client):
|
||||||
|
|
||||||
|
- **Upstream trains** (§3.2, types 0x03/0x04/0x05): per-train bounded columnar buffer (8192
|
||||||
|
rows, head kept on overflow with `Truncated` set — mirrors the schema's `evidence_truncated`
|
||||||
|
honesty), TRAIN_REPORT split across ≤1200-byte datagrams, grant-free with the §3.4 argument
|
||||||
|
spelled out (a 17-byte report row answers a ≥36-byte HMAC-valid packet). Unknown train id
|
||||||
|
gets a zero-row report: "nothing arrived" is an answer. Also surfaced as `udp.trains` in the
|
||||||
|
observations API.
|
||||||
|
- **Real TTL/DSCP/ECN observation** (§3.3): the read loop is `ReadMsgUDPAddrPort` with
|
||||||
|
IP_RECVTTL/IP_RECVTOS/IPV6_RECVHOPLIMIT/IPV6_RECVTCLASS cmsgs on Linux; `0xFF` stays the
|
||||||
|
"not observed" sentinel elsewhere. This unblocks `sec.dscp_ecn_survival` both directions,
|
||||||
|
paired with the new `dscp` parameter on `downtrain` (validated 0–63, refused not clamped,
|
||||||
|
`dscp_applied` in the response).
|
||||||
|
- **Rate limiting** (§2.5, was entirely absent): token buckets keyed per credential AND per
|
||||||
|
source IP; 429 + Retry-After on session/action creation (`/v1/profile` stays ungated), silent
|
||||||
|
drop on the data plane — charged after the HMAC gate so a spoofed flood cannot drain a
|
||||||
|
victim's budget, before the replay window so a dropped seq stays usable. UDP ceilings default
|
||||||
|
above the largest legitimate run (a 200 Mbps throughput test), because a rate limit that
|
||||||
|
clips a real measurement produces a confidently wrong number.
|
||||||
|
- **`action_id` in every granted packet** (§5/§9): payload bytes [8:16] across all granted
|
||||||
|
types, so overlapping actions are attributable. Verified the deployed Kotlin client parses
|
||||||
|
only ECHO_RESP and MTU_ACK payloads, so the reshuffle strands nobody.
|
||||||
|
- **Canary log retention**: the stated 24 h privacy default is now enforced
|
||||||
|
(`ECHOLOT_DNS_LOG_RETENTION_H`), where before the log was time-unbounded.
|
||||||
|
- **`POST /admin/enroll-tokens`** now answers the spec's JSON shape under content negotiation;
|
||||||
|
the README's curl works as documented.
|
||||||
|
- Spec §2.3 registry gained `downtrain` and `tcp-echo`, which the server had been advertising
|
||||||
|
as strings a conformant client must ignore.
|
||||||
|
|
||||||
|
Client-side counterparts still to build: sending 0x03 trains + parsing 0x05 reports
|
||||||
|
(`train.udp_updown`), and passing `dscp` on downtrain actions.
|
||||||
|
|
||||||
|
## ⚠ Version lineage broken: fmr runs v0.11.2, the repo's tags stop at v0.9.x (2026-08-02)
|
||||||
|
|
||||||
|
Discovered while preparing to self-update fmr to the freshly released server-v0.9.2:
|
||||||
|
**fmr runs v0.11.2** (binary installed 2026-08-02 08:51), but this repo's remote has tags only
|
||||||
|
up to `server-v0.9.1`, master fast-forwarded cleanly from this machine, there is no v0.10/v0.11
|
||||||
|
release in Gitea, no source checkout or Go toolchain on fmr, and no deploy script in this repo
|
||||||
|
that stamps versions. Conclusion: v0.11.2 was cross-built from a clone whose commits were never
|
||||||
|
pushed — presumably another dev machine.
|
||||||
|
|
||||||
|
Consequences until resolved:
|
||||||
|
- **Do NOT run `--self-update` on fmr.** Gitea's `/releases/latest` is the *newest-created*
|
||||||
|
release, which is now `server-v0.9.2` — semantically older than the deployed binary; the
|
||||||
|
updater compares strings, not SemVer, and would happily "update" v0.11.2 down to it. No
|
||||||
|
automatic risk exists (fmr has no update timer installed, only the cert timer), but a manual
|
||||||
|
run would downgrade.
|
||||||
|
- The next real release must be tagged **above v0.11.2** (e.g. `server-v0.11.3` or `v0.12.0`)
|
||||||
|
*after* the missing commits are pushed, so "latest" becomes truly latest again.
|
||||||
|
- The unpushed v0.10–v0.11 work needs to be found and pushed from whichever machine built it,
|
||||||
|
or the deployed binary's provenance re-established some other way, before the release channel
|
||||||
|
can be trusted again.
|
||||||
|
|||||||
@@ -0,0 +1,126 @@
|
|||||||
|
<!--
|
||||||
|
SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||||
|
SPDX-License-Identifier: CC-BY-4.0
|
||||||
|
-->
|
||||||
|
|
||||||
|
# Echolot findings registry
|
||||||
|
|
||||||
|
Closes open item 1 of `measurement-schema.md` §9.
|
||||||
|
|
||||||
|
A **finding code** is the stable, machine-readable half of a result. The prose around it changes
|
||||||
|
freely; the code is what a dashboard groups by, what a diff between two runs keys on, and what
|
||||||
|
someone greps a year of archived runs for. That only works if a code means exactly one thing,
|
||||||
|
forever.
|
||||||
|
|
||||||
|
This document is the contract. It is kept in step with
|
||||||
|
`echolot-app/core-measurement/.../FindingRegistry.kt` by a test that fails when either side has a
|
||||||
|
code the other does not — a registry that drifts from its documentation is worse than none,
|
||||||
|
because it looks authoritative.
|
||||||
|
|
||||||
|
## Rules
|
||||||
|
|
||||||
|
1. **The prefix determines the category**, and the category determines which verdict light the
|
||||||
|
finding rolls up into (§7.3). A `nat.*` code appearing under *connectivity* is not a naming
|
||||||
|
quibble; it changes which light turns red. Two codes were renamed from `nat.*` to
|
||||||
|
`connectivity.*` for exactly this reason.
|
||||||
|
2. **One code per concept.** Two emitters independently produced `connectivity.downstream_loss`
|
||||||
|
and `connectivity.loss_downstream` for the same claim before this registry existed. Anyone
|
||||||
|
aggregating either would have silently seen half their data.
|
||||||
|
3. **Codes are declared, not typed.** Emitters reference a `FindingSpec`, so a typo is a compile
|
||||||
|
error and no two call sites can disagree about a finding's category or default severity.
|
||||||
|
4. **Severity in the registry is the default.** An emitter may escalate for a specific run; it may
|
||||||
|
not quietly reclassify the finding in general.
|
||||||
|
5. **Say what is ruled out**, where that is the useful half. "Loss upstream" is worth far more
|
||||||
|
when it also states that the return path is clean, because that halves where to look next.
|
||||||
|
6. **Renaming a code is a breaking change** once runs are archived at scale. Before 1.0 it is
|
||||||
|
cheap; after, it needs an alias and a deprecation window.
|
||||||
|
|
||||||
|
## Registry
|
||||||
|
|
||||||
|
### connectivity
|
||||||
|
|
||||||
|
| code | severity | means | rules out |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `connectivity.udp_unreachable` | high | No UDP echo replies came back from the server at all. | — |
|
||||||
|
| `connectivity.udp_unreachable_upstream` | high | The server received none of the probes, so traffic is dropped on the way out. | The return path: nothing arrived to be replied to. |
|
||||||
|
| `connectivity.udp_loss` | medium | A large fraction of round-trip probes were lost, direction unknown. | — |
|
||||||
|
| `connectivity.loss_upstream` | medium | Probes were lost on the way to the server. | The return path: replies came back for everything that arrived. |
|
||||||
|
| `connectivity.loss_downstream` | medium | Packets were lost on the way back from the server. | The outbound path: the server received what it was answering. |
|
||||||
|
| `connectivity.downstream_blocked` | high | Server-initiated packets never arrive, although round trips work. | Basic reachability: the path forwards replies, just not unsolicited traffic. |
|
||||||
|
| `connectivity.downstream_reorder` | low | Downstream packets arrive in a different order than they were sent. | — |
|
||||||
|
| `connectivity.captive_portal` | medium | A captive portal is intercepting connectivity checks. | — |
|
||||||
|
| `connectivity.no_internet` | high | Android's own connectivity checks fail on this network. | — |
|
||||||
|
|
||||||
|
### mtu
|
||||||
|
|
||||||
|
| code | severity | means | rules out |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `mtu.reduced_downstream` | low | The downstream path MTU is below the usual 1500 bytes. | — |
|
||||||
|
| `mtu.downstream_blackhole` | medium | Datagrams above the path MTU are dropped downstream, fragmented or not. | — |
|
||||||
|
| `mtu.fragments_blocked` | medium | IP fragments do not reach this device even when sent in order. | — |
|
||||||
|
| `mtu.fragment_reorder_sensitive` | low | Fragments are delivered in order but dropped when reordered or delayed. | Fragmentation itself: in-order fragments arrive fine. |
|
||||||
|
|
||||||
|
### nat
|
||||||
|
|
||||||
|
| code | severity | means | rules out |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `nat.udp_rebinding` | medium | A NAT remapped the UDP source port mid-flow. | — |
|
||||||
|
| `nat.symmetric` | medium | The NAT assigns a different external port per destination. | — |
|
||||||
|
|
||||||
|
### perf
|
||||||
|
|
||||||
|
| code | severity | means | rules out |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `perf.throughput_no_delivery` | high | No throughput traffic arrived, although the server sent it. | — |
|
||||||
|
| `perf.throughput_below_offered` | low | Less throughput arrived than the server sent for the whole run. | — |
|
||||||
|
|
||||||
|
### dns
|
||||||
|
|
||||||
|
| code | severity | means | rules out |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `dns.answer_rewritten` | high | A resolver returned an answer that differs from the authoritative record. | — |
|
||||||
|
| `dns.authoritative_unreachable` | medium | The canary zone's authoritative server could not be reached. | — |
|
||||||
|
|
||||||
|
### v6
|
||||||
|
|
||||||
|
The prefix is `v6.`, matching the test-type registry (`v6.brokenness`, `v6.happy_eyeballs`, …).
|
||||||
|
These were `ipv6.*` while declaring `Category.IPV6`; since the prefix map only knows `v6`, they
|
||||||
|
rolled up under *connectivity* instead — the third occurrence of rule 1 being broken.
|
||||||
|
|
||||||
|
| code | severity | means | rules out |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `dns.search_domain_unanswered` | high | The network advertises a DNS search domain that its own server does not answer for. | A fault on this device: the same server answers ordinary names normally. |
|
||||||
|
| `dns.system_resolver_broken` | high | The network's DNS server answers, but this device cannot resolve names through it. | A network fault: the server replied to a query sent from this device. |
|
||||||
|
| `measurement.vpn_constrained` | info | A VPN was active, so the networks underneath it could not be measured. | Nothing — this run says little about the underlying network either way. |
|
||||||
|
| `v6.no_default_route` | medium | The device has a global IPv6 address but no IPv6 default route. | Guesswork: this is read from the routing table, not inferred from silence. |
|
||||||
|
| `v6.route_without_address` | medium | The network advertises an IPv6 default route but the device has no global IPv6 address. | A working IPv6 setup: SLAAC did not produce a usable address on this link. |
|
||||||
|
| `v6.no_icmp_reply` | low | IPv6 is configured but ICMPv6 echo gets no reply. | Nothing on its own: IPv6 may work fine with ICMP filtered. |
|
||||||
|
| `v6.broken` | high | IPv6 is advertised on this network but carries no traffic. | ICMP filtering as the benign explanation: a TCP connection over IPv6 failed too. |
|
||||||
|
| `v6.not_offered` | info | This network does not offer IPv6. | — |
|
||||||
|
|
||||||
|
`v6.no_icmp_reply` was `v6.broken` until a phone reported it while loading an IPv6-only site over
|
||||||
|
TCP perfectly well. The only evidence behind it is ICMPv6 echo, which is widely filtered on
|
||||||
|
networks where IPv6 works — so the finding now states what was observed and names both
|
||||||
|
explanations instead of choosing one. It is still worth reporting: filtered ICMPv6 breaks Path MTU
|
||||||
|
Discovery.
|
||||||
|
|
||||||
|
`v6.broken` returned once that corroboration existed: the `v6.brokenness` test attempts a real TCP
|
||||||
|
connection over IPv6 to the configured server, and only when *both* transports fail on a network
|
||||||
|
that advertises IPv6 is the brokenness claim made — at high severity, because every dual-stack
|
||||||
|
destination pays a timeout before falling back to IPv4. When the TCP connect *succeeds*,
|
||||||
|
`v6.no_icmp_reply` is emitted at high confidence instead, now able to say plainly that ICMPv6 is
|
||||||
|
filtered while IPv6 works. With no server configured there is no corroboration target and the
|
||||||
|
two-explanation `v6.no_icmp_reply` stands unchanged.
|
||||||
|
|
||||||
|
`v6.not_offered` is **info and must stay info**. Most networks still do not offer IPv6 and that is
|
||||||
|
not a fault; reporting it as a warning lights a yellow verdict on a healthy network, which teaches
|
||||||
|
people to ignore the light — the one thing a diagnostic must never do.
|
||||||
|
|
||||||
|
## Adding a finding
|
||||||
|
|
||||||
|
1. Add a `FindingSpec` to `FindingRegistry`, and to its `all` list.
|
||||||
|
2. Add the row here, under the section its prefix names.
|
||||||
|
3. Emit it with `finding(FindingRegistry.YOUR_CODE, …)`.
|
||||||
|
|
||||||
|
The registry test checks 1 and 2 agree, that every prefix maps to the category it claims, and that
|
||||||
|
no two entries share a code.
|
||||||
@@ -52,12 +52,26 @@ Export encoding: UTF-8 JSON, gzip for files (`.echolot.json.gz`), share intent u
|
|||||||
},
|
},
|
||||||
"tiers": { "app": true, "shizuku": true, "root": false },
|
"tiers": { "app": true, "shizuku": true, "root": false },
|
||||||
"profiles_used": ["profile-uuid", ...],
|
"profiles_used": ["profile-uuid", ...],
|
||||||
|
"constraints": {
|
||||||
|
"vpn_active": true,
|
||||||
|
"per_network_blocked": true,
|
||||||
|
"unmeasured_networks": ["net-0", "net-1"]
|
||||||
|
},
|
||||||
"notes": "free-text user annotation"
|
"notes": "free-text user annotation"
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
`tiers` records what was *available*; each test records what it *used*.
|
`tiers` records what was *available*; each test records what it *used*.
|
||||||
|
|
||||||
|
`constraints` records what was *prevented*. A constrained run is neither a failed run nor a normal
|
||||||
|
one, and the distinction has to survive into the data: a run taken through a VPN has the same shape
|
||||||
|
and the same green verdict as a clean run of a healthy network, so without this a reader — or a
|
||||||
|
server aggregating thousands of them — cannot tell that almost nothing was measured. The known case
|
||||||
|
is `per_network_blocked`: Android refuses `Network.bindSocket()` on the underlying networks while a
|
||||||
|
VPN holds the default route, so every per-network test measures the tunnel or nothing at all, and
|
||||||
|
any conclusion about the link underneath is unfounded. Consumers should treat findings from a
|
||||||
|
constrained run as scoped to what was actually reachable, and `unmeasured_networks` names the rest.
|
||||||
|
|
||||||
## 4. `networks[]` — one entry per Android `Network` in play
|
## 4. `networks[]` — one entry per Android `Network` in play
|
||||||
|
|
||||||
A run may exercise several networks simultaneously (Wi-Fi + cellular + USB ethernet). Everything is a snapshot at run start; a `changes[]` list captures mid-run deltas.
|
A run may exercise several networks simultaneously (Wi-Fi + cellular + USB ethernet). Everything is a snapshot at run start; a `changes[]` list captures mid-run deltas.
|
||||||
@@ -269,7 +283,7 @@ The JSON Schema (machine-readable companion, `measurement.schema.json`, generate
|
|||||||
|
|
||||||
| type | example fields | v2 anonymizer transform |
|
| type | example fields | v2 anonymizer transform |
|
||||||
|---|---|---|
|
|---|---|---|
|
||||||
| `ip4`, `ip6` | addresses, routes, hops, DNS answers | prefix-preserving pseudonymization, consistent per document; well-known/reserved ranges kept verbatim |
|
| `ip4`, `ip6` | addresses, routes, hops, DNS answers | prefix-preserving pseudonymization, consistent per document; well-known/reserved ranges kept verbatim. **Exception: ULA (`fc00::/7`) has its whole prefix pseudonymized as a unit.** It resembles RFC1918 but is not analogous: a ULA global ID is 40 random bits, unique to one network by construction (RFC 4193), so the prefix *is* the identifier, whereas `192.168.0.0/16` is shared by millions of networks and identifies none. Pseudonymizing it as a unit keeps "these hosts are on one subnet" while dropping "this is that subnet". |
|
||||||
| `mac`, `bssid` | wifi, arp_watch | OUI kept, NIC part pseudonymized |
|
| `mac`, `bssid` | wifi, arp_watch | OUI kept, NIC part pseudonymized |
|
||||||
| `fqdn` | DNS names, reverse lookups | per-label pseudonyms, public-suffix kept |
|
| `fqdn` | DNS names, reverse lookups | per-label pseudonyms, public-suffix kept |
|
||||||
| `ssid` | wifi | pseudonym |
|
| `ssid` | wifi | pseudonym |
|
||||||
@@ -279,7 +293,8 @@ Free-text fields (`notes`, `error.detail`, dump excerpts from Shizuku parsers) c
|
|||||||
|
|
||||||
## 9. Open items
|
## 9. Open items
|
||||||
|
|
||||||
1. Findings registry document — start alongside the first implemented tests.
|
1. ~~Findings registry document~~ — done: `findings-registry.md`, kept in step with
|
||||||
|
`FindingRegistry.kt` by a test that fails when the two disagree.
|
||||||
2. Whether Shizuku raw-dump excerpts (dumpsys/ip output) are embedded in `evidence` verbatim (auditable, but large and hard to anonymize) or parsed-only with an optional "attach raw dumps" toggle. Proposal: toggle, default on for local archive, default off for export.
|
2. Whether Shizuku raw-dump excerpts (dumpsys/ip output) are embedded in `evidence` verbatim (auditable, but large and hard to anonymize) or parsed-only with an optional "attach raw dumps" toggle. Proposal: toggle, default on for local archive, default off for export.
|
||||||
3. Peer-mode documents: each device produces its own run; the coordinator embeds the peer's findings summary and cross-references by `run.id`. Full merge format deferred.
|
3. Peer-mode documents: each device produces its own run; the coordinator embeds the peer's findings summary and cross-references by `run.id`. Full merge format deferred.
|
||||||
4. Size guardrails: soft cap 20 MB uncompressed per run; trains beyond that downsample evidence (keep aggregates + first/last N + all anomalies) and record `"evidence_truncated": true`.
|
4. Size guardrails: soft cap 20 MB uncompressed per run; trains beyond that downsample evidence (keep aggregates + first/last N + all anomalies) and record `"evidence_truncated": true`.
|
||||||
|
|||||||
+57
-4
@@ -24,11 +24,34 @@ echolot://enroll?v=1&u=<control-URL, urlencoded>&p=pin-sha256:<b64 SPKI hash>&t=
|
|||||||
|
|
||||||
```
|
```
|
||||||
POST /v1/enroll Authorization: Bearer <enrollment-token>
|
POST /v1/enroll Authorization: Bearer <enrollment-token>
|
||||||
→ 200 { "device_credential": "<random 256-bit, b64url>",
|
→ 201 { "device_credential": "<random 256-bit, b64url>",
|
||||||
"device_id": "uuid",
|
"device_id": "uuid" }
|
||||||
"profile": { ... §2.2 ... } }
|
|
||||||
```
|
```
|
||||||
|
|
||||||
|
The **server assembles the bootstrap link**, because it is the only party holding all three parts
|
||||||
|
at once, and the part an operator gets wrong by hand is the base64 pin — which does not fail
|
||||||
|
loudly, it just never matches, and surfaces later as an inscrutable TLS error:
|
||||||
|
|
||||||
|
```
|
||||||
|
POST /admin/enroll-tokens
|
||||||
|
→ { "token": "…", "expires_in_s": 86400,
|
||||||
|
"enroll_uri": "echolot://enroll?v=1&u=…&p=…&t=…" }
|
||||||
|
```
|
||||||
|
|
||||||
|
The control URL in the link comes from `ECHOLOT_PUBLIC_URL`, falling back to the first control
|
||||||
|
listen address. A wildcard bind has no single right answer, so it warns rather than guessing.
|
||||||
|
|
||||||
|
Encoding notes that matter in practice:
|
||||||
|
- `u`, `p` and `t` are **percent-encoded**. The pin is base64, so it contains `+`, `/` and `=`,
|
||||||
|
every one of which means something else in a query string.
|
||||||
|
- A `+` that was *not* encoded decodes to a space. Base64 contains no spaces, so a parser SHOULD
|
||||||
|
restore them — the alternative is a pin wrong by one character and a failure that points nowhere
|
||||||
|
near the cause.
|
||||||
|
- The control URL MUST be `https://`. The pin only protects a TLS connection; a cleartext URL
|
||||||
|
would hand the token to anyone on the path.
|
||||||
|
- **The link is a secret** while it is live: it carries a bearer token, so anyone who sees it
|
||||||
|
before the device does can enroll instead.
|
||||||
|
|
||||||
Enrollment tokens are single-use with expiry, created in the admin UI, scoped `enroll`. The device credential is a long-lived bearer secret, scoped `run-tests`; it is also the HKDF input for session keys. Revocation = deleting the device in the admin UI.
|
Enrollment tokens are single-use with expiry, created in the admin UI, scoped `enroll`. The device credential is a long-lived bearer secret, scoped `run-tests`; it is also the HKDF input for session keys. Revocation = deleting the device in the admin UI.
|
||||||
|
|
||||||
### 2.2 Profile
|
### 2.2 Profile
|
||||||
@@ -61,7 +84,12 @@ The app re-fetches the profile at the start of every run (falling back to the ca
|
|||||||
|
|
||||||
### 2.3 Capabilities (v1 registry)
|
### 2.3 Capabilities (v1 registry)
|
||||||
|
|
||||||
`udp-probe`, `stun-basic`, `stun-5780`, `canary-dns`, `recursive-dns`, `connect-back`, `delayed-echo`, `big-send`, `frag-send`, `tls-echo`, `http-echo`, `throughput`, `ntp`. A server omits what it can't offer (e.g. `stun-5780` without a second IP degrades to `stun-basic`). Clients must skip, and record as `unsupported`, any test whose capability is absent. Unknown capability strings are ignored.
|
`udp-probe`, `stun-basic`, `stun-5780`, `canary-dns`, `recursive-dns`, `connect-back`, `delayed-echo`, `big-send`, `frag-send`, `tls-echo`, `http-echo`, `throughput`, `ntp`, plus:
|
||||||
|
|
||||||
|
- `downtrain` — server-sent downstream trains via the §5 `downtrain` action. Upstream trains need no capability of their own: they are plain client-sent data-plane packets and ride `udp-probe`.
|
||||||
|
- `tcp-echo` — the plain-TCP echo endpoint (§4); `tls-echo` is its ALPN variant on the same port.
|
||||||
|
|
||||||
|
A server omits what it can't offer (e.g. `stun-5780` without a second IP degrades to `stun-basic`). Clients must skip, and record as `unsupported`, any test whose capability is absent. Unknown capability strings are ignored.
|
||||||
|
|
||||||
### 2.4 Sessions
|
### 2.4 Sessions
|
||||||
|
|
||||||
@@ -88,6 +116,31 @@ DELETE /v1/sessions/{id}
|
|||||||
|
|
||||||
Per-credential and per-source-IP token buckets on: session creation, actions, UDP packets, bytes. `429` on control plane; silent drop on data plane (probes must tolerate loss anyway). All reflected/generated traffic goes **only** to the session's observed source address (or, for connect-back, the source address of the session-creating request). Data-plane responses to unauthenticated packets are never larger than the request (§3.4).
|
Per-credential and per-source-IP token buckets on: session creation, actions, UDP packets, bytes. `429` on control plane; silent drop on data plane (probes must tolerate loss anyway). All reflected/generated traffic goes **only** to the session's observed source address (or, for connect-back, the source address of the session-creating request). Data-plane responses to unauthenticated packets are never larger than the request (§3.4).
|
||||||
|
|
||||||
|
### 2.2 `GET /v1/discover` — where the control plane lives
|
||||||
|
|
||||||
|
Unauthenticated, and says almost nothing: the control-plane URL and the server's display name.
|
||||||
|
|
||||||
|
```json
|
||||||
|
{ "control_url": "https://probe.example.net", "name": "example" }
|
||||||
|
```
|
||||||
|
|
||||||
|
It exists so an enrollment link can carry the name a person recognises while the app still connects
|
||||||
|
to the name that selects the pinned certificate. When a server shares port 443 between its admin UI
|
||||||
|
and its control plane, those must be different hostnames — one port and one name is one certificate,
|
||||||
|
and the two need different ones (a browser-trusted certificate, and a long-lived self-signed one the
|
||||||
|
client pins). Without discovery, the difference leaks into every enrollment link an operator hands
|
||||||
|
out.
|
||||||
|
|
||||||
|
**It hands out an address, never a pin.** The pin travels in the link itself. Serving it here would
|
||||||
|
reduce pinning to whatever the certificate authorities are worth, and pinning exists precisely to
|
||||||
|
survive one the operator does not control — a root injected by corporate device management, for
|
||||||
|
instance, which is unremarkable on the networks this tool is pointed at. Because the pin is
|
||||||
|
pre-shared, an intercepted discovery response can only send a device to the wrong host, where the
|
||||||
|
pin will not match: an outage, not a compromise.
|
||||||
|
|
||||||
|
Clients treat it as optional. A server that does not answer, or a link that already names the
|
||||||
|
control endpoint, works unchanged — enrollment must not begin failing because a lookup did.
|
||||||
|
|
||||||
## 3. UDP probe protocol
|
## 3. UDP probe protocol
|
||||||
|
|
||||||
### 3.1 Packet header (fixed 32 bytes, network byte order)
|
### 3.1 Packet header (fixed 32 bytes, network byte order)
|
||||||
|
|||||||
@@ -15,7 +15,7 @@ plugins {
|
|||||||
//
|
//
|
||||||
// major*1_000_000 + minor*10_000 + patch*10 leaves room for 9 patch-level rebuilds (the trailing
|
// major*1_000_000 + minor*10_000 + patch*10 leaves room for 9 patch-level rebuilds (the trailing
|
||||||
// digit) without disturbing the mapping, and stays inside the 2_100_000_000 ceiling until major 2100.
|
// digit) without disturbing the mapping, and stays inside the 2_100_000_000 ceiling until major 2100.
|
||||||
val appVersionName = "0.2.0"
|
val appVersionName = "0.2.3"
|
||||||
|
|
||||||
fun versionCodeOf(semver: String): Int {
|
fun versionCodeOf(semver: String): Int {
|
||||||
val (major, minor, patch) = semver.substringBefore('-').split(".").map(String::toInt)
|
val (major, minor, patch) = semver.substringBefore('-').split(".").map(String::toInt)
|
||||||
@@ -34,8 +34,16 @@ android {
|
|||||||
versionName = appVersionName
|
versionName = appVersionName
|
||||||
// Automation: `adb shell am start -n app.echo_lot.app/.MainActivity --ez autorun true`
|
// Automation: `adb shell am start -n app.echo_lot.app/.MainActivity --ez autorun true`
|
||||||
// runs a measurement immediately and POSTs the report here (dev collection endpoint).
|
// runs a measurement immediately and POSTs the report here (dev collection endpoint).
|
||||||
buildConfigField("String", "REPORT_UPLOAD_URL", "\"http://89.185.109.150:443/report\"")
|
// Empty: the collection endpoint this pointed at was the adb-beacon receiver, which held
|
||||||
buildConfigField("String", "REPORT_UPLOAD_SECRET", "\"D4OmG5gGJsElqVVbtYIZbR\"")
|
// 0.0.0.0:443 in cleartext. That service is gone and echolot-server owns 443 with TLS, so
|
||||||
|
// posting plaintext there now fails as "client sent an HTTP request to an HTTPS server" —
|
||||||
|
// an alarming error for a debugging convenience that is no longer needed, since autorun
|
||||||
|
// reports are read straight off the device with `run-as cat`.
|
||||||
|
//
|
||||||
|
// Deliberately not repointed at /v1/runs. That is the consent-gated upload, and a
|
||||||
|
// debugging shortcut must not be able to satisfy it by accident.
|
||||||
|
buildConfigField("String", "REPORT_UPLOAD_URL", "\"\"")
|
||||||
|
buildConfigField("String", "REPORT_UPLOAD_SECRET", "\"\"")
|
||||||
// The bare SemVer, without the debug build's "-dev" suffix stripped away by the server's
|
// The bare SemVer, without the debug build's "-dev" suffix stripped away by the server's
|
||||||
// parser anyway — sent to servers so they can apply their compatibility window.
|
// parser anyway — sent to servers so they can apply their compatibility window.
|
||||||
buildConfigField("String", "APP_SEMVER", "\"$appVersionName\"")
|
buildConfigField("String", "APP_SEMVER", "\"$appVersionName\"")
|
||||||
|
|||||||
@@ -22,11 +22,35 @@
|
|||||||
|
|
||||||
<activity
|
<activity
|
||||||
android:name=".MainActivity"
|
android:name=".MainActivity"
|
||||||
android:exported="true">
|
android:exported="true"
|
||||||
|
android:launchMode="singleTask">
|
||||||
<intent-filter>
|
<intent-filter>
|
||||||
<action android:name="android.intent.action.MAIN" />
|
<action android:name="android.intent.action.MAIN" />
|
||||||
<category android:name="android.intent.category.LAUNCHER" />
|
<category android:name="android.intent.category.LAUNCHER" />
|
||||||
</intent-filter>
|
</intent-filter>
|
||||||
|
<!--
|
||||||
|
Enrollment bootstrap (probe-protocol.md §2.1): echolot://enroll?v=1&u=…&p=…&t=…
|
||||||
|
Scanning a QR or tapping a link the operator sent configures the server in one
|
||||||
|
action, instead of transcribing a URL, a base64 pin and a token by hand — the pin
|
||||||
|
in particular fails silently when it is wrong by one character.
|
||||||
|
-->
|
||||||
|
<intent-filter android:autoVerify="false">
|
||||||
|
<action android:name="android.intent.action.VIEW" />
|
||||||
|
<category android:name="android.intent.category.DEFAULT" />
|
||||||
|
<category android:name="android.intent.category.BROWSABLE" />
|
||||||
|
<data android:scheme="echolot" android:host="enroll" />
|
||||||
|
</intent-filter>
|
||||||
|
<!--
|
||||||
|
Sign-in redirect. The browser hands the authorization code back through this, which
|
||||||
|
is exactly why the flow uses PKCE: any app may register this scheme, so the code
|
||||||
|
alone must not be enough to complete a sign-in.
|
||||||
|
-->
|
||||||
|
<intent-filter android:autoVerify="false">
|
||||||
|
<action android:name="android.intent.action.VIEW" />
|
||||||
|
<category android:name="android.intent.category.DEFAULT" />
|
||||||
|
<category android:name="android.intent.category.BROWSABLE" />
|
||||||
|
<data android:scheme="echolot" android:host="auth" />
|
||||||
|
</intent-filter>
|
||||||
</activity>
|
</activity>
|
||||||
|
|
||||||
<provider
|
<provider
|
||||||
|
|||||||
@@ -0,0 +1,125 @@
|
|||||||
|
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||||
|
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||||
|
|
||||||
|
package app.echo_lot.app
|
||||||
|
|
||||||
|
import app.echo_lot.protocol.AuthInfo
|
||||||
|
import app.echo_lot.protocol.ControlClient
|
||||||
|
import app.echo_lot.protocol.OidcLogin
|
||||||
|
import kotlinx.serialization.json.Json
|
||||||
|
import kotlinx.serialization.json.jsonObject
|
||||||
|
import kotlinx.serialization.json.jsonPrimitive
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Signing in to the configured server's identity provider.
|
||||||
|
*
|
||||||
|
* The awkward part of a browser-based sign-in on Android is that the app is not running while it
|
||||||
|
* happens. Handing control to a browser puts this process in the background, where it may be
|
||||||
|
* killed at any moment; the callback then arrives at a fresh process with none of the state the
|
||||||
|
* exchange needs. So the PKCE verifier and state are written to storage before the browser opens,
|
||||||
|
* not held in memory — an in-memory value works on a developer's device and fails on a phone under
|
||||||
|
* memory pressure, which is the worst way for this to break.
|
||||||
|
*
|
||||||
|
* Nothing from the identity provider is kept afterwards. The ID token proves who is signing in,
|
||||||
|
* once; the device credential authenticates everything from then on.
|
||||||
|
*/
|
||||||
|
class Account(private val settings: Settings) {
|
||||||
|
|
||||||
|
private val json = Json { ignoreUnknownKeys = true }
|
||||||
|
|
||||||
|
sealed interface SignInStart {
|
||||||
|
/** Open this in a browser. */
|
||||||
|
data class Browser(val url: String) : SignInStart
|
||||||
|
data class Unavailable(val reason: String) : SignInStart
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Fetches the server's auth configuration and builds the authorization URL. */
|
||||||
|
fun begin(): SignInStart {
|
||||||
|
if (!settings.serverConfigured) {
|
||||||
|
return SignInStart.Unavailable(
|
||||||
|
"Enrol with a server first — sign-in belongs to the server's identity provider."
|
||||||
|
)
|
||||||
|
}
|
||||||
|
val auth = runCatching { client().profile(settings.serverCredential).auth }.getOrNull()
|
||||||
|
?: return SignInStart.Unavailable("Could not reach the server to ask how to sign in.")
|
||||||
|
|
||||||
|
auth.discoveryError?.let {
|
||||||
|
// The distinction matters: "the operator configured an IdP that is not answering" is
|
||||||
|
// their problem to fix, and is not the same as "this server has no accounts".
|
||||||
|
return SignInStart.Unavailable("The server's identity provider is not responding: $it")
|
||||||
|
}
|
||||||
|
if (!auth.enabled) {
|
||||||
|
return SignInStart.Unavailable("This server does not offer accounts.")
|
||||||
|
}
|
||||||
|
return try {
|
||||||
|
val pending = OidcLogin.begin(auth)
|
||||||
|
// Written before the browser opens, because after that this process may not survive.
|
||||||
|
settings.pendingVerifier = pending.verifier
|
||||||
|
settings.pendingState = pending.state
|
||||||
|
SignInStart.Browser(pending.authorizationUrl)
|
||||||
|
} catch (t: Throwable) {
|
||||||
|
SignInStart.Unavailable(t.message ?: "Could not start sign-in.")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Completes sign-in from the `echolot://auth` redirect.
|
||||||
|
*
|
||||||
|
* Blocking; callers run it off the main thread.
|
||||||
|
*/
|
||||||
|
fun complete(callbackUri: String): String {
|
||||||
|
val verifier = settings.pendingVerifier
|
||||||
|
val state = settings.pendingState
|
||||||
|
// Cleared first, whatever happens next: these are single-use, and leaving them behind
|
||||||
|
// would let a later callback be completed against a flow nobody started.
|
||||||
|
settings.clearPendingAuth()
|
||||||
|
|
||||||
|
if (verifier.isBlank() || state.isBlank()) {
|
||||||
|
return "That sign-in did not start on this device."
|
||||||
|
}
|
||||||
|
return try {
|
||||||
|
val auth = client().profile(settings.serverCredential).auth
|
||||||
|
val idToken = OidcLogin.complete(
|
||||||
|
auth, OidcLogin.Pending("", verifier, state), callbackUri,
|
||||||
|
)
|
||||||
|
val reply = client().linkAccount(settings.serverCredential, idToken)
|
||||||
|
val o = json.parseToJsonElement(reply).jsonObject
|
||||||
|
val name = o["display_name"]?.jsonPrimitive?.content ?: "signed in"
|
||||||
|
settings.accountName = name
|
||||||
|
settings.accountId = o["account_id"]?.jsonPrimitive?.content ?: ""
|
||||||
|
val admin = o["admin"]?.jsonPrimitive?.content == "true"
|
||||||
|
"Signed in as $name" + if (admin) " (administrator)" else ""
|
||||||
|
} catch (e: OidcLogin.LoginFailed) {
|
||||||
|
e.message ?: "Sign-in failed."
|
||||||
|
} catch (t: Throwable) {
|
||||||
|
"Sign-in failed: ${t.message ?: t.javaClass.simpleName}"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Signs out. The device stays enrolled — signing out should not cost an enrolment. */
|
||||||
|
fun signOut(): String = try {
|
||||||
|
client().unlinkAccount(settings.serverCredential)
|
||||||
|
settings.accountName = ""
|
||||||
|
settings.accountId = ""
|
||||||
|
"Signed out. This device is still enrolled."
|
||||||
|
} catch (t: Throwable) {
|
||||||
|
"Could not sign out: ${t.message ?: t.javaClass.simpleName}"
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Asks the server who it thinks is signed in, so the UI is not trusting stale local state. */
|
||||||
|
fun refresh(): String? = runCatching {
|
||||||
|
val o = json.parseToJsonElement(client().accountStatus(settings.serverCredential)).jsonObject
|
||||||
|
val signedIn = o["signed_in"]?.jsonPrimitive?.content == "true"
|
||||||
|
settings.accountName = if (signedIn) {
|
||||||
|
o["display_name"]?.jsonPrimitive?.content ?: ""
|
||||||
|
} else {
|
||||||
|
""
|
||||||
|
}
|
||||||
|
settings.accountName.takeIf { it.isNotBlank() }
|
||||||
|
}.getOrNull()
|
||||||
|
|
||||||
|
private fun client() = ControlClient(
|
||||||
|
settings.serverUrl, setOf(settings.serverPin), BuildConfig.APP_SEMVER,
|
||||||
|
fallbackAddrs = settings.serverAddrList(),
|
||||||
|
)
|
||||||
|
}
|
||||||
@@ -4,6 +4,7 @@
|
|||||||
package app.echo_lot.app
|
package app.echo_lot.app
|
||||||
|
|
||||||
import androidx.compose.foundation.layout.Arrangement
|
import androidx.compose.foundation.layout.Arrangement
|
||||||
|
import androidx.compose.foundation.layout.safeDrawingPadding
|
||||||
import androidx.compose.foundation.layout.Column
|
import androidx.compose.foundation.layout.Column
|
||||||
import androidx.compose.foundation.layout.Row
|
import androidx.compose.foundation.layout.Row
|
||||||
import androidx.compose.foundation.layout.fillMaxWidth
|
import androidx.compose.foundation.layout.fillMaxWidth
|
||||||
@@ -39,7 +40,7 @@ fun HistoryScreen(
|
|||||||
onDelete: (String) -> Unit,
|
onDelete: (String) -> Unit,
|
||||||
onBack: () -> Unit,
|
onBack: () -> Unit,
|
||||||
) {
|
) {
|
||||||
Column(Modifier.fillMaxWidth().padding(16.dp), verticalArrangement = Arrangement.spacedBy(10.dp)) {
|
Column(Modifier.fillMaxWidth().safeDrawingPadding().padding(16.dp), verticalArrangement = Arrangement.spacedBy(10.dp)) {
|
||||||
Row(verticalAlignment = Alignment.CenterVertically) {
|
Row(verticalAlignment = Alignment.CenterVertically) {
|
||||||
TextButton(onClick = onBack) { Text("‹ Back") }
|
TextButton(onClick = onBack) { Text("‹ Back") }
|
||||||
Text("History", style = MaterialTheme.typography.titleLarge)
|
Text("History", style = MaterialTheme.typography.titleLarge)
|
||||||
@@ -71,12 +72,20 @@ fun HistoryScreen(
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
Text(
|
Text(
|
||||||
"${r.findingCount} finding(s) · ${r.sizeBytes / 1024} kB · ${r.anonymization}",
|
"${r.findingCount} finding(s) · ${r.sizeBytes / 1024} kB · " +
|
||||||
|
"kept complete on this device",
|
||||||
style = MaterialTheme.typography.bodySmall,
|
style = MaterialTheme.typography.bodySmall,
|
||||||
)
|
)
|
||||||
|
// The upload line names the level the upload was made at, not the
|
||||||
|
// archive's. They describe different documents, and showing the archive's
|
||||||
|
// level here claimed more had left the device than actually did.
|
||||||
Text(
|
Text(
|
||||||
if (r.uploaded) "uploaded to ${r.uploadedTo ?: "a server"}"
|
if (r.uploaded) {
|
||||||
else "on this device only",
|
"uploaded to ${r.uploadedTo ?: "a server"}" +
|
||||||
|
(r.uploadedAs?.let { " as $it" } ?: "")
|
||||||
|
} else {
|
||||||
|
"on this device only"
|
||||||
|
},
|
||||||
style = MaterialTheme.typography.bodySmall,
|
style = MaterialTheme.typography.bodySmall,
|
||||||
color = if (r.uploaded) Color(0xFF7FD17F) else Color(0xFFBBBBBB),
|
color = if (r.uploaded) Color(0xFF7FD17F) else Color(0xFFBBBBBB),
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -43,9 +43,27 @@ class MainActivity : ComponentActivity() {
|
|||||||
private val permissionLauncher =
|
private val permissionLauncher =
|
||||||
registerForActivityResult(ActivityResultContracts.RequestMultiplePermissions()) { /* proceed regardless */ }
|
registerForActivityResult(ActivityResultContracts.RequestMultiplePermissions()) { /* proceed regardless */ }
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The intent currently being acted on, so a deep link that arrives while the app is running
|
||||||
|
* is seen by the screen the user is already looking at.
|
||||||
|
*
|
||||||
|
* The activity is singleTask for the same reason. As a standard activity it stacked a second
|
||||||
|
* instance per link, each with its own ViewModel: the enrolment then happened in a throwaway
|
||||||
|
* copy, and pressing back returned to the original screen showing none of it. Silent, and
|
||||||
|
* indistinguishable from the link simply not working.
|
||||||
|
*/
|
||||||
|
private val liveIntent = mutableStateOf<android.content.Intent?>(null)
|
||||||
|
|
||||||
|
override fun onNewIntent(intent: android.content.Intent) {
|
||||||
|
super.onNewIntent(intent)
|
||||||
|
setIntent(intent)
|
||||||
|
liveIntent.value = intent
|
||||||
|
}
|
||||||
|
|
||||||
override fun onCreate(savedInstanceState: Bundle?) {
|
override fun onCreate(savedInstanceState: Bundle?) {
|
||||||
super.onCreate(savedInstanceState)
|
super.onCreate(savedInstanceState)
|
||||||
requestRuntimePermissions()
|
requestRuntimePermissions()
|
||||||
|
liveIntent.value = intent
|
||||||
setContent {
|
setContent {
|
||||||
MaterialTheme(colorScheme = darkColorScheme()) {
|
MaterialTheme(colorScheme = darkColorScheme()) {
|
||||||
Surface(color = MaterialTheme.colorScheme.background) {
|
Surface(color = MaterialTheme.colorScheme.background) {
|
||||||
@@ -59,6 +77,92 @@ class MainActivity : ComponentActivity() {
|
|||||||
// starts a run immediately and uploads the report, so an unattended
|
// starts a run immediately and uploads the report, so an unattended
|
||||||
// measurement needs no UI tapping and no adb round-trip to collect.
|
// measurement needs no UI tapping and no adb round-trip to collect.
|
||||||
val autorun = intent?.getBooleanExtra("autorun", false) == true
|
val autorun = intent?.getBooleanExtra("autorun", false) == true
|
||||||
|
|
||||||
|
// An echolot://enroll link (QR scan, or a link the operator sent) opens the
|
||||||
|
// app straight into settings with the enrollment already done, so the user
|
||||||
|
// sees the result rather than a form they still have to fill in.
|
||||||
|
// Both deep links land here. They are told apart by host, so a sign-in
|
||||||
|
// redirect is never mistaken for an enrolment link — one spends a token, the
|
||||||
|
// other completes an authorization, and confusing them would fail obscurely.
|
||||||
|
val incoming = liveIntent.value?.takeIf { it.action == Intent.ACTION_VIEW }?.dataString
|
||||||
|
val authUri = incoming?.takeIf { it.startsWith("echolot://auth") }
|
||||||
|
val enrollUri = incoming?.takeIf { it.startsWith("echolot://enroll") }
|
||||||
|
androidx.compose.runtime.LaunchedEffect(enrollUri) {
|
||||||
|
if (enrollUri != null) {
|
||||||
|
vm.enroll(enrollUri)
|
||||||
|
screen = Screen.SETTINGS
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Replacing an existing enrollment is asked about, never assumed. Following a
|
||||||
|
// link from a web page is one tap, and the old credential does not survive it.
|
||||||
|
vm.state.pendingEnroll?.let { pending ->
|
||||||
|
androidx.compose.material3.AlertDialog(
|
||||||
|
onDismissRequest = { vm.cancelEnroll() },
|
||||||
|
title = {
|
||||||
|
androidx.compose.material3.Text(
|
||||||
|
if (pending.sameServer) "Enroll again with this server?"
|
||||||
|
else "Replace this device's server?"
|
||||||
|
)
|
||||||
|
},
|
||||||
|
text = {
|
||||||
|
androidx.compose.material3.Text(
|
||||||
|
// Naming the same URL twice reads as a mistake and buries the
|
||||||
|
// one consequence that actually applies: the device is issued a
|
||||||
|
// fresh credential and shows up as a second entry.
|
||||||
|
if (pending.sameServer) {
|
||||||
|
"This device is already enrolled with " +
|
||||||
|
"${pending.currentServer}.\n\n" +
|
||||||
|
"Enrolling again replaces its credential. The old one " +
|
||||||
|
"stops working immediately, and the device appears on " +
|
||||||
|
"the server as a new entry alongside the current one — " +
|
||||||
|
"which you may want to revoke afterwards.\n\n" +
|
||||||
|
"Runs already uploaded, and runs stored on this phone, " +
|
||||||
|
"are not affected."
|
||||||
|
} else {
|
||||||
|
"This device is already enrolled with " +
|
||||||
|
"${pending.currentServer}.\n\n" +
|
||||||
|
"Enrolling with ${pending.newServer} replaces that. Runs " +
|
||||||
|
"already uploaded stay where they are, but this device " +
|
||||||
|
"stops reporting to the old server and appears on the new " +
|
||||||
|
"one as a new device.\n\n" +
|
||||||
|
"Runs stored on this phone are not affected."
|
||||||
|
}
|
||||||
|
)
|
||||||
|
},
|
||||||
|
confirmButton = {
|
||||||
|
androidx.compose.material3.TextButton(onClick = { vm.confirmEnroll() }) {
|
||||||
|
androidx.compose.material3.Text(
|
||||||
|
if (pending.sameServer) "Enroll again" else "Enroll here"
|
||||||
|
)
|
||||||
|
}
|
||||||
|
},
|
||||||
|
dismissButton = {
|
||||||
|
androidx.compose.material3.TextButton(onClick = { vm.cancelEnroll() }) {
|
||||||
|
androidx.compose.material3.Text("Keep current server")
|
||||||
|
}
|
||||||
|
},
|
||||||
|
)
|
||||||
|
}
|
||||||
|
androidx.compose.runtime.LaunchedEffect(authUri) {
|
||||||
|
if (authUri != null) {
|
||||||
|
vm.completeSignIn(authUri)
|
||||||
|
screen = Screen.SETTINGS
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Shizuku can be started, stopped or authorised in its own app, where nothing
|
||||||
|
// calls back into this process. Asking again each time this screen comes
|
||||||
|
// forward is what makes the banner right after the user has been away to fix
|
||||||
|
// it — which is exactly the moment they look at it.
|
||||||
|
val lifecycleOwner = androidx.compose.ui.platform.LocalLifecycleOwner.current
|
||||||
|
androidx.compose.runtime.DisposableEffect(lifecycleOwner) {
|
||||||
|
val obs = androidx.lifecycle.LifecycleEventObserver { _, event ->
|
||||||
|
if (event == androidx.lifecycle.Lifecycle.Event.ON_RESUME) {
|
||||||
|
vm.refreshShizuku()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
lifecycleOwner.lifecycle.addObserver(obs)
|
||||||
|
onDispose { lifecycleOwner.lifecycle.removeObserver(obs) }
|
||||||
|
}
|
||||||
androidx.compose.runtime.LaunchedEffect(autorun) {
|
androidx.compose.runtime.LaunchedEffect(autorun) {
|
||||||
if (autorun) vm.run(devUpload = true)
|
if (autorun) vm.run(devUpload = true)
|
||||||
}
|
}
|
||||||
@@ -74,22 +178,43 @@ class MainActivity : ComponentActivity() {
|
|||||||
finish()
|
finish()
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// Without this, the system Back gesture leaves the activity from Settings or
|
||||||
|
// History instead of returning to the run screen — the screen is a plain state
|
||||||
|
// variable, so nothing connects it to the back stack. Registered only when
|
||||||
|
// there is somewhere to go back to, so Back still exits from the run screen.
|
||||||
|
androidx.activity.compose.BackHandler(enabled = screen != Screen.RUN) {
|
||||||
|
screen = Screen.RUN
|
||||||
|
}
|
||||||
|
|
||||||
when (screen) {
|
when (screen) {
|
||||||
Screen.SETTINGS -> SettingsScreen(
|
Screen.SETTINGS -> SettingsScreen(
|
||||||
settings = vm.settings,
|
settings = vm.settings,
|
||||||
archivedRuns = vm.state.history.size,
|
archivedRuns = vm.archivedRunCount(),
|
||||||
archivedBytes = vm.archivedBytes(),
|
archivedBytes = vm.archivedBytes(),
|
||||||
onApplyRetention = vm::applyRetention,
|
onApplyRetention = vm::applyRetention,
|
||||||
onDeleteAll = vm::deleteAllRuns,
|
onDeleteAll = vm::deleteAllRuns,
|
||||||
onPreviewUpload = {
|
onPreviewUpload = {
|
||||||
// Preview the newest run, since that is the one the user just made
|
// Straight from the archive: the newest run is the one the user just
|
||||||
// and the one they are deciding about.
|
// made and the one they are deciding about. Always shows something,
|
||||||
vm.state.history.firstOrNull()?.let { r ->
|
// even when there is nothing to preview yet.
|
||||||
lifecycleScope.launch { preview = vm.uploadPreview(r.id) }
|
lifecycleScope.launch { preview = vm.previewNewestRun() }
|
||||||
}
|
|
||||||
},
|
},
|
||||||
onCheckServer = vm::checkServer,
|
onCheckServer = vm::checkServer,
|
||||||
|
accountName = vm.accountName,
|
||||||
|
onSignIn = {
|
||||||
|
vm.beginSignIn { url ->
|
||||||
|
// A plain VIEW intent rather than a Custom Tab: the browser is
|
||||||
|
// where the user's existing IdP session already lives, and
|
||||||
|
// androidx.browser would be a dependency for a rounded corner.
|
||||||
|
runCatching {
|
||||||
|
startActivity(Intent(Intent.ACTION_VIEW, android.net.Uri.parse(url)))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
onSignOut = vm::signOut,
|
||||||
|
onEnroll = vm::enroll,
|
||||||
serverStatus = vm.state.archiveStatus,
|
serverStatus = vm.state.archiveStatus,
|
||||||
|
enrollStatus = vm.state.enrollStatus,
|
||||||
onBack = { screen = Screen.RUN },
|
onBack = { screen = Screen.RUN },
|
||||||
)
|
)
|
||||||
Screen.HISTORY -> HistoryScreen(
|
Screen.HISTORY -> HistoryScreen(
|
||||||
@@ -288,6 +413,41 @@ private fun EcholotScreen(
|
|||||||
|
|
||||||
@Composable
|
@Composable
|
||||||
private fun Results(doc: MeasurementDocument) {
|
private fun Results(doc: MeasurementDocument) {
|
||||||
|
// A constrained run is answered before the lights are: the verdict below is INCONCLUSIVE by
|
||||||
|
// §7.3, and without this banner "inconclusive" reads as the app failing rather than the OS
|
||||||
|
// (correctly) refusing to let anything past the VPN be measured.
|
||||||
|
val constraints = doc.run.constraints
|
||||||
|
if (constraints.constrained) {
|
||||||
|
val blocked = constraints.unmeasuredNetworks
|
||||||
|
.mapNotNull { id -> doc.networks.firstOrNull { it.id == id } }
|
||||||
|
.joinToString(", ") { it.iface?.takeIf { s -> s.isNotBlank() } ?: it.transport.name.lowercase() }
|
||||||
|
.ifBlank { "the networks beneath it" }
|
||||||
|
// Same three-way split as the finding: saying "VPN" when the user just disconnected
|
||||||
|
// theirs (the wall lingers during teardown) reads as the app being wrong, not the OS.
|
||||||
|
val (headline, body) = when {
|
||||||
|
constraints.vpnActive && constraints.perNetworkBlocked ->
|
||||||
|
"Measured through a VPN" to
|
||||||
|
("Android does not let apps send on the networks beneath an active VPN, so " +
|
||||||
|
"$blocked could not be measured — these results describe the tunnel. " +
|
||||||
|
"Disconnect the VPN and run again to measure the networks themselves.")
|
||||||
|
constraints.perNetworkBlocked ->
|
||||||
|
"Some networks could not be measured" to
|
||||||
|
("Android refused sends on $blocked — the restriction a VPN leaves in place " +
|
||||||
|
"while it tears down. These networks went unmeasured; wait a few " +
|
||||||
|
"seconds and run again.")
|
||||||
|
else ->
|
||||||
|
"A VPN holds the default route" to
|
||||||
|
("Default-route results describe the tunnel; per-network measurements " +
|
||||||
|
"reached the underlying networks.")
|
||||||
|
}
|
||||||
|
Card(colors = CardDefaults.cardColors(containerColor = Color(0xFF3A2E12))) {
|
||||||
|
Column(Modifier.fillMaxWidth().padding(12.dp)) {
|
||||||
|
Text(headline, color = Color(0xFFFFD08A), fontWeight = FontWeight.SemiBold)
|
||||||
|
Text(body, fontSize = 12.sp, color = Color(0xFFFFD08A))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
val summary = doc.summary
|
val summary = doc.summary
|
||||||
if (summary != null) {
|
if (summary != null) {
|
||||||
Card(colors = CardDefaults.cardColors(containerColor = verdictColor(summary.overall))) {
|
Card(colors = CardDefaults.cardColors(containerColor = verdictColor(summary.overall))) {
|
||||||
|
|||||||
@@ -80,8 +80,16 @@ class RunStore(context: Context, private val settings: Settings) {
|
|||||||
|
|
||||||
private fun client() = ControlClient(
|
private fun client() = ControlClient(
|
||||||
settings.serverUrl, setOf(settings.serverPin), BuildConfig.APP_SEMVER,
|
settings.serverUrl, setOf(settings.serverPin), BuildConfig.APP_SEMVER,
|
||||||
|
fallbackAddrs = settings.serverAddrList(),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
/** Remembers where the server lives, so a later run can reach it without DNS. */
|
||||||
|
private fun rememberAddrs(p: app.echo_lot.protocol.Profile) {
|
||||||
|
val addrs = p.targets.flatMap { listOfNotNull(it.ip4, it.ip6) }
|
||||||
|
.filter { it.isNotBlank() }
|
||||||
|
if (addrs.isNotEmpty()) settings.serverAddrs = addrs.joinToString(",")
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Checks the configured server without uploading anything: reachable, pinned, compatible, and
|
* Checks the configured server without uploading anything: reachable, pinned, compatible, and
|
||||||
* willing to accept runs. Lets the user find out in settings rather than from a failed run.
|
* willing to accept runs. Lets the user find out in settings rather than from a failed run.
|
||||||
@@ -90,6 +98,10 @@ class RunStore(context: Context, private val settings: Settings) {
|
|||||||
if (!settings.serverConfigured) return "Fill in the server URL, pin and credential first."
|
if (!settings.serverConfigured) return "Fill in the server URL, pin and credential first."
|
||||||
return try {
|
return try {
|
||||||
val profile = client().profile(settings.serverCredential)
|
val profile = client().profile(settings.serverCredential)
|
||||||
|
// Learned here so the next run's canary probe knows what to ask for.
|
||||||
|
profile.canaryZone.takeIf { it.isNotBlank() }?.let { settings.canaryZone = it }
|
||||||
|
settings.serverFacts = describeFacts(profile)
|
||||||
|
rememberAddrs(profile)
|
||||||
val compat = Compat.check(profile, BuildConfig.APP_SEMVER)
|
val compat = Compat.check(profile, BuildConfig.APP_SEMVER)
|
||||||
val head = "${profile.name} · server ${profile.serverVersion} · " +
|
val head = "${profile.name} · server ${profile.serverVersion} · " +
|
||||||
"protocol ${profile.compat.protocolVersion.ifBlank { "unstated" }}"
|
"protocol ${profile.compat.protocolVersion.ifBlank { "unstated" }}"
|
||||||
@@ -108,6 +120,69 @@ class RunStore(context: Context, private val settings: Settings) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Redeems an enrollment link and stores the resulting server configuration (§2.1).
|
||||||
|
*
|
||||||
|
* Everything is written at once or not at all: a half-applied server — say a URL and pin with
|
||||||
|
* no credential — fails later, somewhere else, with an error that points at the wrong thing.
|
||||||
|
* Blocking; callers run it off the main thread.
|
||||||
|
*/
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Renders what the server says about itself, for display.
|
||||||
|
*
|
||||||
|
* Only what a person measuring against it would want to check: which addresses the tests will
|
||||||
|
* actually use, on which ports, and what the server admits it can do. Addresses first, because
|
||||||
|
* "which address did this result come from" is the question a report leaves open.
|
||||||
|
*/
|
||||||
|
private fun describeFacts(p: app.echo_lot.protocol.Profile): String {
|
||||||
|
val lines = ArrayList<String>()
|
||||||
|
// "label|value" per line, laid out as real columns by the UI rather than padded with
|
||||||
|
// spaces here. Space padding only lines up in a monospaced font, which makes the layout
|
||||||
|
// depend on a typeface choice made somewhere else entirely.
|
||||||
|
fun row(label: String, value: String) = lines.add("$label|$value")
|
||||||
|
|
||||||
|
row("server", "${p.name} · ${p.serverVersion}")
|
||||||
|
for (t in p.targets) {
|
||||||
|
t.ip4?.let { row("IPv4", it) }
|
||||||
|
t.ip6?.let { row("IPv6", it) }
|
||||||
|
// Marked rather than listed apart: it is the same server, and what matters is being
|
||||||
|
// able to tell which address a NAT-behaviour result came from.
|
||||||
|
t.ip4Alt?.let { row("IPv4 alt", it) }
|
||||||
|
t.ip6Alt?.let { row("IPv6 alt", it) }
|
||||||
|
row("ports", "udp ${t.udpPort} · tcp ${t.tcpPort} · stun ${t.stunPort}")
|
||||||
|
}
|
||||||
|
if (p.canaryZone.isNotBlank()) row("dns zone", p.canaryZone)
|
||||||
|
if (p.capabilities.isNotEmpty()) row("measures", p.capabilities.joinToString(", "))
|
||||||
|
return lines.joinToString(System.lineSeparator())
|
||||||
|
}
|
||||||
|
|
||||||
|
fun enroll(link: String, deviceName: String?): String {
|
||||||
|
val parsed = app.echo_lot.protocol.EnrollmentLink.parse(link)
|
||||||
|
?: return "That does not look like an Echolot enrollment link. It should start with " +
|
||||||
|
"echolot://enroll and carry a URL, a pin and a token."
|
||||||
|
return try {
|
||||||
|
val enrolled = parsed.redeem(deviceName, BuildConfig.APP_SEMVER)
|
||||||
|
val compat = Compat.check(enrolled.profile, BuildConfig.APP_SEMVER)
|
||||||
|
settings.serverFacts = describeFacts(enrolled.profile)
|
||||||
|
rememberAddrs(enrolled.profile)
|
||||||
|
enrolled.profile.canaryZone.takeIf { it.isNotBlank() }?.let { settings.canaryZone = it }
|
||||||
|
settings.serverUrl = enrolled.controlUrl
|
||||||
|
settings.serverPublicUrl = enrolled.publicUrl
|
||||||
|
settings.serverPin = enrolled.pin
|
||||||
|
settings.serverCredential = enrolled.credential
|
||||||
|
val head = "Enrolled with ${enrolled.profile.name} " +
|
||||||
|
"(server ${enrolled.profile.serverVersion})."
|
||||||
|
if (compat.message != null) head + " " + compat.message else head
|
||||||
|
} catch (e: VersionRefused) {
|
||||||
|
"That server will not serve this app: ${e.message}"
|
||||||
|
} catch (t: Throwable) {
|
||||||
|
// The commonest causes are a spent token and a wrong pin, and they look nothing alike
|
||||||
|
// in the message — so pass it through rather than flattening it to "enrollment failed".
|
||||||
|
"Enrollment failed: ${t.message ?: t.javaClass.simpleName}"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Uploads one archived run to the configured server, redacting first.
|
* Uploads one archived run to the configured server, redacting first.
|
||||||
*
|
*
|
||||||
@@ -121,6 +196,9 @@ class RunStore(context: Context, private val settings: Settings) {
|
|||||||
return try {
|
return try {
|
||||||
val client = client()
|
val client = client()
|
||||||
val profile = client.profile(settings.serverCredential)
|
val profile = client.profile(settings.serverCredential)
|
||||||
|
profile.canaryZone.takeIf { it.isNotBlank() }?.let { settings.canaryZone = it }
|
||||||
|
settings.serverFacts = describeFacts(profile)
|
||||||
|
rememberAddrs(profile)
|
||||||
|
|
||||||
// Compatibility before policy: an incompatible server may well advertise an upload
|
// Compatibility before policy: an incompatible server may well advertise an upload
|
||||||
// policy it would never actually apply to us.
|
// policy it would never actually apply to us.
|
||||||
@@ -135,8 +213,10 @@ class RunStore(context: Context, private val settings: Settings) {
|
|||||||
)
|
)
|
||||||
val body = redactedForUpload(docJson, level)
|
val body = redactedForUpload(docJson, level)
|
||||||
val reply = client.uploadRun(settings.serverCredential, body)
|
val reply = client.uploadRun(settings.serverCredential, body)
|
||||||
archive.markUploaded(runId, profile.name)
|
archive.markUploaded(runId, profile.name, level.wire)
|
||||||
UploadOutcome.Sent(profile.name, "as $level, ${body.toByteArray().size} bytes: ${reply.take(120)}")
|
// Deliberately not echoing `reply`: it is the server's index entry as raw JSON, and
|
||||||
|
// it ended up rendered verbatim in the UI. Size and level are what a person wants.
|
||||||
|
UploadOutcome.Sent(profile.name, "as $level, ${body.toByteArray().size} bytes")
|
||||||
} catch (e: VersionRefused) {
|
} catch (e: VersionRefused) {
|
||||||
UploadOutcome.Incompatible(e.message ?: "the server refused this app's version")
|
UploadOutcome.Incompatible(e.message ?: "the server refused this app's version")
|
||||||
} catch (e: UploadRefused) {
|
} catch (e: UploadRefused) {
|
||||||
|
|||||||
@@ -42,6 +42,17 @@ data class UiState(
|
|||||||
val archiveStatus: String? = null,
|
val archiveStatus: String? = null,
|
||||||
/** History, newest first. Refreshed after every run and whenever the history screen opens. */
|
/** History, newest first. Refreshed after every run and whenever the history screen opens. */
|
||||||
val history: List<app.echo_lot.archive.ArchivedRun> = emptyList(),
|
val history: List<app.echo_lot.archive.ArchivedRun> = emptyList(),
|
||||||
|
/**
|
||||||
|
* Result of the last enrollment attempt, shown beside the Enroll button.
|
||||||
|
*
|
||||||
|
* Separate from [archiveStatus]: they are two different actions with two different results,
|
||||||
|
* and sharing one line put the answer to "did enrolling work" at the far end of the card,
|
||||||
|
* below three text fields — or nowhere at all on a fresh install, since that line only
|
||||||
|
* renders once a run exists.
|
||||||
|
*/
|
||||||
|
val enrollStatus: String? = null,
|
||||||
|
/** An enrollment link waiting on confirmation, because this device is already enrolled. */
|
||||||
|
val pendingEnroll: PendingEnroll? = null,
|
||||||
/** Shell-tier readiness, shown before a run; null message = say nothing (Shizuku not installed). */
|
/** Shell-tier readiness, shown before a run; null message = say nothing (Shizuku not installed). */
|
||||||
val shizukuNotice: String? = null,
|
val shizukuNotice: String? = null,
|
||||||
val shizukuReady: Boolean = false,
|
val shizukuReady: Boolean = false,
|
||||||
@@ -49,6 +60,18 @@ data class UiState(
|
|||||||
val shizukuState: ShizukuAvailability.State = ShizukuAvailability.State.NOT_INSTALLED,
|
val shizukuState: ShizukuAvailability.State = ShizukuAvailability.State.NOT_INSTALLED,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
/**
|
||||||
|
* An enrollment link that would replace an existing one, held until the user says so.
|
||||||
|
*
|
||||||
|
* Enrolling is not additive: the new credential replaces the old, and on the previous server this
|
||||||
|
* device simply stops reporting. Following a link is one tap from a web page, which is not enough
|
||||||
|
* deliberation to discard a working enrollment by accident.
|
||||||
|
*/
|
||||||
|
data class PendingEnroll(val link: String, val currentServer: String, val newServer: String) {
|
||||||
|
/** Re-enrolling with the server already configured, rather than moving to a different one. */
|
||||||
|
val sameServer: Boolean get() = currentServer.trimEnd('/') == newServer.trimEnd('/')
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Drives one measurement run: device-tier probes (link snapshot, per-network ICMP) always run;
|
* Drives one measurement run: device-tier probes (link snapshot, per-network ICMP) always run;
|
||||||
* results assemble into a MeasurementDocument with a §7.3 summary. Lives in a ViewModel so a run
|
* results assemble into a MeasurementDocument with a §7.3 summary. Lives in a ViewModel so a run
|
||||||
@@ -80,6 +103,23 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Re-reads the shell tier's state, for when it changed somewhere this process cannot see.
|
||||||
|
*
|
||||||
|
* Permission can be granted inside Shizuku's own app, and Shizuku can be started or stopped
|
||||||
|
* there too; none of that calls back here. Asking again on resume is the only way to be right
|
||||||
|
* after the user has been somewhere else to fix it.
|
||||||
|
*/
|
||||||
|
fun refreshShizuku() {
|
||||||
|
val st = ShizukuAvailability.current(getApplication())
|
||||||
|
state = state.copy(
|
||||||
|
shizukuNotice = ShizukuAvailability.describe(st),
|
||||||
|
shizukuReady = st == ShizukuAvailability.State.READY,
|
||||||
|
shizukuHint = ShizukuAvailability.actionHint(st),
|
||||||
|
shizukuState = st,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
override fun onCleared() {
|
override fun onCleared() {
|
||||||
stopShizukuObserver()
|
stopShizukuObserver()
|
||||||
super.onCleared()
|
super.onCleared()
|
||||||
@@ -90,6 +130,7 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
|||||||
private var runStartWall: String = ""
|
private var runStartWall: String = ""
|
||||||
private var runNetworks: List<app.echo_lot.measurement.Network> = emptyList()
|
private var runNetworks: List<app.echo_lot.measurement.Network> = emptyList()
|
||||||
private var runShizukuOk = false
|
private var runShizukuOk = false
|
||||||
|
private var runConstraints = Constraints()
|
||||||
|
|
||||||
/** Two-clock ids: UUIDs + monotonic ns relative to a per-run origin. */
|
/** Two-clock ids: UUIDs + monotonic ns relative to a per-run origin. */
|
||||||
private class RunIds : ProbeIds {
|
private class RunIds : ProbeIds {
|
||||||
@@ -108,6 +149,7 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
|||||||
fun run(devUpload: Boolean = false) {
|
fun run(devUpload: Boolean = false) {
|
||||||
if (state.running) return
|
if (state.running) return
|
||||||
collected.clear()
|
collected.clear()
|
||||||
|
runConstraints = Constraints()
|
||||||
state = state.copy(running = true, currentStep = "starting", document = null,
|
state = state.copy(running = true, currentStep = "starting", document = null,
|
||||||
uploadStatus = null, archiveStatus = null)
|
uploadStatus = null, archiveStatus = null)
|
||||||
runJob = viewModelScope.launch {
|
runJob = viewModelScope.launch {
|
||||||
@@ -125,7 +167,13 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
|||||||
if (devUpload) {
|
if (devUpload) {
|
||||||
state = state.copy(currentStep = "uploading report")
|
state = state.copy(currentStep = "uploading report")
|
||||||
val r = withContext(Dispatchers.IO) { ReportUploader.upload(doc) }
|
val r = withContext(Dispatchers.IO) { ReportUploader.upload(doc) }
|
||||||
status = if (r.ok) "uploaded ✓ ${r.detail}" else "upload failed: ${r.detail}"
|
status = when {
|
||||||
|
r.ok -> "uploaded ✓ ${r.detail}"
|
||||||
|
// Not a failure worth alarming about: the dev collection endpoint is simply
|
||||||
|
// not configured, and the run is on the device either way.
|
||||||
|
r.detail.startsWith("no upload URL") -> "run complete — read it with adb"
|
||||||
|
else -> "upload failed: ${r.detail}"
|
||||||
|
}
|
||||||
}
|
}
|
||||||
if (archived != null && settings.autoUpload) {
|
if (archived != null && settings.autoUpload) {
|
||||||
state = state.copy(currentStep = "uploading to server")
|
state = state.copy(currentStep = "uploading to server")
|
||||||
@@ -194,8 +242,112 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
|||||||
store.read(id)?.let { store.redactedForUpload(it) }
|
store.read(id)?.let { store.redactedForUpload(it) }
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Preview of the most recent run, read from the archive rather than from [UiState.history].
|
||||||
|
*
|
||||||
|
* The history list is only populated once the History screen has been opened, so a preview
|
||||||
|
* driven from it did nothing at all on a freshly-opened Settings screen — a button that
|
||||||
|
* silently does nothing is worse than one that says why.
|
||||||
|
*/
|
||||||
|
suspend fun previewNewestRun(): String = withContext(Dispatchers.IO) {
|
||||||
|
val newest = store.list().firstOrNull()
|
||||||
|
?: return@withContext "No archived runs yet. Run a measurement first, then this will " +
|
||||||
|
"show exactly what an upload would send."
|
||||||
|
store.read(newest.id)?.let { store.redactedForUpload(it) }
|
||||||
|
?: "That run could not be read back from the archive."
|
||||||
|
}
|
||||||
|
|
||||||
fun archivedBytes(): Long = store.totalBytes()
|
fun archivedBytes(): Long = store.totalBytes()
|
||||||
|
|
||||||
|
/** Counted from the archive itself, not from [UiState.history], which is empty until the
|
||||||
|
* history screen has been opened - the two disagreeing read as data loss. */
|
||||||
|
fun archivedRunCount(): Int = store.list().size
|
||||||
|
|
||||||
|
private val account = Account(settings)
|
||||||
|
|
||||||
|
/** Name of whoever is signed in on this device, for the settings screen. */
|
||||||
|
var accountName by mutableStateOf(settings.accountName)
|
||||||
|
private set
|
||||||
|
|
||||||
|
/** Starts sign-in; the caller opens the returned URL in a browser. */
|
||||||
|
fun beginSignIn(open: (String) -> Unit) {
|
||||||
|
viewModelScope.launch {
|
||||||
|
state = state.copy(archiveStatus = "contacting the server …")
|
||||||
|
when (val r = withContext(Dispatchers.IO) { account.begin() }) {
|
||||||
|
is Account.SignInStart.Browser -> {
|
||||||
|
state = state.copy(archiveStatus = "continue in your browser …")
|
||||||
|
open(r.url)
|
||||||
|
}
|
||||||
|
is Account.SignInStart.Unavailable ->
|
||||||
|
state = state.copy(archiveStatus = r.reason)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Completes sign-in from the echolot://auth redirect. */
|
||||||
|
fun completeSignIn(callbackUri: String) {
|
||||||
|
viewModelScope.launch {
|
||||||
|
val msg = withContext(Dispatchers.IO) { account.complete(callbackUri) }
|
||||||
|
accountName = settings.accountName
|
||||||
|
state = state.copy(archiveStatus = msg)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fun signOut() {
|
||||||
|
viewModelScope.launch {
|
||||||
|
val msg = withContext(Dispatchers.IO) { account.signOut() }
|
||||||
|
accountName = settings.accountName
|
||||||
|
state = state.copy(archiveStatus = msg)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Redeems an enrollment link, from a paste or from an echolot:// deep link. */
|
||||||
|
fun enroll(link: String, deviceName: String? = android.os.Build.MODEL) {
|
||||||
|
// Already enrolled? Ask first. The old credential is gone the moment this succeeds, and a
|
||||||
|
// link followed from a web page is one tap — far too little deliberation for that.
|
||||||
|
if (settings.serverConfigured) {
|
||||||
|
val target = app.echo_lot.protocol.EnrollmentLink.parse(link)?.controlUrl ?: link
|
||||||
|
state = state.copy(
|
||||||
|
pendingEnroll = PendingEnroll(
|
||||||
|
link = link,
|
||||||
|
// Compared against the link's URL, which names the server publicly — so this
|
||||||
|
// has to be the public name too. Using the endpoint made re-enrolling with the
|
||||||
|
// same server look like a move to a different one, because the endpoint and
|
||||||
|
// the public name are deliberately different strings.
|
||||||
|
currentServer = settings.serverPublicUrl,
|
||||||
|
newServer = target,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
doEnroll(link, deviceName)
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The user confirmed replacing an existing enrollment. */
|
||||||
|
fun confirmEnroll(deviceName: String? = android.os.Build.MODEL) {
|
||||||
|
val pending = state.pendingEnroll ?: return
|
||||||
|
state = state.copy(pendingEnroll = null)
|
||||||
|
doEnroll(pending.link, deviceName)
|
||||||
|
}
|
||||||
|
|
||||||
|
fun cancelEnroll() {
|
||||||
|
state = state.copy(
|
||||||
|
pendingEnroll = null,
|
||||||
|
enrollStatus = "Kept the existing enrollment; nothing changed.",
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
private fun doEnroll(link: String, deviceName: String?) {
|
||||||
|
viewModelScope.launch {
|
||||||
|
state = state.copy(enrollStatus = "Enrolling …")
|
||||||
|
val result = withContext(Dispatchers.IO) { store.enroll(link, deviceName) }
|
||||||
|
// A new server means a new canary zone; the old one would describe somebody else's
|
||||||
|
// deployment. Cleared rather than kept, and relearned from the next profile fetch.
|
||||||
|
settings.canaryZone = ""
|
||||||
|
state = state.copy(enrollStatus = result)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/** Settings-screen action: report what the configured server is and whether we can use it. */
|
/** Settings-screen action: report what the configured server is and whether we can use it. */
|
||||||
fun checkServer() {
|
fun checkServer() {
|
||||||
viewModelScope.launch {
|
viewModelScope.launch {
|
||||||
@@ -241,23 +393,45 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
|||||||
val entries = NetworkInventory.snapshot(ctx)
|
val entries = NetworkInventory.snapshot(ctx)
|
||||||
val networks = entries.map { it.model }.also { runNetworks = it }
|
val networks = entries.map { it.model }.also { runNetworks = it }
|
||||||
|
|
||||||
|
// What will this run be prevented from measuring? Decided up front, from one throwaway
|
||||||
|
// bind per network, so the document can say so instead of leaving it to be inferred from
|
||||||
|
// per-test `attempted: false` breadcrumbs (measurement-schema.md §3 `constraints`).
|
||||||
|
runConstraints = app.echo_lot.probe.ConstraintDetector.detect(ctx, entries)
|
||||||
|
|
||||||
val probes: List<Probe> = listOf(
|
val probes: List<Probe> = listOf(
|
||||||
LinkSnapshotProbe(entries),
|
LinkSnapshotProbe(entries),
|
||||||
RouterIdentityProbe(entries),
|
RouterIdentityProbe(entries),
|
||||||
IcmpProbe(entries, v6 = false),
|
IcmpProbe(entries, v6 = false),
|
||||||
IcmpProbe(entries, v6 = true),
|
IcmpProbe(entries, v6 = true),
|
||||||
|
// Folded from the prober after hardware validation: errqueue traceroute (no root,
|
||||||
|
// no JNI) and the mDNS service inventory / VLAN-leakage detector.
|
||||||
|
app.echo_lot.probe.TracerouteProbe(),
|
||||||
|
app.echo_lot.probe.MdnsInventoryProbe(),
|
||||||
CaptivePortalProbe(entries),
|
CaptivePortalProbe(entries),
|
||||||
// Canary zone served by the Echolot probe server (probe-protocol §6.1). Hardcoded to
|
// Canary zone served by the Echolot probe server (probe-protocol §6.1). Hardcoded to
|
||||||
// the reference deployment until profiles/enrollment land in the UI.
|
// the reference deployment until profiles/enrollment land in the UI.
|
||||||
DnsCanaryProbe(canaryZone = "c.echo-lot.app", sessionPrefix = "adhoc"),
|
// Both target whatever server this device is enrolled with, not the deployment the
|
||||||
StunProbe(serverHost = "fmr-1.echo-lot.app"),
|
// app happened to be developed against. With no server configured they get blank
|
||||||
|
// strings and report themselves skipped, which is the honest outcome — the
|
||||||
|
// alternative measures someone else's infrastructure and calls it your network.
|
||||||
|
// Before the canary: "can this device resolve at all" has to be answered before
|
||||||
|
// "are the answers being tampered with" means anything.
|
||||||
|
app.echo_lot.probe.DnsResolverProbe(entries),
|
||||||
|
DnsCanaryProbe(canaryZone = settings.canaryZone, sessionPrefix = "adhoc"),
|
||||||
|
StunProbe(serverHost = settings.serverHost()),
|
||||||
|
// Corroboration for icmp.ping6's silence: a real TCP connection over IPv6. Only its
|
||||||
|
// failure, on a network that advertises IPv6, justifies calling IPv6 broken.
|
||||||
|
app.echo_lot.probe.V6ConnectProbe(entries, serverHost = settings.serverHost()),
|
||||||
)
|
)
|
||||||
|
|
||||||
// Plan the run first: the Shizuku battery is counted alongside the app-tier probes so
|
// Plan the run first: the Shizuku battery is counted alongside the app-tier probes so
|
||||||
// the bar reflects the whole run. Estimates are per-probe (see Probe.estimatedMs).
|
// the bar reflects the whole run. Estimates prefer what THIS device measured on recent
|
||||||
val shizukuEstimateMs = 8_000L
|
// runs (Settings EMA); Probe.estimatedMs is only the cold-start seed — a fixed table
|
||||||
|
// cannot know whether ICMPv6 answers in milliseconds here or waits out its timeout.
|
||||||
|
fun estimateOf(p: Probe) = settings.learnedDurationMs(p.type) ?: p.estimatedMs
|
||||||
|
val shizukuEstimateMs = settings.learnedDurationMs(SHIZUKU_DURATION_KEY) ?: 8_000L
|
||||||
val totalSteps = probes.size + 1
|
val totalSteps = probes.size + 1
|
||||||
var remainingMs = probes.sumOf { it.estimatedMs } + shizukuEstimateMs
|
var remainingMs = probes.sumOf { estimateOf(it) } + shizukuEstimateMs
|
||||||
state = state.copy(stepsDone = 0, stepsTotal = totalSteps,
|
state = state.copy(stepsDone = 0, stepsTotal = totalSteps,
|
||||||
etaSeconds = ((remainingMs + 999) / 1000).toInt())
|
etaSeconds = ((remainingMs + 999) / 1000).toInt())
|
||||||
|
|
||||||
@@ -277,22 +451,33 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
|||||||
}
|
}
|
||||||
)
|
)
|
||||||
tests.add(result); collected.add(result)
|
tests.add(result); collected.add(result)
|
||||||
remainingMs -= p.estimatedMs
|
settings.recordDurationMs(p.type, (result.endedMonoNs - result.startedMonoNs) / 1_000_000)
|
||||||
|
remainingMs -= estimateOf(p)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Shizuku shell tier — self-degrades to UNSUPPORTED when Shizuku isn't running.
|
// Shizuku shell tier — self-degrades to UNSUPPORTED when Shizuku isn't running. One
|
||||||
|
// battery, three tests: the raw captures plus the parsed ra_source/arp_watch views.
|
||||||
step("link.ip_monitor (shizuku)", done = probes.size, total = totalSteps, etaMs = shizukuEstimateMs)
|
step("link.ip_monitor (shizuku)", done = probes.size, total = totalSteps, etaMs = shizukuEstimateMs)
|
||||||
val shizukuTest = try {
|
val shizukuT0 = System.nanoTime()
|
||||||
|
val shizukuTests = try {
|
||||||
ShizukuProbe().run(ctx, ids::uuid, ids::monoNs)
|
ShizukuProbe().run(ctx, ids::uuid, ids::monoNs)
|
||||||
} catch (t: Throwable) {
|
} catch (t: Throwable) {
|
||||||
Test(
|
listOf(Test(
|
||||||
id = ids.uuid(), type = TestType.LINK_IP_MONITOR, tier = Tier.SHIZUKU,
|
id = ids.uuid(), type = TestType.LINK_IP_MONITOR, tier = Tier.SHIZUKU,
|
||||||
startedMonoNs = ids.monoNs(), endedMonoNs = ids.monoNs(),
|
startedMonoNs = ids.monoNs(), endedMonoNs = ids.monoNs(),
|
||||||
status = TestStatus.FAILED, error = TestError("uncaught", t.message ?: t.javaClass.simpleName),
|
status = TestStatus.FAILED, error = TestError("uncaught", t.message ?: t.javaClass.simpleName),
|
||||||
)
|
))
|
||||||
|
}
|
||||||
|
// One key for the whole step: the three tests come out of one shell battery, and the
|
||||||
|
// bar plans them as one step.
|
||||||
|
settings.recordDurationMs(SHIZUKU_DURATION_KEY, (System.nanoTime() - shizukuT0) / 1_000_000)
|
||||||
|
tests.addAll(shizukuTests); collected.addAll(shizukuTests)
|
||||||
|
// "Shizuku tier ran" is the battery's verdict — the derived tests can be PARTIAL on a
|
||||||
|
// perfectly healthy shell tier (e.g. an RA-less v4-only link).
|
||||||
|
runShizukuOk = shizukuTests.any {
|
||||||
|
it.type == TestType.LINK_IP_MONITOR &&
|
||||||
|
(it.status == TestStatus.OK || it.status == TestStatus.PARTIAL)
|
||||||
}
|
}
|
||||||
tests.add(shizukuTest); collected.add(shizukuTest)
|
|
||||||
runShizukuOk = shizukuTest.status == TestStatus.OK || shizukuTest.status == TestStatus.PARTIAL
|
|
||||||
|
|
||||||
return buildDocument(tests)
|
return buildDocument(tests)
|
||||||
}
|
}
|
||||||
@@ -313,39 +498,189 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
|||||||
androidSdk = Build.VERSION.SDK_INT, androidRelease = Build.VERSION.RELEASE,
|
androidSdk = Build.VERSION.SDK_INT, androidRelease = Build.VERSION.RELEASE,
|
||||||
),
|
),
|
||||||
tiers = Tiers(app = true, shizuku = runShizukuOk),
|
tiers = Tiers(app = true, shizuku = runShizukuOk),
|
||||||
|
constraints = runConstraints,
|
||||||
),
|
),
|
||||||
networks = runNetworks,
|
networks = runNetworks,
|
||||||
tests = tests,
|
tests = tests,
|
||||||
findings = findings,
|
findings = findings,
|
||||||
summary = Verdicts.derive(tests, findings),
|
summary = Verdicts.derive(tests, findings, runConstraints),
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Was IPv6 actually provisioned on any network? A global (non-link-local) v6 address or a
|
* Was IPv6 provisioned on the network a test actually ran over?
|
||||||
* v6 default route means the network claims to offer IPv6 — link-local only does not count.
|
*
|
||||||
|
* This deliberately asks about one network rather than about the device. Answering "does any
|
||||||
|
* network here have IPv6" produces a real false positive on a phone, and it is not hypothetical:
|
||||||
|
* an IPv4-only wifi with working cellular alongside it reports "IPv6 is configured, but ICMPv6
|
||||||
|
* gets no reply" — configured on cellular, pinged over wifi, and the two never met.
|
||||||
|
*
|
||||||
|
* A global (non-link-local) address or a v6 default route means the network claims to offer
|
||||||
|
* IPv6; link-local only does not count, since every interface has one.
|
||||||
*/
|
*/
|
||||||
private fun ipv6Provisioned(networks: List<app.echo_lot.measurement.Network>): Boolean =
|
private fun ipv6Provisioned(
|
||||||
networks.any { n ->
|
networks: List<app.echo_lot.measurement.Network>,
|
||||||
|
networkRef: String?,
|
||||||
|
): Boolean {
|
||||||
|
// No reference means the test was not per-network; fall back to the device-wide reading
|
||||||
|
// rather than silently reporting nothing.
|
||||||
|
val scope = networks.filter { networkRef == null || it.id == networkRef }
|
||||||
|
return scope.any { n ->
|
||||||
n.link.addresses.any { a ->
|
n.link.addresses.any { a ->
|
||||||
a.addr.contains(':') &&
|
a.addr.contains(':') &&
|
||||||
!a.addr.startsWith("fe80", ignoreCase = true) &&
|
!a.addr.startsWith("fe80", ignoreCase = true) &&
|
||||||
!a.addr.startsWith("::1")
|
!a.addr.startsWith("::1")
|
||||||
} || n.link.routes.any { it.dst == "::/0" }
|
} || n.link.routes.any { it.dst == "::/0" }
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Per-network ICMP outcomes, keyed by network id.
|
||||||
|
*
|
||||||
|
* Reads the structured evidence the probe records rather than its prose detail — a finding
|
||||||
|
* that depended on the wording of a human-readable string would break silently the first time
|
||||||
|
* that wording improved.
|
||||||
|
*/
|
||||||
|
/** What one network's ICMP attempt did: whether it ran at all, and whether it was answered. */
|
||||||
|
private data class IcmpOutcome(val attempted: Boolean, val ok: Boolean)
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Per-network ICMP outcomes, keyed by network id.
|
||||||
|
*
|
||||||
|
* Reads the structured evidence the probe records rather than its prose detail — a finding
|
||||||
|
* that depended on the wording of a human-readable string would break silently the first time
|
||||||
|
* that wording improved.
|
||||||
|
*/
|
||||||
|
private fun icmpResults(t: Test): Map<String, IcmpOutcome> {
|
||||||
|
val out = HashMap<String, IcmpOutcome>()
|
||||||
|
val ev = t.evidence ?: return out
|
||||||
|
for ((_, v) in ev) {
|
||||||
|
val o = v as? kotlinx.serialization.json.JsonObject ?: continue
|
||||||
|
val ref = (o["network_ref"] as? kotlinx.serialization.json.JsonPrimitive)?.content ?: continue
|
||||||
|
fun flag(k: String) = (o[k] as? kotlinx.serialization.json.JsonPrimitive)?.content == "true"
|
||||||
|
out[ref] = IcmpOutcome(attempted = flag("attempted"), ok = flag("ok"))
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Human-facing name for the network a test ran over; falls back to something readable. */
|
||||||
|
private fun ifaceOf(networks: List<app.echo_lot.measurement.Network>, ref: String?): String =
|
||||||
|
networks.firstOrNull { it.id == ref }?.iface?.takeIf { it.isNotBlank() } ?: "this network"
|
||||||
|
|
||||||
/** Minimal first-pass findings from device-tier evidence; the registry grows with the suite. */
|
/** Minimal first-pass findings from device-tier evidence; the registry grows with the suite. */
|
||||||
private fun deriveFindings(tests: List<Test>, networks: List<app.echo_lot.measurement.Network>): List<Finding> {
|
private fun deriveFindings(tests: List<Test>, networks: List<app.echo_lot.measurement.Network>): List<Finding> {
|
||||||
val out = ArrayList<Finding>()
|
val out = ArrayList<Finding>()
|
||||||
val ids = RunIds()
|
val ids = RunIds()
|
||||||
|
val linkEvidence = tests.filter { it.type == TestType.LINK_SNAPSHOT }.map { EvidenceRef(it.id) }
|
||||||
|
|
||||||
|
// Said as a finding, not only as run.constraints: the constraints block is for machines
|
||||||
|
// aggregating thousands of runs, this is for the person reading this one. Both must exist —
|
||||||
|
// a constrained run with a quiet findings list still reads as "nothing wrong here".
|
||||||
|
if (runConstraints.constrained) {
|
||||||
|
val blocked = runConstraints.unmeasuredNetworks
|
||||||
|
.joinToString(", ") { id -> ifaceOf(networks, id) }
|
||||||
|
.ifBlank { "the underlying networks" }
|
||||||
|
// Three distinct situations share this finding code, and naming the wrong one costs
|
||||||
|
// trust: claiming "a VPN is active" right after the user disconnected theirs is how
|
||||||
|
// this text was first proven wrong on hardware.
|
||||||
|
val (title, description) = when {
|
||||||
|
runConstraints.vpnActive && runConstraints.perNetworkBlocked ->
|
||||||
|
"A VPN is active — $blocked could not be measured" to
|
||||||
|
("Android refuses to let apps send on the networks beneath an active " +
|
||||||
|
"VPN (that is how it prevents traffic leaking around the tunnel), " +
|
||||||
|
"so every per-network test here measured the tunnel or nothing. " +
|
||||||
|
"Nothing in this run says anything about $blocked. To measure them, " +
|
||||||
|
"disconnect the VPN and run again.")
|
||||||
|
runConstraints.perNetworkBlocked ->
|
||||||
|
"The OS refused sends on $blocked" to
|
||||||
|
("Android denied this app permission to send on $blocked (EPERM on " +
|
||||||
|
"bind). That is the restriction a VPN imposes on the networks " +
|
||||||
|
"beneath it — a tunnel disconnected moments ago can still leave it " +
|
||||||
|
"in place while it tears down. Nothing in this run says anything " +
|
||||||
|
"about $blocked; wait a few seconds and run again.")
|
||||||
|
else ->
|
||||||
|
"A VPN holds the default route" to
|
||||||
|
("Everything using the default route in this run describes the tunnel, " +
|
||||||
|
"not the network it rides on. Per-network measurements were " +
|
||||||
|
"permitted and did measure the underlying networks.")
|
||||||
|
}
|
||||||
|
out.add(
|
||||||
|
Finding(
|
||||||
|
id = ids.uuid(),
|
||||||
|
code = FindingRegistry.MEASUREMENT_VPN_CONSTRAINED.code,
|
||||||
|
category = FindingRegistry.MEASUREMENT_VPN_CONSTRAINED.category,
|
||||||
|
severity = FindingRegistry.MEASUREMENT_VPN_CONSTRAINED.severity,
|
||||||
|
confidence = Confidence.HIGH,
|
||||||
|
title = title,
|
||||||
|
description = description,
|
||||||
|
evidenceRefs = linkEvidence,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
val shapes = V6Analysis.classify(networks)
|
||||||
|
// Named per interface: on a phone several networks are up at once, and "IPv6 is broken" is
|
||||||
|
// useless when wifi is the broken one and cellular is fine.
|
||||||
|
for (sh in shapes.filter { it.addressWithoutRoute }) {
|
||||||
|
val where = if (sh.iface.isBlank()) "This device" else sh.iface
|
||||||
|
out.add(
|
||||||
|
Finding(
|
||||||
|
id = ids.uuid(),
|
||||||
|
code = FindingRegistry.V6_NO_DEFAULT_ROUTE.code,
|
||||||
|
category = FindingRegistry.V6_NO_DEFAULT_ROUTE.category,
|
||||||
|
severity = if (sh.tunnel) Severity.INFO else Severity.MEDIUM,
|
||||||
|
confidence = Confidence.HIGH,
|
||||||
|
title = if (sh.tunnel) {
|
||||||
|
"IPv6 reaches only the destinations a tunnel routes ($where)"
|
||||||
|
} else {
|
||||||
|
"IPv6 address with no default route ($where)"
|
||||||
|
},
|
||||||
|
description = "$where has a global IPv6 address but no IPv6 default route, so " +
|
||||||
|
"IPv6 reaches only destinations covered by a specific route. " +
|
||||||
|
if (sh.tunnel) {
|
||||||
|
"A tunnel interface holds those routes, so this looks deliberate. " +
|
||||||
|
"Worth knowing rather than fixing: applications holding a global " +
|
||||||
|
"address will still try IPv6 first and stall for anything outside " +
|
||||||
|
"the tunnel's routes."
|
||||||
|
} else {
|
||||||
|
"Nothing is routing the rest, so the network handed out an address it " +
|
||||||
|
"does not carry traffic for — applications will try IPv6 first " +
|
||||||
|
"and wait for it to fail."
|
||||||
|
},
|
||||||
|
evidenceRefs = linkEvidence,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
for (sh in shapes.filter { it.routeWithoutAddress }) {
|
||||||
|
val where = if (sh.iface.isBlank()) "This network" else sh.iface
|
||||||
|
out.add(
|
||||||
|
Finding(
|
||||||
|
id = ids.uuid(),
|
||||||
|
code = FindingRegistry.V6_ROUTE_WITHOUT_ADDRESS.code,
|
||||||
|
category = FindingRegistry.V6_ROUTE_WITHOUT_ADDRESS.category,
|
||||||
|
severity = Severity.MEDIUM,
|
||||||
|
confidence = Confidence.HIGH,
|
||||||
|
title = "IPv6 router advertised, but no address was configured ($where)",
|
||||||
|
description = "$where has an IPv6 default route but no global IPv6 address. " +
|
||||||
|
"The router is advertising itself as an IPv6 gateway while SLAAC produced " +
|
||||||
|
"no usable address — a missing prefix option, a prefix without the " +
|
||||||
|
"autonomous flag, or DHCPv6-only addressing that did not complete. Hosts " +
|
||||||
|
"believe IPv6 is available and pay a connection timeout on every " +
|
||||||
|
"dual-stack destination before falling back to IPv4, which is felt as " +
|
||||||
|
"general slowness with no packet loss to explain it.",
|
||||||
|
evidenceRefs = linkEvidence,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
for (t in tests) {
|
for (t in tests) {
|
||||||
if (t.type == TestType.NET_CAPTIVE_PORTAL) {
|
if (t.type == TestType.NET_CAPTIVE_PORTAL) {
|
||||||
val ev = t.evidence?.toString() ?: ""
|
val ev = t.evidence?.toString() ?: ""
|
||||||
when {
|
when {
|
||||||
ev.contains("\"captive_portal\"") -> out.add(
|
ev.contains("\"captive_portal\"") -> out.add(
|
||||||
Finding(
|
Finding(
|
||||||
id = ids.uuid(), code = "connectivity.captive_portal", category = Category.CONNECTIVITY,
|
id = ids.uuid(), code = FindingRegistry.CAPTIVE_PORTAL.code,
|
||||||
severity = Severity.MEDIUM, confidence = Confidence.HIGH,
|
category = FindingRegistry.CAPTIVE_PORTAL.category,
|
||||||
|
severity = FindingRegistry.CAPTIVE_PORTAL.severity, confidence = Confidence.HIGH,
|
||||||
title = "Captive portal intercepting connections",
|
title = "Captive portal intercepting connections",
|
||||||
description = "The generate_204 check returned a redirect or a page instead of HTTP 204 — a captive portal (login/splash page) is intercepting traffic on this network.",
|
description = "The generate_204 check returned a redirect or a page instead of HTTP 204 — a captive portal (login/splash page) is intercepting traffic on this network.",
|
||||||
evidenceRefs = listOf(EvidenceRef(t.id)),
|
evidenceRefs = listOf(EvidenceRef(t.id)),
|
||||||
@@ -353,8 +688,9 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
|||||||
)
|
)
|
||||||
t.status == TestStatus.FAILED -> out.add(
|
t.status == TestStatus.FAILED -> out.add(
|
||||||
Finding(
|
Finding(
|
||||||
id = ids.uuid(), code = "connectivity.no_internet", category = Category.CONNECTIVITY,
|
id = ids.uuid(), code = FindingRegistry.NO_INTERNET.code,
|
||||||
severity = Severity.HIGH, confidence = Confidence.HIGH,
|
category = FindingRegistry.NO_INTERNET.category,
|
||||||
|
severity = FindingRegistry.NO_INTERNET.severity, confidence = Confidence.HIGH,
|
||||||
title = "No working internet on any network",
|
title = "No working internet on any network",
|
||||||
description = "Android's own generate_204 connectivity checks failed on every active network (no HTTP 204) — this device has no validated internet path.",
|
description = "Android's own generate_204 connectivity checks failed on every active network (no HTTP 204) — this device has no validated internet path.",
|
||||||
evidenceRefs = listOf(EvidenceRef(t.id)),
|
evidenceRefs = listOf(EvidenceRef(t.id)),
|
||||||
@@ -367,8 +703,9 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
|||||||
if (ev.contains("MISMATCH")) {
|
if (ev.contains("MISMATCH")) {
|
||||||
out.add(
|
out.add(
|
||||||
Finding(
|
Finding(
|
||||||
id = ids.uuid(), code = "dns.answer_rewritten", category = Category.DNS,
|
id = ids.uuid(), code = FindingRegistry.DNS_ANSWER_REWRITTEN.code,
|
||||||
severity = Severity.HIGH, confidence = Confidence.HIGH,
|
category = FindingRegistry.DNS_ANSWER_REWRITTEN.category,
|
||||||
|
severity = FindingRegistry.DNS_ANSWER_REWRITTEN.severity, confidence = Confidence.HIGH,
|
||||||
title = "DNS answers are being rewritten",
|
title = "DNS answers are being rewritten",
|
||||||
description = "A canary reference record returned different RDATA than the spec-defined ground truth — something on the path is rewriting DNS answers (interception, filtering, or a middlebox).",
|
description = "A canary reference record returned different RDATA than the spec-defined ground truth — something on the path is rewriting DNS answers (interception, filtering, or a middlebox).",
|
||||||
evidenceRefs = listOf(EvidenceRef(t.id)),
|
evidenceRefs = listOf(EvidenceRef(t.id)),
|
||||||
@@ -377,8 +714,9 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
|||||||
} else if (ev.contains("\"reached_authoritative\":false")) {
|
} else if (ev.contains("\"reached_authoritative\":false")) {
|
||||||
out.add(
|
out.add(
|
||||||
Finding(
|
Finding(
|
||||||
id = ids.uuid(), code = "dns.authoritative_unreachable", category = Category.DNS,
|
id = ids.uuid(), code = FindingRegistry.DNS_AUTHORITATIVE_UNREACHABLE.code,
|
||||||
severity = Severity.MEDIUM, confidence = Confidence.MEDIUM,
|
category = FindingRegistry.DNS_AUTHORITATIVE_UNREACHABLE.category,
|
||||||
|
severity = FindingRegistry.DNS_AUTHORITATIVE_UNREACHABLE.severity, confidence = Confidence.MEDIUM,
|
||||||
title = "Canary queries don't reach the authoritative server",
|
title = "Canary queries don't reach the authoritative server",
|
||||||
description = "A per-run nonce name (which cannot be cached) was not answered by the canary server — the resolver is intercepting or failing to reach it.",
|
description = "A per-run nonce name (which cannot be cached) was not answered by the canary server — the resolver is intercepting or failing to reach it.",
|
||||||
evidenceRefs = listOf(EvidenceRef(t.id)),
|
evidenceRefs = listOf(EvidenceRef(t.id)),
|
||||||
@@ -391,8 +729,9 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
|||||||
if (ev.contains("address/port-dependent (symmetric NAT")) {
|
if (ev.contains("address/port-dependent (symmetric NAT")) {
|
||||||
out.add(
|
out.add(
|
||||||
Finding(
|
Finding(
|
||||||
id = ids.uuid(), code = "nat.symmetric", category = Category.NAT,
|
id = ids.uuid(), code = FindingRegistry.NAT_SYMMETRIC.code,
|
||||||
severity = Severity.MEDIUM, confidence = Confidence.HIGH,
|
category = FindingRegistry.NAT_SYMMETRIC.category,
|
||||||
|
severity = FindingRegistry.NAT_SYMMETRIC.severity, confidence = Confidence.HIGH,
|
||||||
title = "Symmetric NAT — peer-to-peer connections need a relay",
|
title = "Symmetric NAT — peer-to-peer connections need a relay",
|
||||||
description = "The NAT assigns a different external port per destination (address/port-dependent mapping). Direct peer-to-peer connections (calls, games, file transfer) will usually fail and fall back to relays.",
|
description = "The NAT assigns a different external port per destination (address/port-dependent mapping). Direct peer-to-peer connections (calls, games, file transfer) will usually fail and fall back to relays.",
|
||||||
evidenceRefs = listOf(EvidenceRef(t.id)),
|
evidenceRefs = listOf(EvidenceRef(t.id)),
|
||||||
@@ -400,29 +739,185 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (t.type == TestType.ICMP_PING6 && t.status == TestStatus.FAILED) {
|
if (t.type == TestType.DNS_RESOLVER && t.status == TestStatus.OK) {
|
||||||
// A network with no IPv6 at all is NORMAL — most networks are still IPv4-only,
|
// One finding per network: on a phone the wifi resolver can be wedged while
|
||||||
// and that is not a defect. What IS a defect is IPv6 that the network claims to
|
// cellular is fine, and "DNS is broken" would be wrong about half the device.
|
||||||
// provide (a global address or a default route from RA/DHCPv6) but that does not
|
val ev = t.evidence
|
||||||
// work: that causes Happy-Eyeballs delays, timeouts and hangs. So the severity
|
if (ev != null) {
|
||||||
// depends on whether v6 was provisioned at all.
|
for ((_, v) in ev) {
|
||||||
if (ipv6Provisioned(networks)) {
|
val o = v as? kotlinx.serialization.json.JsonObject ?: continue
|
||||||
out.add(
|
fun str(k: String) =
|
||||||
Finding(
|
(o[k] as? kotlinx.serialization.json.JsonPrimitive)?.content
|
||||||
id = ids.uuid(), code = "ipv6.broken", category = Category.IPV6,
|
val verdict = str("verdict")
|
||||||
severity = Severity.MEDIUM, confidence = Confidence.HIGH,
|
val ref0 = str("network_ref")
|
||||||
title = "IPv6 is configured but not working",
|
val iface0 = networks.firstOrNull { it.id == ref0 }?.iface
|
||||||
description = "This network advertises IPv6 (a global address and/or a default route), but ICMPv6 got no reply on any network. Half-configured IPv6 is worse than none: connections try IPv6 first and stall before falling back.",
|
?.takeIf { it.isNotBlank() } ?: "this network"
|
||||||
evidenceRefs = listOf(EvidenceRef(t.id)),
|
if (verdict == "search domain swallows queries") {
|
||||||
|
// Severity follows the harm, not the shape: the same misconfiguration
|
||||||
|
// is fatal on a resolver that tries the search form and invisible on
|
||||||
|
// one that does not, and saying "high" for a network that currently
|
||||||
|
// resolves fine would be crying wolf.
|
||||||
|
val breaking = str("system_resolves") != "true"
|
||||||
|
out.add(
|
||||||
|
Finding(
|
||||||
|
id = ids.uuid(),
|
||||||
|
code = FindingRegistry.DNS_SEARCH_DOMAIN_UNANSWERED.code,
|
||||||
|
category = FindingRegistry.DNS_SEARCH_DOMAIN_UNANSWERED.category,
|
||||||
|
severity = if (breaking) Severity.HIGH else Severity.MEDIUM,
|
||||||
|
confidence = Confidence.HIGH,
|
||||||
|
title = "The network's search domain swallows DNS queries ($iface0)",
|
||||||
|
description = "This network hands out " +
|
||||||
|
"${str("search_domains") ?: "a search domain"} as a DNS " +
|
||||||
|
"search domain, but its server never answers queries under " +
|
||||||
|
"it — not even to say the name does not exist. Resolvers " +
|
||||||
|
"append that domain to lookups, so they wait for a reply " +
|
||||||
|
"that never comes. " +
|
||||||
|
(if (breaking) {
|
||||||
|
"That is why names are not resolving on this device."
|
||||||
|
} else {
|
||||||
|
"Name resolution still works here, because this " +
|
||||||
|
"resolver tries the plain name first — another " +
|
||||||
|
"device on the same network may fail outright."
|
||||||
|
}) +
|
||||||
|
" Fix it on the router: either stop advertising the search " +
|
||||||
|
"domain, or make the server answer for it, including " +
|
||||||
|
"NXDOMAIN for names it does not have. Note that .local is " +
|
||||||
|
"reserved for mDNS (RFC 6762) and is widely dropped by " +
|
||||||
|
"design; home.arpa (RFC 8375) is the name reserved for this.",
|
||||||
|
evidenceRefs = listOf(EvidenceRef(t.id)),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if (verdict != "server answers, device resolver does not") continue
|
||||||
|
val ref = str("network_ref")
|
||||||
|
val where = networks.firstOrNull { it.id == ref }?.iface
|
||||||
|
?.takeIf { it.isNotBlank() } ?: "this network"
|
||||||
|
out.add(
|
||||||
|
Finding(
|
||||||
|
id = ids.uuid(),
|
||||||
|
code = FindingRegistry.DNS_SYSTEM_RESOLVER_BROKEN.code,
|
||||||
|
category = FindingRegistry.DNS_SYSTEM_RESOLVER_BROKEN.category,
|
||||||
|
severity = FindingRegistry.DNS_SYSTEM_RESOLVER_BROKEN.severity,
|
||||||
|
confidence = Confidence.HIGH,
|
||||||
|
title = "This device cannot resolve names, but the DNS server is fine ($where)",
|
||||||
|
description = "A DNS query sent straight from this device was " +
|
||||||
|
"answered by ${str("servers") ?: "the configured server"} with " +
|
||||||
|
"a valid result, yet asking Android to resolve the same name " +
|
||||||
|
"fails. Whatever is wrong sits between this device's resolver " +
|
||||||
|
"and a server that demonstrably works. " +
|
||||||
|
"Turning wifi off and on, or rejoining the network, clears the " +
|
||||||
|
"common case. If it survives a restart it is not a stuck " +
|
||||||
|
"resolver: look for something on this device that filters DNS " +
|
||||||
|
"— an ad blocker, a private-DNS or VPN app — or a per-device " +
|
||||||
|
"rule on the router aimed at this client.",
|
||||||
|
evidenceRefs = listOf(EvidenceRef(t.id)),
|
||||||
|
)
|
||||||
)
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (t.type == TestType.ICMP_PING6) {
|
||||||
|
// A network with no IPv6 at all is NORMAL — most networks are still IPv4-only, and
|
||||||
|
// that is not a defect. What IS a defect is IPv6 the network claims to provide (a
|
||||||
|
// global address or a default route from RA/DHCPv6) that does not work: that causes
|
||||||
|
// Happy-Eyeballs delays, timeouts and hangs.
|
||||||
|
//
|
||||||
|
// Judged per network, from the per-network evidence rather than the aggregate
|
||||||
|
// status. The aggregate can only say "some network answered", and on a phone with
|
||||||
|
// wifi and cellular up at once that is how "IPv6 is configured but gets no reply"
|
||||||
|
// ends up describing a network where IPv6 was never configured in the first place.
|
||||||
|
val results = icmpResults(t)
|
||||||
|
// The corroborating witness: did a real TCP connection over IPv6 work on this
|
||||||
|
// network? Same evidence shape as the ICMP probe, so the same parser reads it.
|
||||||
|
val v6ConnTest = tests.firstOrNull { it.type == TestType.V6_BROKENNESS }
|
||||||
|
val v6Conn = v6ConnTest?.let { icmpResults(it) } ?: emptyMap()
|
||||||
|
var anyV6Network = false
|
||||||
|
for (n in networks) {
|
||||||
|
val provisioned = ipv6Provisioned(networks, n.id)
|
||||||
|
if (provisioned) anyV6Network = true
|
||||||
|
val r = results[n.id] ?: continue
|
||||||
|
// Silence is only evidence if something was actually sent. A bind that failed
|
||||||
|
// with EPERM says the app could not use the interface, which is a fact about
|
||||||
|
// this app's permissions and says nothing whatsoever about the network.
|
||||||
|
if (!provisioned || !r.attempted || r.ok) continue
|
||||||
|
val where = n.iface?.takeIf { it.isNotBlank() } ?: "this network"
|
||||||
|
val conn = v6Conn[n.id]
|
||||||
|
val evidence = listOfNotNull(
|
||||||
|
EvidenceRef(t.id), v6ConnTest?.let { EvidenceRef(it.id) },
|
||||||
)
|
)
|
||||||
} else {
|
when {
|
||||||
|
// TCP over IPv6 worked: the silence is filtering, and can be said so.
|
||||||
|
conn?.ok == true -> out.add(
|
||||||
|
Finding(
|
||||||
|
id = ids.uuid(), code = FindingRegistry.V6_NO_ICMP_REPLY.code,
|
||||||
|
category = FindingRegistry.V6_NO_ICMP_REPLY.category,
|
||||||
|
severity = FindingRegistry.V6_NO_ICMP_REPLY.severity,
|
||||||
|
confidence = Confidence.HIGH,
|
||||||
|
title = "ICMPv6 is filtered here — IPv6 itself works ($where)",
|
||||||
|
description = "$where answered a real TCP connection over IPv6, " +
|
||||||
|
"so IPv6 works — but ICMPv6 echo got no reply, so something " +
|
||||||
|
"on this network filters ICMPv6. That is a fault in its own " +
|
||||||
|
"right even though connections succeed: Path MTU Discovery " +
|
||||||
|
"depends on ICMPv6, so large packets can vanish rather than " +
|
||||||
|
"being reported as too big.",
|
||||||
|
evidenceRefs = evidence,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
// Both transports failed on a network that advertises IPv6: broken, and
|
||||||
|
// now with the evidence the original v6.broken never had.
|
||||||
|
conn != null && conn.attempted -> out.add(
|
||||||
|
Finding(
|
||||||
|
id = ids.uuid(), code = FindingRegistry.V6_BROKEN.code,
|
||||||
|
category = FindingRegistry.V6_BROKEN.category,
|
||||||
|
severity = FindingRegistry.V6_BROKEN.severity,
|
||||||
|
confidence = Confidence.HIGH,
|
||||||
|
title = "IPv6 is advertised but does not work ($where)",
|
||||||
|
description = "$where advertises IPv6 (a global address and/or a " +
|
||||||
|
"default route), but neither ICMPv6 echo nor a TCP connection " +
|
||||||
|
"over IPv6 got through — two independent transports, both " +
|
||||||
|
"silent. Applications will try IPv6 first and wait out a " +
|
||||||
|
"timeout on every dual-stack destination before falling back " +
|
||||||
|
"to IPv4, felt as everything being slow with no loss to " +
|
||||||
|
"explain it. The network is announcing a service it does not " +
|
||||||
|
"deliver; the fix belongs on the router or upstream.",
|
||||||
|
evidenceRefs = evidence,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
// No corroboration available (no server configured, or the connect never
|
||||||
|
// got as far as sending): the honest two-explanation reading stands.
|
||||||
|
else -> out.add(
|
||||||
|
Finding(
|
||||||
|
id = ids.uuid(), code = FindingRegistry.V6_NO_ICMP_REPLY.code,
|
||||||
|
category = FindingRegistry.V6_NO_ICMP_REPLY.category,
|
||||||
|
severity = FindingRegistry.V6_NO_ICMP_REPLY.severity,
|
||||||
|
confidence = Confidence.MEDIUM,
|
||||||
|
title = "IPv6 is configured, but ICMPv6 gets no reply ($where)",
|
||||||
|
description = "$where advertises IPv6 (a global address and/or a " +
|
||||||
|
"default route), but ICMPv6 echo got no reply over it. That has " +
|
||||||
|
"two explanations which look identical from here: IPv6 is broken, " +
|
||||||
|
"or ICMPv6 is filtered while IPv6 itself works. Filtering is " +
|
||||||
|
"common and is a fault in its own right — it breaks Path MTU " +
|
||||||
|
"Discovery, so large packets vanish rather than being reported as " +
|
||||||
|
"too big.",
|
||||||
|
evidenceRefs = listOf(EvidenceRef(t.id)),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (!anyV6Network) {
|
||||||
|
// Said once for the device, not once per interface: "this network is IPv4-only"
|
||||||
|
// repeated per interface reads as several problems instead of one observation.
|
||||||
out.add(
|
out.add(
|
||||||
Finding(
|
Finding(
|
||||||
id = ids.uuid(), code = "ipv6.not_offered", category = Category.IPV6,
|
id = ids.uuid(), code = FindingRegistry.V6_NOT_OFFERED.code,
|
||||||
severity = Severity.INFO, confidence = Confidence.HIGH,
|
category = FindingRegistry.V6_NOT_OFFERED.category,
|
||||||
|
severity = FindingRegistry.V6_NOT_OFFERED.severity,
|
||||||
|
confidence = Confidence.HIGH,
|
||||||
title = "IPv4-only network (no IPv6 offered)",
|
title = "IPv4-only network (no IPv6 offered)",
|
||||||
description = "No IPv6 address or default route was provisioned, so IPv6 tests could not run. This is normal — many networks are still IPv4-only and it is not a fault.",
|
description = "No IPv6 address or default route was provisioned on " +
|
||||||
|
"any active network, so IPv6 tests could not run. This is normal " +
|
||||||
|
"— many networks are still IPv4-only and it is not a fault.",
|
||||||
evidenceRefs = listOf(EvidenceRef(t.id)),
|
evidenceRefs = listOf(EvidenceRef(t.id)),
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
@@ -438,4 +933,9 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
|||||||
etaSeconds = if (etaMs >= 0) ((etaMs + 999) / 1000).toInt() else state.etaSeconds,
|
etaSeconds = if (etaMs >= 0) ((etaMs + 999) / 1000).toInt() else state.etaSeconds,
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private companion object {
|
||||||
|
/** Duration-learning key for the Shizuku step, which is three tests but one battery. */
|
||||||
|
const val SHIZUKU_DURATION_KEY = "shizuku.battery"
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -50,6 +50,35 @@ class Settings(context: Context) {
|
|||||||
maxTotalBytes = maxTotalMb.toLong() * 1024 * 1024,
|
maxTotalBytes = maxTotalMb.toLong() * 1024 * 1024,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// ---- run-duration learning ---------------------------------------------------------
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Learned duration of one test type on THIS device, or null before the first run.
|
||||||
|
*
|
||||||
|
* The static Probe.estimatedMs values are only cold-start seeds: real durations depend on
|
||||||
|
* the phone and the network it stands in (ICMPv6 answers in milliseconds where IPv6 works
|
||||||
|
* and waits out full timeouts where it does not), so a fixed table is wrong for almost
|
||||||
|
* everyone almost always. What was measured last time is the only estimate that tracks
|
||||||
|
* reality.
|
||||||
|
*/
|
||||||
|
fun learnedDurationMs(type: String): Long? =
|
||||||
|
prefs.getLong("$DURATION_PREFIX$type", -1L).takeIf { it > 0 }
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Feeds one measured duration into the estimate — EMA, 70 % old / 30 % new. Heavy enough
|
||||||
|
* on history that a single odd run (a captive portal stalling DNS) does not whipsaw the
|
||||||
|
* bar, light enough that a real change (enrolling with a server un-skips three probes)
|
||||||
|
* converges within a few runs. Recorded whatever the test's status: a probe that skips in
|
||||||
|
* 2 ms will keep skipping in 2 ms until circumstances change, and then the EMA follows.
|
||||||
|
*/
|
||||||
|
fun recordDurationMs(type: String, ms: Long) {
|
||||||
|
if (ms < 0) return
|
||||||
|
val key = "$DURATION_PREFIX$type"
|
||||||
|
val old = prefs.getLong(key, -1L)
|
||||||
|
val next = if (old <= 0) ms else (old * 7 + ms * 3) / 10
|
||||||
|
prefs.edit().putLong(key, next).apply()
|
||||||
|
}
|
||||||
|
|
||||||
// ---- upload ----------------------------------------------------------------------
|
// ---- upload ----------------------------------------------------------------------
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -97,6 +126,41 @@ class Settings(context: Context) {
|
|||||||
get() = prefs.getString(SERVER_PIN, "") ?: ""
|
get() = prefs.getString(SERVER_PIN, "") ?: ""
|
||||||
set(v) = prefs.edit().putString(SERVER_PIN, v.trim()).apply()
|
set(v) = prefs.edit().putString(SERVER_PIN, v.trim()).apply()
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The address the operator handed out, for showing to a person.
|
||||||
|
*
|
||||||
|
* Separate from [serverUrl], which is the endpoint actually dialled. They differ when the
|
||||||
|
* server publishes one public name and points devices at another to select its pinned
|
||||||
|
* certificate — a detail worth keeping out of the user's face but not out of the settings.
|
||||||
|
*/
|
||||||
|
var serverPublicUrl: String
|
||||||
|
get() = (prefs.getString(SERVER_PUBLIC_URL, "") ?: "").ifBlank { serverUrl }
|
||||||
|
set(v) = prefs.edit().putString(SERVER_PUBLIC_URL, v.trim()).apply()
|
||||||
|
|
||||||
|
/**
|
||||||
|
* What the server said about itself, last time it was asked: addresses, ports, capabilities.
|
||||||
|
*
|
||||||
|
* Cached as a rendered block rather than as fields, because it is shown and never acted on —
|
||||||
|
* these are facts to read, not settings to apply, and storing them as settings would invite
|
||||||
|
* exactly the confusion of an editable box that changes nothing.
|
||||||
|
*/
|
||||||
|
var serverFacts: String
|
||||||
|
get() = prefs.getString(SERVER_FACTS, "") ?: ""
|
||||||
|
set(v) = prefs.edit().putString(SERVER_FACTS, v).apply()
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The server's own addresses, learned from its profile, for reaching it when DNS will not.
|
||||||
|
*
|
||||||
|
* Only the primaries: the alternate pair exists for NAT behaviour discovery and does not carry
|
||||||
|
* the control plane, so falling back to one would fail for a second, unrelated reason.
|
||||||
|
*/
|
||||||
|
var serverAddrs: String
|
||||||
|
get() = prefs.getString(SERVER_ADDRS, "") ?: ""
|
||||||
|
set(v) = prefs.edit().putString(SERVER_ADDRS, v).apply()
|
||||||
|
|
||||||
|
fun serverAddrList(): List<String> =
|
||||||
|
serverAddrs.split(',').map { it.trim() }.filter { it.isNotEmpty() }
|
||||||
|
|
||||||
var serverCredential: String
|
var serverCredential: String
|
||||||
get() = prefs.getString(SERVER_CRED, "") ?: ""
|
get() = prefs.getString(SERVER_CRED, "") ?: ""
|
||||||
set(v) = prefs.edit().putString(SERVER_CRED, v.trim()).apply()
|
set(v) = prefs.edit().putString(SERVER_CRED, v.trim()).apply()
|
||||||
@@ -104,6 +168,58 @@ class Settings(context: Context) {
|
|||||||
val serverConfigured: Boolean
|
val serverConfigured: Boolean
|
||||||
get() = serverUrl.isNotBlank() && serverPin.isNotBlank() && serverCredential.isNotBlank()
|
get() = serverUrl.isNotBlank() && serverPin.isNotBlank() && serverCredential.isNotBlank()
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The DNS zone this server is authoritative for, learned from its profile.
|
||||||
|
*
|
||||||
|
* Cached because the canary probe runs at device tier, before anything has talked to the
|
||||||
|
* server, and a probe that had to make a control-plane call first would fail on exactly the
|
||||||
|
* networks worth measuring. Empty means "not known yet", and the probe reports itself as
|
||||||
|
* skipped rather than inventing a zone.
|
||||||
|
*/
|
||||||
|
var canaryZone: String
|
||||||
|
get() = prefs.getString(CANARY_ZONE, "") ?: ""
|
||||||
|
set(v) = prefs.edit().putString(CANARY_ZONE, v.trim()).apply()
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Host part of the configured server URL, for probes that address it directly (STUN).
|
||||||
|
*
|
||||||
|
* Derived rather than stored: a second copy of the server's name is a second thing to keep in
|
||||||
|
* step, and it would go stale the moment someone re-enrolled against a different server.
|
||||||
|
*/
|
||||||
|
fun serverHost(): String = runCatching {
|
||||||
|
java.net.URI(serverUrl).host?.takeIf { it.isNotBlank() }
|
||||||
|
}.getOrNull() ?: ""
|
||||||
|
|
||||||
|
// ---- account ---------------------------------------------------------------------
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The PKCE verifier and state for a sign-in that is out at the browser.
|
||||||
|
*
|
||||||
|
* Persisted rather than held in memory because handing control to a browser backgrounds this
|
||||||
|
* process, and Android may kill it before the callback returns. An in-memory value works on a
|
||||||
|
* developer's device and fails on a phone under memory pressure.
|
||||||
|
*/
|
||||||
|
var pendingVerifier: String
|
||||||
|
get() = prefs.getString(PENDING_VERIFIER, "") ?: ""
|
||||||
|
set(v) = prefs.edit().putString(PENDING_VERIFIER, v).apply()
|
||||||
|
|
||||||
|
var pendingState: String
|
||||||
|
get() = prefs.getString(PENDING_STATE, "") ?: ""
|
||||||
|
set(v) = prefs.edit().putString(PENDING_STATE, v).apply()
|
||||||
|
|
||||||
|
fun clearPendingAuth() = prefs.edit().remove(PENDING_VERIFIER).remove(PENDING_STATE).apply()
|
||||||
|
|
||||||
|
/** Display name of whoever is signed in on this device; empty when nobody is. */
|
||||||
|
var accountName: String
|
||||||
|
get() = prefs.getString(ACCOUNT_NAME, "") ?: ""
|
||||||
|
set(v) = prefs.edit().putString(ACCOUNT_NAME, v).apply()
|
||||||
|
|
||||||
|
var accountId: String
|
||||||
|
get() = prefs.getString(ACCOUNT_ID, "") ?: ""
|
||||||
|
set(v) = prefs.edit().putString(ACCOUNT_ID, v).apply()
|
||||||
|
|
||||||
|
val signedIn: Boolean get() = accountName.isNotBlank()
|
||||||
|
|
||||||
private fun hex(s: String) = ByteArray(s.length / 2) {
|
private fun hex(s: String) = ByteArray(s.length / 2) {
|
||||||
((Character.digit(s[it * 2], 16) shl 4) or Character.digit(s[it * 2 + 1], 16)).toByte()
|
((Character.digit(s[it * 2], 16) shl 4) or Character.digit(s[it * 2 + 1], 16)).toByte()
|
||||||
}
|
}
|
||||||
@@ -120,5 +236,14 @@ class Settings(context: Context) {
|
|||||||
const val SERVER_URL = "server_url"
|
const val SERVER_URL = "server_url"
|
||||||
const val SERVER_PIN = "server_pin"
|
const val SERVER_PIN = "server_pin"
|
||||||
const val SERVER_CRED = "server_credential"
|
const val SERVER_CRED = "server_credential"
|
||||||
|
const val SERVER_PUBLIC_URL = "server_public_url"
|
||||||
|
const val SERVER_FACTS = "server_facts"
|
||||||
|
const val SERVER_ADDRS = "server_addrs"
|
||||||
|
const val CANARY_ZONE = "server_canary_zone"
|
||||||
|
const val PENDING_VERIFIER = "pending_auth_verifier"
|
||||||
|
const val PENDING_STATE = "pending_auth_state"
|
||||||
|
const val ACCOUNT_NAME = "account_name"
|
||||||
|
const val ACCOUNT_ID = "account_id"
|
||||||
|
const val DURATION_PREFIX = "duration_ms."
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -4,19 +4,24 @@
|
|||||||
package app.echo_lot.app
|
package app.echo_lot.app
|
||||||
|
|
||||||
import androidx.compose.foundation.layout.Arrangement
|
import androidx.compose.foundation.layout.Arrangement
|
||||||
|
import androidx.compose.foundation.layout.safeDrawingPadding
|
||||||
import androidx.compose.foundation.layout.Column
|
import androidx.compose.foundation.layout.Column
|
||||||
import androidx.compose.foundation.layout.Row
|
import androidx.compose.foundation.layout.Row
|
||||||
import androidx.compose.foundation.layout.Spacer
|
import androidx.compose.foundation.layout.Spacer
|
||||||
import androidx.compose.foundation.layout.fillMaxWidth
|
import androidx.compose.foundation.layout.fillMaxWidth
|
||||||
import androidx.compose.foundation.layout.height
|
import androidx.compose.foundation.layout.height
|
||||||
|
import androidx.compose.foundation.layout.width
|
||||||
import androidx.compose.foundation.layout.padding
|
import androidx.compose.foundation.layout.padding
|
||||||
|
import androidx.compose.foundation.shape.RoundedCornerShape
|
||||||
import androidx.compose.foundation.rememberScrollState
|
import androidx.compose.foundation.rememberScrollState
|
||||||
import androidx.compose.foundation.verticalScroll
|
import androidx.compose.foundation.verticalScroll
|
||||||
import androidx.compose.material3.Button
|
import androidx.compose.material3.Button
|
||||||
import androidx.compose.material3.Card
|
import androidx.compose.material3.Card
|
||||||
import androidx.compose.material3.FilterChip
|
import androidx.compose.material3.FilterChip
|
||||||
|
import androidx.compose.material3.LocalContentColor
|
||||||
import androidx.compose.material3.MaterialTheme
|
import androidx.compose.material3.MaterialTheme
|
||||||
import androidx.compose.material3.OutlinedTextField
|
import androidx.compose.material3.OutlinedTextField
|
||||||
|
import androidx.compose.material3.Surface
|
||||||
import androidx.compose.material3.Switch
|
import androidx.compose.material3.Switch
|
||||||
import androidx.compose.material3.Text
|
import androidx.compose.material3.Text
|
||||||
import androidx.compose.material3.TextButton
|
import androidx.compose.material3.TextButton
|
||||||
@@ -29,6 +34,7 @@ import androidx.compose.ui.Alignment
|
|||||||
import androidx.compose.ui.Modifier
|
import androidx.compose.ui.Modifier
|
||||||
import androidx.compose.ui.text.font.FontFamily
|
import androidx.compose.ui.text.font.FontFamily
|
||||||
import androidx.compose.ui.unit.dp
|
import androidx.compose.ui.unit.dp
|
||||||
|
import androidx.compose.ui.unit.sp
|
||||||
import app.echo_lot.privacy.PrivacyLevel
|
import app.echo_lot.privacy.PrivacyLevel
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -47,7 +53,12 @@ fun SettingsScreen(
|
|||||||
onDeleteAll: () -> Unit,
|
onDeleteAll: () -> Unit,
|
||||||
onPreviewUpload: () -> Unit,
|
onPreviewUpload: () -> Unit,
|
||||||
onCheckServer: () -> Unit,
|
onCheckServer: () -> Unit,
|
||||||
|
accountName: String,
|
||||||
|
onSignIn: () -> Unit,
|
||||||
|
onSignOut: () -> Unit,
|
||||||
|
onEnroll: (String) -> Unit,
|
||||||
serverStatus: String?,
|
serverStatus: String?,
|
||||||
|
enrollStatus: String?,
|
||||||
onBack: () -> Unit,
|
onBack: () -> Unit,
|
||||||
) {
|
) {
|
||||||
// SharedPreferences is not observable, so mirror each value into Compose state and write
|
// SharedPreferences is not observable, so mirror each value into Compose state and write
|
||||||
@@ -59,12 +70,26 @@ fun SettingsScreen(
|
|||||||
var autoUpload by remember { mutableStateOf(settings.autoUpload) }
|
var autoUpload by remember { mutableStateOf(settings.autoUpload) }
|
||||||
var privacy by remember { mutableStateOf(settings.privacyLevel) }
|
var privacy by remember { mutableStateOf(settings.privacyLevel) }
|
||||||
var stableSalt by remember { mutableStateOf(settings.stableSalt) }
|
var stableSalt by remember { mutableStateOf(settings.stableSalt) }
|
||||||
var serverUrl by remember { mutableStateOf(settings.serverUrl) }
|
var enrollLink by remember { mutableStateOf("") }
|
||||||
|
// The public name, which is what the operator handed out and what a person recognises. The
|
||||||
|
// endpoint actually dialled is shown beneath it when the two differ, rather than hidden — a
|
||||||
|
// network engineer debugging a connection wants to see where it really goes.
|
||||||
|
var serverUrl by remember { mutableStateOf(settings.serverPublicUrl) }
|
||||||
var serverPin by remember { mutableStateOf(settings.serverPin) }
|
var serverPin by remember { mutableStateOf(settings.serverPin) }
|
||||||
var serverCred by remember { mutableStateOf(settings.serverCredential) }
|
var serverCred by remember { mutableStateOf(settings.serverCredential) }
|
||||||
|
// Enrolling is asynchronous, so these are re-read when its result lands rather than when the
|
||||||
|
// button is pressed — reading them immediately showed the previous server's values and looked
|
||||||
|
// exactly like an enrollment that had silently done nothing.
|
||||||
|
var serverFacts by remember { mutableStateOf(settings.serverFacts) }
|
||||||
|
androidx.compose.runtime.LaunchedEffect(enrollStatus, serverStatus) {
|
||||||
|
serverFacts = settings.serverFacts
|
||||||
|
serverUrl = settings.serverPublicUrl
|
||||||
|
serverPin = settings.serverPin
|
||||||
|
serverCred = settings.serverCredential
|
||||||
|
}
|
||||||
|
|
||||||
Column(
|
Column(
|
||||||
Modifier.fillMaxWidth().verticalScroll(rememberScrollState()).padding(16.dp),
|
Modifier.fillMaxWidth().safeDrawingPadding().verticalScroll(rememberScrollState()).padding(16.dp),
|
||||||
verticalArrangement = Arrangement.spacedBy(12.dp),
|
verticalArrangement = Arrangement.spacedBy(12.dp),
|
||||||
) {
|
) {
|
||||||
Row(verticalAlignment = Alignment.CenterVertically) {
|
Row(verticalAlignment = Alignment.CenterVertically) {
|
||||||
@@ -127,18 +152,59 @@ fun SettingsScreen(
|
|||||||
}
|
}
|
||||||
Text(privacyExplanation(privacy), style = MaterialTheme.typography.bodySmall)
|
Text(privacyExplanation(privacy), style = MaterialTheme.typography.bodySmall)
|
||||||
|
|
||||||
|
// At FULL nothing is pseudonymized, so a salt has nothing to act on. Shown
|
||||||
|
// disabled rather than hidden: the setting is still stored and still applies the
|
||||||
|
// moment the level changes, and a control that vanishes hides that fact.
|
||||||
Toggle(
|
Toggle(
|
||||||
label = "Stable pseudonyms across runs",
|
label = "Stable pseudonyms across runs",
|
||||||
detail = "Lets you compare uploaded runs over time (same SSID reads the same " +
|
detail = if (privacy == PrivacyLevel.FULL) {
|
||||||
"each time). It also links your uploads together, so leave it off on a " +
|
"Not used at this level — nothing is pseudonymized, so there is nothing " +
|
||||||
"server you don't run yourself.",
|
"to keep stable. Choose balanced or strict to use this."
|
||||||
checked = stableSalt,
|
} else {
|
||||||
|
"Lets you compare uploaded runs over time (same SSID reads the same " +
|
||||||
|
"each time). It also links your uploads together, so leave it off on " +
|
||||||
|
"a server you don't run yourself."
|
||||||
|
},
|
||||||
|
checked = stableSalt && privacy != PrivacyLevel.FULL,
|
||||||
|
enabled = privacy != PrivacyLevel.FULL,
|
||||||
) { stableSalt = it; settings.stableSalt = it }
|
) { stableSalt = it; settings.stableSalt = it }
|
||||||
|
|
||||||
TextButton(onClick = onPreviewUpload) { Text("Preview what an upload would send") }
|
TextButton(onClick = onPreviewUpload) { Text("Preview what an upload would send") }
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---- account ----
|
||||||
|
Card(Modifier.fillMaxWidth()) {
|
||||||
|
Column(Modifier.padding(14.dp), verticalArrangement = Arrangement.spacedBy(8.dp)) {
|
||||||
|
Text("Account", style = MaterialTheme.typography.titleMedium)
|
||||||
|
if (accountName.isNotBlank()) {
|
||||||
|
Text("Signed in as $accountName", style = MaterialTheme.typography.bodyMedium)
|
||||||
|
Text(
|
||||||
|
"Runs from every device signed in to this account share one history.",
|
||||||
|
style = MaterialTheme.typography.bodySmall,
|
||||||
|
)
|
||||||
|
TextButton(onClick = onSignOut) { Text("Sign out") }
|
||||||
|
} else {
|
||||||
|
Text(
|
||||||
|
"Signing in is optional. It links this device to an account on your " +
|
||||||
|
"server, so several devices share one history — and some servers only " +
|
||||||
|
"accept uploads from a signed-in device.",
|
||||||
|
style = MaterialTheme.typography.bodySmall,
|
||||||
|
)
|
||||||
|
Button(onClick = onSignIn, enabled = settings.serverConfigured) {
|
||||||
|
Text("Sign in")
|
||||||
|
}
|
||||||
|
if (!settings.serverConfigured) {
|
||||||
|
Text(
|
||||||
|
"Enrol with a server first — the account belongs to the server, not " +
|
||||||
|
"to the app.",
|
||||||
|
style = MaterialTheme.typography.bodySmall,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ---- upload ----
|
// ---- upload ----
|
||||||
Card(Modifier.fillMaxWidth()) {
|
Card(Modifier.fillMaxWidth()) {
|
||||||
Column(Modifier.padding(14.dp), verticalArrangement = Arrangement.spacedBy(8.dp)) {
|
Column(Modifier.padding(14.dp), verticalArrangement = Arrangement.spacedBy(8.dp)) {
|
||||||
@@ -151,10 +217,57 @@ fun SettingsScreen(
|
|||||||
checked = autoUpload,
|
checked = autoUpload,
|
||||||
) { autoUpload = it; settings.autoUpload = it }
|
) { autoUpload = it; settings.autoUpload = it }
|
||||||
|
|
||||||
|
// Enrollment first, because it is the path that works: one link carries the
|
||||||
|
// URL, the pin and a single-use token. The three fields below exist for when
|
||||||
|
// someone has to reconstruct a configuration by hand, not as the normal route.
|
||||||
|
Text(
|
||||||
|
"Paste an enrollment link from your server operator, or scan its QR code. " +
|
||||||
|
"It fills in all three fields below. The link contains a one-time token — " +
|
||||||
|
"treat it like a password until it is used.",
|
||||||
|
style = MaterialTheme.typography.bodySmall,
|
||||||
|
)
|
||||||
OutlinedTextField(
|
OutlinedTextField(
|
||||||
value = serverUrl, onValueChange = { serverUrl = it; settings.serverUrl = it },
|
value = enrollLink, onValueChange = { enrollLink = it },
|
||||||
|
label = { Text("echolot://enroll?…") }, singleLine = true,
|
||||||
|
textStyle = MaterialTheme.typography.bodySmall.copy(fontFamily = FontFamily.Monospace),
|
||||||
|
modifier = Modifier.fillMaxWidth(),
|
||||||
|
)
|
||||||
|
Button(
|
||||||
|
onClick = {
|
||||||
|
onEnroll(enrollLink)
|
||||||
|
enrollLink = "" // spent either way; leaving it around invites a retry
|
||||||
|
},
|
||||||
|
enabled = enrollLink.isNotBlank(),
|
||||||
|
) { Text("Enroll") }
|
||||||
|
// Beside the button that caused it. Enrolling is asynchronous, so without this the
|
||||||
|
// only sign of success is three fields quietly changing further down the card.
|
||||||
|
enrollStatus?.let {
|
||||||
|
Text(it, style = MaterialTheme.typography.bodySmall)
|
||||||
|
}
|
||||||
|
|
||||||
|
OutlinedTextField(
|
||||||
|
value = serverUrl,
|
||||||
|
onValueChange = {
|
||||||
|
serverUrl = it
|
||||||
|
// Typed by hand there is no discovery to consult, so what was entered is
|
||||||
|
// both the public name and the endpoint. Setting only one of them would
|
||||||
|
// leave the app dialling the previous server.
|
||||||
|
settings.serverUrl = it
|
||||||
|
settings.serverPublicUrl = it
|
||||||
|
},
|
||||||
label = { Text("Server URL") }, singleLine = true, modifier = Modifier.fillMaxWidth(),
|
label = { Text("Server URL") }, singleLine = true, modifier = Modifier.fillMaxWidth(),
|
||||||
)
|
)
|
||||||
|
// Directly under the field it explains. Anywhere else it reads as a stray sentence
|
||||||
|
// about some other part of the screen.
|
||||||
|
if (settings.serverUrl.isNotBlank() && settings.serverUrl != settings.serverPublicUrl) {
|
||||||
|
Text(
|
||||||
|
"Connects to ${settings.serverUrl} — this server publishes one name and " +
|
||||||
|
"points devices at another, so its pinned certificate can share a port " +
|
||||||
|
"with its web interface.",
|
||||||
|
style = MaterialTheme.typography.bodySmall,
|
||||||
|
color = LocalContentColor.current.copy(alpha = 0.7f),
|
||||||
|
)
|
||||||
|
}
|
||||||
OutlinedTextField(
|
OutlinedTextField(
|
||||||
value = serverPin, onValueChange = { serverPin = it; settings.serverPin = it },
|
value = serverPin, onValueChange = { serverPin = it; settings.serverPin = it },
|
||||||
label = { Text("Certificate pin (SPKI, base64)") }, singleLine = true,
|
label = { Text("Certificate pin (SPKI, base64)") }, singleLine = true,
|
||||||
@@ -181,6 +294,52 @@ fun SettingsScreen(
|
|||||||
serverStatus?.let {
|
serverStatus?.let {
|
||||||
Text(it, style = MaterialTheme.typography.bodySmall)
|
Text(it, style = MaterialTheme.typography.bodySmall)
|
||||||
}
|
}
|
||||||
|
// What the server reported, placed under the button that asks it rather than among
|
||||||
|
// the fields above: these are facts to read, not settings to apply, and an
|
||||||
|
// editable-looking box that changes nothing is worse than no box at all.
|
||||||
|
//
|
||||||
|
// Monospaced so the addresses line up under each other — column alignment is most
|
||||||
|
// of what makes a list of IPs quicker to read than prose.
|
||||||
|
if (serverFacts.isNotBlank()) {
|
||||||
|
Surface(
|
||||||
|
color = MaterialTheme.colorScheme.surfaceVariant,
|
||||||
|
shape = RoundedCornerShape(8.dp),
|
||||||
|
modifier = Modifier.fillMaxWidth(),
|
||||||
|
) {
|
||||||
|
Column(
|
||||||
|
Modifier.padding(horizontal = 12.dp, vertical = 10.dp),
|
||||||
|
verticalArrangement = Arrangement.spacedBy(2.dp),
|
||||||
|
) {
|
||||||
|
Text(
|
||||||
|
"WHAT THIS SERVER REPORTS",
|
||||||
|
style = MaterialTheme.typography.labelSmall,
|
||||||
|
color = LocalContentColor.current.copy(alpha = 0.7f),
|
||||||
|
)
|
||||||
|
// Real columns rather than padded text: the label column has a fixed
|
||||||
|
// width, so values line up whatever the font does, and a long value
|
||||||
|
// wraps inside its own column instead of under the labels.
|
||||||
|
for (line in serverFacts.lines()) {
|
||||||
|
val label = line.substringBefore('|')
|
||||||
|
val value = line.substringAfter('|', "")
|
||||||
|
Row(Modifier.fillMaxWidth()) {
|
||||||
|
Text(
|
||||||
|
label,
|
||||||
|
style = MaterialTheme.typography.bodySmall,
|
||||||
|
color = LocalContentColor.current.copy(alpha = 0.7f),
|
||||||
|
modifier = Modifier.width(72.dp),
|
||||||
|
)
|
||||||
|
Text(
|
||||||
|
value,
|
||||||
|
style = MaterialTheme.typography.bodySmall.copy(
|
||||||
|
fontFamily = FontFamily.Monospace,
|
||||||
|
),
|
||||||
|
modifier = Modifier.weight(1f),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
Text(
|
Text(
|
||||||
"This app is ${BuildConfig.APP_SEMVER} and speaks probe protocol " +
|
"This app is ${BuildConfig.APP_SEMVER} and speaks probe protocol " +
|
||||||
"${app.echo_lot.protocol.Compat.PROTOCOL_VERSION}. It works with servers " +
|
"${app.echo_lot.protocol.Compat.PROTOCOL_VERSION}. It works with servers " +
|
||||||
@@ -208,13 +367,22 @@ private fun privacyExplanation(level: PrivacyLevel): String = when (level) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
@Composable
|
@Composable
|
||||||
private fun Toggle(label: String, detail: String, checked: Boolean, onChange: (Boolean) -> Unit) {
|
private fun Toggle(
|
||||||
|
label: String,
|
||||||
|
detail: String,
|
||||||
|
checked: Boolean,
|
||||||
|
enabled: Boolean = true,
|
||||||
|
onChange: (Boolean) -> Unit,
|
||||||
|
) {
|
||||||
Row(Modifier.fillMaxWidth(), verticalAlignment = Alignment.Top) {
|
Row(Modifier.fillMaxWidth(), verticalAlignment = Alignment.Top) {
|
||||||
Column(Modifier.weight(1f)) {
|
Column(Modifier.weight(1f)) {
|
||||||
Text(label, style = MaterialTheme.typography.bodyMedium)
|
// Dimmed together with the switch, so "this does nothing right now" reads at a glance
|
||||||
Text(detail, style = MaterialTheme.typography.bodySmall)
|
// instead of only on close inspection.
|
||||||
|
val alpha = if (enabled) 1f else 0.5f
|
||||||
|
Text(label, style = MaterialTheme.typography.bodyMedium, color = LocalContentColor.current.copy(alpha = alpha))
|
||||||
|
Text(detail, style = MaterialTheme.typography.bodySmall, color = LocalContentColor.current.copy(alpha = alpha))
|
||||||
}
|
}
|
||||||
Switch(checked = checked, onCheckedChange = onChange)
|
Switch(checked = checked, onCheckedChange = onChange, enabled = enabled)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -31,10 +31,23 @@ data class ArchivedRun(
|
|||||||
val verdict: String? = null,
|
val verdict: String? = null,
|
||||||
@SerialName("finding_count") val findingCount: Int = 0,
|
@SerialName("finding_count") val findingCount: Int = 0,
|
||||||
@SerialName("size_bytes") val sizeBytes: Long = 0,
|
@SerialName("size_bytes") val sizeBytes: Long = 0,
|
||||||
|
/**
|
||||||
|
* How the *archived* document is redacted. Always "full" in practice, because the archive
|
||||||
|
* deliberately keeps the unredacted run - see the package doc. This is not what was uploaded.
|
||||||
|
*/
|
||||||
val anonymization: String = "full",
|
val anonymization: String = "full",
|
||||||
/** Whether this run has been accepted by a server, so history can show what is backed up. */
|
/** Whether this run has been accepted by a server, so history can show what is backed up. */
|
||||||
val uploaded: Boolean = false,
|
val uploaded: Boolean = false,
|
||||||
@SerialName("uploaded_to") val uploadedTo: String? = null,
|
@SerialName("uploaded_to") val uploadedTo: String? = null,
|
||||||
|
/**
|
||||||
|
* The level the run was *uploaded* at, which is a different document from the archived one.
|
||||||
|
*
|
||||||
|
* Kept separately because conflating the two is actively misleading: the history row showed
|
||||||
|
* the archive's own level ("full") directly beneath "uploaded to fmr", which reads as "the
|
||||||
|
* complete data was uploaded" when a redacted copy had been sent. A privacy display that
|
||||||
|
* overstates what left the device is worse than none.
|
||||||
|
*/
|
||||||
|
@SerialName("uploaded_as") val uploadedAs: String? = null,
|
||||||
)
|
)
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -119,7 +132,7 @@ class RunArchive(private val dir: File, private val now: () -> Long = System::cu
|
|||||||
fun deleteAll(): Int = list().count { delete(it.id) }
|
fun deleteAll(): Int = list().count { delete(it.id) }
|
||||||
|
|
||||||
/** Records that a server accepted this run, so history can distinguish backed-up from local. */
|
/** Records that a server accepted this run, so history can distinguish backed-up from local. */
|
||||||
fun markUploaded(id: String, serverName: String) {
|
fun markUploaded(id: String, serverName: String, uploadedAs: String? = null) {
|
||||||
val f = File(dir, safe(id) + META_EXT)
|
val f = File(dir, safe(id) + META_EXT)
|
||||||
val meta = runCatching { json.decodeFromString(ArchivedRun.serializer(), f.readText()) }.getOrNull()
|
val meta = runCatching { json.decodeFromString(ArchivedRun.serializer(), f.readText()) }.getOrNull()
|
||||||
?: return
|
?: return
|
||||||
@@ -127,7 +140,7 @@ class RunArchive(private val dir: File, private val now: () -> Long = System::cu
|
|||||||
f,
|
f,
|
||||||
json.encodeToString(
|
json.encodeToString(
|
||||||
ArchivedRun.serializer(),
|
ArchivedRun.serializer(),
|
||||||
meta.copy(uploaded = true, uploadedTo = serverName),
|
meta.copy(uploaded = true, uploadedTo = serverName, uploadedAs = uploadedAs),
|
||||||
),
|
),
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
@@ -181,7 +194,10 @@ class RunArchive(private val dir: File, private val now: () -> Long = System::cu
|
|||||||
id = id,
|
id = id,
|
||||||
savedAtEpochMs = now(),
|
savedAtEpochMs = now(),
|
||||||
startedAt = run["started_at"]?.jsonPrimitive?.content,
|
startedAt = run["started_at"]?.jsonPrimitive?.content,
|
||||||
verdict = doc["summary"]?.jsonObject?.get("verdict")?.jsonPrimitive?.content,
|
// The schema calls it `overall` (Summary.overall); reading `verdict` here silently
|
||||||
|
// yielded null for every run, so the history list's most prominent element - the
|
||||||
|
// coloured verdict - was blank on every row.
|
||||||
|
verdict = doc["summary"]?.jsonObject?.get("overall")?.jsonPrimitive?.content,
|
||||||
findingCount = (doc["findings"] as? kotlinx.serialization.json.JsonArray)?.size ?: 0,
|
findingCount = (doc["findings"] as? kotlinx.serialization.json.JsonArray)?.size ?: 0,
|
||||||
sizeBytes = size,
|
sizeBytes = size,
|
||||||
anonymization = run["privacy"]?.jsonObject?.get("anonymization")?.jsonPrimitive?.content ?: "full",
|
anonymization = run["privacy"]?.jsonObject?.get("anonymization")?.jsonPrimitive?.content ?: "full",
|
||||||
|
|||||||
@@ -25,7 +25,7 @@ class RunArchiveTest {
|
|||||||
private fun doc(id: String, findings: Int = 1, pad: Int = 0): String {
|
private fun doc(id: String, findings: Int = 1, pad: Int = 0): String {
|
||||||
val f = (1..findings).joinToString(",") { """{"id":"f$it"}""" }
|
val f = (1..findings).joinToString(",") { """{"id":"f$it"}""" }
|
||||||
return """{"run":{"id":"$id","started_at":"2026-08-01T10:00:00Z","privacy":{"anonymization":"balanced"}},""" +
|
return """{"run":{"id":"$id","started_at":"2026-08-01T10:00:00Z","privacy":{"anonymization":"balanced"}},""" +
|
||||||
""""findings":[$f],"summary":{"verdict":"warn"},"pad":"${"x".repeat(pad)}"}"""
|
""""findings":[$f],"summary":{"overall":"warn"},"pad":"${"x".repeat(pad)}"}"""
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -120,13 +120,40 @@ class RunArchiveTest {
|
|||||||
fun uploadStateIsRecorded() {
|
fun uploadStateIsRecorded() {
|
||||||
val a = archive()
|
val a = archive()
|
||||||
a.save(doc("run-1"))
|
a.save(doc("run-1"))
|
||||||
a.markUploaded("run-1", "fmr")
|
a.markUploaded("run-1", "fmr", "balanced")
|
||||||
val meta = a.list().single()
|
val meta = a.list().single()
|
||||||
assertTrue(meta.uploaded)
|
assertTrue(meta.uploaded)
|
||||||
assertEquals("fmr", meta.uploadedTo)
|
assertEquals("fmr", meta.uploadedTo)
|
||||||
assertEquals("run-1", meta.id, "marking upload must not disturb the rest of the entry")
|
assertEquals("run-1", meta.id, "marking upload must not disturb the rest of the entry")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The archive's own level and the level a run was uploaded at describe *different documents*.
|
||||||
|
// Showing the archive's ("full", because the archive is deliberately unredacted) next to
|
||||||
|
// "uploaded to fmr" reads as "the complete data was uploaded" when a redacted copy was sent —
|
||||||
|
// a privacy display that overstates what left the device is worse than none.
|
||||||
|
@Test
|
||||||
|
fun theUploadedLevelIsRecordedSeparatelyFromTheArchivedOne() {
|
||||||
|
val a = archive()
|
||||||
|
// A real archived document carries no privacy stamp: the anonymizer never runs on the
|
||||||
|
// archive. The shared doc() fixture has one, which is exactly the unrealism that let this
|
||||||
|
// confusion through in the first place.
|
||||||
|
a.save("""{"run":{"id":"run-1"},"findings":[],"summary":{"overall":"green"}}""")
|
||||||
|
a.markUploaded("run-1", "fmr", "balanced")
|
||||||
|
val meta = a.list().single()
|
||||||
|
assertEquals("full", meta.anonymization, "the archived copy is unredacted, by design")
|
||||||
|
assertEquals("balanced", meta.uploadedAs, "the uploaded copy was redacted, and must say so")
|
||||||
|
}
|
||||||
|
|
||||||
|
// The verdict is read from `summary.overall` — the schema's actual field name. Reading
|
||||||
|
// `summary.verdict` silently yielded null for every run, so the history list's most prominent
|
||||||
|
// element was blank on every row while everything else looked fine.
|
||||||
|
@Test
|
||||||
|
fun theVerdictComesFromTheSchemasOverallField() {
|
||||||
|
val a = archive()
|
||||||
|
a.save("""{"run":{"id":"r1"},"findings":[],"summary":{"overall":"yellow"}}""")
|
||||||
|
assertEquals("yellow", a.list().single().verdict)
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
fun deleteRemovesBothFiles() {
|
fun deleteRemovesBothFiles() {
|
||||||
val a = archive()
|
val a = archive()
|
||||||
|
|||||||
@@ -25,6 +25,6 @@ java { sourceCompatibility = JavaVersion.VERSION_17; targetCompatibility = JavaV
|
|||||||
|
|
||||||
tasks.test {
|
tasks.test {
|
||||||
useJUnitPlatform()
|
useJUnitPlatform()
|
||||||
listOf("ECHOLOT_LIVE_URL","ECHOLOT_LIVE_PIN","ECHOLOT_LIVE_CRED","ECHOLOT_LIVE_UDP","ECHOLOT_LIVE_TARGET")
|
listOf("ECHOLOT_LIVE_URL","ECHOLOT_LIVE_PIN","ECHOLOT_LIVE_CRED","ECHOLOT_LIVE_UDP","ECHOLOT_LIVE_TARGET","ECHOLOT_ENROLL_URI")
|
||||||
.forEach { k -> System.getenv(k)?.let { environment(k, it) } }
|
.forEach { k -> System.getenv(k)?.let { environment(k, it) } }
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,107 @@
|
|||||||
|
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||||
|
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||||
|
|
||||||
|
package app.echo_lot.engine
|
||||||
|
|
||||||
|
import kotlinx.serialization.SerialName
|
||||||
|
import kotlinx.serialization.Serializable
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Splits a round-trip train into its two directions using what the server witnessed.
|
||||||
|
*
|
||||||
|
* A round trip can only report that *something* was lost somewhere. That is the least useful form
|
||||||
|
* of the answer: "3 % loss" sends an engineer looking in both directions at once. The server
|
||||||
|
* records every packet it received, per sequence number (probe-protocol.md §6), so the two cases
|
||||||
|
* are actually distinguishable:
|
||||||
|
*
|
||||||
|
* - sent, never seen by the server → **upstream** loss
|
||||||
|
* - seen by the server, reply never arrived → **downstream** loss
|
||||||
|
*
|
||||||
|
* The same records give one-way delay *variation* per direction. Absolute one-way delay would
|
||||||
|
* need synchronised clocks and we deliberately have none (measurement-schema.md's two-clock rule),
|
||||||
|
* but the variation does not: (server_rx − client_tx) contains an unknown constant clock offset,
|
||||||
|
* and differencing successive samples cancels it. So jitter is honestly attributable to a
|
||||||
|
* direction even though latency is not.
|
||||||
|
*/
|
||||||
|
object Directional {
|
||||||
|
|
||||||
|
/** One probe as the client saw it. [tRxNs] null means no reply came back. */
|
||||||
|
data class Sample(val seq: Int, val tTxNs: Long, val tRxNs: Long?)
|
||||||
|
|
||||||
|
/** One probe as the server saw it: its own receive and transmit stamps, on its own clock. */
|
||||||
|
data class ServerSighting(val seq: Int, val tRxNs: Long, val tTxNs: Long)
|
||||||
|
|
||||||
|
fun analyse(sent: List<Sample>, seen: List<ServerSighting>): DirectionalMetrics {
|
||||||
|
val byServerSeq = seen.associateBy { it.seq }
|
||||||
|
// Only sequences we actually sent count. A server record for a sequence we have no note
|
||||||
|
// of is not evidence about this train — it is a bug or a stray, and silently folding it
|
||||||
|
// in would produce loss percentages above 100 or below zero.
|
||||||
|
val relevant = sent.filter { byServerSeq.containsKey(it.seq) }
|
||||||
|
|
||||||
|
val nSent = sent.size
|
||||||
|
val nSeen = relevant.size
|
||||||
|
val nReplied = sent.count { it.tRxNs != null }
|
||||||
|
|
||||||
|
// A reply can only exist if the request arrived, so downstream loss is measured against
|
||||||
|
// what the server saw, not against what we sent — otherwise upstream loss is counted twice.
|
||||||
|
val lostUp = nSent - nSeen
|
||||||
|
val lostDown = (nSeen - nReplied).coerceAtLeast(0)
|
||||||
|
|
||||||
|
val upDeltas = relevant.sortedBy { it.seq }
|
||||||
|
.map { byServerSeq.getValue(it.seq).tRxNs - it.tTxNs }
|
||||||
|
val downDeltas = sent.filter { it.tRxNs != null && byServerSeq.containsKey(it.seq) }
|
||||||
|
.sortedBy { it.seq }
|
||||||
|
.map { it.tRxNs!! - byServerSeq.getValue(it.seq).tTxNs }
|
||||||
|
|
||||||
|
return DirectionalMetrics(
|
||||||
|
sent = nSent,
|
||||||
|
seenByServer = nSeen,
|
||||||
|
repliesReceived = nReplied,
|
||||||
|
lostUpstream = lostUp,
|
||||||
|
lostDownstream = lostDown,
|
||||||
|
lossUpstreamPct = pct(lostUp, nSent),
|
||||||
|
// Denominator is what reached the server: of the packets that got there, how many
|
||||||
|
// replies came back.
|
||||||
|
lossDownstreamPct = pct(lostDown, nSeen),
|
||||||
|
jitterUpstreamMs = jitterMs(upDeltas),
|
||||||
|
jitterDownstreamMs = jitterMs(downDeltas),
|
||||||
|
/** True when the server saw nothing at all, which is a different fault from loss. */
|
||||||
|
noneReachedServer = nSent > 0 && nSeen == 0,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Mean absolute difference between consecutive one-way samples (RFC 3393 IPDV, averaged).
|
||||||
|
*
|
||||||
|
* Differencing is what makes this legitimate without synchronised clocks: each sample carries
|
||||||
|
* the same unknown offset between the two clocks, and the difference cancels it. Fewer than
|
||||||
|
* two samples yields null rather than zero — "no jitter" and "not enough data to say" are
|
||||||
|
* different claims and only one of them is true here.
|
||||||
|
*/
|
||||||
|
private fun jitterMs(oneWayNs: List<Long>): Double? {
|
||||||
|
if (oneWayNs.size < 2) return null
|
||||||
|
val deltas = oneWayNs.zipWithNext { a, b -> kotlin.math.abs(b - a) }
|
||||||
|
return round2(deltas.average() / 1_000_000.0)
|
||||||
|
}
|
||||||
|
|
||||||
|
private fun pct(part: Int, whole: Int): Double =
|
||||||
|
if (whole <= 0) 0.0 else round2(part * 100.0 / whole)
|
||||||
|
|
||||||
|
private fun round2(v: Double) = Math.round(v * 100.0) / 100.0
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Directional metrics for train.udp_updown; recomputable from the columnar evidence. */
|
||||||
|
@Serializable
|
||||||
|
data class DirectionalMetrics(
|
||||||
|
val sent: Int,
|
||||||
|
@SerialName("seen_by_server") val seenByServer: Int,
|
||||||
|
@SerialName("replies_received") val repliesReceived: Int,
|
||||||
|
@SerialName("lost_upstream") val lostUpstream: Int,
|
||||||
|
@SerialName("lost_downstream") val lostDownstream: Int,
|
||||||
|
@SerialName("loss_upstream_pct") val lossUpstreamPct: Double,
|
||||||
|
@SerialName("loss_downstream_pct") val lossDownstreamPct: Double,
|
||||||
|
/** One-way delay variation (RFC 3393), per direction. Null when there were too few samples. */
|
||||||
|
@SerialName("jitter_upstream_ms") val jitterUpstreamMs: Double? = null,
|
||||||
|
@SerialName("jitter_downstream_ms") val jitterDownstreamMs: Double? = null,
|
||||||
|
@SerialName("none_reached_server") val noneReachedServer: Boolean = false,
|
||||||
|
)
|
||||||
+176
-10
@@ -37,6 +37,120 @@ class DownstreamMeasurement(private val ids: IdSource) {
|
|||||||
/** How long to wait for a granted burst after the server accepts the action. */
|
/** How long to wait for a granted burst after the server accepts the action. */
|
||||||
private val collectWindowMs = 4_000L
|
private val collectWindowMs = 4_000L
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Shorter, but long enough to cover the first_last mode's deliberate 250 ms hold plus a
|
||||||
|
* reassembly. A fragment burst is one datagram: it is here quickly or not at all.
|
||||||
|
*/
|
||||||
|
private val fragWindowMs = 1_500L
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Asks the server to send one deliberately-fragmented datagram per ordering, and reports
|
||||||
|
* which orderings survive the path.
|
||||||
|
*
|
||||||
|
* Kernel fragmentation always emits fragments in order, first one first, so an oversized
|
||||||
|
* datagram can only answer "do fragments get through at all". The interesting fault is about
|
||||||
|
* ordering: only the *first* fragment carries the UDP ports, so a stateful firewall that has
|
||||||
|
* not seen it has nothing to match the rest against, and many drop them. That failure is
|
||||||
|
* invisible to every in-order test and shows up in the field as "large DNS answers fail here"
|
||||||
|
* or "the tunnel breaks when the MTU drops".
|
||||||
|
*/
|
||||||
|
fun fragmentOrdering(
|
||||||
|
credential: String,
|
||||||
|
sessionId: String,
|
||||||
|
control: ControlClient,
|
||||||
|
probe: ProbeSession,
|
||||||
|
sessionRef: String,
|
||||||
|
sizeBytes: Int = 2000,
|
||||||
|
fragBytes: Int = 576,
|
||||||
|
): Pair<Test, List<Finding>> {
|
||||||
|
val testId = ids.uuid()
|
||||||
|
val started = ids.monoNs()
|
||||||
|
val delivered = LinkedHashMap<String, Boolean>()
|
||||||
|
val fragmentCounts = LinkedHashMap<String, Int>()
|
||||||
|
var unsupported = false
|
||||||
|
|
||||||
|
for (mode in FRAG_MODES) {
|
||||||
|
val reply = runCatching {
|
||||||
|
control.action(
|
||||||
|
credential, sessionId,
|
||||||
|
"""{"action":"frag_send","size_bytes":$sizeBytes,"mode":"$mode","frag_bytes":$fragBytes}""",
|
||||||
|
)
|
||||||
|
}
|
||||||
|
if (reply.isFailure) {
|
||||||
|
// A server without a raw socket says so; that is a missing capability, not a
|
||||||
|
// property of the network, and must not be recorded as a failed delivery.
|
||||||
|
unsupported = true
|
||||||
|
break
|
||||||
|
}
|
||||||
|
parseInt(reply.getOrNull(), "fragments")?.let { fragmentCounts[mode] = it }
|
||||||
|
// The burst is already on the wire when the action returns (it is sent
|
||||||
|
// synchronously), so anything that survived is either here or lost.
|
||||||
|
val got = probe.collectGranted(fragWindowMs).any { it.type == Wire.TYPE_FRAG_DATA }
|
||||||
|
delivered[mode] = got
|
||||||
|
}
|
||||||
|
|
||||||
|
if (unsupported) {
|
||||||
|
return Test(
|
||||||
|
id = testId, type = TestType.MTU_FRAG_ORDERING, sessionRef = sessionRef, tier = Tier.APP,
|
||||||
|
startedMonoNs = started, endedMonoNs = ids.monoNs(),
|
||||||
|
status = TestStatus.UNSUPPORTED,
|
||||||
|
error = TestError("no_raw_socket", "this server cannot craft fragments"),
|
||||||
|
) to emptyList()
|
||||||
|
}
|
||||||
|
|
||||||
|
val metrics = json.encodeToJsonElement(
|
||||||
|
FragOrderingMetrics(
|
||||||
|
sizeBytes = sizeBytes,
|
||||||
|
fragBytes = fragBytes,
|
||||||
|
fragmentsPerBurst = fragmentCounts,
|
||||||
|
deliveredByMode = delivered,
|
||||||
|
inOrderDelivered = delivered[FRAG_IN_ORDER] == true,
|
||||||
|
reorderedDelivered = delivered[FRAG_REVERSED] == true,
|
||||||
|
delayedFirstDelivered = delivered[FRAG_FIRST_LAST] == true,
|
||||||
|
),
|
||||||
|
) as JsonObject
|
||||||
|
|
||||||
|
val findings = ArrayList<Finding>()
|
||||||
|
val inOrder = delivered[FRAG_IN_ORDER] == true
|
||||||
|
val reversed = delivered[FRAG_REVERSED] == true
|
||||||
|
val firstLast = delivered[FRAG_FIRST_LAST] == true
|
||||||
|
|
||||||
|
if (!inOrder) {
|
||||||
|
findings.add(
|
||||||
|
finding(
|
||||||
|
FindingRegistry.FRAGMENTS_BLOCKED, testId,
|
||||||
|
"IP fragments do not reach this device",
|
||||||
|
"A fragmented datagram sent in the normal order never arrived. Anything that " +
|
||||||
|
"relies on fragmentation — large DNS answers over UDP, some VPN traffic — " +
|
||||||
|
"will fail here rather than slow down.",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
} else if (!reversed || !firstLast) {
|
||||||
|
// The precise and useful finding: fragments work, but only if they arrive tidily.
|
||||||
|
val which = buildList {
|
||||||
|
if (!reversed) add("out of order")
|
||||||
|
if (!firstLast) add("with the first fragment delayed")
|
||||||
|
}.joinToString(" or ")
|
||||||
|
findings.add(
|
||||||
|
finding(
|
||||||
|
FindingRegistry.FRAGMENT_REORDER_SENSITIVE, testId,
|
||||||
|
"Fragments are dropped when they arrive $which",
|
||||||
|
"In-order fragments are delivered, but the same datagram sent $which is not. " +
|
||||||
|
"Something on the path only reassembles when the first fragment (the one " +
|
||||||
|
"carrying the UDP ports) arrives first — typical of a stateful firewall " +
|
||||||
|
"or NAT. It works until the network reorders, then fails intermittently, " +
|
||||||
|
"which is the hardest kind of fault to chase.",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
return Test(
|
||||||
|
id = testId, type = TestType.MTU_FRAG_ORDERING, sessionRef = sessionRef, tier = Tier.APP,
|
||||||
|
startedMonoNs = started, endedMonoNs = ids.monoNs(),
|
||||||
|
status = if (inOrder) TestStatus.OK else TestStatus.PARTIAL,
|
||||||
|
metrics = metrics,
|
||||||
|
) to findings
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Runs all three against an already-primed session.
|
* Runs all three against an already-primed session.
|
||||||
*
|
*
|
||||||
@@ -54,6 +168,10 @@ class DownstreamMeasurement(private val ids: IdSource) {
|
|||||||
trainCount: Int = 100,
|
trainCount: Int = 100,
|
||||||
trainSizeBytes: Int = 300,
|
trainSizeBytes: Int = 300,
|
||||||
trainIntervalUs: Int = 3_000,
|
trainIntervalUs: Int = 3_000,
|
||||||
|
// DSCP to mark the train with (0-63), or -1 to leave packets unmarked. Pairing a marked
|
||||||
|
// downtrain with the server-observed DSCP of an upstream train is the two-direction
|
||||||
|
// sec.dscp_ecn_survival measurement.
|
||||||
|
trainDscp: Int = -1,
|
||||||
): Pair<List<Test>, List<Finding>> {
|
): Pair<List<Test>, List<Finding>> {
|
||||||
val tests = ArrayList<Test>()
|
val tests = ArrayList<Test>()
|
||||||
val findings = ArrayList<Finding>()
|
val findings = ArrayList<Finding>()
|
||||||
@@ -61,11 +179,22 @@ class DownstreamMeasurement(private val ids: IdSource) {
|
|||||||
val df = bigSend(credential, sessionId, control, probe, sessionRef, sizes, df = true)
|
val df = bigSend(credential, sessionId, control, probe, sessionRef, sizes, df = true)
|
||||||
val frag = bigSend(credential, sessionId, control, probe, sessionRef, sizes, df = false)
|
val frag = bigSend(credential, sessionId, control, probe, sessionRef, sizes, df = false)
|
||||||
val train = downTrain(
|
val train = downTrain(
|
||||||
credential, sessionId, control, probe, sessionRef, trainCount, trainSizeBytes, trainIntervalUs,
|
credential, sessionId, control, probe, sessionRef, trainCount, trainSizeBytes,
|
||||||
|
trainIntervalUs, trainDscp,
|
||||||
)
|
)
|
||||||
|
|
||||||
tests.add(df.test); tests.add(frag.test); tests.add(train.test)
|
tests.add(df.test); tests.add(frag.test); tests.add(train.test)
|
||||||
|
|
||||||
|
// Fragment ordering only makes sense once we know fragments arrive at all; when they do
|
||||||
|
// not, the ordering variants would all report "not delivered" and read as three faults
|
||||||
|
// instead of one.
|
||||||
|
if (frag.largestDelivered != null) {
|
||||||
|
val (fragTest, fragFindings) =
|
||||||
|
fragmentOrdering(credential, sessionId, control, probe, sessionRef)
|
||||||
|
tests.add(fragTest)
|
||||||
|
findings.addAll(fragFindings)
|
||||||
|
}
|
||||||
|
|
||||||
// A downstream MTU below the classic 1500-byte Ethernet payload is worth saying out loud:
|
// A downstream MTU below the classic 1500-byte Ethernet payload is worth saying out loud:
|
||||||
// it is the usual cause of "small requests work, large responses hang".
|
// it is the usual cause of "small requests work, large responses hang".
|
||||||
val pathMtu = df.largestDelivered
|
val pathMtu = df.largestDelivered
|
||||||
@@ -74,7 +203,7 @@ class DownstreamMeasurement(private val ids: IdSource) {
|
|||||||
if (ipMtu < 1500) {
|
if (ipMtu < 1500) {
|
||||||
findings.add(
|
findings.add(
|
||||||
finding(
|
finding(
|
||||||
"mtu.reduced_downstream", Category.MTU, Severity.LOW, df.test.id,
|
FindingRegistry.MTU_REDUCED_DOWNSTREAM, df.test.id,
|
||||||
"Downstream path MTU is $ipMtu bytes, below 1500",
|
"Downstream path MTU is $ipMtu bytes, below 1500",
|
||||||
"The largest datagram that reached this device without fragmenting was " +
|
"The largest datagram that reached this device without fragmenting was " +
|
||||||
"$pathMtu bytes of payload ($ipMtu on the wire). Tunnels (PPPoE, VPN, " +
|
"$pathMtu bytes of payload ($ipMtu on the wire). Tunnels (PPPoE, VPN, " +
|
||||||
@@ -89,7 +218,7 @@ class DownstreamMeasurement(private val ids: IdSource) {
|
|||||||
if (fragLargest <= pathMtu && sizes.any { it > pathMtu }) {
|
if (fragLargest <= pathMtu && sizes.any { it > pathMtu }) {
|
||||||
findings.add(
|
findings.add(
|
||||||
finding(
|
finding(
|
||||||
"mtu.downstream_blackhole", Category.MTU, Severity.MEDIUM, frag.test.id,
|
FindingRegistry.MTU_DOWNSTREAM_BLACKHOLE, frag.test.id,
|
||||||
"Datagrams above $pathMtu bytes are dropped downstream, fragmented or not",
|
"Datagrams above $pathMtu bytes are dropped downstream, fragmented or not",
|
||||||
"Nothing larger than $pathMtu bytes arrived, even when the network was " +
|
"Nothing larger than $pathMtu bytes arrived, even when the network was " +
|
||||||
"free to fragment it. Traffic that relies on large responses will " +
|
"free to fragment it. Traffic that relies on large responses will " +
|
||||||
@@ -102,7 +231,7 @@ class DownstreamMeasurement(private val ids: IdSource) {
|
|||||||
if (train.received == 0) {
|
if (train.received == 0) {
|
||||||
findings.add(
|
findings.add(
|
||||||
finding(
|
finding(
|
||||||
"connectivity.downstream_blocked", Category.CONNECTIVITY, Severity.HIGH, train.test.id,
|
FindingRegistry.DOWNSTREAM_BLOCKED, train.test.id,
|
||||||
"No server-initiated packets arrived",
|
"No server-initiated packets arrived",
|
||||||
"The server sent ${train.sent} packets toward this device and none arrived, " +
|
"The server sent ${train.sent} packets toward this device and none arrived, " +
|
||||||
"while the round-trip echo worked. Something on the path forwards replies " +
|
"while the round-trip echo worked. Something on the path forwards replies " +
|
||||||
@@ -112,7 +241,7 @@ class DownstreamMeasurement(private val ids: IdSource) {
|
|||||||
} else if (train.lossPct >= 5.0) {
|
} else if (train.lossPct >= 5.0) {
|
||||||
findings.add(
|
findings.add(
|
||||||
finding(
|
finding(
|
||||||
"connectivity.downstream_loss", Category.CONNECTIVITY, Severity.MEDIUM, train.test.id,
|
FindingRegistry.LOSS_DOWNSTREAM, train.test.id,
|
||||||
"Downstream loss of ${round1(train.lossPct)}%",
|
"Downstream loss of ${round1(train.lossPct)}%",
|
||||||
"${train.sent - train.received} of ${train.sent} packets sent toward this " +
|
"${train.sent - train.received} of ${train.sent} packets sent toward this " +
|
||||||
"device were lost. Downstream loss is invisible to a round-trip test, " +
|
"device were lost. Downstream loss is invisible to a round-trip test, " +
|
||||||
@@ -123,7 +252,7 @@ class DownstreamMeasurement(private val ids: IdSource) {
|
|||||||
if (train.reordered > 0) {
|
if (train.reordered > 0) {
|
||||||
findings.add(
|
findings.add(
|
||||||
finding(
|
finding(
|
||||||
"connectivity.downstream_reorder", Category.CONNECTIVITY, Severity.LOW, train.test.id,
|
FindingRegistry.DOWNSTREAM_REORDER, train.test.id,
|
||||||
"${train.reordered} downstream packet(s) arrived out of order",
|
"${train.reordered} downstream packet(s) arrived out of order",
|
||||||
"Packets arrived in a different order than they were sent. Usually per-packet " +
|
"Packets arrived in a different order than they were sent. Usually per-packet " +
|
||||||
"load balancing across links; harmless for most traffic, not for all of it.",
|
"load balancing across links; harmless for most traffic, not for all of it.",
|
||||||
@@ -214,15 +343,19 @@ class DownstreamMeasurement(private val ids: IdSource) {
|
|||||||
|
|
||||||
private fun downTrain(
|
private fun downTrain(
|
||||||
credential: String, sessionId: String, control: ControlClient, probe: ProbeSession,
|
credential: String, sessionId: String, control: ControlClient, probe: ProbeSession,
|
||||||
sessionRef: String, count: Int, sizeBytes: Int, intervalUs: Int,
|
sessionRef: String, count: Int, sizeBytes: Int, intervalUs: Int, dscp: Int = -1,
|
||||||
): TrainResult {
|
): TrainResult {
|
||||||
val testId = ids.uuid()
|
val testId = ids.uuid()
|
||||||
val started = ids.monoNs()
|
val started = ids.monoNs()
|
||||||
|
|
||||||
|
// dscp is only sent when requested: an older server rejects unknown-value problems
|
||||||
|
// louder than absent keys, and unmarked is the correct default for a plain loss train.
|
||||||
|
val dscpField = if (dscp in 0..63) ""","dscp":$dscp""" else ""
|
||||||
val reply = runCatching {
|
val reply = runCatching {
|
||||||
control.action(
|
control.action(
|
||||||
credential, sessionId,
|
credential, sessionId,
|
||||||
"""{"action":"downtrain","count":$count,"size_bytes":$sizeBytes,"interval_us":$intervalUs}""",
|
"""{"action":"downtrain","count":$count,"size_bytes":$sizeBytes,""" +
|
||||||
|
""""interval_us":$intervalUs$dscpField}""",
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
if (reply.isFailure) {
|
if (reply.isFailure) {
|
||||||
@@ -270,6 +403,12 @@ class DownstreamMeasurement(private val ids: IdSource) {
|
|||||||
interArrivalMsAvg = interArrival.average().takeIf { interArrival.isNotEmpty() }?.let(::round1),
|
interArrivalMsAvg = interArrival.average().takeIf { interArrival.isNotEmpty() }?.let(::round1),
|
||||||
interArrivalMsMax = interArrival.maxOrNull()?.let(::round1),
|
interArrivalMsMax = interArrival.maxOrNull()?.let(::round1),
|
||||||
sendIntervalUs = intervalUs,
|
sendIntervalUs = intervalUs,
|
||||||
|
dscpRequested = dscp.takeIf { it in 0..63 },
|
||||||
|
// The server says whether it could actually mark (dscp_applied); recorded so a
|
||||||
|
// survival comparison never blames the path for a marking the sender skipped.
|
||||||
|
dscpApplied = reply.getOrNull()?.let {
|
||||||
|
Regex("\"dscp_applied\"\\s*:\\s*(true|false)").find(it)?.groupValues?.get(1)?.toBoolean()
|
||||||
|
},
|
||||||
),
|
),
|
||||||
) as JsonObject
|
) as JsonObject
|
||||||
|
|
||||||
@@ -290,9 +429,17 @@ class DownstreamMeasurement(private val ids: IdSource) {
|
|||||||
|
|
||||||
// ---- helpers ----------------------------------------------------------------------
|
// ---- helpers ----------------------------------------------------------------------
|
||||||
|
|
||||||
private fun finding(code: String, cat: Category, sev: Severity, testId: String, title: String, desc: String) =
|
/**
|
||||||
|
* Builds a finding from a registry entry, which supplies the code, category and severity.
|
||||||
|
*
|
||||||
|
* Taking a [FindingSpec] rather than three loose values is the point: a typo becomes a
|
||||||
|
* compile error, and two call sites cannot disagree about which category a finding belongs
|
||||||
|
* to - a disagreement that would split one fault across two verdict lights.
|
||||||
|
*/
|
||||||
|
private fun finding(spec: FindingSpec, testId: String, title: String, desc: String) =
|
||||||
Finding(
|
Finding(
|
||||||
id = ids.uuid(), code = code, category = cat, severity = sev, confidence = Confidence.HIGH,
|
id = ids.uuid(), code = spec.code, category = spec.category, severity = spec.severity,
|
||||||
|
confidence = Confidence.HIGH,
|
||||||
title = title, description = desc, evidenceRefs = listOf(EvidenceRef(testId)),
|
title = title, description = desc, evidenceRefs = listOf(EvidenceRef(testId)),
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -310,6 +457,11 @@ class DownstreamMeasurement(private val ids: IdSource) {
|
|||||||
/** IPv4 (20) + UDP (8). The v6 case is 48; reported per-family once v6 sessions land. */
|
/** IPv4 (20) + UDP (8). The v6 case is 48; reported per-family once v6 sessions land. */
|
||||||
const val IP_UDP_OVERHEAD4 = 28
|
const val IP_UDP_OVERHEAD4 = 28
|
||||||
|
|
||||||
|
const val FRAG_IN_ORDER = "in_order"
|
||||||
|
const val FRAG_REVERSED = "reversed"
|
||||||
|
const val FRAG_FIRST_LAST = "first_last"
|
||||||
|
val FRAG_MODES = listOf(FRAG_IN_ORDER, FRAG_REVERSED, FRAG_FIRST_LAST)
|
||||||
|
|
||||||
/** Straddles the usual suspects: 1500 Ethernet, 1492 PPPoE, 1400-ish tunnels. */
|
/** Straddles the usual suspects: 1500 Ethernet, 1492 PPPoE, 1400-ish tunnels. */
|
||||||
val DEFAULT_SIZES = listOf(600, 1200, 1372, 1400, 1450, 1472, 1500, 2000, 4000)
|
val DEFAULT_SIZES = listOf(600, 1200, 1372, 1400, 1450, 1472, 1500, 2000, 4000)
|
||||||
|
|
||||||
@@ -330,6 +482,18 @@ data class BigSendMetrics(
|
|||||||
@SerialName("path_mtu_bytes") val pathMtuBytes: Int? = null,
|
@SerialName("path_mtu_bytes") val pathMtuBytes: Int? = null,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
/** Metrics for mtu.frag_ordering. */
|
||||||
|
@Serializable
|
||||||
|
data class FragOrderingMetrics(
|
||||||
|
@SerialName("size_bytes") val sizeBytes: Int,
|
||||||
|
@SerialName("frag_bytes") val fragBytes: Int,
|
||||||
|
@SerialName("fragments_per_burst") val fragmentsPerBurst: Map<String, Int>,
|
||||||
|
@SerialName("delivered_by_mode") val deliveredByMode: Map<String, Boolean>,
|
||||||
|
@SerialName("in_order_delivered") val inOrderDelivered: Boolean,
|
||||||
|
@SerialName("reordered_delivered") val reorderedDelivered: Boolean,
|
||||||
|
@SerialName("delayed_first_delivered") val delayedFirstDelivered: Boolean,
|
||||||
|
)
|
||||||
|
|
||||||
/** Metrics for train.udp_downstream. */
|
/** Metrics for train.udp_downstream. */
|
||||||
@Serializable
|
@Serializable
|
||||||
data class DownTrainMetrics(
|
data class DownTrainMetrics(
|
||||||
@@ -341,4 +505,6 @@ data class DownTrainMetrics(
|
|||||||
@SerialName("inter_arrival_ms_avg") val interArrivalMsAvg: Double? = null,
|
@SerialName("inter_arrival_ms_avg") val interArrivalMsAvg: Double? = null,
|
||||||
@SerialName("inter_arrival_ms_max") val interArrivalMsMax: Double? = null,
|
@SerialName("inter_arrival_ms_max") val interArrivalMsMax: Double? = null,
|
||||||
@SerialName("send_interval_us") val sendIntervalUs: Int,
|
@SerialName("send_interval_us") val sendIntervalUs: Int,
|
||||||
|
@SerialName("dscp_requested") val dscpRequested: Int? = null,
|
||||||
|
@SerialName("dscp_applied") val dscpApplied: Boolean? = null,
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -10,8 +10,13 @@ import app.echo_lot.measurement.*
|
|||||||
import app.echo_lot.protocol.ControlClient
|
import app.echo_lot.protocol.ControlClient
|
||||||
import app.echo_lot.protocol.ProbeSession
|
import app.echo_lot.protocol.ProbeSession
|
||||||
import kotlinx.serialization.json.Json
|
import kotlinx.serialization.json.Json
|
||||||
|
import kotlinx.serialization.json.JsonArray
|
||||||
import kotlinx.serialization.json.JsonObject
|
import kotlinx.serialization.json.JsonObject
|
||||||
import kotlinx.serialization.json.encodeToJsonElement
|
import kotlinx.serialization.json.encodeToJsonElement
|
||||||
|
import kotlinx.serialization.json.intOrNull
|
||||||
|
import kotlinx.serialization.json.jsonObject
|
||||||
|
import kotlinx.serialization.json.jsonPrimitive
|
||||||
|
import kotlinx.serialization.json.longOrNull
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Runs the server-facing measurements against one target and assembles a [MeasurementDocument]:
|
* Runs the server-facing measurements against one target and assembles a [MeasurementDocument]:
|
||||||
@@ -42,6 +47,14 @@ class ServerMeasurement(
|
|||||||
* it is a flag rather than an assumption.
|
* it is a flag rather than an assumption.
|
||||||
*/
|
*/
|
||||||
val downstream: Boolean = true,
|
val downstream: Boolean = true,
|
||||||
|
/**
|
||||||
|
* Throughput moves real data — a 5-second run at 50 Mbps is about 30 MB — so it is off
|
||||||
|
* unless asked for. On a metered mobile connection that is the user's money, and a
|
||||||
|
* measurement tool that spends it without being told to is not one people keep installed.
|
||||||
|
*/
|
||||||
|
val throughput: Boolean = false,
|
||||||
|
@Suppress("unused") val throughputSeconds: Int = 5,
|
||||||
|
@Suppress("unused") val throughputKbps: Int = 50_000,
|
||||||
)
|
)
|
||||||
|
|
||||||
fun run(cfg: Config): MeasurementDocument {
|
fun run(cfg: Config): MeasurementDocument {
|
||||||
@@ -71,10 +84,18 @@ class ServerMeasurement(
|
|||||||
// re-primed source is never recorded and every granted send goes to the old, closed port.
|
// re-primed source is never recorded and every granted send goes to the old, closed port.
|
||||||
// Session identity lives on the server; the socket must live as long as it does.
|
// Session identity lives on the server; the socket must live as long as it does.
|
||||||
ProbeSession(cfg.credential, session, cfg.udpHost, cfg.udpPort).use { ps ->
|
ProbeSession(cfg.credential, session, cfg.udpHost, cfg.udpPort).use { ps ->
|
||||||
val (test, findings) = echoTrain(cfg, ps, startMono)
|
val (test, findings) = echoTrain(cfg, ps, startMono, control, session.sessionId)
|
||||||
tests.add(test)
|
tests.add(test)
|
||||||
allFindings.addAll(findings)
|
allFindings.addAll(findings)
|
||||||
|
|
||||||
|
// The upstream train needs no grant and no capability beyond udp-probe itself; a
|
||||||
|
// server that predates trains simply never answers the report request, which the
|
||||||
|
// measurement reports as exactly that ambiguity rather than as network loss.
|
||||||
|
val (utTest, utFindings) = UpstreamTrainMeasurement(ids)
|
||||||
|
.run(ps, sessionRef = "sess-1")
|
||||||
|
tests.add(utTest)
|
||||||
|
allFindings.addAll(utFindings)
|
||||||
|
|
||||||
// Downstream needs a session the server has already seen traffic from — the echo
|
// Downstream needs a session the server has already seen traffic from — the echo
|
||||||
// train just provided that — and a server that advertises the grants. Skipped
|
// train just provided that — and a server that advertises the grants. Skipped
|
||||||
// quietly against an older server rather than reported as a failure of the network.
|
// quietly against an older server rather than reported as a failure of the network.
|
||||||
@@ -84,6 +105,15 @@ class ServerMeasurement(
|
|||||||
tests.addAll(dsTests)
|
tests.addAll(dsTests)
|
||||||
allFindings.addAll(dsFindings)
|
allFindings.addAll(dsFindings)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (cfg.throughput && profile.supports("throughput")) {
|
||||||
|
val (tpTest, tpFindings) = ThroughputMeasurement(ids).run(
|
||||||
|
cfg.credential, session.sessionId, control, ps, sessionRef = "sess-1",
|
||||||
|
durationS = cfg.throughputSeconds, kbps = cfg.throughputKbps,
|
||||||
|
)
|
||||||
|
tests.add(tpTest)
|
||||||
|
allFindings.addAll(tpFindings)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
control.deleteSession(cfg.credential, session.sessionId)
|
control.deleteSession(cfg.credential, session.sessionId)
|
||||||
@@ -105,6 +135,7 @@ class ServerMeasurement(
|
|||||||
|
|
||||||
private fun echoTrain(
|
private fun echoTrain(
|
||||||
cfg: Config, ps: ProbeSession, startMono: Long,
|
cfg: Config, ps: ProbeSession, startMono: Long,
|
||||||
|
control: ControlClient? = null, sessionId: String? = null,
|
||||||
): Pair<Test, List<Finding>> {
|
): Pair<Test, List<Finding>> {
|
||||||
val testId = ids.uuid()
|
val testId = ids.uuid()
|
||||||
val seqs = ArrayList<Int>()
|
val seqs = ArrayList<Int>()
|
||||||
@@ -114,9 +145,15 @@ class ServerMeasurement(
|
|||||||
val rtts = ArrayList<Double>()
|
val rtts = ArrayList<Double>()
|
||||||
val observedPorts = LinkedHashSet<Int>()
|
val observedPorts = LinkedHashSet<Int>()
|
||||||
|
|
||||||
|
// Wire sequence numbers, kept so the server's observations can be correlated packet by
|
||||||
|
// packet. They are not 0..n-1: the counter is shared with every other packet type on the
|
||||||
|
// session, so "the nth echo" is not "sequence n".
|
||||||
|
val wireSeqs = ArrayList<Int>()
|
||||||
|
|
||||||
for (i in 0 until cfg.echoCount) {
|
for (i in 0 until cfg.echoCount) {
|
||||||
val txMono = ids.monoNs() - startMono
|
val txMono = ids.monoNs() - startMono
|
||||||
val r = ps.echo(cfg.echoPaddingBytes)
|
val r = ps.echo(cfg.echoPaddingBytes)
|
||||||
|
wireSeqs.add(ps.lastSeq)
|
||||||
seqs.add(i)
|
seqs.add(i)
|
||||||
tTx.add(txMono)
|
tTx.add(txMono)
|
||||||
sizes.add(Wire_HEADER + cfg.echoPaddingBytes)
|
sizes.add(Wire_HEADER + cfg.echoPaddingBytes)
|
||||||
@@ -129,6 +166,20 @@ class ServerMeasurement(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Ask the server what it actually received. This is what turns "3 % loss somewhere" into
|
||||||
|
// "3 % loss upstream" - the least useful form of the answer into a usable one.
|
||||||
|
val directional: DirectionalMetrics? =
|
||||||
|
if (control != null && sessionId != null) {
|
||||||
|
runCatching {
|
||||||
|
val samples = wireSeqs.indices.map {
|
||||||
|
Directional.Sample(wireSeqs[it], tTx[it] ?: 0L, tRx[it])
|
||||||
|
}
|
||||||
|
Directional.analyse(samples, serverSightings(control, cfg, sessionId))
|
||||||
|
}.getOrNull() // an older server without the endpoint simply yields no split
|
||||||
|
} else {
|
||||||
|
null
|
||||||
|
}
|
||||||
|
|
||||||
val sent = cfg.echoCount
|
val sent = cfg.echoCount
|
||||||
val received = rtts.size
|
val received = rtts.size
|
||||||
val lossPct = if (sent == 0) 0.0 else (sent - received) * 100.0 / sent
|
val lossPct = if (sent == 0) 0.0 else (sent - received) * 100.0 / sent
|
||||||
@@ -138,6 +189,9 @@ class ServerMeasurement(
|
|||||||
epochMonoNs = startMono, seq = seqs, tTxNs = tTx, tRxNs = tRx, sizeBytes = sizes,
|
epochMonoNs = startMono, seq = seqs, tTxNs = tTx, tRxNs = tRx, sizeBytes = sizes,
|
||||||
).toEvidence()
|
).toEvidence()
|
||||||
|
|
||||||
|
val directionalJson = directional?.let {
|
||||||
|
json.encodeToJsonElement(DirectionalMetrics.serializer(), it) as JsonObject
|
||||||
|
}
|
||||||
val metrics: JsonObject = json.encodeToJsonElement(
|
val metrics: JsonObject = json.encodeToJsonElement(
|
||||||
EchoMetrics(
|
EchoMetrics(
|
||||||
sent = sent, received = received, lossPct = round1(lossPct),
|
sent = sent, received = received, lossPct = round1(lossPct),
|
||||||
@@ -147,7 +201,7 @@ class ServerMeasurement(
|
|||||||
observedPorts = observedPorts.toList(),
|
observedPorts = observedPorts.toList(),
|
||||||
natRebindingDetected = natRebinding,
|
natRebindingDetected = natRebinding,
|
||||||
)
|
)
|
||||||
) as JsonObject
|
).let { base -> JsonObject((base as JsonObject) + (directionalJson ?: JsonObject(emptyMap()))) }
|
||||||
|
|
||||||
val status = when {
|
val status = when {
|
||||||
received == 0 -> TestStatus.FAILED
|
received == 0 -> TestStatus.FAILED
|
||||||
@@ -162,30 +216,91 @@ class ServerMeasurement(
|
|||||||
|
|
||||||
val findings = ArrayList<Finding>()
|
val findings = ArrayList<Finding>()
|
||||||
if (received == 0) {
|
if (received == 0) {
|
||||||
findings.add(finding("nat.udp_unreachable", Category.CONNECTIVITY, Severity.HIGH, testId,
|
findings.add(finding(FindingRegistry.UDP_UNREACHABLE, testId,
|
||||||
"No UDP echo replies from the server",
|
"No UDP echo replies from the server",
|
||||||
"Every ECHO probe to the server's UDP data plane was lost — the path blocks or drops the session's UDP traffic."))
|
"Every ECHO probe to the server's UDP data plane was lost — the path blocks or drops the session's UDP traffic."))
|
||||||
} else if (lossPct >= 20.0) {
|
} else if (lossPct >= 20.0) {
|
||||||
findings.add(finding("connectivity.udp_loss", Category.CONNECTIVITY, Severity.MEDIUM, testId,
|
findings.add(finding(FindingRegistry.UDP_LOSS, testId,
|
||||||
"High UDP loss to the server (${round1(lossPct)}%)",
|
"High UDP loss to the server (${round1(lossPct)}%)",
|
||||||
"A large fraction of ECHO probes were lost, indicating an unreliable UDP path."))
|
"A large fraction of ECHO probes were lost, indicating an unreliable UDP path."))
|
||||||
}
|
}
|
||||||
|
// Naming the direction is the entire value of the split, so the findings do.
|
||||||
|
directional?.let { d ->
|
||||||
|
when {
|
||||||
|
d.noneReachedServer && received == 0 -> findings.add(
|
||||||
|
finding(FindingRegistry.UDP_UNREACHABLE_UPSTREAM, testId,
|
||||||
|
"Nothing reached the server",
|
||||||
|
"The server received none of the ${d.sent} probes, so the traffic is being " +
|
||||||
|
"dropped on the way out, not on the way back. A firewall or NAT on " +
|
||||||
|
"this side of the path is the place to look."),
|
||||||
|
)
|
||||||
|
d.lossUpstreamPct >= 2.0 -> findings.add(
|
||||||
|
finding(FindingRegistry.LOSS_UPSTREAM, testId,
|
||||||
|
"${d.lossUpstreamPct} % of probes were lost on the way to the server",
|
||||||
|
"${d.lostUpstream} of ${d.sent} probes never reached the server. The " +
|
||||||
|
"return path is not implicated: replies came back for everything that " +
|
||||||
|
"arrived."),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
if (d.lossDownstreamPct >= 2.0) {
|
||||||
|
findings.add(
|
||||||
|
finding(FindingRegistry.LOSS_DOWNSTREAM, testId,
|
||||||
|
"${d.lossDownstreamPct} % of replies were lost on the way back",
|
||||||
|
"The server received ${d.seenByServer} probes and answered them, but " +
|
||||||
|
"${d.lostDownstream} of those replies never arrived. The outbound path " +
|
||||||
|
"is fine; the fault is on the return leg."),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
if (natRebinding) {
|
if (natRebinding) {
|
||||||
findings.add(finding("nat.udp_rebinding", Category.NAT, Severity.MEDIUM, testId,
|
findings.add(finding(FindingRegistry.NAT_UDP_REBINDING, testId,
|
||||||
"NAT remapped the UDP source port mid-flow",
|
"NAT remapped the UDP source port mid-flow",
|
||||||
"The server observed more than one source port for this session (${observedPorts.joinToString()}), i.e. a NAT with a short UDP mapping or per-packet remapping."))
|
"The server observed more than one source port for this session (${observedPorts.joinToString()}), i.e. a NAT with a short UDP mapping or per-packet remapping."))
|
||||||
}
|
}
|
||||||
return test to findings
|
return test to findings
|
||||||
}
|
}
|
||||||
|
|
||||||
private fun finding(code: String, cat: Category, sev: Severity, testId: String, title: String, desc: String) =
|
/**
|
||||||
|
* The server's per-packet record of this session's echoes (spec section 6). Filtered to
|
||||||
|
* ECHO_REQ, because the observation list also holds MTU probes and anything else we sent -
|
||||||
|
* counting those as train packets would invent loss that is not there.
|
||||||
|
*/
|
||||||
|
private fun serverSightings(
|
||||||
|
control: ControlClient, cfg: Config, sessionId: String,
|
||||||
|
): List<Directional.ServerSighting> {
|
||||||
|
val body = control.observations(cfg.credential, sessionId)
|
||||||
|
val packets = Json.parseToJsonElement(body).jsonObject["udp"]
|
||||||
|
?.jsonObject?.get("packets") as? JsonArray ?: return emptyList()
|
||||||
|
return packets.mapNotNull { el ->
|
||||||
|
val o = el as? JsonObject ?: return@mapNotNull null
|
||||||
|
val type = o["type"]?.jsonPrimitive?.intOrNull ?: return@mapNotNull null
|
||||||
|
if (type != ECHO_REQ_TYPE) return@mapNotNull null
|
||||||
|
Directional.ServerSighting(
|
||||||
|
seq = o["seq"]?.jsonPrimitive?.intOrNull ?: return@mapNotNull null,
|
||||||
|
tRxNs = o["t_rx_ns"]?.jsonPrimitive?.longOrNull ?: return@mapNotNull null,
|
||||||
|
tTxNs = o["t_tx_ns"]?.jsonPrimitive?.longOrNull ?: return@mapNotNull null,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Builds a finding from a registry entry, which supplies the code, category and severity.
|
||||||
|
*
|
||||||
|
* Taking a [FindingSpec] rather than three loose values is the point: a typo becomes a
|
||||||
|
* compile error, and two call sites cannot disagree about which category a finding belongs
|
||||||
|
* to - a disagreement that would split one fault across two verdict lights.
|
||||||
|
*/
|
||||||
|
private fun finding(spec: FindingSpec, testId: String, title: String, desc: String) =
|
||||||
Finding(
|
Finding(
|
||||||
id = ids.uuid(), code = code, category = cat, severity = sev, confidence = Confidence.HIGH,
|
id = ids.uuid(), code = spec.code, category = spec.category, severity = spec.severity,
|
||||||
|
confidence = Confidence.HIGH,
|
||||||
title = title, description = desc, evidenceRefs = listOf(EvidenceRef(testId)),
|
title = title, description = desc, evidenceRefs = listOf(EvidenceRef(testId)),
|
||||||
)
|
)
|
||||||
|
|
||||||
private companion object {
|
private companion object {
|
||||||
const val Wire_HEADER = 32
|
const val Wire_HEADER = 32
|
||||||
|
const val ECHO_REQ_TYPE = 0x01
|
||||||
fun round1(v: Double) = Math.round(v * 10.0) / 10.0
|
fun round1(v: Double) = Math.round(v * 10.0) / 10.0
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,338 @@
|
|||||||
|
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||||
|
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||||
|
|
||||||
|
package app.echo_lot.engine
|
||||||
|
|
||||||
|
import app.echo_lot.measurement.*
|
||||||
|
import app.echo_lot.protocol.ControlClient
|
||||||
|
import app.echo_lot.protocol.ProbeSession
|
||||||
|
import app.echo_lot.protocol.Wire
|
||||||
|
import kotlinx.serialization.SerialName
|
||||||
|
import kotlinx.serialization.Serializable
|
||||||
|
import kotlinx.serialization.json.Json
|
||||||
|
import kotlinx.serialization.json.JsonObject
|
||||||
|
import kotlinx.serialization.json.encodeToJsonElement
|
||||||
|
import kotlinx.serialization.json.jsonArray
|
||||||
|
import kotlinx.serialization.json.jsonObject
|
||||||
|
import kotlinx.serialization.json.jsonPrimitive
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Downstream throughput: the server sends at a paced rate for a bounded time and the client
|
||||||
|
* measures what arrives (`perf.throughput_udp`).
|
||||||
|
*
|
||||||
|
* The number this produces is only meaningful with a qualifier attached, and getting that
|
||||||
|
* qualifier right is most of the work here. A throughput test reports the *smallest* limit on the
|
||||||
|
* path, and the sender's own ceiling is one of the candidates: if the server was asked for 50 Mbps
|
||||||
|
* and 50 Mbps arrived, the network was never the constraint and "50 Mbps" says nothing about it.
|
||||||
|
* Reporting that as a capacity measurement would be a confident lie, so the result always carries
|
||||||
|
* [ThroughputMetrics.limitedBy] and a finding is only raised when the network is actually
|
||||||
|
* implicated.
|
||||||
|
*
|
||||||
|
* Comparing against the *sender's* count rather than the requested rate is the other half: the
|
||||||
|
* server reports how much it actually put on the wire, and the gap between that and what arrived
|
||||||
|
* is the loss. A receiver alone cannot tell "the network dropped it" from "the sender never sent
|
||||||
|
* it", and guessing turns a healthy server-side limit into a phantom network fault.
|
||||||
|
*/
|
||||||
|
class ThroughputMeasurement(private val ids: IdSource) {
|
||||||
|
|
||||||
|
private val json = Json { encodeDefaults = true; explicitNulls = true }
|
||||||
|
|
||||||
|
fun run(
|
||||||
|
credential: String,
|
||||||
|
sessionId: String,
|
||||||
|
control: ControlClient,
|
||||||
|
probe: ProbeSession,
|
||||||
|
sessionRef: String,
|
||||||
|
durationS: Int = 5,
|
||||||
|
kbps: Int = 50_000,
|
||||||
|
sizeBytes: Int = 1200,
|
||||||
|
): Pair<Test, List<Finding>> {
|
||||||
|
val testId = ids.uuid()
|
||||||
|
val started = ids.monoNs()
|
||||||
|
|
||||||
|
val reply = runCatching {
|
||||||
|
control.action(
|
||||||
|
credential, sessionId,
|
||||||
|
"""{"action":"throughput","direction":"down","duration_s":$durationS,""" +
|
||||||
|
""""kbps":$kbps,"size_bytes":$sizeBytes}""",
|
||||||
|
)
|
||||||
|
}
|
||||||
|
if (reply.isFailure) {
|
||||||
|
return Test(
|
||||||
|
id = testId, type = TestType.PERF_THROUGHPUT_UDP, sessionRef = sessionRef, tier = Tier.APP,
|
||||||
|
startedMonoNs = started, endedMonoNs = ids.monoNs(),
|
||||||
|
status = TestStatus.UNSUPPORTED,
|
||||||
|
error = TestError("action_refused", reply.exceptionOrNull()?.message ?: "throughput refused"),
|
||||||
|
) to emptyList()
|
||||||
|
}
|
||||||
|
|
||||||
|
// The server may have shortened the run to fit its own byte budget; listen for what it
|
||||||
|
// actually promised, not for what we asked.
|
||||||
|
val plannedMs = parseInt(reply.getOrNull(), "duration_ms") ?: (durationS * 1000)
|
||||||
|
|
||||||
|
// A margin past the planned end so the tail of the run is not counted as loss: packets
|
||||||
|
// still in flight when we stop listening were not dropped, they were merely late.
|
||||||
|
val received = probe.collectGranted(plannedMs + 1_500L)
|
||||||
|
.filter { it.type == Wire.TYPE_THROUGHPUT_DATA }
|
||||||
|
|
||||||
|
val bytes = received.sumOf { it.sizeBytes.toLong() }
|
||||||
|
val spanNs = if (received.size >= 2) {
|
||||||
|
received.maxOf { it.tRxNs } - received.minOf { it.tRxNs }
|
||||||
|
} else {
|
||||||
|
0L
|
||||||
|
}
|
||||||
|
// Measured over the arrival span rather than our listening window, which includes the
|
||||||
|
// request round trip and the trailing margin and would understate the rate.
|
||||||
|
val receivedKbps = if (spanNs > 0) (bytes * 8 * 1_000_000 / spanNs).toInt() else 0
|
||||||
|
|
||||||
|
val sender = senderReport(control, credential, sessionId)
|
||||||
|
val sentPackets = sender?.packets ?: 0
|
||||||
|
val lossPct = if (sentPackets > 0) {
|
||||||
|
round2((sentPackets - received.size).coerceAtLeast(0) * 100.0 / sentPackets)
|
||||||
|
} else {
|
||||||
|
null
|
||||||
|
}
|
||||||
|
|
||||||
|
// Only a run the *clock* ended measured the network. One stopped by our own byte budget
|
||||||
|
// or rate ceiling measured this server.
|
||||||
|
val limitedBy = sender?.limitedBy ?: "unknown"
|
||||||
|
val networkLimited = limitedBy == "duration" &&
|
||||||
|
sender != null && receivedKbps > 0 && receivedKbps < sender.kbps * 9 / 10
|
||||||
|
|
||||||
|
val metrics = json.encodeToJsonElement(
|
||||||
|
ThroughputMetrics(
|
||||||
|
requestedKbps = kbps,
|
||||||
|
plannedDurationMs = plannedMs,
|
||||||
|
packetsReceived = received.size,
|
||||||
|
bytesReceived = bytes,
|
||||||
|
receivedKbps = receivedKbps,
|
||||||
|
senderPackets = sender?.packets,
|
||||||
|
senderBytes = sender?.bytes,
|
||||||
|
senderKbps = sender?.kbps,
|
||||||
|
lossPct = lossPct,
|
||||||
|
limitedBy = limitedBy,
|
||||||
|
measuresNetwork = networkLimited,
|
||||||
|
),
|
||||||
|
) as JsonObject
|
||||||
|
|
||||||
|
val findings = ArrayList<Finding>()
|
||||||
|
when {
|
||||||
|
sender == null -> Unit // no sender report: nothing can be concluded, so nothing is
|
||||||
|
received.isEmpty() -> findings.add(
|
||||||
|
finding(
|
||||||
|
FindingRegistry.THROUGHPUT_NO_DELIVERY, testId,
|
||||||
|
"No throughput traffic arrived",
|
||||||
|
"The server sent ${sender.packets} packets and none arrived. This is a " +
|
||||||
|
"connectivity fault rather than a slow link.",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
networkLimited -> findings.add(
|
||||||
|
finding(
|
||||||
|
FindingRegistry.THROUGHPUT_BELOW_OFFERED, testId,
|
||||||
|
"Downstream throughput ${receivedKbps / 1000} Mbit/s, below the " +
|
||||||
|
"${sender.kbps / 1000} Mbit/s offered",
|
||||||
|
"The server sent at ${sender.kbps / 1000} Mbit/s for the full run and " +
|
||||||
|
"${receivedKbps / 1000} Mbit/s arrived" +
|
||||||
|
(lossPct?.let { ", losing $it % of packets" } ?: "") +
|
||||||
|
". The path could not carry what was offered.",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
return Test(
|
||||||
|
id = testId, type = TestType.PERF_THROUGHPUT_UDP, sessionRef = sessionRef, tier = Tier.APP,
|
||||||
|
startedMonoNs = started, endedMonoNs = ids.monoNs(),
|
||||||
|
status = if (received.isEmpty()) TestStatus.FAILED else TestStatus.OK,
|
||||||
|
metrics = metrics,
|
||||||
|
) to findings
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Upstream throughput: the client sends, the server counts.
|
||||||
|
*
|
||||||
|
* The mirror image of the downstream case, and it needs no grant — the client is generating
|
||||||
|
* its own traffic, so there is no amplification to gate. What it does need is the server's
|
||||||
|
* count: only the far end knows how much arrived, and without that number a sender can
|
||||||
|
* measure how fast it can *transmit*, which is not the same question and is usually just the
|
||||||
|
* speed of the local NIC.
|
||||||
|
*/
|
||||||
|
fun runUpstream(
|
||||||
|
credential: String,
|
||||||
|
sessionId: String,
|
||||||
|
control: ControlClient,
|
||||||
|
probe: ProbeSession,
|
||||||
|
sessionRef: String,
|
||||||
|
durationS: Int = 5,
|
||||||
|
kbps: Int = 20_000,
|
||||||
|
sizeBytes: Int = 1200,
|
||||||
|
): Pair<Test, List<Finding>> {
|
||||||
|
val testId = ids.uuid()
|
||||||
|
val started = ids.monoNs()
|
||||||
|
|
||||||
|
// Zeroes the server's counter so this run measures itself rather than inheriting the
|
||||||
|
// packets of an earlier one on the same session.
|
||||||
|
val reply = runCatching {
|
||||||
|
control.action(credential, sessionId, """{"action":"throughput","direction":"up"}""")
|
||||||
|
}
|
||||||
|
if (reply.isFailure) {
|
||||||
|
return Test(
|
||||||
|
id = testId, type = TestType.PERF_THROUGHPUT_UDP, sessionRef = sessionRef, tier = Tier.APP,
|
||||||
|
startedMonoNs = started, endedMonoNs = ids.monoNs(),
|
||||||
|
status = TestStatus.UNSUPPORTED,
|
||||||
|
error = TestError("action_refused", reply.exceptionOrNull()?.message ?: "refused"),
|
||||||
|
) to emptyList()
|
||||||
|
}
|
||||||
|
|
||||||
|
val sent = probe.sendThroughput(durationS * 1000L, kbps, sizeBytes)
|
||||||
|
// A moment for the tail of the run to arrive; counting still-in-flight packets as lost
|
||||||
|
// would inflate the loss figure by whatever the path's delay happens to be.
|
||||||
|
Thread.sleep(500)
|
||||||
|
val seen = upstreamCount(control, credential, sessionId)
|
||||||
|
|
||||||
|
val lossPct = if (sent.packets > 0 && seen != null) {
|
||||||
|
round2((sent.packets - seen.packets).coerceAtLeast(0) * 100.0 / sent.packets)
|
||||||
|
} else {
|
||||||
|
null
|
||||||
|
}
|
||||||
|
// The receiver's rate is the measurement. The sender's is what we managed to emit, which
|
||||||
|
// is a property of this phone and its radio, not of the network.
|
||||||
|
val achievedKbps = seen?.kbps ?: 0
|
||||||
|
|
||||||
|
val metrics = json.encodeToJsonElement(
|
||||||
|
UpstreamThroughputMetrics(
|
||||||
|
requestedKbps = kbps,
|
||||||
|
sentPackets = sent.packets,
|
||||||
|
sentBytes = sent.bytes,
|
||||||
|
sentKbps = sent.kbps,
|
||||||
|
receivedPackets = seen?.packets,
|
||||||
|
receivedBytes = seen?.bytes,
|
||||||
|
receivedKbps = achievedKbps,
|
||||||
|
lossPct = lossPct,
|
||||||
|
// Same honesty rule as downstream: if what arrived matches what we offered, the
|
||||||
|
// path was never the constraint and this number says nothing about it.
|
||||||
|
measuresNetwork = seen != null && achievedKbps > 0 && achievedKbps < sent.kbps * 9 / 10,
|
||||||
|
),
|
||||||
|
) as JsonObject
|
||||||
|
|
||||||
|
val findings = ArrayList<Finding>()
|
||||||
|
if (seen != null && seen.packets == 0 && sent.packets > 0) {
|
||||||
|
findings.add(
|
||||||
|
finding(
|
||||||
|
FindingRegistry.THROUGHPUT_NO_DELIVERY, testId,
|
||||||
|
"No upstream traffic reached the server",
|
||||||
|
"This device sent ${sent.packets} packets and the server received none. " +
|
||||||
|
"That is a connectivity fault on the outbound path rather than a slow link.",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
} else if (lossPct != null && lossPct >= 2.0) {
|
||||||
|
findings.add(
|
||||||
|
finding(
|
||||||
|
FindingRegistry.THROUGHPUT_BELOW_OFFERED, testId,
|
||||||
|
"Upstream loss of $lossPct % at ${sent.kbps / 1000} Mbit/s",
|
||||||
|
"The server received ${seen?.packets} of the ${sent.packets} packets this " +
|
||||||
|
"device sent. The outbound path could not carry what was offered.",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
return Test(
|
||||||
|
id = testId, type = TestType.PERF_THROUGHPUT_UDP, sessionRef = sessionRef, tier = Tier.APP,
|
||||||
|
startedMonoNs = started, endedMonoNs = ids.monoNs(),
|
||||||
|
status = if (seen == null || seen.packets == 0) TestStatus.FAILED else TestStatus.OK,
|
||||||
|
metrics = metrics,
|
||||||
|
) to findings
|
||||||
|
}
|
||||||
|
|
||||||
|
private data class UpstreamCount(val packets: Int, val bytes: Long, val kbps: Int)
|
||||||
|
|
||||||
|
/** The server's tally for this session's upstream run. */
|
||||||
|
private fun upstreamCount(
|
||||||
|
control: ControlClient, credential: String, sessionId: String,
|
||||||
|
): UpstreamCount? = runCatching {
|
||||||
|
val o = Json.parseToJsonElement(control.observations(credential, sessionId))
|
||||||
|
.jsonObject["throughput_up"]?.jsonObject ?: return null
|
||||||
|
UpstreamCount(
|
||||||
|
packets = o["packets"]?.jsonPrimitive?.content?.toIntOrNull() ?: 0,
|
||||||
|
bytes = o["bytes"]?.jsonPrimitive?.content?.toLongOrNull() ?: 0,
|
||||||
|
kbps = o["kbps"]?.jsonPrimitive?.content?.toIntOrNull() ?: 0,
|
||||||
|
)
|
||||||
|
}.getOrNull()
|
||||||
|
|
||||||
|
private data class SenderReport(
|
||||||
|
val packets: Int, val bytes: Long, val kbps: Int, val limitedBy: String,
|
||||||
|
)
|
||||||
|
|
||||||
|
/** The server's own account of the run, from the observations API. */
|
||||||
|
private fun senderReport(
|
||||||
|
control: ControlClient, credential: String, sessionId: String,
|
||||||
|
): SenderReport? = runCatching {
|
||||||
|
val arr = Json.parseToJsonElement(control.observations(credential, sessionId))
|
||||||
|
.jsonObject["throughput"]?.jsonArray ?: return null
|
||||||
|
val last = arr.lastOrNull()?.jsonObject ?: return null
|
||||||
|
SenderReport(
|
||||||
|
packets = last["packets"]?.jsonPrimitive?.content?.toIntOrNull() ?: 0,
|
||||||
|
bytes = last["bytes"]?.jsonPrimitive?.content?.toLongOrNull() ?: 0,
|
||||||
|
kbps = last["kbps"]?.jsonPrimitive?.content?.toIntOrNull() ?: 0,
|
||||||
|
limitedBy = last["limited_by"]?.jsonPrimitive?.content ?: "unknown",
|
||||||
|
)
|
||||||
|
}.getOrNull()
|
||||||
|
|
||||||
|
private fun parseInt(body: String?, key: String): Int? =
|
||||||
|
body?.let { Regex("\"$key\"\\s*:\\s*(-?\\d+)").find(it)?.groupValues?.get(1)?.toIntOrNull() }
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Builds a finding from a registry entry, which supplies the code, category and severity.
|
||||||
|
*
|
||||||
|
* Taking a [FindingSpec] rather than three loose values is the point: a typo becomes a
|
||||||
|
* compile error, and two call sites cannot disagree about which category a finding belongs
|
||||||
|
* to - a disagreement that would split one fault across two verdict lights.
|
||||||
|
*/
|
||||||
|
private fun finding(spec: FindingSpec, testId: String, title: String, desc: String) =
|
||||||
|
Finding(
|
||||||
|
id = ids.uuid(), code = spec.code, category = spec.category, severity = spec.severity,
|
||||||
|
confidence = Confidence.HIGH,
|
||||||
|
title = title, description = desc, evidenceRefs = listOf(EvidenceRef(testId)),
|
||||||
|
)
|
||||||
|
|
||||||
|
private fun round2(v: Double) = Math.round(v * 100.0) / 100.0
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Metrics for perf.throughput_udp in the upstream direction. */
|
||||||
|
@Serializable
|
||||||
|
data class UpstreamThroughputMetrics(
|
||||||
|
val direction: String = "up",
|
||||||
|
@SerialName("requested_kbps") val requestedKbps: Int,
|
||||||
|
@SerialName("sent_packets") val sentPackets: Int,
|
||||||
|
@SerialName("sent_bytes") val sentBytes: Long,
|
||||||
|
/** What this device managed to emit — a property of the phone and its radio, not the path. */
|
||||||
|
@SerialName("sent_kbps") val sentKbps: Int,
|
||||||
|
@SerialName("received_packets") val receivedPackets: Int? = null,
|
||||||
|
@SerialName("received_bytes") val receivedBytes: Long? = null,
|
||||||
|
/** What arrived, measured by the only party that can measure it. This is the result. */
|
||||||
|
@SerialName("received_kbps") val receivedKbps: Int,
|
||||||
|
@SerialName("loss_pct") val lossPct: Double? = null,
|
||||||
|
@SerialName("measures_network") val measuresNetwork: Boolean,
|
||||||
|
)
|
||||||
|
|
||||||
|
/** Metrics for perf.throughput_udp. */
|
||||||
|
@Serializable
|
||||||
|
data class ThroughputMetrics(
|
||||||
|
val direction: String = "down",
|
||||||
|
@SerialName("requested_kbps") val requestedKbps: Int,
|
||||||
|
@SerialName("planned_duration_ms") val plannedDurationMs: Int,
|
||||||
|
@SerialName("packets_received") val packetsReceived: Int,
|
||||||
|
@SerialName("bytes_received") val bytesReceived: Long,
|
||||||
|
@SerialName("received_kbps") val receivedKbps: Int,
|
||||||
|
@SerialName("sender_packets") val senderPackets: Int? = null,
|
||||||
|
@SerialName("sender_bytes") val senderBytes: Long? = null,
|
||||||
|
@SerialName("sender_kbps") val senderKbps: Int? = null,
|
||||||
|
/** Against the sender's count, so a server-side limit is never counted as network loss. */
|
||||||
|
@SerialName("loss_pct") val lossPct: Double? = null,
|
||||||
|
/** What ended the run: duration | budget | rate | send_error | unknown. */
|
||||||
|
@SerialName("limited_by") val limitedBy: String,
|
||||||
|
/**
|
||||||
|
* Whether this number says anything about the network. False when the sender's own ceiling
|
||||||
|
* was the binding constraint — in which case the rate is a property of the test, not the path.
|
||||||
|
*/
|
||||||
|
@SerialName("measures_network") val measuresNetwork: Boolean,
|
||||||
|
)
|
||||||
+159
@@ -0,0 +1,159 @@
|
|||||||
|
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||||
|
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||||
|
|
||||||
|
package app.echo_lot.engine
|
||||||
|
|
||||||
|
import app.echo_lot.measurement.*
|
||||||
|
import app.echo_lot.protocol.ProbeSession
|
||||||
|
import kotlinx.serialization.SerialName
|
||||||
|
import kotlinx.serialization.Serializable
|
||||||
|
import kotlinx.serialization.json.Json
|
||||||
|
import kotlinx.serialization.json.JsonObject
|
||||||
|
import kotlinx.serialization.json.encodeToJsonElement
|
||||||
|
|
||||||
|
/**
|
||||||
|
* train.udp_updown — the client sends a paced train (types 0x03), then asks the server what
|
||||||
|
* arrived (0x04 → 0x05) and lines both views up per sequence number.
|
||||||
|
*
|
||||||
|
* This is the measurement a round trip cannot make: an echo run only says "lost somewhere", the
|
||||||
|
* train's two ledgers say lost on the way OUT, specifically, because the server's report names
|
||||||
|
* exactly which sequence numbers reached it. The downstream direction has its own test
|
||||||
|
* (train.udp_downstream) under a grant; this one needs none, since the client generates all the
|
||||||
|
* traffic itself.
|
||||||
|
*
|
||||||
|
* The evidence is the schema's columnar TrainEvidence: one index per sent packet, with the
|
||||||
|
* server-side columns null where a packet never arrived. Server timestamps are on the server's
|
||||||
|
* own clock — only differences within that clock mean anything unless time.server_offset maps
|
||||||
|
* them (two-clock rule).
|
||||||
|
*/
|
||||||
|
class UpstreamTrainMeasurement(private val ids: IdSource) {
|
||||||
|
|
||||||
|
private val json = Json { encodeDefaults = true; explicitNulls = true }
|
||||||
|
|
||||||
|
fun run(
|
||||||
|
probe: ProbeSession,
|
||||||
|
sessionRef: String,
|
||||||
|
count: Int = 200,
|
||||||
|
sizeBytes: Int = 200,
|
||||||
|
interPacketMs: Long = 5,
|
||||||
|
): Pair<Test, List<Finding>> {
|
||||||
|
val testId = ids.uuid()
|
||||||
|
val started = ids.monoNs()
|
||||||
|
// The id only needs to be unique within this session; a clash across sessions is
|
||||||
|
// meaningless because trains are buffered per session on the server.
|
||||||
|
val trainId = (System.nanoTime() and 0x7FFFFFFF).toInt()
|
||||||
|
|
||||||
|
val sent = probe.sendTrain(trainId, count, sizeBytes, interPacketMs)
|
||||||
|
// Let the tail arrive before asking for the ledger; packets still in flight when the
|
||||||
|
// report is cut would read as upstream loss.
|
||||||
|
Thread.sleep(300)
|
||||||
|
val report = probe.trainReport(trainId)
|
||||||
|
|
||||||
|
if (report == null) {
|
||||||
|
return Test(
|
||||||
|
id = testId, type = TestType.TRAIN_UDP_UPDOWN, sessionRef = sessionRef,
|
||||||
|
tier = Tier.APP, startedMonoNs = started, endedMonoNs = ids.monoNs(),
|
||||||
|
status = TestStatus.FAILED,
|
||||||
|
// Honest ambiguity: an old server drops 0x04 silently, and a lost report looks
|
||||||
|
// identical from here. Neither says anything about the train itself.
|
||||||
|
error = TestError(
|
||||||
|
"no_report",
|
||||||
|
"no train report arrived — the report was lost, or the server predates trains",
|
||||||
|
),
|
||||||
|
) to emptyList()
|
||||||
|
}
|
||||||
|
|
||||||
|
val bySeq = report.rows.associateBy { it.seq }
|
||||||
|
fun col255(v: Int): Int? = v.takeIf { it != 255 } // 255 = "not observed" on the wire
|
||||||
|
|
||||||
|
val evidence = TrainEvidence(
|
||||||
|
epochMonoNs = started,
|
||||||
|
seq = sent.map { it.seq },
|
||||||
|
tTxNs = sent.map { it.tTxNs },
|
||||||
|
tSrvRxNs = sent.map { bySeq[it.seq]?.tRxNs },
|
||||||
|
tRxNs = sent.map { null }, // upstream only: nothing comes back per packet
|
||||||
|
sizeBytes = sent.map { it.sizeBytes },
|
||||||
|
ttlSeenByServer = sent.map { bySeq[it.seq]?.let { r -> col255(r.ttl) } },
|
||||||
|
dscpSeenByServer = sent.map { bySeq[it.seq]?.let { r -> col255(r.dscp) } },
|
||||||
|
ecnSeenByServer = sent.map { bySeq[it.seq]?.let { r -> col255(r.ecn) } },
|
||||||
|
evidenceTruncated = report.truncated,
|
||||||
|
).toEvidence()
|
||||||
|
|
||||||
|
// Loss against the server's total count, not its row list: rows past the server's buffer
|
||||||
|
// cap are counted but not kept, and treating them as lost would invent loss exactly on
|
||||||
|
// the biggest trains.
|
||||||
|
val lossPct = if (sent.isEmpty()) 0.0 else {
|
||||||
|
(sent.size - report.received).coerceAtLeast(0) * 100.0 / sent.size
|
||||||
|
}
|
||||||
|
val metrics = json.encodeToJsonElement(
|
||||||
|
UpstreamTrainMetrics(
|
||||||
|
sent = sent.size,
|
||||||
|
receivedByServer = report.received,
|
||||||
|
lossPct = round1(lossPct),
|
||||||
|
reportPartsExpected = report.partsExpected,
|
||||||
|
reportPartsReceived = report.partsReceived,
|
||||||
|
truncated = report.truncated,
|
||||||
|
),
|
||||||
|
) as JsonObject
|
||||||
|
|
||||||
|
val findings = ArrayList<Finding>()
|
||||||
|
if (sent.isNotEmpty() && report.received == 0) {
|
||||||
|
findings.add(
|
||||||
|
Finding(
|
||||||
|
id = ids.uuid(),
|
||||||
|
code = FindingRegistry.UDP_UNREACHABLE_UPSTREAM.code,
|
||||||
|
category = FindingRegistry.UDP_UNREACHABLE_UPSTREAM.category,
|
||||||
|
severity = FindingRegistry.UDP_UNREACHABLE_UPSTREAM.severity,
|
||||||
|
confidence = Confidence.HIGH,
|
||||||
|
title = "The server received none of ${sent.size} upstream packets",
|
||||||
|
description = "Every train packet vanished on the way out, while the " +
|
||||||
|
"report request's reply made it back — the outbound path drops this " +
|
||||||
|
"traffic, the return path works.",
|
||||||
|
evidenceRefs = listOf(EvidenceRef(testId)),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
} else if (lossPct >= 2.0) {
|
||||||
|
findings.add(
|
||||||
|
Finding(
|
||||||
|
id = ids.uuid(),
|
||||||
|
code = FindingRegistry.LOSS_UPSTREAM.code,
|
||||||
|
category = FindingRegistry.LOSS_UPSTREAM.category,
|
||||||
|
severity = FindingRegistry.LOSS_UPSTREAM.severity,
|
||||||
|
confidence = Confidence.HIGH,
|
||||||
|
title = "Upstream loss of ${round1(lossPct)} %",
|
||||||
|
description = "The server received ${report.received} of the ${sent.size} " +
|
||||||
|
"packets this device sent, and its per-sequence ledger names the " +
|
||||||
|
"missing ones. This is outbound loss specifically; the return path " +
|
||||||
|
"delivered the report.",
|
||||||
|
evidenceRefs = listOf(EvidenceRef(testId)),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
val status = when {
|
||||||
|
report.received == 0 && sent.isNotEmpty() -> TestStatus.FAILED
|
||||||
|
report.partsReceived < report.partsExpected -> TestStatus.PARTIAL
|
||||||
|
else -> TestStatus.OK
|
||||||
|
}
|
||||||
|
return Test(
|
||||||
|
id = testId, type = TestType.TRAIN_UDP_UPDOWN, sessionRef = sessionRef, tier = Tier.APP,
|
||||||
|
startedMonoNs = started, endedMonoNs = ids.monoNs(),
|
||||||
|
status = status, evidence = evidence, metrics = metrics,
|
||||||
|
) to findings
|
||||||
|
}
|
||||||
|
|
||||||
|
private fun round1(v: Double) = Math.round(v * 10.0) / 10.0
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Metrics for train.udp_updown. */
|
||||||
|
@Serializable
|
||||||
|
data class UpstreamTrainMetrics(
|
||||||
|
val sent: Int,
|
||||||
|
/** The server's total count — includes packets past its row buffer (counted, not listed). */
|
||||||
|
@SerialName("received_by_server") val receivedByServer: Int,
|
||||||
|
@SerialName("loss_pct") val lossPct: Double,
|
||||||
|
@SerialName("report_parts_expected") val reportPartsExpected: Int,
|
||||||
|
@SerialName("report_parts_received") val reportPartsReceived: Int,
|
||||||
|
/** The server's row buffer overflowed: rows are a sample, the count is still complete. */
|
||||||
|
val truncated: Boolean,
|
||||||
|
)
|
||||||
@@ -0,0 +1,155 @@
|
|||||||
|
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||||
|
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||||
|
|
||||||
|
package app.echo_lot.engine
|
||||||
|
|
||||||
|
import app.echo_lot.engine.Directional.Sample
|
||||||
|
import app.echo_lot.engine.Directional.ServerSighting
|
||||||
|
import kotlin.test.Test
|
||||||
|
import kotlin.test.assertEquals
|
||||||
|
import kotlin.test.assertFalse
|
||||||
|
import kotlin.test.assertNotNull
|
||||||
|
import kotlin.test.assertNull
|
||||||
|
import kotlin.test.assertTrue
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The arithmetic that turns "3 % loss somewhere" into "3 % loss upstream". Getting a denominator
|
||||||
|
* wrong here does not crash anything — it produces a plausible number pointing at the wrong half
|
||||||
|
* of the network, which is worse than no number at all. Hence a test per claim.
|
||||||
|
*/
|
||||||
|
class DirectionalTest {
|
||||||
|
|
||||||
|
/** A clean train: every packet sent, seen and answered. Server clock offset by a constant. */
|
||||||
|
private fun clean(n: Int, offsetNs: Long = 5_000_000_000L): Pair<List<Sample>, List<ServerSighting>> {
|
||||||
|
val sent = (1..n).map { Sample(it, tTxNs = it * 10_000_000L, tRxNs = it * 10_000_000L + 4_000_000L) }
|
||||||
|
val seen = (1..n).map {
|
||||||
|
ServerSighting(it, tRxNs = offsetNs + it * 10_000_000L + 2_000_000L,
|
||||||
|
tTxNs = offsetNs + it * 10_000_000L + 2_100_000L)
|
||||||
|
}
|
||||||
|
return sent to seen
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun aCleanTrainReportsNoLossInEitherDirection() {
|
||||||
|
val (sent, seen) = clean(10)
|
||||||
|
val m = Directional.analyse(sent, seen)
|
||||||
|
assertEquals(10, m.sent)
|
||||||
|
assertEquals(10, m.seenByServer)
|
||||||
|
assertEquals(10, m.repliesReceived)
|
||||||
|
assertEquals(0.0, m.lossUpstreamPct)
|
||||||
|
assertEquals(0.0, m.lossDownstreamPct)
|
||||||
|
assertFalse(m.noneReachedServer)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The whole point: a packet the server never saw was lost on the way there.
|
||||||
|
@Test
|
||||||
|
fun packetsTheServerNeverSawAreUpstreamLoss() {
|
||||||
|
val (sent, seen) = clean(10)
|
||||||
|
val m = Directional.analyse(sent, seen.filter { it.seq !in setOf(3, 7) })
|
||||||
|
assertEquals(2, m.lostUpstream)
|
||||||
|
assertEquals(0, m.lostDownstream)
|
||||||
|
assertEquals(20.0, m.lossUpstreamPct)
|
||||||
|
assertEquals(0.0, m.lossDownstreamPct, "a packet that never arrived cannot be lost coming back")
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun repliesThatNeverArrivedAreDownstreamLoss() {
|
||||||
|
val (sent, seen) = clean(10)
|
||||||
|
val withHoles = sent.map { if (it.seq in setOf(2, 5)) it.copy(tRxNs = null) else it }
|
||||||
|
val m = Directional.analyse(withHoles, seen)
|
||||||
|
assertEquals(0, m.lostUpstream)
|
||||||
|
assertEquals(2, m.lostDownstream)
|
||||||
|
assertEquals(20.0, m.lossDownstreamPct)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Downstream loss is measured against what actually reached the server. Using "sent" as the
|
||||||
|
// denominator would count every upstream loss a second time and overstate the return path.
|
||||||
|
@Test
|
||||||
|
fun downstreamLossIsRelativeToWhatReachedTheServer() {
|
||||||
|
val (sent, seen) = clean(10)
|
||||||
|
// 5 lost on the way there; of the 5 that arrived, 1 reply is lost coming back.
|
||||||
|
val seenPartial = seen.filter { it.seq > 5 }
|
||||||
|
val withHole = sent.map {
|
||||||
|
when {
|
||||||
|
it.seq <= 5 -> it.copy(tRxNs = null) // never got there, so never came back
|
||||||
|
it.seq == 6 -> it.copy(tRxNs = null) // arrived, reply lost
|
||||||
|
else -> it
|
||||||
|
}
|
||||||
|
}
|
||||||
|
val m = Directional.analyse(withHole, seenPartial)
|
||||||
|
assertEquals(5, m.lostUpstream)
|
||||||
|
assertEquals(50.0, m.lossUpstreamPct)
|
||||||
|
assertEquals(1, m.lostDownstream)
|
||||||
|
assertEquals(20.0, m.lossDownstreamPct, "1 of the 5 that arrived, not 1 of 10")
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun aServerThatSawNothingIsCalledOutSeparately() {
|
||||||
|
val (sent, _) = clean(6)
|
||||||
|
val m = Directional.analyse(sent.map { it.copy(tRxNs = null) }, emptyList())
|
||||||
|
assertTrue(m.noneReachedServer)
|
||||||
|
assertEquals(100.0, m.lossUpstreamPct)
|
||||||
|
assertEquals(0.0, m.lossDownstreamPct, "with nothing arriving there is no return path to blame")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Jitter is legitimate without synchronised clocks because the offset cancels when successive
|
||||||
|
// one-way samples are differenced. This pins that: a huge constant offset must not show up.
|
||||||
|
@Test
|
||||||
|
fun jitterIsUnaffectedByTheClockOffsetBetweenTheTwoMachines() {
|
||||||
|
val (sent, near) = clean(10, offsetNs = 0)
|
||||||
|
val (_, far) = clean(10, offsetNs = 9_999_999_999L)
|
||||||
|
val a = Directional.analyse(sent, near)
|
||||||
|
val b = Directional.analyse(sent, far)
|
||||||
|
assertEquals(a.jitterUpstreamMs, b.jitterUpstreamMs,
|
||||||
|
"a constant clock offset must cancel when consecutive samples are differenced")
|
||||||
|
assertEquals(0.0, assertNotNull(a.jitterUpstreamMs), "an evenly spaced train has no jitter")
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun jitterReflectsUnevenArrival() {
|
||||||
|
val sent = listOf(
|
||||||
|
Sample(1, 0, 10_000_000),
|
||||||
|
Sample(2, 10_000_000, 20_000_000),
|
||||||
|
Sample(3, 20_000_000, 30_000_000),
|
||||||
|
)
|
||||||
|
// Server receive times drift: +2ms, +7ms, +3ms relative to send.
|
||||||
|
val seen = listOf(
|
||||||
|
ServerSighting(1, 2_000_000, 2_100_000),
|
||||||
|
ServerSighting(2, 17_000_000, 17_100_000),
|
||||||
|
ServerSighting(3, 23_000_000, 23_100_000),
|
||||||
|
)
|
||||||
|
val m = Directional.analyse(sent, seen)
|
||||||
|
// one-way samples: 2ms, 7ms, 3ms → |7-2| and |3-7| → mean 4.5ms
|
||||||
|
assertEquals(4.5, assertNotNull(m.jitterUpstreamMs))
|
||||||
|
}
|
||||||
|
|
||||||
|
// "No jitter" and "not enough data to say" are different claims, and only one is true here.
|
||||||
|
@Test
|
||||||
|
fun tooFewSamplesReportsNoJitterRatherThanZero() {
|
||||||
|
val m = Directional.analyse(
|
||||||
|
listOf(Sample(1, 0, 10_000_000)),
|
||||||
|
listOf(ServerSighting(1, 2_000_000, 2_100_000)),
|
||||||
|
)
|
||||||
|
assertNull(m.jitterUpstreamMs)
|
||||||
|
assertNull(m.jitterDownstreamMs)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A server record for a sequence we never sent is not evidence about this train; folding it
|
||||||
|
// in would yield loss percentages outside 0–100.
|
||||||
|
@Test
|
||||||
|
fun strayServerRecordsAreIgnored() {
|
||||||
|
val (sent, seen) = clean(5)
|
||||||
|
val m = Directional.analyse(sent, seen + ServerSighting(99, 1, 2) + ServerSighting(100, 3, 4))
|
||||||
|
assertEquals(5, m.seenByServer)
|
||||||
|
assertEquals(0.0, m.lossUpstreamPct)
|
||||||
|
assertTrue(m.lossDownstreamPct in 0.0..100.0)
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun anEmptyTrainDoesNotDivideByZero() {
|
||||||
|
val m = Directional.analyse(emptyList(), emptyList())
|
||||||
|
assertEquals(0.0, m.lossUpstreamPct)
|
||||||
|
assertEquals(0.0, m.lossDownstreamPct)
|
||||||
|
assertFalse(m.noneReachedServer, "nothing sent is not the same as nothing arriving")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -44,7 +44,9 @@ class LiveDownstreamTest {
|
|||||||
for (t in tests) println("${t.type} status=${t.status} metrics=${t.metrics}")
|
for (t in tests) println("${t.type} status=${t.status} metrics=${t.metrics}")
|
||||||
for (f in findings) println("finding ${f.code} [${f.severity}] ${f.title}")
|
for (f in findings) println("finding ${f.code} [${f.severity}] ${f.title}")
|
||||||
|
|
||||||
assertEquals(3, tests.size, "expected pmtud_down, frag_delivery and a downstream train")
|
// Assert on what is present, not on how many: adding a measurement should not be a
|
||||||
|
// test edit. (It was, once — hence the note.)
|
||||||
|
assertTrue(tests.size >= 3, "expected at least the three downstream tests, got ${tests.size}")
|
||||||
val byType = tests.associateBy { it.type }
|
val byType = tests.associateBy { it.type }
|
||||||
|
|
||||||
val pmtud = assertNotNull(byType[TestType.MTU_PMTUD_DOWN], "no mtu.pmtud_down test")
|
val pmtud = assertNotNull(byType[TestType.MTU_PMTUD_DOWN], "no mtu.pmtud_down test")
|
||||||
@@ -58,6 +60,17 @@ class LiveDownstreamTest {
|
|||||||
val frag = assertNotNull(byType[TestType.MTU_FRAG_DELIVERY], "no mtu.frag_delivery test")
|
val frag = assertNotNull(byType[TestType.MTU_FRAG_DELIVERY], "no mtu.frag_delivery test")
|
||||||
assertNotNull(frag.metrics?.get("largest_delivered_bytes"))
|
assertNotNull(frag.metrics?.get("largest_delivered_bytes"))
|
||||||
|
|
||||||
|
// Fragment ordering runs only when fragments arrive at all, and only against a server
|
||||||
|
// that can craft them — so it is checked when present rather than required.
|
||||||
|
byType[TestType.MTU_FRAG_ORDERING]?.let { fo ->
|
||||||
|
val m = fo.metrics?.toString() ?: ""
|
||||||
|
println("fragment ordering: ${fo.status} $m")
|
||||||
|
if (fo.status != TestStatus.UNSUPPORTED) {
|
||||||
|
assertTrue(m.contains("in_order"), "no per-ordering result: $m")
|
||||||
|
assertTrue(m.contains("reversed"), "reversed ordering was never attempted: $m")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
val train = assertNotNull(byType[TestType.TRAIN_UDP_DOWNSTREAM], "no downstream train")
|
val train = assertNotNull(byType[TestType.TRAIN_UDP_DOWNSTREAM], "no downstream train")
|
||||||
assertNotNull(train.evidence, "a train without columnar evidence is not recomputable")
|
assertNotNull(train.evidence, "a train without columnar evidence is not recomputable")
|
||||||
val received = train.metrics?.get("received")?.toString()?.toIntOrNull() ?: 0
|
val received = train.metrics?.get("received")?.toString()?.toIntOrNull() ?: 0
|
||||||
|
|||||||
@@ -0,0 +1,63 @@
|
|||||||
|
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||||
|
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||||
|
|
||||||
|
package app.echo_lot.engine
|
||||||
|
|
||||||
|
import app.echo_lot.protocol.EnrollmentLink
|
||||||
|
import kotlin.test.Test
|
||||||
|
import kotlin.test.assertEquals
|
||||||
|
import kotlin.test.assertNotNull
|
||||||
|
import kotlin.test.assertTrue
|
||||||
|
import kotlin.test.fail
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Enrolls against a LIVE server using the link the server itself minted (probe-protocol.md §2.1).
|
||||||
|
*
|
||||||
|
* This is the test that matters for enrollment, because the failure mode it guards against is a
|
||||||
|
* *disagreement* between two programs: the Go side assembles the link, the Kotlin side takes it
|
||||||
|
* apart, and if they differ by one percent-encoding the pin is wrong by one character — which
|
||||||
|
* does not fail loudly, it fails as an inscrutable TLS error days later. A unit test on either
|
||||||
|
* side alone cannot see that.
|
||||||
|
*
|
||||||
|
* Needs ECHOLOT_ENROLL_URI (minted over SSH by scripts/test-fmr.sh); self-skips without it.
|
||||||
|
*/
|
||||||
|
class LiveEnrollmentTest {
|
||||||
|
|
||||||
|
private val enrollUri = System.getenv("ECHOLOT_ENROLL_URI")
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun enrollsFromTheServersOwnLink() {
|
||||||
|
if (enrollUri.isNullOrBlank()) {
|
||||||
|
println("LiveEnrollmentTest skipped (no ECHOLOT_ENROLL_URI)"); return
|
||||||
|
}
|
||||||
|
println("link: ${enrollUri.take(60)}…")
|
||||||
|
|
||||||
|
val link = assertNotNull(
|
||||||
|
EnrollmentLink.parse(enrollUri),
|
||||||
|
"the client could not parse a link the server produced — the two sides disagree",
|
||||||
|
)
|
||||||
|
println("parsed: url=${link.controlUrl} pin=${link.pin.take(12)}… token=${link.token.take(8)}…")
|
||||||
|
|
||||||
|
// Redeeming applies the pin to the very request that spends the token, so a wrong pin
|
||||||
|
// fails here at the handshake rather than after the token is gone.
|
||||||
|
val enrolled = link.redeem(deviceName = "live-test", appVersion = "0.2.0")
|
||||||
|
assertTrue(enrolled.credential.isNotBlank(), "no credential came back")
|
||||||
|
assertTrue(enrolled.deviceId.isNotBlank(), "no device id came back")
|
||||||
|
println("enrolled: device=${enrolled.deviceId} server=${enrolled.profile.name} " +
|
||||||
|
"${enrolled.profile.serverVersion}")
|
||||||
|
|
||||||
|
// The credential must actually work, and the pin from the link must be the one that
|
||||||
|
// verifies the server — that is the whole claim the link is making.
|
||||||
|
assertEquals(link.controlUrl, enrolled.controlUrl)
|
||||||
|
assertTrue(enrolled.profile.capabilities.contains("udp-probe"),
|
||||||
|
"profile fetched with the new credential looks wrong: ${enrolled.profile.capabilities}")
|
||||||
|
|
||||||
|
// Single-use: a token that still works after redemption is a token an attacker can reuse.
|
||||||
|
try {
|
||||||
|
link.redeem(deviceName = "should-not-happen", appVersion = "0.2.0")
|
||||||
|
fail("the enrollment token was accepted twice — it must be single-use")
|
||||||
|
} catch (t: Throwable) {
|
||||||
|
println("second redemption correctly refused: ${t.message?.take(120)}")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -7,6 +7,7 @@ import app.echo_lot.measurement.*
|
|||||||
import kotlinx.serialization.json.Json
|
import kotlinx.serialization.json.Json
|
||||||
import kotlin.test.Test
|
import kotlin.test.Test
|
||||||
import kotlin.test.assertEquals
|
import kotlin.test.assertEquals
|
||||||
|
import kotlin.test.assertNotNull
|
||||||
import kotlin.test.assertTrue
|
import kotlin.test.assertTrue
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -59,6 +60,18 @@ class LiveMeasurementTest {
|
|||||||
println("metrics: $metrics")
|
println("metrics: $metrics")
|
||||||
assertTrue(metrics.toString().contains("rtt_ms_avg"))
|
assertTrue(metrics.toString().contains("rtt_ms_avg"))
|
||||||
|
|
||||||
|
// The directional split is the point of asking the server what it saw: without it a
|
||||||
|
// lossy path is reported as "loss" with no direction, which sends an engineer looking
|
||||||
|
// in both at once. Correlation is by wire sequence number, so a mismatch here means the
|
||||||
|
// two sides disagree about which packet is which.
|
||||||
|
val m = metrics.toString()
|
||||||
|
assertTrue(m.contains("seen_by_server"), "no directional split in the metrics: $m")
|
||||||
|
val seen = Regex(""""seen_by_server":(\d+)""").find(m)?.groupValues?.get(1)?.toInt()
|
||||||
|
assertNotNull(seen, "seen_by_server missing")
|
||||||
|
assertEquals(20, seen, "the server should have seen every probe on a healthy path")
|
||||||
|
assertTrue(m.contains("jitter_upstream_ms"), "no per-direction jitter: $m")
|
||||||
|
println("directional: $m")
|
||||||
|
|
||||||
assertTrue(doc.summary != null)
|
assertTrue(doc.summary != null)
|
||||||
// A healthy local->fmr path should be green (no loss, no rebinding) or yellow.
|
// A healthy local->fmr path should be green (no loss, no rebinding) or yellow.
|
||||||
println("summary: ${doc.summary}")
|
println("summary: ${doc.summary}")
|
||||||
|
|||||||
@@ -0,0 +1,105 @@
|
|||||||
|
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||||
|
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||||
|
|
||||||
|
package app.echo_lot.engine
|
||||||
|
|
||||||
|
import app.echo_lot.measurement.TestStatus
|
||||||
|
import app.echo_lot.protocol.ControlClient
|
||||||
|
import app.echo_lot.protocol.ProbeSession
|
||||||
|
import kotlin.test.Test
|
||||||
|
import kotlin.test.assertEquals
|
||||||
|
import kotlin.test.assertNotNull
|
||||||
|
import kotlin.test.assertTrue
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Downstream throughput against a LIVE server. Self-skips without ECHOLOT_LIVE_*.
|
||||||
|
*
|
||||||
|
* The assertions are about *honesty* rather than speed: a rate is only a measurement if the run
|
||||||
|
* was ended by the clock and the sender's own count backs it up. A test that just asserted "some
|
||||||
|
* Mbps arrived" would pass equally well against a broken implementation.
|
||||||
|
*/
|
||||||
|
class LiveThroughputTest {
|
||||||
|
|
||||||
|
private val url = System.getenv("ECHOLOT_LIVE_URL")
|
||||||
|
private val pin = System.getenv("ECHOLOT_LIVE_PIN")
|
||||||
|
private val cred = System.getenv("ECHOLOT_LIVE_CRED")
|
||||||
|
private val udp = System.getenv("ECHOLOT_LIVE_UDP")
|
||||||
|
private val target = System.getenv("ECHOLOT_LIVE_TARGET") ?: "fmr"
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun measuresDownstreamRateAndSaysWhatLimitedIt() {
|
||||||
|
if (url == null || pin == null || cred == null || udp == null) {
|
||||||
|
println("LiveThroughputTest skipped (no ECHOLOT_LIVE_* env)"); return
|
||||||
|
}
|
||||||
|
val control = ControlClient(url, setOf(pin), "0.2.0")
|
||||||
|
val session = control.createSession(cred, target)
|
||||||
|
val (host, port) = udp.split(":").let { it[0] to it[1].toInt() }
|
||||||
|
|
||||||
|
val (test, findings) = ProbeSession(cred, session, host, port).use { ps ->
|
||||||
|
ps.echo() // prime: the grant binds to the observed source
|
||||||
|
ThroughputMeasurement(SystemIdSource()).run(
|
||||||
|
cred, session.sessionId, control, ps, sessionRef = "sess-1",
|
||||||
|
durationS = 3, kbps = 20_000,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
control.deleteSession(cred, session.sessionId)
|
||||||
|
|
||||||
|
val m = assertNotNull(test.metrics).toString()
|
||||||
|
println("throughput: ${test.status} $m")
|
||||||
|
for (f in findings) println("finding ${f.code} [${f.severity}] ${f.title}")
|
||||||
|
|
||||||
|
assertEquals(TestStatus.OK, test.status, "no throughput traffic arrived: $m")
|
||||||
|
|
||||||
|
// The sender's own count must be present — without it, loss cannot be attributed and the
|
||||||
|
// number is not a measurement.
|
||||||
|
assertTrue(m.contains("sender_packets"), "no sender report to compare against: $m")
|
||||||
|
assertTrue(m.contains("limited_by"), "the result must say what ended the run: $m")
|
||||||
|
|
||||||
|
val received = Regex(""""received_kbps":(\d+)""").find(m)?.groupValues?.get(1)?.toInt()
|
||||||
|
assertNotNull(received)
|
||||||
|
assertTrue(received > 0, "measured 0 kbps: $m")
|
||||||
|
println("received ${received / 1000} Mbit/s")
|
||||||
|
|
||||||
|
// A run this short and this far below the ceiling should end on the clock. Anything else
|
||||||
|
// means the grant was the constraint, and then the rate says nothing about the path.
|
||||||
|
assertTrue(m.contains(""""limited_by":"duration""""),
|
||||||
|
"the run did not end on the clock, so the rate measures the server, not the path: $m")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Upstream is the direction only the far end can measure. The assertion that matters is that
|
||||||
|
// the server's count is present and plausible against what we sent — a test that only checked
|
||||||
|
// "we transmitted some Mbps" would pass against a server that counted nothing at all.
|
||||||
|
@Test
|
||||||
|
fun measuresUpstreamAgainstTheServersCount() {
|
||||||
|
if (url == null || pin == null || cred == null || udp == null) {
|
||||||
|
println("LiveThroughputTest(up) skipped"); return
|
||||||
|
}
|
||||||
|
val control = ControlClient(url, setOf(pin), "0.2.0")
|
||||||
|
val session = control.createSession(cred, target)
|
||||||
|
val (host, port) = udp.split(":").let { it[0] to it[1].toInt() }
|
||||||
|
|
||||||
|
val (test, findings) = ProbeSession(cred, session, host, port).use { ps ->
|
||||||
|
ps.echo()
|
||||||
|
ThroughputMeasurement(SystemIdSource()).runUpstream(
|
||||||
|
cred, session.sessionId, control, ps, sessionRef = "sess-1",
|
||||||
|
durationS = 3, kbps = 10_000,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
control.deleteSession(cred, session.sessionId)
|
||||||
|
|
||||||
|
val m = assertNotNull(test.metrics).toString()
|
||||||
|
println("upstream: ${test.status} $m")
|
||||||
|
for (f in findings) println("finding ${f.code} [${f.severity}] ${f.title}")
|
||||||
|
|
||||||
|
assertEquals(TestStatus.OK, test.status, "the server counted nothing: $m")
|
||||||
|
val recv = Regex(""""received_packets":(\d+)""").find(m)?.groupValues?.get(1)?.toInt()
|
||||||
|
val sent = Regex(""""sent_packets":(\d+)""").find(m)?.groupValues?.get(1)?.toInt()
|
||||||
|
assertNotNull(recv); assertNotNull(sent)
|
||||||
|
assertTrue(sent > 100, "barely anything was sent, so the rate means nothing: $m")
|
||||||
|
assertTrue(recv > 0, "the server received none of $sent packets: $m")
|
||||||
|
// The counts should be close on a healthy path; wildly different means the two sides are
|
||||||
|
// counting different things rather than the network losing packets.
|
||||||
|
assertTrue(recv <= sent, "the server counted MORE than we sent — the counter is not being reset")
|
||||||
|
println("sent $sent, server saw $recv")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,65 @@
|
|||||||
|
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||||
|
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||||
|
|
||||||
|
package app.echo_lot.engine
|
||||||
|
|
||||||
|
import app.echo_lot.measurement.TestStatus
|
||||||
|
import app.echo_lot.protocol.ControlClient
|
||||||
|
import app.echo_lot.protocol.ProbeSession
|
||||||
|
import kotlin.test.Test
|
||||||
|
import kotlin.test.assertEquals
|
||||||
|
import kotlin.test.assertNotNull
|
||||||
|
import kotlin.test.assertTrue
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Upstream train (types 0x03-0x05) against a LIVE server. Self-skips without ECHOLOT_LIVE_*.
|
||||||
|
*
|
||||||
|
* What is asserted is the ledger property: the server's report must account for what was sent,
|
||||||
|
* per sequence number, because directional loss attribution is the entire reason trains exist —
|
||||||
|
* a test that only checked "a report came back" would pass against a server that counts nothing.
|
||||||
|
*/
|
||||||
|
class LiveUpstreamTrainTest {
|
||||||
|
|
||||||
|
private val url = System.getenv("ECHOLOT_LIVE_URL")
|
||||||
|
private val pin = System.getenv("ECHOLOT_LIVE_PIN")
|
||||||
|
private val cred = System.getenv("ECHOLOT_LIVE_CRED")
|
||||||
|
private val udp = System.getenv("ECHOLOT_LIVE_UDP")
|
||||||
|
private val target = System.getenv("ECHOLOT_LIVE_TARGET") ?: "fmr"
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun serverLedgerAccountsForTheTrain() {
|
||||||
|
if (url == null || pin == null || cred == null || udp == null) {
|
||||||
|
println("LiveUpstreamTrainTest skipped (no ECHOLOT_LIVE_* env)"); return
|
||||||
|
}
|
||||||
|
val control = ControlClient(url, setOf(pin), "0.2.0")
|
||||||
|
val session = control.createSession(cred, target)
|
||||||
|
val (host, port) = udp.split(":").let { it[0] to it[1].toInt() }
|
||||||
|
|
||||||
|
val (test, findings) = ProbeSession(cred, session, host, port).use { ps ->
|
||||||
|
ps.echo() // prime the session so its source is known
|
||||||
|
UpstreamTrainMeasurement(SystemIdSource()).run(
|
||||||
|
ps, sessionRef = "sess-1", count = 120, sizeBytes = 200, interPacketMs = 3,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
control.deleteSession(cred, session.sessionId)
|
||||||
|
|
||||||
|
val m = assertNotNull(test.metrics).toString()
|
||||||
|
println("updown: ${test.status} $m")
|
||||||
|
for (f in findings) println("finding ${f.code} [${f.severity}] ${f.title}")
|
||||||
|
|
||||||
|
assertEquals(TestStatus.OK, test.status, "train report incomplete or absent: $m")
|
||||||
|
|
||||||
|
val sent = Regex(""""sent":(\d+)""").find(m)?.groupValues?.get(1)?.toInt()
|
||||||
|
val received = Regex(""""received_by_server":(\d+)""").find(m)?.groupValues?.get(1)?.toInt()
|
||||||
|
assertNotNull(sent); assertNotNull(received)
|
||||||
|
assertTrue(sent > 0, "nothing was sent: $m")
|
||||||
|
// Over a working path the ledger must be near-complete; a lossy wifi may drop a few, but
|
||||||
|
// a server that fails to count would show up as massive phantom loss here.
|
||||||
|
assertTrue(received >= sent * 9 / 10, "server counted $received of $sent: $m")
|
||||||
|
|
||||||
|
// The columnar evidence must carry a server timestamp for arrived packets — that column
|
||||||
|
// is what one-way delay math consumes after timesync.
|
||||||
|
val ev = assertNotNull(test.evidence).toString()
|
||||||
|
assertTrue(ev.contains("t_srv_rx_ns"), "no server rx column in evidence")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -35,9 +35,37 @@ data class Run(
|
|||||||
val device: DeviceInfo,
|
val device: DeviceInfo,
|
||||||
val tiers: Tiers,
|
val tiers: Tiers,
|
||||||
@SerialName("profiles_used") val profilesUsed: List<String> = emptyList(),
|
@SerialName("profiles_used") val profilesUsed: List<String> = emptyList(),
|
||||||
|
val constraints: Constraints = Constraints(),
|
||||||
val notes: String? = null,
|
val notes: String? = null,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
/**
|
||||||
|
* What limited this run — the counterpart to [Tiers], which records what was available.
|
||||||
|
*
|
||||||
|
* A constrained run is not a failed run, and it is not a normal one either. Without this, a run
|
||||||
|
* taken through a VPN looks exactly like a clean run of a healthy network: the same shape, the
|
||||||
|
* same green verdict, and no way for a reader — or a server aggregating thousands of these — to
|
||||||
|
* know that almost nothing was actually measured.
|
||||||
|
*/
|
||||||
|
@Serializable
|
||||||
|
data class Constraints(
|
||||||
|
/** A VPN held the default route while this ran. */
|
||||||
|
@SerialName("vpn_active") val vpnActive: Boolean = false,
|
||||||
|
/**
|
||||||
|
* Per-network probing was refused by the OS.
|
||||||
|
*
|
||||||
|
* Android blocks `Network.bindSocket()` on the underlying networks whenever a VPN is up, to
|
||||||
|
* stop apps leaking around the tunnel. Every per-network test then measures nothing, so any
|
||||||
|
* conclusion drawn about the wifi or cellular link underneath is unfounded.
|
||||||
|
*/
|
||||||
|
@SerialName("per_network_blocked") val perNetworkBlocked: Boolean = false,
|
||||||
|
/** Networks that could not be measured, by id. */
|
||||||
|
@SerialName("unmeasured_networks") val unmeasuredNetworks: List<String> = emptyList(),
|
||||||
|
) {
|
||||||
|
/** True when this run's results mean something different from an unconstrained one. */
|
||||||
|
val constrained: Boolean get() = vpnActive || perNetworkBlocked
|
||||||
|
}
|
||||||
|
|
||||||
@Serializable
|
@Serializable
|
||||||
enum class Trigger {
|
enum class Trigger {
|
||||||
@SerialName("manual") MANUAL,
|
@SerialName("manual") MANUAL,
|
||||||
|
|||||||
+321
@@ -0,0 +1,321 @@
|
|||||||
|
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||||
|
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||||
|
|
||||||
|
package app.echo_lot.measurement
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The registry of finding codes (measurement-schema.md §9, open item 1).
|
||||||
|
*
|
||||||
|
* A finding code is the stable, machine-readable half of a result: the prose changes, the code is
|
||||||
|
* what a dashboard groups by and what someone greps a year of archived runs for. That only holds
|
||||||
|
* if a code means exactly one thing forever — which is not something ad-hoc string literals at
|
||||||
|
* fifteen call sites can promise.
|
||||||
|
*
|
||||||
|
* The failure this exists to prevent had already happened by the time it was written. Two
|
||||||
|
* independently-added emitters produced `connectivity.downstream_loss` and
|
||||||
|
* `connectivity.loss_downstream` for the same concept, and nothing anywhere objected. Anyone
|
||||||
|
* aggregating either one would have silently seen half their data.
|
||||||
|
*
|
||||||
|
* So codes are declared here as typed specs, each carrying its category and default severity, and
|
||||||
|
* emitters reference the spec rather than retyping the string. That makes a typo a compile error,
|
||||||
|
* and makes it impossible for two call sites to disagree about which category a finding belongs
|
||||||
|
* to — a disagreement that would otherwise split one fault across two verdict lights.
|
||||||
|
*/
|
||||||
|
data class FindingSpec(
|
||||||
|
val code: String,
|
||||||
|
val category: Category,
|
||||||
|
/** Severity when nothing about the specific run argues otherwise; emitters may escalate. */
|
||||||
|
val severity: Severity,
|
||||||
|
/** One line: what this finding asserts. Present tense, no hedging. */
|
||||||
|
val meaning: String,
|
||||||
|
/**
|
||||||
|
* What the finding rules *out*, where that is the useful half. "Loss upstream" is worth much
|
||||||
|
* more when it also says the return path is fine, because that halves where to look next.
|
||||||
|
*/
|
||||||
|
val rulesOut: String? = null,
|
||||||
|
)
|
||||||
|
|
||||||
|
object FindingRegistry {
|
||||||
|
|
||||||
|
// ---- connectivity ----------------------------------------------------------------
|
||||||
|
|
||||||
|
// Renamed from nat.* before anything shipped: neither of these is about NAT, and the
|
||||||
|
// prefix is what decides which category - and therefore which verdict light - a finding
|
||||||
|
// rolls up into. A nat.* code landing under connectivity would be a permanent puzzle.
|
||||||
|
val UDP_UNREACHABLE = FindingSpec(
|
||||||
|
"connectivity.udp_unreachable", Category.CONNECTIVITY, Severity.HIGH,
|
||||||
|
"No UDP echo replies came back from the server at all.",
|
||||||
|
)
|
||||||
|
|
||||||
|
val UDP_UNREACHABLE_UPSTREAM = FindingSpec(
|
||||||
|
"connectivity.udp_unreachable_upstream", Category.CONNECTIVITY, Severity.HIGH,
|
||||||
|
"The server received none of the probes, so traffic is dropped on the way out.",
|
||||||
|
rulesOut = "The return path: nothing arrived to be replied to.",
|
||||||
|
)
|
||||||
|
|
||||||
|
val UDP_LOSS = FindingSpec(
|
||||||
|
"connectivity.udp_loss", Category.CONNECTIVITY, Severity.MEDIUM,
|
||||||
|
"A large fraction of round-trip probes were lost, direction unknown.",
|
||||||
|
)
|
||||||
|
|
||||||
|
val LOSS_UPSTREAM = FindingSpec(
|
||||||
|
"connectivity.loss_upstream", Category.CONNECTIVITY, Severity.MEDIUM,
|
||||||
|
"Probes were lost on the way to the server.",
|
||||||
|
rulesOut = "The return path: replies came back for everything that arrived.",
|
||||||
|
)
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The single code for "lost on the return path", whichever measurement found it.
|
||||||
|
*
|
||||||
|
* Two emitters had independently invented `connectivity.downstream_loss` and
|
||||||
|
* `connectivity.loss_downstream` for this, and nothing objected. Anyone aggregating either
|
||||||
|
* one would have silently seen half their data. Paired with [LOSS_UPSTREAM] so the two
|
||||||
|
* directions read as a set.
|
||||||
|
*/
|
||||||
|
val LOSS_DOWNSTREAM = FindingSpec(
|
||||||
|
"connectivity.loss_downstream", Category.CONNECTIVITY, Severity.MEDIUM,
|
||||||
|
"Packets were lost on the way back from the server.",
|
||||||
|
rulesOut = "The outbound path: the server received what it was answering.",
|
||||||
|
)
|
||||||
|
|
||||||
|
val DOWNSTREAM_BLOCKED = FindingSpec(
|
||||||
|
"connectivity.downstream_blocked", Category.CONNECTIVITY, Severity.HIGH,
|
||||||
|
"Server-initiated packets never arrive, although round trips work.",
|
||||||
|
rulesOut = "Basic reachability: the path forwards replies, just not unsolicited traffic.",
|
||||||
|
)
|
||||||
|
|
||||||
|
val DOWNSTREAM_REORDER = FindingSpec(
|
||||||
|
"connectivity.downstream_reorder", Category.CONNECTIVITY, Severity.LOW,
|
||||||
|
"Downstream packets arrive in a different order than they were sent.",
|
||||||
|
)
|
||||||
|
|
||||||
|
// MEDIUM, not HIGH: a captive portal is a condition to report, not necessarily a fault - on
|
||||||
|
// hotel or cafe wifi it is exactly what should be there, and logging in clears it. NO_INTERNET
|
||||||
|
// is the HIGH one, because nothing the user does locally fixes that. The registry first said
|
||||||
|
// HIGH; the probe emitting it had always said MEDIUM, and the probe was the considered value.
|
||||||
|
val CAPTIVE_PORTAL = FindingSpec(
|
||||||
|
"connectivity.captive_portal", Category.CONNECTIVITY, Severity.MEDIUM,
|
||||||
|
"A captive portal is intercepting connectivity checks.",
|
||||||
|
)
|
||||||
|
|
||||||
|
val NO_INTERNET = FindingSpec(
|
||||||
|
"connectivity.no_internet", Category.CONNECTIVITY, Severity.HIGH,
|
||||||
|
"Android's own connectivity checks fail on this network.",
|
||||||
|
)
|
||||||
|
|
||||||
|
// ---- mtu -------------------------------------------------------------------------
|
||||||
|
|
||||||
|
val MTU_REDUCED_DOWNSTREAM = FindingSpec(
|
||||||
|
"mtu.reduced_downstream", Category.MTU, Severity.LOW,
|
||||||
|
"The downstream path MTU is below the usual 1500 bytes.",
|
||||||
|
)
|
||||||
|
|
||||||
|
val MTU_DOWNSTREAM_BLACKHOLE = FindingSpec(
|
||||||
|
"mtu.downstream_blackhole", Category.MTU, Severity.MEDIUM,
|
||||||
|
"Datagrams above the path MTU are dropped downstream, fragmented or not.",
|
||||||
|
)
|
||||||
|
|
||||||
|
val FRAGMENTS_BLOCKED = FindingSpec(
|
||||||
|
"mtu.fragments_blocked", Category.MTU, Severity.MEDIUM,
|
||||||
|
"IP fragments do not reach this device even when sent in order.",
|
||||||
|
)
|
||||||
|
|
||||||
|
val FRAGMENT_REORDER_SENSITIVE = FindingSpec(
|
||||||
|
"mtu.fragment_reorder_sensitive", Category.MTU, Severity.LOW,
|
||||||
|
"Fragments are delivered in order but dropped when reordered or delayed.",
|
||||||
|
rulesOut = "Fragmentation itself: in-order fragments arrive fine.",
|
||||||
|
)
|
||||||
|
|
||||||
|
// ---- nat -------------------------------------------------------------------------
|
||||||
|
|
||||||
|
val NAT_UDP_REBINDING = FindingSpec(
|
||||||
|
"nat.udp_rebinding", Category.NAT, Severity.MEDIUM,
|
||||||
|
"A NAT remapped the UDP source port mid-flow.",
|
||||||
|
)
|
||||||
|
|
||||||
|
val NAT_SYMMETRIC = FindingSpec(
|
||||||
|
"nat.symmetric", Category.NAT, Severity.MEDIUM,
|
||||||
|
"The NAT assigns a different external port per destination.",
|
||||||
|
)
|
||||||
|
|
||||||
|
// ---- perf ------------------------------------------------------------------------
|
||||||
|
|
||||||
|
val THROUGHPUT_NO_DELIVERY = FindingSpec(
|
||||||
|
"perf.throughput_no_delivery", Category.PERFORMANCE, Severity.HIGH,
|
||||||
|
"No throughput traffic arrived, although the server sent it.",
|
||||||
|
)
|
||||||
|
|
||||||
|
val THROUGHPUT_BELOW_OFFERED = FindingSpec(
|
||||||
|
"perf.throughput_below_offered", Category.PERFORMANCE, Severity.LOW,
|
||||||
|
"Less throughput arrived than the server sent for the whole run.",
|
||||||
|
)
|
||||||
|
|
||||||
|
// ---- dns -------------------------------------------------------------------------
|
||||||
|
|
||||||
|
val DNS_ANSWER_REWRITTEN = FindingSpec(
|
||||||
|
"dns.answer_rewritten", Category.DNS, Severity.HIGH,
|
||||||
|
"A resolver returned an answer that differs from the authoritative record.",
|
||||||
|
)
|
||||||
|
|
||||||
|
val DNS_AUTHORITATIVE_UNREACHABLE = FindingSpec(
|
||||||
|
"dns.authoritative_unreachable", Category.DNS, Severity.MEDIUM,
|
||||||
|
"The canary zone's authoritative server could not be reached.",
|
||||||
|
)
|
||||||
|
|
||||||
|
// ---- v6 ----------------------------------------------------------------------------
|
||||||
|
//
|
||||||
|
// Prefix is `v6.`, matching the test-type registry (v6.brokenness, v6.happy_eyeballs, ...).
|
||||||
|
// These were `ipv6.*` while declaring Category.IPV6, but the prefix map only knows "v6", so
|
||||||
|
// they silently rolled up under connectivity: the third instance of a prefix disagreeing with
|
||||||
|
// its category and quietly moving a fault to a different verdict light.
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Renamed from `v6.broken`, which claimed more than the evidence supports.
|
||||||
|
*
|
||||||
|
* The only signal behind it is ICMPv6 echo getting no reply — and ICMPv6 echo is widely
|
||||||
|
* filtered on networks where IPv6 otherwise works perfectly. A phone that reported this while
|
||||||
|
* happily loading an IPv6-only site over TCP is what caught it. From here the two cases look
|
||||||
|
* identical, so the finding now says what was observed and names both explanations rather than
|
||||||
|
* picking one.
|
||||||
|
*
|
||||||
|
* It is worth reporting either way: filtered ICMPv6 breaks Path MTU Discovery, which is its
|
||||||
|
* own fault even when IPv6 works.
|
||||||
|
*/
|
||||||
|
/**
|
||||||
|
* A global IPv6 address with no default route.
|
||||||
|
*
|
||||||
|
* This is the structural version of the same complaint, and it is worth far more than the
|
||||||
|
* ICMP one because it admits no other explanation: the device has an address it cannot route
|
||||||
|
* with. Nothing is filtered, nothing is inferred — the routing table says so directly, and it
|
||||||
|
* is already in the link snapshot.
|
||||||
|
*
|
||||||
|
* Not always a fault. A VPN that installs host routes to specific destinations produces
|
||||||
|
* exactly this shape on purpose, and it works. What makes it worth reporting either way is
|
||||||
|
* that applications cannot tell: having a global address, they will try IPv6 first and stall
|
||||||
|
* for every destination the routes do not cover.
|
||||||
|
*/
|
||||||
|
/**
|
||||||
|
* An IPv6 default route with no global address to use it from — the mirror of
|
||||||
|
* [V6_NO_DEFAULT_ROUTE], and the more common misconfiguration of the two.
|
||||||
|
*
|
||||||
|
* The router is sending RAs that name it as a default gateway, but SLAAC produced no address:
|
||||||
|
* no prefix information option, or a prefix without the autonomous flag, or DHCPv6-only
|
||||||
|
* addressing the device did not complete. The network is announcing IPv6 service it does not
|
||||||
|
* actually deliver.
|
||||||
|
*
|
||||||
|
* This is worth flagging above the ICMP signal because it is both certain and consequential.
|
||||||
|
* Hosts see router advertisements, believe IPv6 is available, and pay a connection-attempt
|
||||||
|
* timeout on every dual-stack destination before falling back to IPv4 — the classic "the
|
||||||
|
* internet feels slow" complaint with no packet loss anywhere to explain it.
|
||||||
|
*/
|
||||||
|
val V6_ROUTE_WITHOUT_ADDRESS = FindingSpec(
|
||||||
|
"v6.route_without_address", Category.IPV6, Severity.MEDIUM,
|
||||||
|
"The network advertises an IPv6 default route but the device has no global IPv6 address.",
|
||||||
|
rulesOut = "A working IPv6 setup: SLAAC did not produce a usable address on this link.",
|
||||||
|
)
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A VPN prevented the underlying networks from being measured.
|
||||||
|
*
|
||||||
|
* Reported rather than worked around: Android refuses `Network.bindSocket()` on the networks
|
||||||
|
* beneath a VPN precisely so apps cannot leak around the tunnel, and that is correct
|
||||||
|
* behaviour. What is not acceptable is a run that quietly measures nothing and calls the
|
||||||
|
* result healthy, so this says plainly which networks went unmeasured and why.
|
||||||
|
*/
|
||||||
|
/**
|
||||||
|
* The network's DNS server answers, but this device cannot resolve through it.
|
||||||
|
*
|
||||||
|
* Worth separating from every other DNS failure because the remedy is somewhere else entirely.
|
||||||
|
* A name that will not resolve looks identical to a user whatever the cause, and the two causes
|
||||||
|
* pull in opposite directions: a server that does not answer means the network is broken and
|
||||||
|
* the router is the thing to examine, while a server that answers a direct query on a device
|
||||||
|
* that still cannot resolve means the platform resolver has wedged — fixed by toggling wifi,
|
||||||
|
* and nothing to do with the network at all.
|
||||||
|
*
|
||||||
|
* Proven rather than inferred: the probe sends its own UDP query, bypassing the component under
|
||||||
|
* suspicion, and compares that against what the platform returns for the same name.
|
||||||
|
*/
|
||||||
|
/**
|
||||||
|
* The network hands out a search domain its DNS server will not answer for.
|
||||||
|
*
|
||||||
|
* A resolver appends search domains to lookups, so every name a client asks about can stall on
|
||||||
|
* a domain the server ignores. The failure mode is silence rather than a negative answer, and
|
||||||
|
* silence is indistinguishable from packet loss: clients retry instead of moving on, and some
|
||||||
|
* give up on the lookup entirely. That makes it look like the device is broken when the
|
||||||
|
* network is.
|
||||||
|
*
|
||||||
|
* Whether it bites depends on the resolver — some try the bare name first and never notice —
|
||||||
|
* which is why two devices on the same network can disagree about whether DNS works.
|
||||||
|
*/
|
||||||
|
val DNS_SEARCH_DOMAIN_UNANSWERED = FindingSpec(
|
||||||
|
"dns.search_domain_unanswered", Category.DNS, Severity.HIGH,
|
||||||
|
"The network advertises a DNS search domain that its own server does not answer for.",
|
||||||
|
rulesOut = "A fault on this device: the same server answers ordinary names normally.",
|
||||||
|
)
|
||||||
|
|
||||||
|
val DNS_SYSTEM_RESOLVER_BROKEN = FindingSpec(
|
||||||
|
"dns.system_resolver_broken", Category.DNS, Severity.HIGH,
|
||||||
|
"The network's DNS server answers, but this device cannot resolve names through it.",
|
||||||
|
rulesOut = "A network fault: the server replied to a query sent from this device.",
|
||||||
|
)
|
||||||
|
|
||||||
|
val MEASUREMENT_VPN_CONSTRAINED = FindingSpec(
|
||||||
|
"measurement.vpn_constrained", Category.CONNECTIVITY, Severity.INFO,
|
||||||
|
"A VPN was active, so the networks underneath it could not be measured.",
|
||||||
|
rulesOut = "Nothing — this run says little about the underlying network either way.",
|
||||||
|
)
|
||||||
|
|
||||||
|
val V6_NO_DEFAULT_ROUTE = FindingSpec(
|
||||||
|
"v6.no_default_route", Category.IPV6, Severity.MEDIUM,
|
||||||
|
"The device has a global IPv6 address but no IPv6 default route.",
|
||||||
|
rulesOut = "Guesswork: this is read from the routing table, not inferred from silence.",
|
||||||
|
)
|
||||||
|
|
||||||
|
val V6_NO_ICMP_REPLY = FindingSpec(
|
||||||
|
"v6.no_icmp_reply", Category.IPV6, Severity.LOW,
|
||||||
|
"IPv6 is configured but ICMPv6 echo gets no reply.",
|
||||||
|
rulesOut = "Nothing on its own: IPv6 may work fine with ICMP filtered.",
|
||||||
|
)
|
||||||
|
|
||||||
|
/**
|
||||||
|
* IPv6 is advertised and does not work — the claim `v6.broken` originally made on ICMP
|
||||||
|
* silence alone, now reinstated because it can finally be backed: it is only emitted when a
|
||||||
|
* real IPv6 TCP connection (v6.brokenness) failed on the same network whose ICMPv6 went
|
||||||
|
* unanswered. Two independent transports failing on a network that advertises IPv6 is what
|
||||||
|
* "broken" actually means; either signal alone still gets [V6_NO_ICMP_REPLY].
|
||||||
|
*/
|
||||||
|
val V6_BROKEN = FindingSpec(
|
||||||
|
"v6.broken", Category.IPV6, Severity.HIGH,
|
||||||
|
"IPv6 is advertised on this network but carries no traffic.",
|
||||||
|
rulesOut = "ICMP filtering as the benign explanation: a TCP connection over IPv6 failed too.",
|
||||||
|
)
|
||||||
|
|
||||||
|
/**
|
||||||
|
* INFO deliberately, and it needs to stay that way.
|
||||||
|
*
|
||||||
|
* Most networks still do not offer IPv6, and that is not a fault. Reporting it as a warning
|
||||||
|
* lights a yellow verdict on a perfectly healthy network, which teaches people to ignore the
|
||||||
|
* light — the one thing a diagnostic must never do.
|
||||||
|
*/
|
||||||
|
val V6_NOT_OFFERED = FindingSpec(
|
||||||
|
"v6.not_offered", Category.IPV6, Severity.INFO,
|
||||||
|
"This network does not offer IPv6.",
|
||||||
|
)
|
||||||
|
|
||||||
|
/** Every registered finding, in declaration order. */
|
||||||
|
val all: List<FindingSpec> = listOf(
|
||||||
|
UDP_UNREACHABLE, UDP_UNREACHABLE_UPSTREAM, UDP_LOSS, LOSS_UPSTREAM, LOSS_DOWNSTREAM,
|
||||||
|
DOWNSTREAM_BLOCKED, DOWNSTREAM_REORDER, CAPTIVE_PORTAL, NO_INTERNET,
|
||||||
|
MTU_REDUCED_DOWNSTREAM, MTU_DOWNSTREAM_BLACKHOLE, FRAGMENTS_BLOCKED,
|
||||||
|
FRAGMENT_REORDER_SENSITIVE,
|
||||||
|
NAT_UDP_REBINDING, NAT_SYMMETRIC,
|
||||||
|
THROUGHPUT_NO_DELIVERY, THROUGHPUT_BELOW_OFFERED,
|
||||||
|
DNS_ANSWER_REWRITTEN, DNS_AUTHORITATIVE_UNREACHABLE,
|
||||||
|
DNS_SEARCH_DOMAIN_UNANSWERED, DNS_SYSTEM_RESOLVER_BROKEN, MEASUREMENT_VPN_CONSTRAINED,
|
||||||
|
V6_NO_DEFAULT_ROUTE, V6_ROUTE_WITHOUT_ADDRESS, V6_NO_ICMP_REPLY, V6_BROKEN, V6_NOT_OFFERED,
|
||||||
|
)
|
||||||
|
|
||||||
|
private val byCode: Map<String, FindingSpec> = all.associateBy { it.code }
|
||||||
|
|
||||||
|
fun byCode(code: String): FindingSpec? = byCode[code]
|
||||||
|
}
|
||||||
@@ -18,6 +18,28 @@ data class Network(
|
|||||||
val wifi: Wifi? = null,
|
val wifi: Wifi? = null,
|
||||||
val cellular: Cellular? = null,
|
val cellular: Cellular? = null,
|
||||||
val changes: List<NetworkChange> = emptyList(),
|
val changes: List<NetworkChange> = emptyList(),
|
||||||
|
@SerialName("system_verdict") val systemVerdict: SystemVerdict? = null,
|
||||||
|
)
|
||||||
|
|
||||||
|
/**
|
||||||
|
* What Android itself concluded about a network, as opposed to what we measured.
|
||||||
|
*
|
||||||
|
* Recorded because it is the verdict the user can see — the "no internet" warning in the status
|
||||||
|
* bar — and because it is free: the platform has already done the work by the time a run starts.
|
||||||
|
*
|
||||||
|
* Its real value is disagreement. When Android says a network is unusable and our own probes reach
|
||||||
|
* the internet regardless, the fault is in the device rather than the network, and that distinction
|
||||||
|
* is the difference between "fix your router" and "toggle your wifi". Neither number alone can say
|
||||||
|
* that; only the two together.
|
||||||
|
*/
|
||||||
|
@Serializable
|
||||||
|
data class SystemVerdict(
|
||||||
|
/** Android's own connectivity check passed. Null when the platform did not say. */
|
||||||
|
val validated: Boolean? = null,
|
||||||
|
/** Android believes a captive portal is intercepting this network. */
|
||||||
|
@SerialName("captive_portal") val captivePortal: Boolean? = null,
|
||||||
|
/** Some traffic works and some does not — Android's own hedge. */
|
||||||
|
@SerialName("partial_connectivity") val partialConnectivity: Boolean? = null,
|
||||||
)
|
)
|
||||||
|
|
||||||
@Serializable
|
@Serializable
|
||||||
|
|||||||
@@ -35,6 +35,9 @@ data class CategorySummary(
|
|||||||
* (critical|high → red, medium|low → yellow, info/none → green).
|
* (critical|high → red, medium|low → yellow, info/none → green).
|
||||||
* - A category is `inconclusive` when > 50% of its tests are failed/unsupported.
|
* - A category is `inconclusive` when > 50% of its tests are failed/unsupported.
|
||||||
* - Overall = the worst category light; `inconclusive` only when ALL categories are.
|
* - Overall = the worst category light; `inconclusive` only when ALL categories are.
|
||||||
|
* - A run whose per-network probing was blocked is `inconclusive` outright, whatever the
|
||||||
|
* categories say. The lights describe what the tests found; when the OS refused to let the
|
||||||
|
* tests run, a green light would describe nothing at all.
|
||||||
*
|
*
|
||||||
* The mapping test-type → category comes from [TestType.category]. Only categories that have
|
* The mapping test-type → category comes from [TestType.category]. Only categories that have
|
||||||
* findings or tests appear in the summary.
|
* findings or tests appear in the summary.
|
||||||
@@ -44,7 +47,10 @@ object Verdicts {
|
|||||||
private fun isInconclusiveTest(s: TestStatus) =
|
private fun isInconclusiveTest(s: TestStatus) =
|
||||||
s == TestStatus.FAILED || s == TestStatus.UNSUPPORTED
|
s == TestStatus.FAILED || s == TestStatus.UNSUPPORTED
|
||||||
|
|
||||||
fun derive(tests: List<Test>, findings: List<Finding>): Summary {
|
fun derive(tests: List<Test>, findings: List<Finding>): Summary =
|
||||||
|
derive(tests, findings, Constraints())
|
||||||
|
|
||||||
|
fun derive(tests: List<Test>, findings: List<Finding>, constraints: Constraints): Summary {
|
||||||
val testsByCat = tests.groupBy { TestType.category(it.type) }
|
val testsByCat = tests.groupBy { TestType.category(it.type) }
|
||||||
val findingsByCat = findings.groupBy { it.category }
|
val findingsByCat = findings.groupBy { it.category }
|
||||||
val categories = (testsByCat.keys + findingsByCat.keys)
|
val categories = (testsByCat.keys + findingsByCat.keys)
|
||||||
@@ -72,7 +78,14 @@ object Verdicts {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
val overall = deriveOverall(perCat.values)
|
// A run that could not measure the networks it was asked about has not found them
|
||||||
|
// healthy; it has found out nothing. Reporting that as green is the single most
|
||||||
|
// misleading thing this function could do, so the constraint outranks the lights.
|
||||||
|
val overall = if (constraints.perNetworkBlocked) {
|
||||||
|
Verdict.INCONCLUSIVE
|
||||||
|
} else {
|
||||||
|
deriveOverall(perCat.values)
|
||||||
|
}
|
||||||
return Summary(overall = overall, categories = perCat)
|
return Summary(overall = overall, categories = perCat)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -77,6 +77,8 @@ object TestType {
|
|||||||
const val MTU_BLACKHOLE = "mtu.blackhole"
|
const val MTU_BLACKHOLE = "mtu.blackhole"
|
||||||
const val MTU_MSS_OBSERVED = "mtu.mss_observed"
|
const val MTU_MSS_OBSERVED = "mtu.mss_observed"
|
||||||
const val MTU_FRAG_DELIVERY = "mtu.frag_delivery"
|
const val MTU_FRAG_DELIVERY = "mtu.frag_delivery"
|
||||||
|
/** Whether fragments survive arriving out of order, not merely whether they survive. */
|
||||||
|
const val MTU_FRAG_ORDERING = "mtu.frag_ordering"
|
||||||
// nat
|
// nat
|
||||||
const val NAT_STUN_5780 = "nat.stun_5780"
|
const val NAT_STUN_5780 = "nat.stun_5780"
|
||||||
const val NAT_MAPPING_LIFETIME_UDP = "nat.mapping_lifetime_udp"
|
const val NAT_MAPPING_LIFETIME_UDP = "nat.mapping_lifetime_udp"
|
||||||
@@ -87,6 +89,13 @@ object TestType {
|
|||||||
// dns
|
// dns
|
||||||
const val DNS_RESOLVER_INVENTORY = "dns.resolver_inventory"
|
const val DNS_RESOLVER_INVENTORY = "dns.resolver_inventory"
|
||||||
const val DNS_CANARY = "dns.canary"
|
const val DNS_CANARY = "dns.canary"
|
||||||
|
/**
|
||||||
|
* Does this device's own resolver work, as distinct from the network's DNS.
|
||||||
|
*
|
||||||
|
* Registry addition, v1.1. Kept apart from [DNS_CANARY], which asks whether answers are being
|
||||||
|
* tampered with; this asks whether answers arrive at all, and where the failure sits.
|
||||||
|
*/
|
||||||
|
const val DNS_RESOLVER = "dns.resolver"
|
||||||
const val DNS_INTERCEPTION = "dns.interception"
|
const val DNS_INTERCEPTION = "dns.interception"
|
||||||
const val DNS_TTL_INTEGRITY = "dns.ttl_integrity"
|
const val DNS_TTL_INTEGRITY = "dns.ttl_integrity"
|
||||||
const val DNS_ANSWER_INTEGRITY = "dns.answer_integrity"
|
const val DNS_ANSWER_INTEGRITY = "dns.answer_integrity"
|
||||||
|
|||||||
@@ -0,0 +1,73 @@
|
|||||||
|
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||||
|
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||||
|
|
||||||
|
package app.echo_lot.measurement
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The two ways a network can be half-configured for IPv6, read from the link snapshot.
|
||||||
|
*
|
||||||
|
* Pure model logic rather than something a ViewModel does, because "is this network's IPv6
|
||||||
|
* broken, and in which direction" is exactly the kind of judgement that should be checkable
|
||||||
|
* against a captured routing table without a phone in the loop.
|
||||||
|
*/
|
||||||
|
object V6Analysis {
|
||||||
|
|
||||||
|
/** Linux tunnel interfaces: WireGuard/Netbird (tun*, wg*), plus the usual VPN names. */
|
||||||
|
private val TUNNEL_IFACE = Regex("""^(tun|tap|wg|ppp|ipsec|utun)\d*$""")
|
||||||
|
|
||||||
|
/** What one network's IPv6 configuration looks like. */
|
||||||
|
data class Shape(
|
||||||
|
val iface: String,
|
||||||
|
/** A global address with no ::/0 route: an address the device cannot route with. */
|
||||||
|
val addressWithoutRoute: Boolean,
|
||||||
|
/** A ::/0 route with no global address: a route the device cannot source from. */
|
||||||
|
val routeWithoutAddress: Boolean,
|
||||||
|
/** The routes belong to a tunnel, so a partial view of IPv6 is likely deliberate. */
|
||||||
|
val tunnel: Boolean,
|
||||||
|
)
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Classifies each network's IPv6 configuration.
|
||||||
|
*
|
||||||
|
* Both shapes are read straight from the link snapshot rather than inferred from silence, so
|
||||||
|
* unlike an ICMP signal there is no competing explanation for what was observed — and both
|
||||||
|
* matter for the same reason: an application cannot tell in advance, so it tries IPv6 first
|
||||||
|
* and waits.
|
||||||
|
*
|
||||||
|
* They differ in what they mean. An address with no route is what a VPN installing host routes
|
||||||
|
* to specific destinations produces on purpose, and it works; calling that a fault would be the
|
||||||
|
* "lack of IPv6 is a yellow condition" mistake in a new costume, so a tunnel downgrades it to
|
||||||
|
* information. A route with no address is the opposite: the router advertised itself as a
|
||||||
|
* default gateway but SLAAC produced nothing usable, so the network is announcing IPv6 service
|
||||||
|
* it does not deliver. That one is a real misconfiguration however it arises.
|
||||||
|
*/
|
||||||
|
fun classify(networks: List<Network>): List<Shape> = networks.map { n ->
|
||||||
|
val globalV6 = n.link.addresses.any { isGlobalV6(it.addr) }
|
||||||
|
val v6Routes = n.link.routes.filter { it.dst.contains(':') }
|
||||||
|
val hasDefault = v6Routes.any { it.dst == "::/0" }
|
||||||
|
Shape(
|
||||||
|
iface = n.iface ?: v6Routes.firstOrNull()?.iface.orEmpty(),
|
||||||
|
addressWithoutRoute = globalV6 && !hasDefault,
|
||||||
|
routeWithoutAddress = hasDefault && !globalV6,
|
||||||
|
// Android labels the transport itself, which beats guessing from a name; the regex
|
||||||
|
// stays as a backstop for tunnels Android does not own (a userspace WireGuard, say,
|
||||||
|
// or anything seen through the shell tier).
|
||||||
|
tunnel = n.transport == Transport.VPN ||
|
||||||
|
v6Routes.any { TUNNEL_IFACE.containsMatchIn(it.iface.orEmpty()) },
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Whether an address is IPv6 and usable as a source for off-link traffic.
|
||||||
|
*
|
||||||
|
* ULAs count. A ULA is not globally routable, but it is a global-*scope* address the stack
|
||||||
|
* will happily select as a source, which is the property that matters here — an overlay
|
||||||
|
* network handing out fc00::/7 addresses is providing working IPv6 to the destinations it
|
||||||
|
* carries, and treating that as "no address" would misreport every VPN as broken.
|
||||||
|
*/
|
||||||
|
private fun isGlobalV6(addr: String): Boolean {
|
||||||
|
if (!addr.contains(':')) return false
|
||||||
|
val a = addr.substringBefore('%').lowercase() // strip any zone index
|
||||||
|
return !a.startsWith("fe80") && a != "::1" && a != "::"
|
||||||
|
}
|
||||||
|
}
|
||||||
+133
@@ -0,0 +1,133 @@
|
|||||||
|
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||||
|
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||||
|
|
||||||
|
package app.echo_lot.measurement
|
||||||
|
|
||||||
|
import java.io.File
|
||||||
|
import kotlin.test.Test
|
||||||
|
import kotlin.test.assertEquals
|
||||||
|
import kotlin.test.assertTrue
|
||||||
|
import kotlin.test.fail
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Keeps the finding registry honest.
|
||||||
|
*
|
||||||
|
* The interesting test is the last one: it reads `docs/findings-registry.md` and fails when the
|
||||||
|
* document and the code disagree. Documentation that drifts from its implementation is worse than
|
||||||
|
* none, because it still looks authoritative — and a finding registry is precisely the artifact
|
||||||
|
* other people build tooling against.
|
||||||
|
*/
|
||||||
|
class FindingRegistryTest {
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun codesAreUnique() {
|
||||||
|
val dupes = FindingRegistry.all.groupBy { it.code }.filterValues { it.size > 1 }.keys
|
||||||
|
assertTrue(dupes.isEmpty(), "duplicate finding codes: $dupes")
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun everyDeclaredSpecIsInTheAllList() {
|
||||||
|
// Reflection over the object's properties: a spec that is declared but left out of `all`
|
||||||
|
// is invisible to the doc check and to any consumer enumerating the registry.
|
||||||
|
val declared = FindingRegistry::class.java.declaredMethods
|
||||||
|
.filter { it.parameterCount == 0 && it.returnType == FindingSpec::class.java }
|
||||||
|
.mapNotNull { runCatching { it.invoke(FindingRegistry) as FindingSpec }.getOrNull() }
|
||||||
|
.map { it.code }
|
||||||
|
.toSet()
|
||||||
|
val listed = FindingRegistry.all.map { it.code }.toSet()
|
||||||
|
assertEquals(declared, listed, "declared specs and the `all` list disagree")
|
||||||
|
}
|
||||||
|
|
||||||
|
// The prefix decides the category, and the category decides which verdict light the finding
|
||||||
|
// rolls up into. A code whose prefix disagrees with its category silently moves a fault to a
|
||||||
|
// different light — the exact bug that got two codes renamed out of nat.*.
|
||||||
|
@Test
|
||||||
|
fun everyPrefixMatchesItsCategory() {
|
||||||
|
for (spec in FindingRegistry.all) {
|
||||||
|
val fromPrefix = TestType.category(spec.code)
|
||||||
|
assertEquals(
|
||||||
|
fromPrefix, spec.category,
|
||||||
|
"${spec.code} is declared as ${spec.category} but its prefix maps to $fromPrefix",
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun codesFollowTheNamingConvention() {
|
||||||
|
val shape = Regex("^[a-z0-9]+\\.[a-z0-9_]+$")
|
||||||
|
for (spec in FindingRegistry.all) {
|
||||||
|
assertTrue(shape.matches(spec.code), "malformed code: ${spec.code}")
|
||||||
|
assertTrue(spec.meaning.isNotBlank(), "${spec.code} has no meaning")
|
||||||
|
assertTrue(
|
||||||
|
spec.meaning.trimEnd().endsWith("."),
|
||||||
|
"${spec.code}'s meaning should be a sentence: '${spec.meaning}'",
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Two near-identical codes are how one fault ends up split across two dashboards. This is a
|
||||||
|
// blunt check — it will not catch every synonym — but it catches the shape that already
|
||||||
|
// happened: the same words in a different order.
|
||||||
|
@Test
|
||||||
|
fun noTwoCodesAreAnagramsOfEachOther() {
|
||||||
|
val normalised = FindingRegistry.all.associate { spec ->
|
||||||
|
spec.code to spec.code.substringAfter('.').split('_').sorted().joinToString("_")
|
||||||
|
}
|
||||||
|
val clashes = normalised.entries.groupBy { it.value }.filterValues { it.size > 1 }
|
||||||
|
if (clashes.isNotEmpty()) {
|
||||||
|
fail("codes differing only in word order: ${clashes.values.map { g -> g.map { it.key } }}")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun theDocumentAndTheRegistryAgree() {
|
||||||
|
val doc = findDoc() ?: run {
|
||||||
|
println("findings-registry.md not found from ${File(".").absolutePath} — skipping")
|
||||||
|
return
|
||||||
|
}
|
||||||
|
val text = doc.readText()
|
||||||
|
|
||||||
|
// Only table rows count as "documented". Prose may legitimately mention a code that no
|
||||||
|
// longer exists — the rules section explains why two were merged — and treating that as
|
||||||
|
// a registry entry would force the document to forget its own history.
|
||||||
|
val documented = text.lines()
|
||||||
|
.filter { it.trimStart().startsWith("|") }
|
||||||
|
.flatMap { row -> Regex("`([a-z0-9]+\\.[a-z0-9_]+)`").findAll(row).map { it.groupValues[1] } }
|
||||||
|
.toSet()
|
||||||
|
val registered = FindingRegistry.all.map { it.code }.toSet()
|
||||||
|
|
||||||
|
val missingFromDoc = registered - documented
|
||||||
|
val missingFromCode = documented - registered
|
||||||
|
assertTrue(
|
||||||
|
missingFromDoc.isEmpty(),
|
||||||
|
"these codes exist in FindingRegistry but not in docs/findings-registry.md: $missingFromDoc",
|
||||||
|
)
|
||||||
|
assertTrue(
|
||||||
|
missingFromCode.isEmpty(),
|
||||||
|
"docs/findings-registry.md documents codes that no longer exist: $missingFromCode",
|
||||||
|
)
|
||||||
|
|
||||||
|
// And the severities must match, or the document is describing a different system.
|
||||||
|
for (spec in FindingRegistry.all) {
|
||||||
|
val row = text.lines().firstOrNull {
|
||||||
|
it.trimStart().startsWith("|") && it.contains("`${spec.code}`")
|
||||||
|
} ?: continue
|
||||||
|
val severity = spec.severity.name.lowercase()
|
||||||
|
assertTrue(
|
||||||
|
row.contains("| $severity |"),
|
||||||
|
"${spec.code} is ${severity} in code but the doc row says otherwise: $row",
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Walks up from the test's working directory to find the repo's docs/ folder. */
|
||||||
|
private fun findDoc(): File? {
|
||||||
|
var dir: File? = File(".").absoluteFile
|
||||||
|
repeat(6) {
|
||||||
|
val candidate = File(dir, "docs/findings-registry.md")
|
||||||
|
if (candidate.isFile) return candidate
|
||||||
|
dir = dir?.parentFile
|
||||||
|
}
|
||||||
|
return null
|
||||||
|
}
|
||||||
|
}
|
||||||
+125
@@ -0,0 +1,125 @@
|
|||||||
|
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||||
|
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||||
|
|
||||||
|
package app.echo_lot.measurement
|
||||||
|
|
||||||
|
import kotlin.test.Test
|
||||||
|
import kotlin.test.assertEquals
|
||||||
|
import kotlin.test.assertFalse
|
||||||
|
import kotlin.test.assertTrue
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The fixtures here are a real device's routing table, transcribed from `dumpsys connectivity`
|
||||||
|
* on a OnePlus 15 with a Netbird tunnel up: wifi advertising a default route it cannot source
|
||||||
|
* from, cellular working properly, and a VPN carrying host routes to two destinations.
|
||||||
|
*
|
||||||
|
* Using a captured table rather than invented ones matters, because the bug this guards against
|
||||||
|
* is not "the boolean logic is wrong" — it is "the shapes I imagined are not the shapes real
|
||||||
|
* networks produce".
|
||||||
|
*/
|
||||||
|
class V6AnalysisTest {
|
||||||
|
|
||||||
|
private fun net(
|
||||||
|
id: String,
|
||||||
|
transport: Transport,
|
||||||
|
iface: String,
|
||||||
|
addrs: List<String>,
|
||||||
|
routes: List<Pair<String, String>>,
|
||||||
|
) = Network(
|
||||||
|
id = id,
|
||||||
|
transport = transport,
|
||||||
|
iface = iface,
|
||||||
|
link = Link(
|
||||||
|
addresses = addrs.map { Address(addr = it.substringBefore('/'), prefixLen = 64) },
|
||||||
|
routes = routes.map { (dst, dev) -> Route(dst = dst, iface = dev) },
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
/** wlan0: an IPv6 default route via a link-local gateway, but SLAAC produced no address. */
|
||||||
|
private val wifi = net(
|
||||||
|
"w", Transport.WIFI, "wlan0",
|
||||||
|
addrs = listOf("fe80::bcf6:edff:fe67:b139", "10.13.102.122"),
|
||||||
|
routes = listOf(
|
||||||
|
"fe80::/64" to "wlan0",
|
||||||
|
"::/0" to "wlan0",
|
||||||
|
"0.0.0.0/0" to "wlan0",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
/** rmnet_data1: a properly configured cellular link — global address and a default route. */
|
||||||
|
private val cellular = net(
|
||||||
|
"c", Transport.CELLULAR, "rmnet_data1",
|
||||||
|
addrs = listOf("2001:4bb8:46a:e724:289d:87ff:feb6:ebd3"),
|
||||||
|
routes = listOf("::/0" to "rmnet_data1", "2001:4bb8:46a:e724::/64" to "rmnet_data1"),
|
||||||
|
)
|
||||||
|
|
||||||
|
/** tun1: Netbird, with a ULA and host routes to exactly two destinations. */
|
||||||
|
private val vpn = net(
|
||||||
|
"v", Transport.VPN, "tun1",
|
||||||
|
addrs = listOf("100.64.158.131", "fdfd:c4fe:c4fe:c4fe:1f3c:98a0:dd66:ac7"),
|
||||||
|
routes = listOf(
|
||||||
|
"2001:1ad0:c4fe:6767::2/128" to "tun1",
|
||||||
|
"2001:1ad0:c4fe:a::136/128" to "tun1",
|
||||||
|
"fdfd:c4fe:c4fe:c4fe::/64" to "tun1",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `wifi advertising a route it cannot source from is reported`() {
|
||||||
|
val s = V6Analysis.classify(listOf(wifi)).single()
|
||||||
|
assertTrue(s.routeWithoutAddress, "::/0 with only a link-local address is the RA-without-SLAAC case")
|
||||||
|
assertFalse(s.addressWithoutRoute)
|
||||||
|
assertFalse(s.tunnel, "wifi is not a tunnel")
|
||||||
|
assertEquals("wlan0", s.iface)
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `a properly configured link produces no finding`() {
|
||||||
|
val s = V6Analysis.classify(listOf(cellular)).single()
|
||||||
|
assertFalse(s.routeWithoutAddress)
|
||||||
|
assertFalse(s.addressWithoutRoute)
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `a tunnel with host routes is deliberate, not broken`() {
|
||||||
|
val s = V6Analysis.classify(listOf(vpn)).single()
|
||||||
|
assertTrue(s.addressWithoutRoute, "a ULA and no ::/0 is an address with nothing to route it")
|
||||||
|
assertTrue(s.tunnel, "so it must be reported as information, not as a fault")
|
||||||
|
assertFalse(s.routeWithoutAddress)
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `each network is judged on its own`() {
|
||||||
|
// The whole point of per-network classification: "IPv6 is broken" is useless advice when
|
||||||
|
// wifi is the broken one and cellular is fine.
|
||||||
|
val shapes = V6Analysis.classify(listOf(wifi, cellular, vpn)).associateBy { it.iface }
|
||||||
|
assertTrue(shapes.getValue("wlan0").routeWithoutAddress)
|
||||||
|
assertFalse(shapes.getValue("rmnet_data1").routeWithoutAddress)
|
||||||
|
assertFalse(shapes.getValue("rmnet_data1").addressWithoutRoute)
|
||||||
|
assertTrue(shapes.getValue("tun1").addressWithoutRoute)
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `a link-local-only network with no v6 route says nothing either way`() {
|
||||||
|
// Plain IPv4-only wifi: no IPv6 offered at all. That is v6.not_offered's business, and
|
||||||
|
// reporting it here as well would double up on a network that is merely legacy, not broken.
|
||||||
|
val v4only = net(
|
||||||
|
"4", Transport.WIFI, "wlan0",
|
||||||
|
addrs = listOf("fe80::1", "192.168.1.5"),
|
||||||
|
routes = listOf("0.0.0.0/0" to "wlan0"),
|
||||||
|
)
|
||||||
|
val s = V6Analysis.classify(listOf(v4only)).single()
|
||||||
|
assertFalse(s.routeWithoutAddress)
|
||||||
|
assertFalse(s.addressWithoutRoute)
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `a zone index does not hide a link-local address`() {
|
||||||
|
val zoned = net(
|
||||||
|
"z", Transport.WIFI, "wlan0",
|
||||||
|
addrs = listOf("fe80::1%wlan0"),
|
||||||
|
routes = listOf("::/0" to "wlan0"),
|
||||||
|
)
|
||||||
|
assertTrue(V6Analysis.classify(listOf(zoned)).single().routeWithoutAddress)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -20,4 +20,8 @@ kotlin {
|
|||||||
}
|
}
|
||||||
java { sourceCompatibility = JavaVersion.VERSION_17; targetCompatibility = JavaVersion.VERSION_17 }
|
java { sourceCompatibility = JavaVersion.VERSION_17; targetCompatibility = JavaVersion.VERSION_17 }
|
||||||
|
|
||||||
tasks.test { useJUnitPlatform() }
|
tasks.test {
|
||||||
|
useJUnitPlatform()
|
||||||
|
// Opt-in: point this at a captured run to check the anonymizer against real data.
|
||||||
|
System.getenv("ECHOLOT_REAL_RUN")?.let { environment("ECHOLOT_REAL_RUN", it) }
|
||||||
|
}
|
||||||
|
|||||||
@@ -108,12 +108,22 @@ class Anonymizer(private val level: PrivacyLevel, private val salt: Salt) {
|
|||||||
is JsonObject -> walkObject(v, path)
|
is JsonObject -> walkObject(v, path)
|
||||||
is JsonArray -> JsonArray(v.map { walk(key, it, path) })
|
is JsonArray -> JsonArray(v.map { walk(key, it, path) })
|
||||||
is JsonPrimitive ->
|
is JsonPrimitive ->
|
||||||
if (v.isString) transform(Classification.typeOf(key, path), v.content).let(::JsonPrimitive)
|
if (v.isString) {
|
||||||
else v
|
// Name first (it is precise), then shape (it is exhaustive). A field nobody
|
||||||
|
// classified must not be a field that leaks.
|
||||||
|
val type = Classification.typeOf(key, path) ?: Classification.inferFromValue(v.content)
|
||||||
|
JsonPrimitive(transform(type, v.content))
|
||||||
|
} else {
|
||||||
|
v
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
private fun transform(type: LogicalType?, value: String): String = when (type) {
|
private fun transform(type: LogicalType?, value: String): String = when (type) {
|
||||||
null -> value
|
// Unclassified strings still get their *embedded* identifiers scrubbed. A whole-value
|
||||||
|
// check cannot see them: raw shell output is one long string that is neither a MAC nor an
|
||||||
|
// address, so it sailed through both the name table and the shape check carrying every
|
||||||
|
// MAC on the user's LAN.
|
||||||
|
null -> scrubEmbedded(value)
|
||||||
LogicalType.SSID -> pseudo("ssid", value) { "net-" + it.take(6) }
|
LogicalType.SSID -> pseudo("ssid", value) { "net-" + it.take(6) }
|
||||||
LogicalType.MAC, LogicalType.BSSID -> macPreservingOui(value)
|
LogicalType.MAC, LogicalType.BSSID -> macPreservingOui(value)
|
||||||
LogicalType.IP4 -> ip4(value)
|
LogicalType.IP4 -> ip4(value)
|
||||||
@@ -123,6 +133,38 @@ class Anonymizer(private val level: PrivacyLevel, private val salt: Salt) {
|
|||||||
LogicalType.FREETEXT -> "[removed: may contain identifying text]"
|
LogicalType.FREETEXT -> "[removed: may contain identifying text]"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Replaces addresses and MACs found *inside* a longer string.
|
||||||
|
*
|
||||||
|
* Shizuku probes embed raw command output verbatim — `ip neigh`, `ip route`, `dumpsys` — which
|
||||||
|
* is genuinely valuable evidence and also a complete inventory of every device on the user's
|
||||||
|
* network, with hardware addresses. measurement-schema.md §9 flagged these as "hard to
|
||||||
|
* anonymize" and proposed dropping them from exports.
|
||||||
|
*
|
||||||
|
* Scrubbing beats dropping: the output stays readable and auditable — you can still see the
|
||||||
|
* shape of the neighbour table and how many hosts there were — while the identifiers become
|
||||||
|
* the same pseudonyms used everywhere else in the document. So a MAC appearing both in a
|
||||||
|
* parsed field and in a raw dump still reads as one device.
|
||||||
|
*
|
||||||
|
* Only addresses and MACs are touched, for the same reason as [Classification.inferFromValue]:
|
||||||
|
* they are the patterns that cannot be mistaken for something else in free text.
|
||||||
|
*/
|
||||||
|
private fun scrubEmbedded(value: String): String {
|
||||||
|
// Cheap bail-out: the overwhelming majority of strings are short and contain neither.
|
||||||
|
if (value.length < 7 || (!value.contains(':') && !value.contains('.'))) return value
|
||||||
|
// One pass, not three. Sequential passes re-process their own output: after a MAC became
|
||||||
|
// 78:9a:18:xx:yy:zz the IPv6 pattern matched it — six hex groups separated by colons is
|
||||||
|
// exactly an address — and mangled the vendor prefix that the MAC rule had just taken
|
||||||
|
// care to preserve. Ordered alternation resolves each position once, MAC first.
|
||||||
|
return EMBEDDED.replace(value) { m ->
|
||||||
|
when {
|
||||||
|
m.groups[1] != null -> macPreservingOui(m.value)
|
||||||
|
m.groups[2] != null -> ip6(m.value)
|
||||||
|
else -> ip4(m.value)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ---- per-type transforms -------------------------------------------------------------
|
// ---- per-type transforms -------------------------------------------------------------
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -147,6 +189,11 @@ class Anonymizer(private val level: PrivacyLevel, private val salt: Salt) {
|
|||||||
* Public addresses keep only their /16 so the network is still locatable at ISP granularity.
|
* Public addresses keep only their /16 so the network is still locatable at ISP granularity.
|
||||||
*/
|
*/
|
||||||
private fun ip4(value: String): String {
|
private fun ip4(value: String): String {
|
||||||
|
// A route destination carries a prefix length; pseudonymize the address and put it back,
|
||||||
|
// or "0.0.0.0/0" turns into nonsense and the routing table becomes unreadable.
|
||||||
|
value.substringAfter('/', "").takeIf { it.isNotEmpty() && value.contains('/') }?.let { len ->
|
||||||
|
return ip4(value.substringBefore('/')) + "/" + len
|
||||||
|
}
|
||||||
val o = value.split(".")
|
val o = value.split(".")
|
||||||
if (o.size != 4 || o.any { it.toIntOrNull() == null }) return value
|
if (o.size != 4 || o.any { it.toIntOrNull() == null }) return value
|
||||||
val n = o.map { it.toInt() }
|
val n = o.map { it.toInt() }
|
||||||
@@ -167,8 +214,36 @@ class Anonymizer(private val level: PrivacyLevel, private val salt: Salt) {
|
|||||||
* is a device fingerprint, especially with EUI-64.
|
* is a device fingerprint, especially with EUI-64.
|
||||||
*/
|
*/
|
||||||
private fun ip6(value: String): String {
|
private fun ip6(value: String): String {
|
||||||
|
// Dotted quads reach here through the family-agnostic field names (addr, gateway, dst);
|
||||||
|
// hand them to the IPv4 path rather than mangling them as if they were v6.
|
||||||
|
if (value.count { it == ':' } < 2) return ip4(value)
|
||||||
|
if (value.contains('/')) {
|
||||||
|
return ip6(value.substringBefore('/')) + "/" + value.substringAfter('/')
|
||||||
|
}
|
||||||
val v = value.lowercase(Locale.ROOT)
|
val v = value.lowercase(Locale.ROOT)
|
||||||
|
// The unspecified address and the default route are not identities; mangling them would
|
||||||
|
// make a routing table unreadable for no privacy gain.
|
||||||
if (v == "::1" || v == "::" || v.startsWith("fe80:") || v.startsWith("ff")) return v
|
if (v == "::1" || v == "::" || v.startsWith("fe80:") || v.startsWith("ff")) return v
|
||||||
|
|
||||||
|
// Unique local addresses (fc00::/7) need the *whole* prefix replaced, not the tail.
|
||||||
|
//
|
||||||
|
// They look like the v6 equivalent of RFC1918, and the first instinct is to keep them for
|
||||||
|
// the same reason: private, topological, says nothing about anyone. That reasoning does
|
||||||
|
// not carry over. An RFC1918 prefix is shared by millions of networks and identifies
|
||||||
|
// none of them; a ULA global ID is 40 *random* bits, unique to one network by
|
||||||
|
// construction (RFC 4193). It is a network fingerprint. Passing the leading groups
|
||||||
|
// through - which is what the general path does - leaked 32 of those 40 bits.
|
||||||
|
//
|
||||||
|
// The prefix is pseudonymized as a unit, so two addresses on the same ULA subnet still
|
||||||
|
// land on the same pseudonymous prefix. "These hosts are on one network" survives;
|
||||||
|
// "this is *that* network" does not.
|
||||||
|
if (v.startsWith("fc") || v.startsWith("fd")) {
|
||||||
|
val groups = v.substringBefore('%').split(":")
|
||||||
|
val prefix = pseudo("ula-prefix", groups.take(3).joinToString(":")) { it }
|
||||||
|
val host = pseudo("ula-host", v) { it }
|
||||||
|
return "fd${prefix.substring(0, 2)}:${prefix.substring(2, 6)}:${prefix.substring(6, 10)}" +
|
||||||
|
"::${host.substring(0, 4)}"
|
||||||
|
}
|
||||||
val groups = v.substringBefore('%').split(":")
|
val groups = v.substringBefore('%').split(":")
|
||||||
if (groups.size < 3) return v
|
if (groups.size < 3) return v
|
||||||
val h = pseudo("ip6", value) { it }
|
val h = pseudo("ip6", value) { it }
|
||||||
@@ -219,7 +294,7 @@ class Anonymizer(private val level: PrivacyLevel, private val salt: Salt) {
|
|||||||
|
|
||||||
/** Deterministic per (domain, value, salt); memoized so one value maps to one pseudonym. */
|
/** Deterministic per (domain, value, salt); memoized so one value maps to one pseudonym. */
|
||||||
private fun pseudo(domain: String, value: String, shape: (String) -> String): String =
|
private fun pseudo(domain: String, value: String, shape: (String) -> String): String =
|
||||||
cache.getOrPut("$domain | |||||||