Compare commits
45
Commits
@@ -10,12 +10,18 @@
|
||||
# releases don't trigger each other's pipelines.
|
||||
#
|
||||
# Required secrets:
|
||||
# REGISTRY_TOKEN personal access token with read+write package scope —
|
||||
# the built-in Actions token is NOT accepted by the
|
||||
# container registry (docker login → unauthorized).
|
||||
# Create: user Settings → Applications → Generate token.
|
||||
# REGISTRY_USER optional; defaults to the pushing actor's username.
|
||||
# The release job needs only the built-in GITHUB_TOKEN.
|
||||
# REGISTRY_TOKEN personal access token with read+write package scope —
|
||||
# the built-in Actions token is NOT accepted by the
|
||||
# container registry (docker login → unauthorized).
|
||||
# Create: user Settings → Applications → Generate token.
|
||||
# REGISTRY_USER optional; defaults to the pushing actor's username.
|
||||
# RELEASE_SIGNING_KEY base64 ed25519 seed that signs SHA256SUMS. Self-updating
|
||||
# servers verify the signature against the public key baked
|
||||
# into the binary (selfupdate.DefaultPublicKeyB64) and REFUSE
|
||||
# unsigned releases, so this job hard-fails without it —
|
||||
# a release nobody can install is better failed loudly here.
|
||||
# Mint a pair with: go run ./cmd/release-sign -gen
|
||||
# The release job otherwise needs only the built-in GITHUB_TOKEN.
|
||||
|
||||
name: server-release
|
||||
on:
|
||||
@@ -44,6 +50,18 @@ jobs:
|
||||
done
|
||||
(cd ../dist && sha256sum * > SHA256SUMS)
|
||||
|
||||
- name: Sign SHA256SUMS
|
||||
working-directory: server
|
||||
env:
|
||||
RELEASE_SIGNING_KEY: ${{ secrets.RELEASE_SIGNING_KEY }}
|
||||
run: |
|
||||
[ -n "$RELEASE_SIGNING_KEY" ] || { echo "::error::secret RELEASE_SIGNING_KEY is missing — self-updating servers refuse unsigned releases, so publishing one would strand the fleet. Add it under Settings → Actions → Secrets."; exit 1; }
|
||||
go run ./cmd/release-sign ../dist/SHA256SUMS
|
||||
# Verify with the key baked into the binary we just built — catches a
|
||||
# secret that does not match DefaultPublicKeyB64 before it ships.
|
||||
PUB=$(grep -o 'DefaultPublicKeyB64 = "[^"]*"' internal/selfupdate/selfupdate.go | cut -d'"' -f2)
|
||||
go run ./cmd/release-sign -verify -pub "$PUB" ../dist/SHA256SUMS
|
||||
|
||||
- name: Create release + attach binaries
|
||||
env:
|
||||
TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
@@ -145,6 +145,17 @@ First build downloads AGP/Compose/Shizuku from Google Maven + Maven Central.
|
||||
Shizuku, **toggle Wireless debugging off/on** — Shizuku keeps running (separate process), a fresh
|
||||
port + mDNS record appear, and the beacon/connector recover. Plan the Shizuku-tier dev loop
|
||||
around this (or USB, if ever available).
|
||||
- **A poisoned Gradle *build cache* entry can silently drop a whole module from the APK.**
|
||||
Symptom: the app dies with `ClassNotFoundException` for a class that plainly exists, while the
|
||||
build is green and `./gradlew :app:dependencies` lists the module on `debugRuntimeClasspath`.
|
||||
The module's own jar is correct; its code simply never reaches AGP's intermediates. `clean`,
|
||||
`rm -rf */build` and `--rerun-tasks` all fail to fix it, because **none of them touch the build
|
||||
cache** — look for `compileKotlin FROM-CACHE` in the log. Fix: rebuild with `--no-build-cache`.
|
||||
Verify by grepping the APK's dex for a string literal that only that module defines; grepping for
|
||||
a *class name* proves nothing, because callers carry the name as a reference whether or not the
|
||||
class is packaged:
|
||||
`unzip -o -q app-debug.apk "classes*.dex" && grep -a "pin-sha256:" *.dex`
|
||||
Suspect this whenever a runtime failure contradicts a successful build.
|
||||
- **Empty-jar race with the IDE.** VSCodium's Java/Kotlin extension runs its own Gradle daemon on
|
||||
the same project; when it overlaps a CLI build, a module's `build/libs/*.jar` can end up
|
||||
containing only a manifest, and Gradle then considers `jar` up-to-date. Dependent modules fail
|
||||
|
||||
+368
-3
@@ -211,8 +211,8 @@ Two collection-loop gotchas found while driving the phone over USB:
|
||||
2. ~~If `trace.errqueue_reachable` = PARTIAL, add a C-over-JNI errqueue shim.~~ **Retired** —
|
||||
SUPPORTED on both known devices; `traceroute.udp4` reads real hops via `Os.recvmsg` +
|
||||
`StructMsghdr` reflection, so no `:native` module is needed.
|
||||
3. Start the Go server skeleton (enrollment + profile + sessions + UDP echo with observation
|
||||
blocks + canary-DNS reference records) per probe-protocol.md.
|
||||
3. ~~Start the Go server skeleton per probe-protocol.md.~~ **Shipped** — live on fmr since
|
||||
v0.2.0 (2026-07-31); see the server sections below.
|
||||
4. Fold confirmed capabilities into the production `core-probe` / `core-shizuku` modules.
|
||||
|
||||
## Production probe server — LIVE on dedicated VM "fmr" (2026-07-31)
|
||||
@@ -222,7 +222,7 @@ SSH only — verified untouched by the daemon (explicit multi-address binds, no
|
||||
Control: fmr-1:8443 (SPKI pin `zRV9qkiLnRexAeh4RrSfJzbPWO+U/2Oj2/NVM/KfXlg=`, verified
|
||||
externally over v4+v6). UDP data plane on all four service addresses :8442 — the second IP is
|
||||
the stun-5780 substrate. Daily randomized self-update timer installed (checksum-verified
|
||||
against SHA256SUMS; signature verification still TODO before treating the source as untrusted).
|
||||
against SHA256SUMS; signature verification landed 2026-08-02 — see "Release signing" below).
|
||||
Host config in `/etc/echolot-server.env`. SSH access for sessions: `ssh claude-echolot`.
|
||||
|
||||
## Server v0.3.0 — STUN + TCP echo + observations + actions (2026-07-31)
|
||||
@@ -1002,3 +1002,368 @@ Also fixed: the Settings *Preview what an upload would send* button did nothing.
|
||||
`UiState.history`, which is empty until the History screen has been opened — the same root cause as
|
||||
the "0 run(s)" count. It now reads the archive directly, and says so when there is nothing to
|
||||
preview rather than silently ignoring the tap.
|
||||
|
||||
### Security: the admin listener was publicly exposed for ~15 minutes (2026-08-01)
|
||||
Moving the admin listener to `[::2]:443` for the UI exposed `/admin/enroll-tokens` and
|
||||
`/admin/selftest` to the internet **with no authentication**. Anyone who could reach
|
||||
`fmr.echo-lot.app` could mint enrolment tokens.
|
||||
|
||||
The listener was designed localhost-only — its own flag help says *"keep localhost"* — and that
|
||||
assumption travelled with it when the address changed. The compounding error: `checkAdminExposure`,
|
||||
added the same day, verifies **encryption** and says nothing about **authentication**. It passed,
|
||||
and a green light on an adjacent property is worse than no check, because it invites you to stop
|
||||
looking.
|
||||
|
||||
Closed by returning to loopback (the TLS and ACME work is retained, just not exposed). All 68 device
|
||||
enrolments matched the timestamps of test runs, so there is no evidence of abuse — but the window
|
||||
existed on a freshly published hostname and absence cannot be proven. 39 unused enrolment tokens
|
||||
were purged, since any could have been minted by someone else and they cost nothing to replace, and
|
||||
63 test devices removed.
|
||||
|
||||
**The admin listener does not become reachable again until it authenticates.** That reorders the UI
|
||||
work: auth on the listener first, everything else after.
|
||||
|
||||
### Open: encrypted uploads, where the operator cannot read the data
|
||||
Not built. Recorded because the shape is decided by a few early choices, and the current design
|
||||
happens to leave the door open.
|
||||
|
||||
The goal: hand someone an account, let them upload, and be unable to read what they uploaded.
|
||||
|
||||
Sketch: a random per-account **master key**, generated on the first device and wrapped under a
|
||||
key derived from a passphrase (PBKDF2-HMAC-SHA256 — stdlib on both sides). The wrapped key is
|
||||
stored server-side as an opaque blob, so a new device signs in, fetches it, and unwraps locally;
|
||||
the server never sees either key. Runs are encrypted client-side with AES-256-GCM, fresh nonce per
|
||||
run. All of this is stdlib in Go and `javax.crypto` in Kotlin — no dependency either side.
|
||||
|
||||
Four consequences that decide whether it is worth it:
|
||||
|
||||
1. **What stays readable determines what the UI can do.** The server builds its index by *parsing*
|
||||
the document — verdict, finding count, started_at. An opaque payload means the client supplies
|
||||
that metadata or the index disappears, and with it retention-by-verdict and any "runs with
|
||||
findings" view. The honest version supplies only run id, timestamp and size, and moves the rest
|
||||
client-side.
|
||||
2. **Lose the passphrase, lose the data.** That is the feature working, and also the support
|
||||
burden. It needs a recovery code printed at setup, not a reset flow — there is nothing to reset.
|
||||
3. **Metadata is not hidden.** The operator still sees which account uploaded, when, how often and
|
||||
how large. "Cannot see it" is about content, not existence, and saying otherwise would oversell.
|
||||
4. **It makes `min_anonymization` unenforceable** — a server cannot check a level it cannot read.
|
||||
That is not a conflict so much as a redundancy: the anonymization floor exists to protect the
|
||||
user from the operator, and encryption does that better. The two should not both be demanded of
|
||||
one upload.
|
||||
|
||||
What keeps this possible: uploads are already stored byte-for-byte as received, and every index
|
||||
field is derived in one function (`runs.Put`). The thing to avoid is admin features that *require*
|
||||
reading content — those would have to be unbuilt later.
|
||||
|
||||
### App sign-in, and an undisclosed dependency it surfaced (2026-08-01)
|
||||
The app can now sign in to the server's identity provider: authorization code with PKCE, a
|
||||
`Sign in` card in settings, and the `echolot://auth` redirect handled alongside the enrolment one
|
||||
(told apart by host, since one spends a token and the other completes an authorization).
|
||||
|
||||
The detail that decides whether this works on a real phone: **the PKCE verifier is written to
|
||||
storage before the browser opens**, not held in memory. Handing control to a browser backgrounds
|
||||
the process and Android may kill it; the callback then arrives at a fresh process. An in-memory
|
||||
verifier works on a developer's device and fails under memory pressure, which is the worst way for
|
||||
a sign-in to break.
|
||||
|
||||
Nothing from the IdP is retained. The ID token proves who is signing in, once, and the device
|
||||
credential authenticates everything after — no access tokens stored, no refresh tokens rotated.
|
||||
|
||||
**A server remains entirely optional.** All eight probes are device-tier; `serverConfigured` gates
|
||||
only upload and the account. But answering that question exposed something worth fixing: two probes
|
||||
hardcode the reference deployment —
|
||||
|
||||
```kotlin
|
||||
DnsCanaryProbe(canaryZone = "c.echo-lot.app", ...) // "Hardcoded to the reference deployment"
|
||||
StunProbe(serverHost = "fmr-1.echo-lot.app")
|
||||
```
|
||||
|
||||
so a user with no server of their own still sends DNS and STUN traffic to fmr without being told.
|
||||
For a tool that goes to this much trouble over what leaves the device, an undisclosed dependency on
|
||||
a third party's infrastructure is the wrong default. It should prefer the configured server, and be
|
||||
explicit when there is none. **Closed** — both probes take the enrolled server from settings
|
||||
(canary zone learned from the profile, cleared on re-enroll) and report themselves SKIPPED with
|
||||
the reason when none is configured.
|
||||
|
||||
### v6.broken was a false positive waiting to happen (2026-08-01)
|
||||
A phone could not open `https://fmr.echo-lot.app` while loading the same server by IP literal
|
||||
perfectly well. Two things came out of chasing it.
|
||||
|
||||
**The admin UI is IPv6-only, by consequence rather than intent.** `fmr.echo-lot.app` has an AAAA
|
||||
and no A record — verified identical at Cloudflare, Google and Quad9, so DNS itself is healthy.
|
||||
That follows from reserving all four measurement addresses for testing, which left only `::2` for
|
||||
management, and `::2` has no IPv4 counterpart. Any client without working IPv6 sees an unreachable
|
||||
admin interface — a poor property for the interface you reach *from the networks you are debugging*.
|
||||
|
||||
**And the app's own `v6.broken` finding was unsound.** It fired on exactly one signal — ICMPv6 echo
|
||||
getting no reply — with `Confidence.HIGH`. ICMPv6 echo is widely filtered on networks where IPv6
|
||||
works fine, which is precisely what that phone demonstrated: no ICMPv6 replies, working IPv6 TCP.
|
||||
The finding asserted a cause it had no evidence for, which is the same class of error as the
|
||||
multi-homed `100 % downstream loss` earlier: a confident measurement of something that was not
|
||||
happening.
|
||||
|
||||
Now `v6.no_icmp_reply`, severity low, confidence medium, and the text names *both* explanations
|
||||
instead of choosing one. It is still worth reporting, because filtered ICMPv6 breaks Path MTU
|
||||
Discovery — large packets vanish rather than being reported as too big — which is a real fault even
|
||||
when IPv6 works.
|
||||
|
||||
The proper fix is corroboration: attempt a real IPv6 connection and only call it broken when that
|
||||
fails too. That needs a target, which runs into the hardcoded-reference-deployment issue already
|
||||
open above. **Both closed 2026-08-02** — see "Corroborated IPv6 findings" below.
|
||||
|
||||
## Per-network probing is blocked while a VPN is up (2026-08-01)
|
||||
|
||||
`Network.bindSocket()` fails with `EPERM` for every underlying network when a VPN holds the
|
||||
default route — verified on the OnePlus 15 with Netbird active: `Binding socket to network 101
|
||||
failed: EPERM` for both cellular and wifi. This is Android preventing VPN leaks, not a bug to work
|
||||
around, and it means the whole per-network measurement approach is unavailable to any user with a
|
||||
VPN connected. Worth deciding deliberately rather than discovering per report:
|
||||
|
||||
- The run currently succeeds and simply measures nothing per network. Honest, but silent — the
|
||||
document records `attempted: false` and the UI says green.
|
||||
- A user with a corporate VPN permanently on would get a green run that measured almost nothing.
|
||||
|
||||
Options are to detect the VPN and say so plainly ("this network cannot be measured while a VPN is
|
||||
active"), to measure the tunnel itself as the network under test, or both. **Decided and built
|
||||
2026-08-02**: say so plainly, everywhere the run is read — see "Constrained runs" below.
|
||||
|
||||
Related: `icmp.ping6` now records `attempted` alongside `ok` per network, because collapsing them
|
||||
made the app report "IPv6 is configured, but ICMPv6 gets no reply" about an interface it had never
|
||||
succeeded in sending on — a claim about the user's carrier with no evidence behind it.
|
||||
|
||||
## Reserved measurement addresses, and the web UI on both families (2026-08-01)
|
||||
|
||||
fmr has two IPv4 (.150/.151) and three IPv6 (::150/::151/::2) addresses. `.150`/`::150` now carry
|
||||
the services; `.151`/`::151` are reserved for measurement, declared in `ECHOLOT_RESERVED_ADDRS`.
|
||||
|
||||
Reserved does **not** mean silent. The UDP data plane, the canary DNS and STUN's RFC 5780 alternate
|
||||
all belong there — reserving an address and then forbidding the measurements that need it would
|
||||
defeat the purpose. What must never appear is a service, and above all not ports 80 or 443: a
|
||||
handshake completing on a port known not to be listening is what proves interception, and that
|
||||
proof survives exactly as long as nothing binds those ports. `config.CheckReserved` enforces it at
|
||||
startup, refusing wildcard binds outright (every listener defaults to `:port`, so the next one added
|
||||
will claim reserved addresses without anyone deciding to).
|
||||
|
||||
The first version of the guard was too strict and the live config caught it: it would have refused
|
||||
the existing UDP and DNS binds on `.151`. The rule is about services and web ports, not about
|
||||
listening at all.
|
||||
|
||||
**The adb-beacon receiver was wildcard-bound to `0.0.0.0:443`**, occupying port 443 on every IPv4
|
||||
address including the reserved one — so the IPv4 interception test had been compromised for as long
|
||||
as it had been running, silently. It is now `systemctl disable --now echolot-adb-beacon`; restore
|
||||
with `systemctl enable --now`. Note what this implies: the guard covers this server's own listeners,
|
||||
and a stray process outside its config can still pollute a reserved address. A startup probe that
|
||||
*verifies* 80/443 are actually free on the reserved addresses would be a stronger guarantee than
|
||||
checking our own configuration — built 2026-08-02 (`selftest.ReservedWebPortsFree`, fatal at
|
||||
startup when anything is listening there).
|
||||
|
||||
The admin UI and the ACME responder now take comma-separated addresses like every other listener;
|
||||
they were single-address, which is why the UI could only ever live on `::2`. It serves on `.150:443` and
|
||||
`[::150]:443`; sshd on `.150:2322` and `[::150]:2322`.
|
||||
|
||||
`::2` is gone entirely — unbound, then removed from `/etc/systemd/network/ext.network`. The
|
||||
transition kept it bound throughout and dropped it only after the CNAME landed, because removing it
|
||||
first would have broken both the UI and ACME renewal for the very name the certificate is issued
|
||||
to. Listeners came off before the address did, in that order, or the services would have failed to
|
||||
bind on restart.
|
||||
|
||||
Verified after a full reboot: `fmr.echo-lot.app` answers 200 over both families, `.151`/`::151` are
|
||||
closed on 80 and 443, canary DNS is still up on `.151`, and neither `::2` nor the beacon returns.
|
||||
(`echolot-server` is `After=network-online.target` with `Restart=on-failure`, which is what makes
|
||||
binding specific addresses safe across a boot — a wildcard bind would not have needed it, and that
|
||||
is the trade for the reserved addresses being meaningful.)
|
||||
|
||||
The point of all this: `fmr.echo-lot.app` gained an A record, so the server stopped being reachable
|
||||
only over IPv6 — which is what made it unreachable from a phone with no working IPv6, presenting as
|
||||
"this host does not exist" in two different browsers.
|
||||
|
||||
### If the beacon comes back, it belongs in the web UI
|
||||
|
||||
Not as a separate listener. The receiver being its own Python service on `0.0.0.0:443` is exactly
|
||||
what silently compromised the reserved address, and a second process racing for a port is a
|
||||
recurring problem rather than a one-off: whoever loses the race simply fails to start, and on a
|
||||
reboot which one that is comes down to unit ordering.
|
||||
|
||||
Folding it in costs little and settles several things at once. It would be two routes on the admin
|
||||
UI (`POST` the observed adb port, `GET /apk` for the staged build), behind the TLS the UI already
|
||||
terminates and the certificate it already renews, with no extra port and no wildcard. It also gets
|
||||
authentication for free — the current receiver accepts a port report from anyone who can reach it,
|
||||
which is tolerable for a dev tool on a trusted network and not something to keep once it lives
|
||||
beside an admin session.
|
||||
|
||||
The one thing that changes on the device side is that the POST becomes HTTPS. That is a real
|
||||
certificate rather than a self-signed one, so it costs a URL scheme rather than any trust plumbing.
|
||||
|
||||
## The control plane shares port 443 (2026-08-01)
|
||||
|
||||
`fmr-1.echo-lot.app:443` is the control plane, `fmr.echo-lot.app:443` the admin UI, both on
|
||||
`.150`/`::150`, one listener, selected by SNI for the certificate and by `Host` for the handler.
|
||||
|
||||
The reason is not tidiness, it is reachability. Captive portals, hotel wifi and corporate firewalls
|
||||
routinely permit only 80 and 443 — which is exactly the population of networks this tool exists to
|
||||
diagnose. A control plane on 8443 is unreachable precisely when it matters most, and it fails as
|
||||
"cannot reach server", which tells the user nothing.
|
||||
|
||||
They cannot share a certificate, which is why this needs two names. The control plane is trusted by
|
||||
SPKI pin and so uses a long-lived self-signed certificate; a browser needs one a CA vouches for.
|
||||
One name on one port is one certificate, so the port can only be shared by splitting the names.
|
||||
Pinning the Let's Encrypt key instead was considered and rejected: it survives renewal only while
|
||||
key reuse holds, so a routine key rotation would brick the whole fleet.
|
||||
|
||||
Verified per SNI on 443: `fmr.echo-lot.app` serves `issuer=Let's Encrypt`, `fmr-1.echo-lot.app`
|
||||
serves the self-signed cert whose pin is unchanged (`zRV9…Xlg=`), `/v1/profile` answers 401 on the
|
||||
control name and 303 to the login page on the UI name.
|
||||
|
||||
**8443 stays open.** Devices enrolled before this carry that URL in their settings, and closing it
|
||||
for the sake of a port number would strand every one of them. It can go once no enrolled device
|
||||
still points at it — not before.
|
||||
|
||||
The rule from the naming change still binds: `fmr` may be a CNAME to exactly one host and never a
|
||||
multi-address record, because a pinned client that reaches a different key does not fail over.
|
||||
|
||||
## Constrained runs: a VPN'd run now says so, everywhere (2026-08-02, app 0.2.1)
|
||||
|
||||
The measurement schema gained a top-level `constraints` block (§3) and the app now fills it.
|
||||
`ConstraintDetector` (core-probe) runs before any probe: one throwaway `Network.bindSocket()` per
|
||||
non-VPN network, plus a transport check for an active VPN. The result lands in three places, and
|
||||
all three are deliberate:
|
||||
|
||||
- **`run.constraints`** — for machines. A server aggregating thousands of runs can now separate
|
||||
"measured a healthy network" from "measured almost nothing through a tunnel"; the shapes were
|
||||
identical before.
|
||||
- **A `measurement.vpn_constrained` finding** — for the person reading this run, naming the
|
||||
interfaces that went unmeasured. A constrained run with a quiet findings list still reads as
|
||||
"nothing wrong here".
|
||||
- **The §7.3 verdict** — `Verdicts.derive` takes the constraints and returns INCONCLUSIVE
|
||||
outright for a per-network-blocked run, whatever the category lights say; the run screen shows
|
||||
an amber "Measured through a VPN" banner above the verdict so INCONCLUSIVE reads as the OS
|
||||
refusing, not the app failing.
|
||||
|
||||
Detection is one bind per network rather than parsing per-test `attempted:false` breadcrumbs, so
|
||||
it cannot drift when probe evidence formats change.
|
||||
|
||||
## Corroborated IPv6 findings: v6.broken is back, with evidence (2026-08-02, app 0.2.1)
|
||||
|
||||
The new `V6ConnectProbe` (test type `v6.brokenness`) attempts a real TCP connection over IPv6 to
|
||||
the configured server's :443, per network that *claims* IPv6 (global address or v6 default
|
||||
route) — IPv4-only networks are not attempted, since their failure is by design and would
|
||||
manufacture the exact false positive this exists to kill. The finding derivation is now three-way:
|
||||
|
||||
- ICMPv6 silent, TCP works → `v6.no_icmp_reply` at **high** confidence, retitled "ICMPv6 is
|
||||
filtered here — IPv6 itself works" (still reported: filtered ICMPv6 breaks PMTUD).
|
||||
- ICMPv6 silent, TCP fails too → **`v6.broken`** (high severity, reinstated in the registry +
|
||||
findings-registry.md): two independent transports silent on a network advertising IPv6.
|
||||
- No corroboration (no server configured, or the connect never got as far as sending) → the
|
||||
two-explanation `v6.no_icmp_reply` at medium confidence, unchanged.
|
||||
|
||||
Like STUN and the canary, the probe SKIPs honestly when no server is configured — corroboration
|
||||
is a benefit of enrollment, not a reason to borrow fmr.
|
||||
|
||||
## Server: reserved 80/443 verified against the OS, and signed releases (2026-08-02)
|
||||
|
||||
**Reserved-address startup probe.** `serve()` now proves 80/443 are actually free on every
|
||||
`ECHOLOT_RESERVED_ADDRS` address before starting: a throwaway bind per port
|
||||
(`selftest.ReservedWebPortsFree`), fatal on EADDRINUSE with the offending address named — the
|
||||
check `CheckReserved` cannot do, because a stray process outside our config (the adb-beacon
|
||||
receiver on `0.0.0.0:443` was exactly that) is invisible to configuration checks. Bind errors
|
||||
that are not "in use" (typo'd address, address not on this host) warn instead of refusing —
|
||||
they are config problems, not pollution.
|
||||
|
||||
**Release signing.** Self-update now trusts a signature, not a host. CI signs `SHA256SUMS` with
|
||||
an ed25519 key (`relsign` package, `cmd/release-sign`) and the updater refuses any release whose
|
||||
`SHA256SUMS.sig` is missing or does not verify against the public key baked into the binary
|
||||
(`selfupdate.DefaultPublicKeyB64`; operators with their own pipeline override via
|
||||
`ECHOLOT_SELF_UPDATE_PUBKEY`). The private key exists in exactly two places: the Gitea Actions
|
||||
secret `RELEASE_SIGNING_KEY`, and the offline original on the dev PC at
|
||||
`~/.echolot/release-signing-key`. It is deliberately NOT on fmr and NOT in the repo — a
|
||||
compromised release host can withhold updates but no longer inject one. CI hard-fails when the
|
||||
secret is missing (an unsigned release would strand every verifying server) and cross-checks the
|
||||
signature against the key in the source it just built.
|
||||
|
||||
**ACTION REQUIRED before the next `server-v*` tag:** add the Gitea repo secret
|
||||
`RELEASE_SIGNING_KEY` (Settings → Actions → Secrets) with the contents of
|
||||
`~/.echolot/release-signing-key` from the dev PC. Ordering is safe: the currently deployed
|
||||
v0.3.x updater does not verify, so it will happily install the first signed release; every
|
||||
release after that is verified. **Done 2026-08-02** — the secret is in place. (The "v0.3.x"
|
||||
above should read "the currently deployed release": deployments had moved on to v0.9.x by the
|
||||
time signing landed; the point — the deployed updater predates verification and will accept the
|
||||
first signed release — is unchanged.)
|
||||
|
||||
## Prober fold: traceroute.udp4 and the mDNS inventory go production (2026-08-02, app 0.2.2)
|
||||
|
||||
The two highest-value validated capabilities moved from the prober into `core-probe`:
|
||||
|
||||
- **`traceroute.udp4`** (`TracerouteProbe`): UDP traceroute reading ICMP time-exceeded off the
|
||||
socket error queue via `Os.recvmsg(MSG_ERRQUEUE)` through the reflection facade — no root, no
|
||||
raw socket, no JNI, ~250 ms for six hops. Emits the schema's `TracerouteEvidence` (rtt in ns).
|
||||
`OsAbi` came with it, including the measured fact that `Os.getsockoptInt` exists on neither
|
||||
known device, so PMTU must always be read from the errqueue (`ee_info`), never
|
||||
`getsockopt(IP_MTU)`. The load-bearing line survived the port: EAGAIN out of the reflected
|
||||
`recvmsg` means "queue empty", not failure.
|
||||
- **`local.mdns_inventory`** (`MdnsInventoryProbe`): MulticastLock + NSD discovery, the service
|
||||
inventory that doubles as the VLAN-leakage detector. Both hardware lessons kept: the
|
||||
`_services._dns-sd._udp.` meta-query returns 0 beside live services on both devices (so the
|
||||
concrete types are the measurement and the meta-query result is itself evidence), and the
|
||||
listen window is 10 s because 4 s missed services.
|
||||
|
||||
Still to fold, in order: the Shizuku dump *parsers* (the raw `link.ip_monitor` captures already
|
||||
hold two divergent vendor formats that could feed `link.ra_source` and `sec.arp_watch`);
|
||||
`multinetwork.request_and_bind` (extend ConstraintDetector to *request* transports rather than
|
||||
only probing present ones); `peer.ble_advertise` (needs three new permissions and a peer mode to
|
||||
exist first).
|
||||
|
||||
## Server v0.9.2: trains, real TTL/DSCP/ECN, rate limits (2026-08-02)
|
||||
|
||||
The spec-vs-implementation gap audit closed its top items; protocol_version 1.0.0 → 1.0.1
|
||||
(additive — below 1.0.0 the minor is the breaking axis, and nothing here breaks an old client):
|
||||
|
||||
- **Upstream trains** (§3.2, types 0x03/0x04/0x05): per-train bounded columnar buffer (8192
|
||||
rows, head kept on overflow with `Truncated` set — mirrors the schema's `evidence_truncated`
|
||||
honesty), TRAIN_REPORT split across ≤1200-byte datagrams, grant-free with the §3.4 argument
|
||||
spelled out (a 17-byte report row answers a ≥36-byte HMAC-valid packet). Unknown train id
|
||||
gets a zero-row report: "nothing arrived" is an answer. Also surfaced as `udp.trains` in the
|
||||
observations API.
|
||||
- **Real TTL/DSCP/ECN observation** (§3.3): the read loop is `ReadMsgUDPAddrPort` with
|
||||
IP_RECVTTL/IP_RECVTOS/IPV6_RECVHOPLIMIT/IPV6_RECVTCLASS cmsgs on Linux; `0xFF` stays the
|
||||
"not observed" sentinel elsewhere. This unblocks `sec.dscp_ecn_survival` both directions,
|
||||
paired with the new `dscp` parameter on `downtrain` (validated 0–63, refused not clamped,
|
||||
`dscp_applied` in the response).
|
||||
- **Rate limiting** (§2.5, was entirely absent): token buckets keyed per credential AND per
|
||||
source IP; 429 + Retry-After on session/action creation (`/v1/profile` stays ungated), silent
|
||||
drop on the data plane — charged after the HMAC gate so a spoofed flood cannot drain a
|
||||
victim's budget, before the replay window so a dropped seq stays usable. UDP ceilings default
|
||||
above the largest legitimate run (a 200 Mbps throughput test), because a rate limit that
|
||||
clips a real measurement produces a confidently wrong number.
|
||||
- **`action_id` in every granted packet** (§5/§9): payload bytes [8:16] across all granted
|
||||
types, so overlapping actions are attributable. Verified the deployed Kotlin client parses
|
||||
only ECHO_RESP and MTU_ACK payloads, so the reshuffle strands nobody.
|
||||
- **Canary log retention**: the stated 24 h privacy default is now enforced
|
||||
(`ECHOLOT_DNS_LOG_RETENTION_H`), where before the log was time-unbounded.
|
||||
- **`POST /admin/enroll-tokens`** now answers the spec's JSON shape under content negotiation;
|
||||
the README's curl works as documented.
|
||||
- Spec §2.3 registry gained `downtrain` and `tcp-echo`, which the server had been advertising
|
||||
as strings a conformant client must ignore.
|
||||
|
||||
Client-side counterparts still to build: sending 0x03 trains + parsing 0x05 reports
|
||||
(`train.udp_updown`), and passing `dscp` on downtrain actions.
|
||||
|
||||
## ⚠ Version lineage broken: fmr runs v0.11.2, the repo's tags stop at v0.9.x (2026-08-02)
|
||||
|
||||
Discovered while preparing to self-update fmr to the freshly released server-v0.9.2:
|
||||
**fmr runs v0.11.2** (binary installed 2026-08-02 08:51), but this repo's remote has tags only
|
||||
up to `server-v0.9.1`, master fast-forwarded cleanly from this machine, there is no v0.10/v0.11
|
||||
release in Gitea, no source checkout or Go toolchain on fmr, and no deploy script in this repo
|
||||
that stamps versions. Conclusion: v0.11.2 was cross-built from a clone whose commits were never
|
||||
pushed — presumably another dev machine.
|
||||
|
||||
Consequences until resolved:
|
||||
- **Do NOT run `--self-update` on fmr.** Gitea's `/releases/latest` is the *newest-created*
|
||||
release, which is now `server-v0.9.2` — semantically older than the deployed binary; the
|
||||
updater compares strings, not SemVer, and would happily "update" v0.11.2 down to it. No
|
||||
automatic risk exists (fmr has no update timer installed, only the cert timer), but a manual
|
||||
run would downgrade.
|
||||
- The next real release must be tagged **above v0.11.2** (e.g. `server-v0.11.3` or `v0.12.0`)
|
||||
*after* the missing commits are pushed, so "latest" becomes truly latest again.
|
||||
- The unpushed v0.10–v0.11 work needs to be found and pushed from whichever machine built it,
|
||||
or the deployed binary's provenance re-established some other way, before the release channel
|
||||
can be trusted again.
|
||||
|
||||
@@ -89,9 +89,29 @@ rolled up under *connectivity* instead — the third occurrence of rule 1 being
|
||||
|
||||
| code | severity | means | rules out |
|
||||
|---|---|---|---|
|
||||
| `v6.broken` | medium | IPv6 is configured on this network but does not work. | Absence of IPv6: it is provisioned, it simply fails. |
|
||||
| `dns.search_domain_unanswered` | high | The network advertises a DNS search domain that its own server does not answer for. | A fault on this device: the same server answers ordinary names normally. |
|
||||
| `dns.system_resolver_broken` | high | The network's DNS server answers, but this device cannot resolve names through it. | A network fault: the server replied to a query sent from this device. |
|
||||
| `measurement.vpn_constrained` | info | A VPN was active, so the networks underneath it could not be measured. | Nothing — this run says little about the underlying network either way. |
|
||||
| `v6.no_default_route` | medium | The device has a global IPv6 address but no IPv6 default route. | Guesswork: this is read from the routing table, not inferred from silence. |
|
||||
| `v6.route_without_address` | medium | The network advertises an IPv6 default route but the device has no global IPv6 address. | A working IPv6 setup: SLAAC did not produce a usable address on this link. |
|
||||
| `v6.no_icmp_reply` | low | IPv6 is configured but ICMPv6 echo gets no reply. | Nothing on its own: IPv6 may work fine with ICMP filtered. |
|
||||
| `v6.broken` | high | IPv6 is advertised on this network but carries no traffic. | ICMP filtering as the benign explanation: a TCP connection over IPv6 failed too. |
|
||||
| `v6.not_offered` | info | This network does not offer IPv6. | — |
|
||||
|
||||
`v6.no_icmp_reply` was `v6.broken` until a phone reported it while loading an IPv6-only site over
|
||||
TCP perfectly well. The only evidence behind it is ICMPv6 echo, which is widely filtered on
|
||||
networks where IPv6 works — so the finding now states what was observed and names both
|
||||
explanations instead of choosing one. It is still worth reporting: filtered ICMPv6 breaks Path MTU
|
||||
Discovery.
|
||||
|
||||
`v6.broken` returned once that corroboration existed: the `v6.brokenness` test attempts a real TCP
|
||||
connection over IPv6 to the configured server, and only when *both* transports fail on a network
|
||||
that advertises IPv6 is the brokenness claim made — at high severity, because every dual-stack
|
||||
destination pays a timeout before falling back to IPv4. When the TCP connect *succeeds*,
|
||||
`v6.no_icmp_reply` is emitted at high confidence instead, now able to say plainly that ICMPv6 is
|
||||
filtered while IPv6 works. With no server configured there is no corroboration target and the
|
||||
two-explanation `v6.no_icmp_reply` stands unchanged.
|
||||
|
||||
`v6.not_offered` is **info and must stay info**. Most networks still do not offer IPv6 and that is
|
||||
not a fault; reporting it as a warning lights a yellow verdict on a healthy network, which teaches
|
||||
people to ignore the light — the one thing a diagnostic must never do.
|
||||
|
||||
@@ -52,12 +52,26 @@ Export encoding: UTF-8 JSON, gzip for files (`.echolot.json.gz`), share intent u
|
||||
},
|
||||
"tiers": { "app": true, "shizuku": true, "root": false },
|
||||
"profiles_used": ["profile-uuid", ...],
|
||||
"constraints": {
|
||||
"vpn_active": true,
|
||||
"per_network_blocked": true,
|
||||
"unmeasured_networks": ["net-0", "net-1"]
|
||||
},
|
||||
"notes": "free-text user annotation"
|
||||
}
|
||||
```
|
||||
|
||||
`tiers` records what was *available*; each test records what it *used*.
|
||||
|
||||
`constraints` records what was *prevented*. A constrained run is neither a failed run nor a normal
|
||||
one, and the distinction has to survive into the data: a run taken through a VPN has the same shape
|
||||
and the same green verdict as a clean run of a healthy network, so without this a reader — or a
|
||||
server aggregating thousands of them — cannot tell that almost nothing was measured. The known case
|
||||
is `per_network_blocked`: Android refuses `Network.bindSocket()` on the underlying networks while a
|
||||
VPN holds the default route, so every per-network test measures the tunnel or nothing at all, and
|
||||
any conclusion about the link underneath is unfounded. Consumers should treat findings from a
|
||||
constrained run as scoped to what was actually reachable, and `unmeasured_networks` names the rest.
|
||||
|
||||
## 4. `networks[]` — one entry per Android `Network` in play
|
||||
|
||||
A run may exercise several networks simultaneously (Wi-Fi + cellular + USB ethernet). Everything is a snapshot at run start; a `changes[]` list captures mid-run deltas.
|
||||
|
||||
+31
-1
@@ -84,7 +84,12 @@ The app re-fetches the profile at the start of every run (falling back to the ca
|
||||
|
||||
### 2.3 Capabilities (v1 registry)
|
||||
|
||||
`udp-probe`, `stun-basic`, `stun-5780`, `canary-dns`, `recursive-dns`, `connect-back`, `delayed-echo`, `big-send`, `frag-send`, `tls-echo`, `http-echo`, `throughput`, `ntp`. A server omits what it can't offer (e.g. `stun-5780` without a second IP degrades to `stun-basic`). Clients must skip, and record as `unsupported`, any test whose capability is absent. Unknown capability strings are ignored.
|
||||
`udp-probe`, `stun-basic`, `stun-5780`, `canary-dns`, `recursive-dns`, `connect-back`, `delayed-echo`, `big-send`, `frag-send`, `tls-echo`, `http-echo`, `throughput`, `ntp`, plus:
|
||||
|
||||
- `downtrain` — server-sent downstream trains via the §5 `downtrain` action. Upstream trains need no capability of their own: they are plain client-sent data-plane packets and ride `udp-probe`.
|
||||
- `tcp-echo` — the plain-TCP echo endpoint (§4); `tls-echo` is its ALPN variant on the same port.
|
||||
|
||||
A server omits what it can't offer (e.g. `stun-5780` without a second IP degrades to `stun-basic`). Clients must skip, and record as `unsupported`, any test whose capability is absent. Unknown capability strings are ignored.
|
||||
|
||||
### 2.4 Sessions
|
||||
|
||||
@@ -111,6 +116,31 @@ DELETE /v1/sessions/{id}
|
||||
|
||||
Per-credential and per-source-IP token buckets on: session creation, actions, UDP packets, bytes. `429` on control plane; silent drop on data plane (probes must tolerate loss anyway). All reflected/generated traffic goes **only** to the session's observed source address (or, for connect-back, the source address of the session-creating request). Data-plane responses to unauthenticated packets are never larger than the request (§3.4).
|
||||
|
||||
### 2.2 `GET /v1/discover` — where the control plane lives
|
||||
|
||||
Unauthenticated, and says almost nothing: the control-plane URL and the server's display name.
|
||||
|
||||
```json
|
||||
{ "control_url": "https://probe.example.net", "name": "example" }
|
||||
```
|
||||
|
||||
It exists so an enrollment link can carry the name a person recognises while the app still connects
|
||||
to the name that selects the pinned certificate. When a server shares port 443 between its admin UI
|
||||
and its control plane, those must be different hostnames — one port and one name is one certificate,
|
||||
and the two need different ones (a browser-trusted certificate, and a long-lived self-signed one the
|
||||
client pins). Without discovery, the difference leaks into every enrollment link an operator hands
|
||||
out.
|
||||
|
||||
**It hands out an address, never a pin.** The pin travels in the link itself. Serving it here would
|
||||
reduce pinning to whatever the certificate authorities are worth, and pinning exists precisely to
|
||||
survive one the operator does not control — a root injected by corporate device management, for
|
||||
instance, which is unremarkable on the networks this tool is pointed at. Because the pin is
|
||||
pre-shared, an intercepted discovery response can only send a device to the wrong host, where the
|
||||
pin will not match: an outage, not a compromise.
|
||||
|
||||
Clients treat it as optional. A server that does not answer, or a link that already names the
|
||||
control endpoint, works unchanged — enrollment must not begin failing because a lookup did.
|
||||
|
||||
## 3. UDP probe protocol
|
||||
|
||||
### 3.1 Packet header (fixed 32 bytes, network byte order)
|
||||
|
||||
@@ -15,7 +15,7 @@ plugins {
|
||||
//
|
||||
// major*1_000_000 + minor*10_000 + patch*10 leaves room for 9 patch-level rebuilds (the trailing
|
||||
// digit) without disturbing the mapping, and stays inside the 2_100_000_000 ceiling until major 2100.
|
||||
val appVersionName = "0.2.0"
|
||||
val appVersionName = "0.2.3"
|
||||
|
||||
fun versionCodeOf(semver: String): Int {
|
||||
val (major, minor, patch) = semver.substringBefore('-').split(".").map(String::toInt)
|
||||
@@ -34,8 +34,16 @@ android {
|
||||
versionName = appVersionName
|
||||
// Automation: `adb shell am start -n app.echo_lot.app/.MainActivity --ez autorun true`
|
||||
// runs a measurement immediately and POSTs the report here (dev collection endpoint).
|
||||
buildConfigField("String", "REPORT_UPLOAD_URL", "\"http://89.185.109.150:443/report\"")
|
||||
buildConfigField("String", "REPORT_UPLOAD_SECRET", "\"D4OmG5gGJsElqVVbtYIZbR\"")
|
||||
// Empty: the collection endpoint this pointed at was the adb-beacon receiver, which held
|
||||
// 0.0.0.0:443 in cleartext. That service is gone and echolot-server owns 443 with TLS, so
|
||||
// posting plaintext there now fails as "client sent an HTTP request to an HTTPS server" —
|
||||
// an alarming error for a debugging convenience that is no longer needed, since autorun
|
||||
// reports are read straight off the device with `run-as cat`.
|
||||
//
|
||||
// Deliberately not repointed at /v1/runs. That is the consent-gated upload, and a
|
||||
// debugging shortcut must not be able to satisfy it by accident.
|
||||
buildConfigField("String", "REPORT_UPLOAD_URL", "\"\"")
|
||||
buildConfigField("String", "REPORT_UPLOAD_SECRET", "\"\"")
|
||||
// The bare SemVer, without the debug build's "-dev" suffix stripped away by the server's
|
||||
// parser anyway — sent to servers so they can apply their compatibility window.
|
||||
buildConfigField("String", "APP_SEMVER", "\"$appVersionName\"")
|
||||
|
||||
@@ -22,7 +22,8 @@
|
||||
|
||||
<activity
|
||||
android:name=".MainActivity"
|
||||
android:exported="true">
|
||||
android:exported="true"
|
||||
android:launchMode="singleTask">
|
||||
<intent-filter>
|
||||
<action android:name="android.intent.action.MAIN" />
|
||||
<category android:name="android.intent.category.LAUNCHER" />
|
||||
@@ -39,6 +40,17 @@
|
||||
<category android:name="android.intent.category.BROWSABLE" />
|
||||
<data android:scheme="echolot" android:host="enroll" />
|
||||
</intent-filter>
|
||||
<!--
|
||||
Sign-in redirect. The browser hands the authorization code back through this, which
|
||||
is exactly why the flow uses PKCE: any app may register this scheme, so the code
|
||||
alone must not be enough to complete a sign-in.
|
||||
-->
|
||||
<intent-filter android:autoVerify="false">
|
||||
<action android:name="android.intent.action.VIEW" />
|
||||
<category android:name="android.intent.category.DEFAULT" />
|
||||
<category android:name="android.intent.category.BROWSABLE" />
|
||||
<data android:scheme="echolot" android:host="auth" />
|
||||
</intent-filter>
|
||||
</activity>
|
||||
|
||||
<provider
|
||||
|
||||
@@ -0,0 +1,125 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package app.echo_lot.app
|
||||
|
||||
import app.echo_lot.protocol.AuthInfo
|
||||
import app.echo_lot.protocol.ControlClient
|
||||
import app.echo_lot.protocol.OidcLogin
|
||||
import kotlinx.serialization.json.Json
|
||||
import kotlinx.serialization.json.jsonObject
|
||||
import kotlinx.serialization.json.jsonPrimitive
|
||||
|
||||
/**
|
||||
* Signing in to the configured server's identity provider.
|
||||
*
|
||||
* The awkward part of a browser-based sign-in on Android is that the app is not running while it
|
||||
* happens. Handing control to a browser puts this process in the background, where it may be
|
||||
* killed at any moment; the callback then arrives at a fresh process with none of the state the
|
||||
* exchange needs. So the PKCE verifier and state are written to storage before the browser opens,
|
||||
* not held in memory — an in-memory value works on a developer's device and fails on a phone under
|
||||
* memory pressure, which is the worst way for this to break.
|
||||
*
|
||||
* Nothing from the identity provider is kept afterwards. The ID token proves who is signing in,
|
||||
* once; the device credential authenticates everything from then on.
|
||||
*/
|
||||
class Account(private val settings: Settings) {
|
||||
|
||||
private val json = Json { ignoreUnknownKeys = true }
|
||||
|
||||
sealed interface SignInStart {
|
||||
/** Open this in a browser. */
|
||||
data class Browser(val url: String) : SignInStart
|
||||
data class Unavailable(val reason: String) : SignInStart
|
||||
}
|
||||
|
||||
/** Fetches the server's auth configuration and builds the authorization URL. */
|
||||
fun begin(): SignInStart {
|
||||
if (!settings.serverConfigured) {
|
||||
return SignInStart.Unavailable(
|
||||
"Enrol with a server first — sign-in belongs to the server's identity provider."
|
||||
)
|
||||
}
|
||||
val auth = runCatching { client().profile(settings.serverCredential).auth }.getOrNull()
|
||||
?: return SignInStart.Unavailable("Could not reach the server to ask how to sign in.")
|
||||
|
||||
auth.discoveryError?.let {
|
||||
// The distinction matters: "the operator configured an IdP that is not answering" is
|
||||
// their problem to fix, and is not the same as "this server has no accounts".
|
||||
return SignInStart.Unavailable("The server's identity provider is not responding: $it")
|
||||
}
|
||||
if (!auth.enabled) {
|
||||
return SignInStart.Unavailable("This server does not offer accounts.")
|
||||
}
|
||||
return try {
|
||||
val pending = OidcLogin.begin(auth)
|
||||
// Written before the browser opens, because after that this process may not survive.
|
||||
settings.pendingVerifier = pending.verifier
|
||||
settings.pendingState = pending.state
|
||||
SignInStart.Browser(pending.authorizationUrl)
|
||||
} catch (t: Throwable) {
|
||||
SignInStart.Unavailable(t.message ?: "Could not start sign-in.")
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Completes sign-in from the `echolot://auth` redirect.
|
||||
*
|
||||
* Blocking; callers run it off the main thread.
|
||||
*/
|
||||
fun complete(callbackUri: String): String {
|
||||
val verifier = settings.pendingVerifier
|
||||
val state = settings.pendingState
|
||||
// Cleared first, whatever happens next: these are single-use, and leaving them behind
|
||||
// would let a later callback be completed against a flow nobody started.
|
||||
settings.clearPendingAuth()
|
||||
|
||||
if (verifier.isBlank() || state.isBlank()) {
|
||||
return "That sign-in did not start on this device."
|
||||
}
|
||||
return try {
|
||||
val auth = client().profile(settings.serverCredential).auth
|
||||
val idToken = OidcLogin.complete(
|
||||
auth, OidcLogin.Pending("", verifier, state), callbackUri,
|
||||
)
|
||||
val reply = client().linkAccount(settings.serverCredential, idToken)
|
||||
val o = json.parseToJsonElement(reply).jsonObject
|
||||
val name = o["display_name"]?.jsonPrimitive?.content ?: "signed in"
|
||||
settings.accountName = name
|
||||
settings.accountId = o["account_id"]?.jsonPrimitive?.content ?: ""
|
||||
val admin = o["admin"]?.jsonPrimitive?.content == "true"
|
||||
"Signed in as $name" + if (admin) " (administrator)" else ""
|
||||
} catch (e: OidcLogin.LoginFailed) {
|
||||
e.message ?: "Sign-in failed."
|
||||
} catch (t: Throwable) {
|
||||
"Sign-in failed: ${t.message ?: t.javaClass.simpleName}"
|
||||
}
|
||||
}
|
||||
|
||||
/** Signs out. The device stays enrolled — signing out should not cost an enrolment. */
|
||||
fun signOut(): String = try {
|
||||
client().unlinkAccount(settings.serverCredential)
|
||||
settings.accountName = ""
|
||||
settings.accountId = ""
|
||||
"Signed out. This device is still enrolled."
|
||||
} catch (t: Throwable) {
|
||||
"Could not sign out: ${t.message ?: t.javaClass.simpleName}"
|
||||
}
|
||||
|
||||
/** Asks the server who it thinks is signed in, so the UI is not trusting stale local state. */
|
||||
fun refresh(): String? = runCatching {
|
||||
val o = json.parseToJsonElement(client().accountStatus(settings.serverCredential)).jsonObject
|
||||
val signedIn = o["signed_in"]?.jsonPrimitive?.content == "true"
|
||||
settings.accountName = if (signedIn) {
|
||||
o["display_name"]?.jsonPrimitive?.content ?: ""
|
||||
} else {
|
||||
""
|
||||
}
|
||||
settings.accountName.takeIf { it.isNotBlank() }
|
||||
}.getOrNull()
|
||||
|
||||
private fun client() = ControlClient(
|
||||
settings.serverUrl, setOf(settings.serverPin), BuildConfig.APP_SEMVER,
|
||||
fallbackAddrs = settings.serverAddrList(),
|
||||
)
|
||||
}
|
||||
@@ -43,9 +43,27 @@ class MainActivity : ComponentActivity() {
|
||||
private val permissionLauncher =
|
||||
registerForActivityResult(ActivityResultContracts.RequestMultiplePermissions()) { /* proceed regardless */ }
|
||||
|
||||
/**
|
||||
* The intent currently being acted on, so a deep link that arrives while the app is running
|
||||
* is seen by the screen the user is already looking at.
|
||||
*
|
||||
* The activity is singleTask for the same reason. As a standard activity it stacked a second
|
||||
* instance per link, each with its own ViewModel: the enrolment then happened in a throwaway
|
||||
* copy, and pressing back returned to the original screen showing none of it. Silent, and
|
||||
* indistinguishable from the link simply not working.
|
||||
*/
|
||||
private val liveIntent = mutableStateOf<android.content.Intent?>(null)
|
||||
|
||||
override fun onNewIntent(intent: android.content.Intent) {
|
||||
super.onNewIntent(intent)
|
||||
setIntent(intent)
|
||||
liveIntent.value = intent
|
||||
}
|
||||
|
||||
override fun onCreate(savedInstanceState: Bundle?) {
|
||||
super.onCreate(savedInstanceState)
|
||||
requestRuntimePermissions()
|
||||
liveIntent.value = intent
|
||||
setContent {
|
||||
MaterialTheme(colorScheme = darkColorScheme()) {
|
||||
Surface(color = MaterialTheme.colorScheme.background) {
|
||||
@@ -63,13 +81,88 @@ class MainActivity : ComponentActivity() {
|
||||
// An echolot://enroll link (QR scan, or a link the operator sent) opens the
|
||||
// app straight into settings with the enrollment already done, so the user
|
||||
// sees the result rather than a form they still have to fill in.
|
||||
val enrollUri = intent?.takeIf { it.action == Intent.ACTION_VIEW }?.dataString
|
||||
// Both deep links land here. They are told apart by host, so a sign-in
|
||||
// redirect is never mistaken for an enrolment link — one spends a token, the
|
||||
// other completes an authorization, and confusing them would fail obscurely.
|
||||
val incoming = liveIntent.value?.takeIf { it.action == Intent.ACTION_VIEW }?.dataString
|
||||
val authUri = incoming?.takeIf { it.startsWith("echolot://auth") }
|
||||
val enrollUri = incoming?.takeIf { it.startsWith("echolot://enroll") }
|
||||
androidx.compose.runtime.LaunchedEffect(enrollUri) {
|
||||
if (enrollUri != null) {
|
||||
vm.enroll(enrollUri)
|
||||
screen = Screen.SETTINGS
|
||||
}
|
||||
}
|
||||
// Replacing an existing enrollment is asked about, never assumed. Following a
|
||||
// link from a web page is one tap, and the old credential does not survive it.
|
||||
vm.state.pendingEnroll?.let { pending ->
|
||||
androidx.compose.material3.AlertDialog(
|
||||
onDismissRequest = { vm.cancelEnroll() },
|
||||
title = {
|
||||
androidx.compose.material3.Text(
|
||||
if (pending.sameServer) "Enroll again with this server?"
|
||||
else "Replace this device's server?"
|
||||
)
|
||||
},
|
||||
text = {
|
||||
androidx.compose.material3.Text(
|
||||
// Naming the same URL twice reads as a mistake and buries the
|
||||
// one consequence that actually applies: the device is issued a
|
||||
// fresh credential and shows up as a second entry.
|
||||
if (pending.sameServer) {
|
||||
"This device is already enrolled with " +
|
||||
"${pending.currentServer}.\n\n" +
|
||||
"Enrolling again replaces its credential. The old one " +
|
||||
"stops working immediately, and the device appears on " +
|
||||
"the server as a new entry alongside the current one — " +
|
||||
"which you may want to revoke afterwards.\n\n" +
|
||||
"Runs already uploaded, and runs stored on this phone, " +
|
||||
"are not affected."
|
||||
} else {
|
||||
"This device is already enrolled with " +
|
||||
"${pending.currentServer}.\n\n" +
|
||||
"Enrolling with ${pending.newServer} replaces that. Runs " +
|
||||
"already uploaded stay where they are, but this device " +
|
||||
"stops reporting to the old server and appears on the new " +
|
||||
"one as a new device.\n\n" +
|
||||
"Runs stored on this phone are not affected."
|
||||
}
|
||||
)
|
||||
},
|
||||
confirmButton = {
|
||||
androidx.compose.material3.TextButton(onClick = { vm.confirmEnroll() }) {
|
||||
androidx.compose.material3.Text(
|
||||
if (pending.sameServer) "Enroll again" else "Enroll here"
|
||||
)
|
||||
}
|
||||
},
|
||||
dismissButton = {
|
||||
androidx.compose.material3.TextButton(onClick = { vm.cancelEnroll() }) {
|
||||
androidx.compose.material3.Text("Keep current server")
|
||||
}
|
||||
},
|
||||
)
|
||||
}
|
||||
androidx.compose.runtime.LaunchedEffect(authUri) {
|
||||
if (authUri != null) {
|
||||
vm.completeSignIn(authUri)
|
||||
screen = Screen.SETTINGS
|
||||
}
|
||||
}
|
||||
// Shizuku can be started, stopped or authorised in its own app, where nothing
|
||||
// calls back into this process. Asking again each time this screen comes
|
||||
// forward is what makes the banner right after the user has been away to fix
|
||||
// it — which is exactly the moment they look at it.
|
||||
val lifecycleOwner = androidx.compose.ui.platform.LocalLifecycleOwner.current
|
||||
androidx.compose.runtime.DisposableEffect(lifecycleOwner) {
|
||||
val obs = androidx.lifecycle.LifecycleEventObserver { _, event ->
|
||||
if (event == androidx.lifecycle.Lifecycle.Event.ON_RESUME) {
|
||||
vm.refreshShizuku()
|
||||
}
|
||||
}
|
||||
lifecycleOwner.lifecycle.addObserver(obs)
|
||||
onDispose { lifecycleOwner.lifecycle.removeObserver(obs) }
|
||||
}
|
||||
androidx.compose.runtime.LaunchedEffect(autorun) {
|
||||
if (autorun) vm.run(devUpload = true)
|
||||
}
|
||||
@@ -107,8 +200,21 @@ class MainActivity : ComponentActivity() {
|
||||
lifecycleScope.launch { preview = vm.previewNewestRun() }
|
||||
},
|
||||
onCheckServer = vm::checkServer,
|
||||
accountName = vm.accountName,
|
||||
onSignIn = {
|
||||
vm.beginSignIn { url ->
|
||||
// A plain VIEW intent rather than a Custom Tab: the browser is
|
||||
// where the user's existing IdP session already lives, and
|
||||
// androidx.browser would be a dependency for a rounded corner.
|
||||
runCatching {
|
||||
startActivity(Intent(Intent.ACTION_VIEW, android.net.Uri.parse(url)))
|
||||
}
|
||||
}
|
||||
},
|
||||
onSignOut = vm::signOut,
|
||||
onEnroll = vm::enroll,
|
||||
serverStatus = vm.state.archiveStatus,
|
||||
enrollStatus = vm.state.enrollStatus,
|
||||
onBack = { screen = Screen.RUN },
|
||||
)
|
||||
Screen.HISTORY -> HistoryScreen(
|
||||
@@ -307,6 +413,41 @@ private fun EcholotScreen(
|
||||
|
||||
@Composable
|
||||
private fun Results(doc: MeasurementDocument) {
|
||||
// A constrained run is answered before the lights are: the verdict below is INCONCLUSIVE by
|
||||
// §7.3, and without this banner "inconclusive" reads as the app failing rather than the OS
|
||||
// (correctly) refusing to let anything past the VPN be measured.
|
||||
val constraints = doc.run.constraints
|
||||
if (constraints.constrained) {
|
||||
val blocked = constraints.unmeasuredNetworks
|
||||
.mapNotNull { id -> doc.networks.firstOrNull { it.id == id } }
|
||||
.joinToString(", ") { it.iface?.takeIf { s -> s.isNotBlank() } ?: it.transport.name.lowercase() }
|
||||
.ifBlank { "the networks beneath it" }
|
||||
// Same three-way split as the finding: saying "VPN" when the user just disconnected
|
||||
// theirs (the wall lingers during teardown) reads as the app being wrong, not the OS.
|
||||
val (headline, body) = when {
|
||||
constraints.vpnActive && constraints.perNetworkBlocked ->
|
||||
"Measured through a VPN" to
|
||||
("Android does not let apps send on the networks beneath an active VPN, so " +
|
||||
"$blocked could not be measured — these results describe the tunnel. " +
|
||||
"Disconnect the VPN and run again to measure the networks themselves.")
|
||||
constraints.perNetworkBlocked ->
|
||||
"Some networks could not be measured" to
|
||||
("Android refused sends on $blocked — the restriction a VPN leaves in place " +
|
||||
"while it tears down. These networks went unmeasured; wait a few " +
|
||||
"seconds and run again.")
|
||||
else ->
|
||||
"A VPN holds the default route" to
|
||||
("Default-route results describe the tunnel; per-network measurements " +
|
||||
"reached the underlying networks.")
|
||||
}
|
||||
Card(colors = CardDefaults.cardColors(containerColor = Color(0xFF3A2E12))) {
|
||||
Column(Modifier.fillMaxWidth().padding(12.dp)) {
|
||||
Text(headline, color = Color(0xFFFFD08A), fontWeight = FontWeight.SemiBold)
|
||||
Text(body, fontSize = 12.sp, color = Color(0xFFFFD08A))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
val summary = doc.summary
|
||||
if (summary != null) {
|
||||
Card(colors = CardDefaults.cardColors(containerColor = verdictColor(summary.overall))) {
|
||||
|
||||
@@ -80,8 +80,16 @@ class RunStore(context: Context, private val settings: Settings) {
|
||||
|
||||
private fun client() = ControlClient(
|
||||
settings.serverUrl, setOf(settings.serverPin), BuildConfig.APP_SEMVER,
|
||||
fallbackAddrs = settings.serverAddrList(),
|
||||
)
|
||||
|
||||
/** Remembers where the server lives, so a later run can reach it without DNS. */
|
||||
private fun rememberAddrs(p: app.echo_lot.protocol.Profile) {
|
||||
val addrs = p.targets.flatMap { listOfNotNull(it.ip4, it.ip6) }
|
||||
.filter { it.isNotBlank() }
|
||||
if (addrs.isNotEmpty()) settings.serverAddrs = addrs.joinToString(",")
|
||||
}
|
||||
|
||||
/**
|
||||
* Checks the configured server without uploading anything: reachable, pinned, compatible, and
|
||||
* willing to accept runs. Lets the user find out in settings rather than from a failed run.
|
||||
@@ -90,6 +98,10 @@ class RunStore(context: Context, private val settings: Settings) {
|
||||
if (!settings.serverConfigured) return "Fill in the server URL, pin and credential first."
|
||||
return try {
|
||||
val profile = client().profile(settings.serverCredential)
|
||||
// Learned here so the next run's canary probe knows what to ask for.
|
||||
profile.canaryZone.takeIf { it.isNotBlank() }?.let { settings.canaryZone = it }
|
||||
settings.serverFacts = describeFacts(profile)
|
||||
rememberAddrs(profile)
|
||||
val compat = Compat.check(profile, BuildConfig.APP_SEMVER)
|
||||
val head = "${profile.name} · server ${profile.serverVersion} · " +
|
||||
"protocol ${profile.compat.protocolVersion.ifBlank { "unstated" }}"
|
||||
@@ -115,6 +127,36 @@ class RunStore(context: Context, private val settings: Settings) {
|
||||
* no credential — fails later, somewhere else, with an error that points at the wrong thing.
|
||||
* Blocking; callers run it off the main thread.
|
||||
*/
|
||||
|
||||
/**
|
||||
* Renders what the server says about itself, for display.
|
||||
*
|
||||
* Only what a person measuring against it would want to check: which addresses the tests will
|
||||
* actually use, on which ports, and what the server admits it can do. Addresses first, because
|
||||
* "which address did this result come from" is the question a report leaves open.
|
||||
*/
|
||||
private fun describeFacts(p: app.echo_lot.protocol.Profile): String {
|
||||
val lines = ArrayList<String>()
|
||||
// "label|value" per line, laid out as real columns by the UI rather than padded with
|
||||
// spaces here. Space padding only lines up in a monospaced font, which makes the layout
|
||||
// depend on a typeface choice made somewhere else entirely.
|
||||
fun row(label: String, value: String) = lines.add("$label|$value")
|
||||
|
||||
row("server", "${p.name} · ${p.serverVersion}")
|
||||
for (t in p.targets) {
|
||||
t.ip4?.let { row("IPv4", it) }
|
||||
t.ip6?.let { row("IPv6", it) }
|
||||
// Marked rather than listed apart: it is the same server, and what matters is being
|
||||
// able to tell which address a NAT-behaviour result came from.
|
||||
t.ip4Alt?.let { row("IPv4 alt", it) }
|
||||
t.ip6Alt?.let { row("IPv6 alt", it) }
|
||||
row("ports", "udp ${t.udpPort} · tcp ${t.tcpPort} · stun ${t.stunPort}")
|
||||
}
|
||||
if (p.canaryZone.isNotBlank()) row("dns zone", p.canaryZone)
|
||||
if (p.capabilities.isNotEmpty()) row("measures", p.capabilities.joinToString(", "))
|
||||
return lines.joinToString(System.lineSeparator())
|
||||
}
|
||||
|
||||
fun enroll(link: String, deviceName: String?): String {
|
||||
val parsed = app.echo_lot.protocol.EnrollmentLink.parse(link)
|
||||
?: return "That does not look like an Echolot enrollment link. It should start with " +
|
||||
@@ -122,7 +164,11 @@ class RunStore(context: Context, private val settings: Settings) {
|
||||
return try {
|
||||
val enrolled = parsed.redeem(deviceName, BuildConfig.APP_SEMVER)
|
||||
val compat = Compat.check(enrolled.profile, BuildConfig.APP_SEMVER)
|
||||
settings.serverFacts = describeFacts(enrolled.profile)
|
||||
rememberAddrs(enrolled.profile)
|
||||
enrolled.profile.canaryZone.takeIf { it.isNotBlank() }?.let { settings.canaryZone = it }
|
||||
settings.serverUrl = enrolled.controlUrl
|
||||
settings.serverPublicUrl = enrolled.publicUrl
|
||||
settings.serverPin = enrolled.pin
|
||||
settings.serverCredential = enrolled.credential
|
||||
val head = "Enrolled with ${enrolled.profile.name} " +
|
||||
@@ -150,6 +196,9 @@ class RunStore(context: Context, private val settings: Settings) {
|
||||
return try {
|
||||
val client = client()
|
||||
val profile = client.profile(settings.serverCredential)
|
||||
profile.canaryZone.takeIf { it.isNotBlank() }?.let { settings.canaryZone = it }
|
||||
settings.serverFacts = describeFacts(profile)
|
||||
rememberAddrs(profile)
|
||||
|
||||
// Compatibility before policy: an incompatible server may well advertise an upload
|
||||
// policy it would never actually apply to us.
|
||||
|
||||
@@ -42,6 +42,17 @@ data class UiState(
|
||||
val archiveStatus: String? = null,
|
||||
/** History, newest first. Refreshed after every run and whenever the history screen opens. */
|
||||
val history: List<app.echo_lot.archive.ArchivedRun> = emptyList(),
|
||||
/**
|
||||
* Result of the last enrollment attempt, shown beside the Enroll button.
|
||||
*
|
||||
* Separate from [archiveStatus]: they are two different actions with two different results,
|
||||
* and sharing one line put the answer to "did enrolling work" at the far end of the card,
|
||||
* below three text fields — or nowhere at all on a fresh install, since that line only
|
||||
* renders once a run exists.
|
||||
*/
|
||||
val enrollStatus: String? = null,
|
||||
/** An enrollment link waiting on confirmation, because this device is already enrolled. */
|
||||
val pendingEnroll: PendingEnroll? = null,
|
||||
/** Shell-tier readiness, shown before a run; null message = say nothing (Shizuku not installed). */
|
||||
val shizukuNotice: String? = null,
|
||||
val shizukuReady: Boolean = false,
|
||||
@@ -49,6 +60,18 @@ data class UiState(
|
||||
val shizukuState: ShizukuAvailability.State = ShizukuAvailability.State.NOT_INSTALLED,
|
||||
)
|
||||
|
||||
/**
|
||||
* An enrollment link that would replace an existing one, held until the user says so.
|
||||
*
|
||||
* Enrolling is not additive: the new credential replaces the old, and on the previous server this
|
||||
* device simply stops reporting. Following a link is one tap from a web page, which is not enough
|
||||
* deliberation to discard a working enrollment by accident.
|
||||
*/
|
||||
data class PendingEnroll(val link: String, val currentServer: String, val newServer: String) {
|
||||
/** Re-enrolling with the server already configured, rather than moving to a different one. */
|
||||
val sameServer: Boolean get() = currentServer.trimEnd('/') == newServer.trimEnd('/')
|
||||
}
|
||||
|
||||
/**
|
||||
* Drives one measurement run: device-tier probes (link snapshot, per-network ICMP) always run;
|
||||
* results assemble into a MeasurementDocument with a §7.3 summary. Lives in a ViewModel so a run
|
||||
@@ -80,6 +103,23 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-reads the shell tier's state, for when it changed somewhere this process cannot see.
|
||||
*
|
||||
* Permission can be granted inside Shizuku's own app, and Shizuku can be started or stopped
|
||||
* there too; none of that calls back here. Asking again on resume is the only way to be right
|
||||
* after the user has been somewhere else to fix it.
|
||||
*/
|
||||
fun refreshShizuku() {
|
||||
val st = ShizukuAvailability.current(getApplication())
|
||||
state = state.copy(
|
||||
shizukuNotice = ShizukuAvailability.describe(st),
|
||||
shizukuReady = st == ShizukuAvailability.State.READY,
|
||||
shizukuHint = ShizukuAvailability.actionHint(st),
|
||||
shizukuState = st,
|
||||
)
|
||||
}
|
||||
|
||||
override fun onCleared() {
|
||||
stopShizukuObserver()
|
||||
super.onCleared()
|
||||
@@ -90,6 +130,7 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
||||
private var runStartWall: String = ""
|
||||
private var runNetworks: List<app.echo_lot.measurement.Network> = emptyList()
|
||||
private var runShizukuOk = false
|
||||
private var runConstraints = Constraints()
|
||||
|
||||
/** Two-clock ids: UUIDs + monotonic ns relative to a per-run origin. */
|
||||
private class RunIds : ProbeIds {
|
||||
@@ -108,6 +149,7 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
||||
fun run(devUpload: Boolean = false) {
|
||||
if (state.running) return
|
||||
collected.clear()
|
||||
runConstraints = Constraints()
|
||||
state = state.copy(running = true, currentStep = "starting", document = null,
|
||||
uploadStatus = null, archiveStatus = null)
|
||||
runJob = viewModelScope.launch {
|
||||
@@ -125,7 +167,13 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
||||
if (devUpload) {
|
||||
state = state.copy(currentStep = "uploading report")
|
||||
val r = withContext(Dispatchers.IO) { ReportUploader.upload(doc) }
|
||||
status = if (r.ok) "uploaded ✓ ${r.detail}" else "upload failed: ${r.detail}"
|
||||
status = when {
|
||||
r.ok -> "uploaded ✓ ${r.detail}"
|
||||
// Not a failure worth alarming about: the dev collection endpoint is simply
|
||||
// not configured, and the run is on the device either way.
|
||||
r.detail.startsWith("no upload URL") -> "run complete — read it with adb"
|
||||
else -> "upload failed: ${r.detail}"
|
||||
}
|
||||
}
|
||||
if (archived != null && settings.autoUpload) {
|
||||
state = state.copy(currentStep = "uploading to server")
|
||||
@@ -215,11 +263,88 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
||||
* history screen has been opened - the two disagreeing read as data loss. */
|
||||
fun archivedRunCount(): Int = store.list().size
|
||||
|
||||
private val account = Account(settings)
|
||||
|
||||
/** Name of whoever is signed in on this device, for the settings screen. */
|
||||
var accountName by mutableStateOf(settings.accountName)
|
||||
private set
|
||||
|
||||
/** Starts sign-in; the caller opens the returned URL in a browser. */
|
||||
fun beginSignIn(open: (String) -> Unit) {
|
||||
viewModelScope.launch {
|
||||
state = state.copy(archiveStatus = "contacting the server …")
|
||||
when (val r = withContext(Dispatchers.IO) { account.begin() }) {
|
||||
is Account.SignInStart.Browser -> {
|
||||
state = state.copy(archiveStatus = "continue in your browser …")
|
||||
open(r.url)
|
||||
}
|
||||
is Account.SignInStart.Unavailable ->
|
||||
state = state.copy(archiveStatus = r.reason)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Completes sign-in from the echolot://auth redirect. */
|
||||
fun completeSignIn(callbackUri: String) {
|
||||
viewModelScope.launch {
|
||||
val msg = withContext(Dispatchers.IO) { account.complete(callbackUri) }
|
||||
accountName = settings.accountName
|
||||
state = state.copy(archiveStatus = msg)
|
||||
}
|
||||
}
|
||||
|
||||
fun signOut() {
|
||||
viewModelScope.launch {
|
||||
val msg = withContext(Dispatchers.IO) { account.signOut() }
|
||||
accountName = settings.accountName
|
||||
state = state.copy(archiveStatus = msg)
|
||||
}
|
||||
}
|
||||
|
||||
/** Redeems an enrollment link, from a paste or from an echolot:// deep link. */
|
||||
fun enroll(link: String, deviceName: String? = android.os.Build.MODEL) {
|
||||
// Already enrolled? Ask first. The old credential is gone the moment this succeeds, and a
|
||||
// link followed from a web page is one tap — far too little deliberation for that.
|
||||
if (settings.serverConfigured) {
|
||||
val target = app.echo_lot.protocol.EnrollmentLink.parse(link)?.controlUrl ?: link
|
||||
state = state.copy(
|
||||
pendingEnroll = PendingEnroll(
|
||||
link = link,
|
||||
// Compared against the link's URL, which names the server publicly — so this
|
||||
// has to be the public name too. Using the endpoint made re-enrolling with the
|
||||
// same server look like a move to a different one, because the endpoint and
|
||||
// the public name are deliberately different strings.
|
||||
currentServer = settings.serverPublicUrl,
|
||||
newServer = target,
|
||||
)
|
||||
)
|
||||
return
|
||||
}
|
||||
doEnroll(link, deviceName)
|
||||
}
|
||||
|
||||
/** The user confirmed replacing an existing enrollment. */
|
||||
fun confirmEnroll(deviceName: String? = android.os.Build.MODEL) {
|
||||
val pending = state.pendingEnroll ?: return
|
||||
state = state.copy(pendingEnroll = null)
|
||||
doEnroll(pending.link, deviceName)
|
||||
}
|
||||
|
||||
fun cancelEnroll() {
|
||||
state = state.copy(
|
||||
pendingEnroll = null,
|
||||
enrollStatus = "Kept the existing enrollment; nothing changed.",
|
||||
)
|
||||
}
|
||||
|
||||
private fun doEnroll(link: String, deviceName: String?) {
|
||||
viewModelScope.launch {
|
||||
state = state.copy(archiveStatus = "enrolling …")
|
||||
state = state.copy(archiveStatus = withContext(Dispatchers.IO) { store.enroll(link, deviceName) })
|
||||
state = state.copy(enrollStatus = "Enrolling …")
|
||||
val result = withContext(Dispatchers.IO) { store.enroll(link, deviceName) }
|
||||
// A new server means a new canary zone; the old one would describe somebody else's
|
||||
// deployment. Cleared rather than kept, and relearned from the next profile fetch.
|
||||
settings.canaryZone = ""
|
||||
state = state.copy(enrollStatus = result)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -268,23 +393,45 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
||||
val entries = NetworkInventory.snapshot(ctx)
|
||||
val networks = entries.map { it.model }.also { runNetworks = it }
|
||||
|
||||
// What will this run be prevented from measuring? Decided up front, from one throwaway
|
||||
// bind per network, so the document can say so instead of leaving it to be inferred from
|
||||
// per-test `attempted: false` breadcrumbs (measurement-schema.md §3 `constraints`).
|
||||
runConstraints = app.echo_lot.probe.ConstraintDetector.detect(ctx, entries)
|
||||
|
||||
val probes: List<Probe> = listOf(
|
||||
LinkSnapshotProbe(entries),
|
||||
RouterIdentityProbe(entries),
|
||||
IcmpProbe(entries, v6 = false),
|
||||
IcmpProbe(entries, v6 = true),
|
||||
// Folded from the prober after hardware validation: errqueue traceroute (no root,
|
||||
// no JNI) and the mDNS service inventory / VLAN-leakage detector.
|
||||
app.echo_lot.probe.TracerouteProbe(),
|
||||
app.echo_lot.probe.MdnsInventoryProbe(),
|
||||
CaptivePortalProbe(entries),
|
||||
// Canary zone served by the Echolot probe server (probe-protocol §6.1). Hardcoded to
|
||||
// the reference deployment until profiles/enrollment land in the UI.
|
||||
DnsCanaryProbe(canaryZone = "c.echo-lot.app", sessionPrefix = "adhoc"),
|
||||
StunProbe(serverHost = "fmr-1.echo-lot.app"),
|
||||
// Both target whatever server this device is enrolled with, not the deployment the
|
||||
// app happened to be developed against. With no server configured they get blank
|
||||
// strings and report themselves skipped, which is the honest outcome — the
|
||||
// alternative measures someone else's infrastructure and calls it your network.
|
||||
// Before the canary: "can this device resolve at all" has to be answered before
|
||||
// "are the answers being tampered with" means anything.
|
||||
app.echo_lot.probe.DnsResolverProbe(entries),
|
||||
DnsCanaryProbe(canaryZone = settings.canaryZone, sessionPrefix = "adhoc"),
|
||||
StunProbe(serverHost = settings.serverHost()),
|
||||
// Corroboration for icmp.ping6's silence: a real TCP connection over IPv6. Only its
|
||||
// failure, on a network that advertises IPv6, justifies calling IPv6 broken.
|
||||
app.echo_lot.probe.V6ConnectProbe(entries, serverHost = settings.serverHost()),
|
||||
)
|
||||
|
||||
// Plan the run first: the Shizuku battery is counted alongside the app-tier probes so
|
||||
// the bar reflects the whole run. Estimates are per-probe (see Probe.estimatedMs).
|
||||
val shizukuEstimateMs = 8_000L
|
||||
// the bar reflects the whole run. Estimates prefer what THIS device measured on recent
|
||||
// runs (Settings EMA); Probe.estimatedMs is only the cold-start seed — a fixed table
|
||||
// cannot know whether ICMPv6 answers in milliseconds here or waits out its timeout.
|
||||
fun estimateOf(p: Probe) = settings.learnedDurationMs(p.type) ?: p.estimatedMs
|
||||
val shizukuEstimateMs = settings.learnedDurationMs(SHIZUKU_DURATION_KEY) ?: 8_000L
|
||||
val totalSteps = probes.size + 1
|
||||
var remainingMs = probes.sumOf { it.estimatedMs } + shizukuEstimateMs
|
||||
var remainingMs = probes.sumOf { estimateOf(it) } + shizukuEstimateMs
|
||||
state = state.copy(stepsDone = 0, stepsTotal = totalSteps,
|
||||
etaSeconds = ((remainingMs + 999) / 1000).toInt())
|
||||
|
||||
@@ -304,22 +451,33 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
||||
}
|
||||
)
|
||||
tests.add(result); collected.add(result)
|
||||
remainingMs -= p.estimatedMs
|
||||
settings.recordDurationMs(p.type, (result.endedMonoNs - result.startedMonoNs) / 1_000_000)
|
||||
remainingMs -= estimateOf(p)
|
||||
}
|
||||
|
||||
// Shizuku shell tier — self-degrades to UNSUPPORTED when Shizuku isn't running.
|
||||
// Shizuku shell tier — self-degrades to UNSUPPORTED when Shizuku isn't running. One
|
||||
// battery, three tests: the raw captures plus the parsed ra_source/arp_watch views.
|
||||
step("link.ip_monitor (shizuku)", done = probes.size, total = totalSteps, etaMs = shizukuEstimateMs)
|
||||
val shizukuTest = try {
|
||||
val shizukuT0 = System.nanoTime()
|
||||
val shizukuTests = try {
|
||||
ShizukuProbe().run(ctx, ids::uuid, ids::monoNs)
|
||||
} catch (t: Throwable) {
|
||||
Test(
|
||||
listOf(Test(
|
||||
id = ids.uuid(), type = TestType.LINK_IP_MONITOR, tier = Tier.SHIZUKU,
|
||||
startedMonoNs = ids.monoNs(), endedMonoNs = ids.monoNs(),
|
||||
status = TestStatus.FAILED, error = TestError("uncaught", t.message ?: t.javaClass.simpleName),
|
||||
)
|
||||
))
|
||||
}
|
||||
// One key for the whole step: the three tests come out of one shell battery, and the
|
||||
// bar plans them as one step.
|
||||
settings.recordDurationMs(SHIZUKU_DURATION_KEY, (System.nanoTime() - shizukuT0) / 1_000_000)
|
||||
tests.addAll(shizukuTests); collected.addAll(shizukuTests)
|
||||
// "Shizuku tier ran" is the battery's verdict — the derived tests can be PARTIAL on a
|
||||
// perfectly healthy shell tier (e.g. an RA-less v4-only link).
|
||||
runShizukuOk = shizukuTests.any {
|
||||
it.type == TestType.LINK_IP_MONITOR &&
|
||||
(it.status == TestStatus.OK || it.status == TestStatus.PARTIAL)
|
||||
}
|
||||
tests.add(shizukuTest); collected.add(shizukuTest)
|
||||
runShizukuOk = shizukuTest.status == TestStatus.OK || shizukuTest.status == TestStatus.PARTIAL
|
||||
|
||||
return buildDocument(tests)
|
||||
}
|
||||
@@ -340,31 +498,180 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
||||
androidSdk = Build.VERSION.SDK_INT, androidRelease = Build.VERSION.RELEASE,
|
||||
),
|
||||
tiers = Tiers(app = true, shizuku = runShizukuOk),
|
||||
constraints = runConstraints,
|
||||
),
|
||||
networks = runNetworks,
|
||||
tests = tests,
|
||||
findings = findings,
|
||||
summary = Verdicts.derive(tests, findings),
|
||||
summary = Verdicts.derive(tests, findings, runConstraints),
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* Was IPv6 actually provisioned on any network? A global (non-link-local) v6 address or a
|
||||
* v6 default route means the network claims to offer IPv6 — link-local only does not count.
|
||||
* Was IPv6 provisioned on the network a test actually ran over?
|
||||
*
|
||||
* This deliberately asks about one network rather than about the device. Answering "does any
|
||||
* network here have IPv6" produces a real false positive on a phone, and it is not hypothetical:
|
||||
* an IPv4-only wifi with working cellular alongside it reports "IPv6 is configured, but ICMPv6
|
||||
* gets no reply" — configured on cellular, pinged over wifi, and the two never met.
|
||||
*
|
||||
* A global (non-link-local) address or a v6 default route means the network claims to offer
|
||||
* IPv6; link-local only does not count, since every interface has one.
|
||||
*/
|
||||
private fun ipv6Provisioned(networks: List<app.echo_lot.measurement.Network>): Boolean =
|
||||
networks.any { n ->
|
||||
private fun ipv6Provisioned(
|
||||
networks: List<app.echo_lot.measurement.Network>,
|
||||
networkRef: String?,
|
||||
): Boolean {
|
||||
// No reference means the test was not per-network; fall back to the device-wide reading
|
||||
// rather than silently reporting nothing.
|
||||
val scope = networks.filter { networkRef == null || it.id == networkRef }
|
||||
return scope.any { n ->
|
||||
n.link.addresses.any { a ->
|
||||
a.addr.contains(':') &&
|
||||
!a.addr.startsWith("fe80", ignoreCase = true) &&
|
||||
!a.addr.startsWith("::1")
|
||||
} || n.link.routes.any { it.dst == "::/0" }
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Per-network ICMP outcomes, keyed by network id.
|
||||
*
|
||||
* Reads the structured evidence the probe records rather than its prose detail — a finding
|
||||
* that depended on the wording of a human-readable string would break silently the first time
|
||||
* that wording improved.
|
||||
*/
|
||||
/** What one network's ICMP attempt did: whether it ran at all, and whether it was answered. */
|
||||
private data class IcmpOutcome(val attempted: Boolean, val ok: Boolean)
|
||||
|
||||
/**
|
||||
* Per-network ICMP outcomes, keyed by network id.
|
||||
*
|
||||
* Reads the structured evidence the probe records rather than its prose detail — a finding
|
||||
* that depended on the wording of a human-readable string would break silently the first time
|
||||
* that wording improved.
|
||||
*/
|
||||
private fun icmpResults(t: Test): Map<String, IcmpOutcome> {
|
||||
val out = HashMap<String, IcmpOutcome>()
|
||||
val ev = t.evidence ?: return out
|
||||
for ((_, v) in ev) {
|
||||
val o = v as? kotlinx.serialization.json.JsonObject ?: continue
|
||||
val ref = (o["network_ref"] as? kotlinx.serialization.json.JsonPrimitive)?.content ?: continue
|
||||
fun flag(k: String) = (o[k] as? kotlinx.serialization.json.JsonPrimitive)?.content == "true"
|
||||
out[ref] = IcmpOutcome(attempted = flag("attempted"), ok = flag("ok"))
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
/** Human-facing name for the network a test ran over; falls back to something readable. */
|
||||
private fun ifaceOf(networks: List<app.echo_lot.measurement.Network>, ref: String?): String =
|
||||
networks.firstOrNull { it.id == ref }?.iface?.takeIf { it.isNotBlank() } ?: "this network"
|
||||
|
||||
/** Minimal first-pass findings from device-tier evidence; the registry grows with the suite. */
|
||||
private fun deriveFindings(tests: List<Test>, networks: List<app.echo_lot.measurement.Network>): List<Finding> {
|
||||
val out = ArrayList<Finding>()
|
||||
val ids = RunIds()
|
||||
val linkEvidence = tests.filter { it.type == TestType.LINK_SNAPSHOT }.map { EvidenceRef(it.id) }
|
||||
|
||||
// Said as a finding, not only as run.constraints: the constraints block is for machines
|
||||
// aggregating thousands of runs, this is for the person reading this one. Both must exist —
|
||||
// a constrained run with a quiet findings list still reads as "nothing wrong here".
|
||||
if (runConstraints.constrained) {
|
||||
val blocked = runConstraints.unmeasuredNetworks
|
||||
.joinToString(", ") { id -> ifaceOf(networks, id) }
|
||||
.ifBlank { "the underlying networks" }
|
||||
// Three distinct situations share this finding code, and naming the wrong one costs
|
||||
// trust: claiming "a VPN is active" right after the user disconnected theirs is how
|
||||
// this text was first proven wrong on hardware.
|
||||
val (title, description) = when {
|
||||
runConstraints.vpnActive && runConstraints.perNetworkBlocked ->
|
||||
"A VPN is active — $blocked could not be measured" to
|
||||
("Android refuses to let apps send on the networks beneath an active " +
|
||||
"VPN (that is how it prevents traffic leaking around the tunnel), " +
|
||||
"so every per-network test here measured the tunnel or nothing. " +
|
||||
"Nothing in this run says anything about $blocked. To measure them, " +
|
||||
"disconnect the VPN and run again.")
|
||||
runConstraints.perNetworkBlocked ->
|
||||
"The OS refused sends on $blocked" to
|
||||
("Android denied this app permission to send on $blocked (EPERM on " +
|
||||
"bind). That is the restriction a VPN imposes on the networks " +
|
||||
"beneath it — a tunnel disconnected moments ago can still leave it " +
|
||||
"in place while it tears down. Nothing in this run says anything " +
|
||||
"about $blocked; wait a few seconds and run again.")
|
||||
else ->
|
||||
"A VPN holds the default route" to
|
||||
("Everything using the default route in this run describes the tunnel, " +
|
||||
"not the network it rides on. Per-network measurements were " +
|
||||
"permitted and did measure the underlying networks.")
|
||||
}
|
||||
out.add(
|
||||
Finding(
|
||||
id = ids.uuid(),
|
||||
code = FindingRegistry.MEASUREMENT_VPN_CONSTRAINED.code,
|
||||
category = FindingRegistry.MEASUREMENT_VPN_CONSTRAINED.category,
|
||||
severity = FindingRegistry.MEASUREMENT_VPN_CONSTRAINED.severity,
|
||||
confidence = Confidence.HIGH,
|
||||
title = title,
|
||||
description = description,
|
||||
evidenceRefs = linkEvidence,
|
||||
)
|
||||
)
|
||||
}
|
||||
val shapes = V6Analysis.classify(networks)
|
||||
// Named per interface: on a phone several networks are up at once, and "IPv6 is broken" is
|
||||
// useless when wifi is the broken one and cellular is fine.
|
||||
for (sh in shapes.filter { it.addressWithoutRoute }) {
|
||||
val where = if (sh.iface.isBlank()) "This device" else sh.iface
|
||||
out.add(
|
||||
Finding(
|
||||
id = ids.uuid(),
|
||||
code = FindingRegistry.V6_NO_DEFAULT_ROUTE.code,
|
||||
category = FindingRegistry.V6_NO_DEFAULT_ROUTE.category,
|
||||
severity = if (sh.tunnel) Severity.INFO else Severity.MEDIUM,
|
||||
confidence = Confidence.HIGH,
|
||||
title = if (sh.tunnel) {
|
||||
"IPv6 reaches only the destinations a tunnel routes ($where)"
|
||||
} else {
|
||||
"IPv6 address with no default route ($where)"
|
||||
},
|
||||
description = "$where has a global IPv6 address but no IPv6 default route, so " +
|
||||
"IPv6 reaches only destinations covered by a specific route. " +
|
||||
if (sh.tunnel) {
|
||||
"A tunnel interface holds those routes, so this looks deliberate. " +
|
||||
"Worth knowing rather than fixing: applications holding a global " +
|
||||
"address will still try IPv6 first and stall for anything outside " +
|
||||
"the tunnel's routes."
|
||||
} else {
|
||||
"Nothing is routing the rest, so the network handed out an address it " +
|
||||
"does not carry traffic for — applications will try IPv6 first " +
|
||||
"and wait for it to fail."
|
||||
},
|
||||
evidenceRefs = linkEvidence,
|
||||
)
|
||||
)
|
||||
}
|
||||
for (sh in shapes.filter { it.routeWithoutAddress }) {
|
||||
val where = if (sh.iface.isBlank()) "This network" else sh.iface
|
||||
out.add(
|
||||
Finding(
|
||||
id = ids.uuid(),
|
||||
code = FindingRegistry.V6_ROUTE_WITHOUT_ADDRESS.code,
|
||||
category = FindingRegistry.V6_ROUTE_WITHOUT_ADDRESS.category,
|
||||
severity = Severity.MEDIUM,
|
||||
confidence = Confidence.HIGH,
|
||||
title = "IPv6 router advertised, but no address was configured ($where)",
|
||||
description = "$where has an IPv6 default route but no global IPv6 address. " +
|
||||
"The router is advertising itself as an IPv6 gateway while SLAAC produced " +
|
||||
"no usable address — a missing prefix option, a prefix without the " +
|
||||
"autonomous flag, or DHCPv6-only addressing that did not complete. Hosts " +
|
||||
"believe IPv6 is available and pay a connection timeout on every " +
|
||||
"dual-stack destination before falling back to IPv4, which is felt as " +
|
||||
"general slowness with no packet loss to explain it.",
|
||||
evidenceRefs = linkEvidence,
|
||||
)
|
||||
)
|
||||
}
|
||||
|
||||
for (t in tests) {
|
||||
if (t.type == TestType.NET_CAPTIVE_PORTAL) {
|
||||
val ev = t.evidence?.toString() ?: ""
|
||||
@@ -432,31 +739,185 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
||||
)
|
||||
}
|
||||
}
|
||||
if (t.type == TestType.ICMP_PING6 && t.status == TestStatus.FAILED) {
|
||||
// A network with no IPv6 at all is NORMAL — most networks are still IPv4-only,
|
||||
// and that is not a defect. What IS a defect is IPv6 that the network claims to
|
||||
// provide (a global address or a default route from RA/DHCPv6) but that does not
|
||||
// work: that causes Happy-Eyeballs delays, timeouts and hangs. So the severity
|
||||
// depends on whether v6 was provisioned at all.
|
||||
if (ipv6Provisioned(networks)) {
|
||||
out.add(
|
||||
Finding(
|
||||
id = ids.uuid(), code = FindingRegistry.V6_BROKEN.code,
|
||||
category = FindingRegistry.V6_BROKEN.category,
|
||||
severity = FindingRegistry.V6_BROKEN.severity, confidence = Confidence.HIGH,
|
||||
title = "IPv6 is configured but not working",
|
||||
description = "This network advertises IPv6 (a global address and/or a default route), but ICMPv6 got no reply on any network. Half-configured IPv6 is worse than none: connections try IPv6 first and stall before falling back.",
|
||||
evidenceRefs = listOf(EvidenceRef(t.id)),
|
||||
if (t.type == TestType.DNS_RESOLVER && t.status == TestStatus.OK) {
|
||||
// One finding per network: on a phone the wifi resolver can be wedged while
|
||||
// cellular is fine, and "DNS is broken" would be wrong about half the device.
|
||||
val ev = t.evidence
|
||||
if (ev != null) {
|
||||
for ((_, v) in ev) {
|
||||
val o = v as? kotlinx.serialization.json.JsonObject ?: continue
|
||||
fun str(k: String) =
|
||||
(o[k] as? kotlinx.serialization.json.JsonPrimitive)?.content
|
||||
val verdict = str("verdict")
|
||||
val ref0 = str("network_ref")
|
||||
val iface0 = networks.firstOrNull { it.id == ref0 }?.iface
|
||||
?.takeIf { it.isNotBlank() } ?: "this network"
|
||||
if (verdict == "search domain swallows queries") {
|
||||
// Severity follows the harm, not the shape: the same misconfiguration
|
||||
// is fatal on a resolver that tries the search form and invisible on
|
||||
// one that does not, and saying "high" for a network that currently
|
||||
// resolves fine would be crying wolf.
|
||||
val breaking = str("system_resolves") != "true"
|
||||
out.add(
|
||||
Finding(
|
||||
id = ids.uuid(),
|
||||
code = FindingRegistry.DNS_SEARCH_DOMAIN_UNANSWERED.code,
|
||||
category = FindingRegistry.DNS_SEARCH_DOMAIN_UNANSWERED.category,
|
||||
severity = if (breaking) Severity.HIGH else Severity.MEDIUM,
|
||||
confidence = Confidence.HIGH,
|
||||
title = "The network's search domain swallows DNS queries ($iface0)",
|
||||
description = "This network hands out " +
|
||||
"${str("search_domains") ?: "a search domain"} as a DNS " +
|
||||
"search domain, but its server never answers queries under " +
|
||||
"it — not even to say the name does not exist. Resolvers " +
|
||||
"append that domain to lookups, so they wait for a reply " +
|
||||
"that never comes. " +
|
||||
(if (breaking) {
|
||||
"That is why names are not resolving on this device."
|
||||
} else {
|
||||
"Name resolution still works here, because this " +
|
||||
"resolver tries the plain name first — another " +
|
||||
"device on the same network may fail outright."
|
||||
}) +
|
||||
" Fix it on the router: either stop advertising the search " +
|
||||
"domain, or make the server answer for it, including " +
|
||||
"NXDOMAIN for names it does not have. Note that .local is " +
|
||||
"reserved for mDNS (RFC 6762) and is widely dropped by " +
|
||||
"design; home.arpa (RFC 8375) is the name reserved for this.",
|
||||
evidenceRefs = listOf(EvidenceRef(t.id)),
|
||||
)
|
||||
)
|
||||
continue
|
||||
}
|
||||
if (verdict != "server answers, device resolver does not") continue
|
||||
val ref = str("network_ref")
|
||||
val where = networks.firstOrNull { it.id == ref }?.iface
|
||||
?.takeIf { it.isNotBlank() } ?: "this network"
|
||||
out.add(
|
||||
Finding(
|
||||
id = ids.uuid(),
|
||||
code = FindingRegistry.DNS_SYSTEM_RESOLVER_BROKEN.code,
|
||||
category = FindingRegistry.DNS_SYSTEM_RESOLVER_BROKEN.category,
|
||||
severity = FindingRegistry.DNS_SYSTEM_RESOLVER_BROKEN.severity,
|
||||
confidence = Confidence.HIGH,
|
||||
title = "This device cannot resolve names, but the DNS server is fine ($where)",
|
||||
description = "A DNS query sent straight from this device was " +
|
||||
"answered by ${str("servers") ?: "the configured server"} with " +
|
||||
"a valid result, yet asking Android to resolve the same name " +
|
||||
"fails. Whatever is wrong sits between this device's resolver " +
|
||||
"and a server that demonstrably works. " +
|
||||
"Turning wifi off and on, or rejoining the network, clears the " +
|
||||
"common case. If it survives a restart it is not a stuck " +
|
||||
"resolver: look for something on this device that filters DNS " +
|
||||
"— an ad blocker, a private-DNS or VPN app — or a per-device " +
|
||||
"rule on the router aimed at this client.",
|
||||
evidenceRefs = listOf(EvidenceRef(t.id)),
|
||||
)
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
if (t.type == TestType.ICMP_PING6) {
|
||||
// A network with no IPv6 at all is NORMAL — most networks are still IPv4-only, and
|
||||
// that is not a defect. What IS a defect is IPv6 the network claims to provide (a
|
||||
// global address or a default route from RA/DHCPv6) that does not work: that causes
|
||||
// Happy-Eyeballs delays, timeouts and hangs.
|
||||
//
|
||||
// Judged per network, from the per-network evidence rather than the aggregate
|
||||
// status. The aggregate can only say "some network answered", and on a phone with
|
||||
// wifi and cellular up at once that is how "IPv6 is configured but gets no reply"
|
||||
// ends up describing a network where IPv6 was never configured in the first place.
|
||||
val results = icmpResults(t)
|
||||
// The corroborating witness: did a real TCP connection over IPv6 work on this
|
||||
// network? Same evidence shape as the ICMP probe, so the same parser reads it.
|
||||
val v6ConnTest = tests.firstOrNull { it.type == TestType.V6_BROKENNESS }
|
||||
val v6Conn = v6ConnTest?.let { icmpResults(it) } ?: emptyMap()
|
||||
var anyV6Network = false
|
||||
for (n in networks) {
|
||||
val provisioned = ipv6Provisioned(networks, n.id)
|
||||
if (provisioned) anyV6Network = true
|
||||
val r = results[n.id] ?: continue
|
||||
// Silence is only evidence if something was actually sent. A bind that failed
|
||||
// with EPERM says the app could not use the interface, which is a fact about
|
||||
// this app's permissions and says nothing whatsoever about the network.
|
||||
if (!provisioned || !r.attempted || r.ok) continue
|
||||
val where = n.iface?.takeIf { it.isNotBlank() } ?: "this network"
|
||||
val conn = v6Conn[n.id]
|
||||
val evidence = listOfNotNull(
|
||||
EvidenceRef(t.id), v6ConnTest?.let { EvidenceRef(it.id) },
|
||||
)
|
||||
} else {
|
||||
when {
|
||||
// TCP over IPv6 worked: the silence is filtering, and can be said so.
|
||||
conn?.ok == true -> out.add(
|
||||
Finding(
|
||||
id = ids.uuid(), code = FindingRegistry.V6_NO_ICMP_REPLY.code,
|
||||
category = FindingRegistry.V6_NO_ICMP_REPLY.category,
|
||||
severity = FindingRegistry.V6_NO_ICMP_REPLY.severity,
|
||||
confidence = Confidence.HIGH,
|
||||
title = "ICMPv6 is filtered here — IPv6 itself works ($where)",
|
||||
description = "$where answered a real TCP connection over IPv6, " +
|
||||
"so IPv6 works — but ICMPv6 echo got no reply, so something " +
|
||||
"on this network filters ICMPv6. That is a fault in its own " +
|
||||
"right even though connections succeed: Path MTU Discovery " +
|
||||
"depends on ICMPv6, so large packets can vanish rather than " +
|
||||
"being reported as too big.",
|
||||
evidenceRefs = evidence,
|
||||
)
|
||||
)
|
||||
// Both transports failed on a network that advertises IPv6: broken, and
|
||||
// now with the evidence the original v6.broken never had.
|
||||
conn != null && conn.attempted -> out.add(
|
||||
Finding(
|
||||
id = ids.uuid(), code = FindingRegistry.V6_BROKEN.code,
|
||||
category = FindingRegistry.V6_BROKEN.category,
|
||||
severity = FindingRegistry.V6_BROKEN.severity,
|
||||
confidence = Confidence.HIGH,
|
||||
title = "IPv6 is advertised but does not work ($where)",
|
||||
description = "$where advertises IPv6 (a global address and/or a " +
|
||||
"default route), but neither ICMPv6 echo nor a TCP connection " +
|
||||
"over IPv6 got through — two independent transports, both " +
|
||||
"silent. Applications will try IPv6 first and wait out a " +
|
||||
"timeout on every dual-stack destination before falling back " +
|
||||
"to IPv4, felt as everything being slow with no loss to " +
|
||||
"explain it. The network is announcing a service it does not " +
|
||||
"deliver; the fix belongs on the router or upstream.",
|
||||
evidenceRefs = evidence,
|
||||
)
|
||||
)
|
||||
// No corroboration available (no server configured, or the connect never
|
||||
// got as far as sending): the honest two-explanation reading stands.
|
||||
else -> out.add(
|
||||
Finding(
|
||||
id = ids.uuid(), code = FindingRegistry.V6_NO_ICMP_REPLY.code,
|
||||
category = FindingRegistry.V6_NO_ICMP_REPLY.category,
|
||||
severity = FindingRegistry.V6_NO_ICMP_REPLY.severity,
|
||||
confidence = Confidence.MEDIUM,
|
||||
title = "IPv6 is configured, but ICMPv6 gets no reply ($where)",
|
||||
description = "$where advertises IPv6 (a global address and/or a " +
|
||||
"default route), but ICMPv6 echo got no reply over it. That has " +
|
||||
"two explanations which look identical from here: IPv6 is broken, " +
|
||||
"or ICMPv6 is filtered while IPv6 itself works. Filtering is " +
|
||||
"common and is a fault in its own right — it breaks Path MTU " +
|
||||
"Discovery, so large packets vanish rather than being reported as " +
|
||||
"too big.",
|
||||
evidenceRefs = listOf(EvidenceRef(t.id)),
|
||||
)
|
||||
)
|
||||
}
|
||||
}
|
||||
if (!anyV6Network) {
|
||||
// Said once for the device, not once per interface: "this network is IPv4-only"
|
||||
// repeated per interface reads as several problems instead of one observation.
|
||||
out.add(
|
||||
Finding(
|
||||
id = ids.uuid(), code = FindingRegistry.V6_NOT_OFFERED.code,
|
||||
category = FindingRegistry.V6_NOT_OFFERED.category,
|
||||
severity = FindingRegistry.V6_NOT_OFFERED.severity, confidence = Confidence.HIGH,
|
||||
severity = FindingRegistry.V6_NOT_OFFERED.severity,
|
||||
confidence = Confidence.HIGH,
|
||||
title = "IPv4-only network (no IPv6 offered)",
|
||||
description = "No IPv6 address or default route was provisioned, so IPv6 tests could not run. This is normal — many networks are still IPv4-only and it is not a fault.",
|
||||
description = "No IPv6 address or default route was provisioned on " +
|
||||
"any active network, so IPv6 tests could not run. This is normal " +
|
||||
"— many networks are still IPv4-only and it is not a fault.",
|
||||
evidenceRefs = listOf(EvidenceRef(t.id)),
|
||||
)
|
||||
)
|
||||
@@ -472,4 +933,9 @@ class RunViewModel(app: Application) : AndroidViewModel(app) {
|
||||
etaSeconds = if (etaMs >= 0) ((etaMs + 999) / 1000).toInt() else state.etaSeconds,
|
||||
)
|
||||
}
|
||||
|
||||
private companion object {
|
||||
/** Duration-learning key for the Shizuku step, which is three tests but one battery. */
|
||||
const val SHIZUKU_DURATION_KEY = "shizuku.battery"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -50,6 +50,35 @@ class Settings(context: Context) {
|
||||
maxTotalBytes = maxTotalMb.toLong() * 1024 * 1024,
|
||||
)
|
||||
|
||||
// ---- run-duration learning ---------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Learned duration of one test type on THIS device, or null before the first run.
|
||||
*
|
||||
* The static Probe.estimatedMs values are only cold-start seeds: real durations depend on
|
||||
* the phone and the network it stands in (ICMPv6 answers in milliseconds where IPv6 works
|
||||
* and waits out full timeouts where it does not), so a fixed table is wrong for almost
|
||||
* everyone almost always. What was measured last time is the only estimate that tracks
|
||||
* reality.
|
||||
*/
|
||||
fun learnedDurationMs(type: String): Long? =
|
||||
prefs.getLong("$DURATION_PREFIX$type", -1L).takeIf { it > 0 }
|
||||
|
||||
/**
|
||||
* Feeds one measured duration into the estimate — EMA, 70 % old / 30 % new. Heavy enough
|
||||
* on history that a single odd run (a captive portal stalling DNS) does not whipsaw the
|
||||
* bar, light enough that a real change (enrolling with a server un-skips three probes)
|
||||
* converges within a few runs. Recorded whatever the test's status: a probe that skips in
|
||||
* 2 ms will keep skipping in 2 ms until circumstances change, and then the EMA follows.
|
||||
*/
|
||||
fun recordDurationMs(type: String, ms: Long) {
|
||||
if (ms < 0) return
|
||||
val key = "$DURATION_PREFIX$type"
|
||||
val old = prefs.getLong(key, -1L)
|
||||
val next = if (old <= 0) ms else (old * 7 + ms * 3) / 10
|
||||
prefs.edit().putLong(key, next).apply()
|
||||
}
|
||||
|
||||
// ---- upload ----------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@@ -97,6 +126,41 @@ class Settings(context: Context) {
|
||||
get() = prefs.getString(SERVER_PIN, "") ?: ""
|
||||
set(v) = prefs.edit().putString(SERVER_PIN, v.trim()).apply()
|
||||
|
||||
/**
|
||||
* The address the operator handed out, for showing to a person.
|
||||
*
|
||||
* Separate from [serverUrl], which is the endpoint actually dialled. They differ when the
|
||||
* server publishes one public name and points devices at another to select its pinned
|
||||
* certificate — a detail worth keeping out of the user's face but not out of the settings.
|
||||
*/
|
||||
var serverPublicUrl: String
|
||||
get() = (prefs.getString(SERVER_PUBLIC_URL, "") ?: "").ifBlank { serverUrl }
|
||||
set(v) = prefs.edit().putString(SERVER_PUBLIC_URL, v.trim()).apply()
|
||||
|
||||
/**
|
||||
* What the server said about itself, last time it was asked: addresses, ports, capabilities.
|
||||
*
|
||||
* Cached as a rendered block rather than as fields, because it is shown and never acted on —
|
||||
* these are facts to read, not settings to apply, and storing them as settings would invite
|
||||
* exactly the confusion of an editable box that changes nothing.
|
||||
*/
|
||||
var serverFacts: String
|
||||
get() = prefs.getString(SERVER_FACTS, "") ?: ""
|
||||
set(v) = prefs.edit().putString(SERVER_FACTS, v).apply()
|
||||
|
||||
/**
|
||||
* The server's own addresses, learned from its profile, for reaching it when DNS will not.
|
||||
*
|
||||
* Only the primaries: the alternate pair exists for NAT behaviour discovery and does not carry
|
||||
* the control plane, so falling back to one would fail for a second, unrelated reason.
|
||||
*/
|
||||
var serverAddrs: String
|
||||
get() = prefs.getString(SERVER_ADDRS, "") ?: ""
|
||||
set(v) = prefs.edit().putString(SERVER_ADDRS, v).apply()
|
||||
|
||||
fun serverAddrList(): List<String> =
|
||||
serverAddrs.split(',').map { it.trim() }.filter { it.isNotEmpty() }
|
||||
|
||||
var serverCredential: String
|
||||
get() = prefs.getString(SERVER_CRED, "") ?: ""
|
||||
set(v) = prefs.edit().putString(SERVER_CRED, v.trim()).apply()
|
||||
@@ -104,6 +168,58 @@ class Settings(context: Context) {
|
||||
val serverConfigured: Boolean
|
||||
get() = serverUrl.isNotBlank() && serverPin.isNotBlank() && serverCredential.isNotBlank()
|
||||
|
||||
/**
|
||||
* The DNS zone this server is authoritative for, learned from its profile.
|
||||
*
|
||||
* Cached because the canary probe runs at device tier, before anything has talked to the
|
||||
* server, and a probe that had to make a control-plane call first would fail on exactly the
|
||||
* networks worth measuring. Empty means "not known yet", and the probe reports itself as
|
||||
* skipped rather than inventing a zone.
|
||||
*/
|
||||
var canaryZone: String
|
||||
get() = prefs.getString(CANARY_ZONE, "") ?: ""
|
||||
set(v) = prefs.edit().putString(CANARY_ZONE, v.trim()).apply()
|
||||
|
||||
/**
|
||||
* Host part of the configured server URL, for probes that address it directly (STUN).
|
||||
*
|
||||
* Derived rather than stored: a second copy of the server's name is a second thing to keep in
|
||||
* step, and it would go stale the moment someone re-enrolled against a different server.
|
||||
*/
|
||||
fun serverHost(): String = runCatching {
|
||||
java.net.URI(serverUrl).host?.takeIf { it.isNotBlank() }
|
||||
}.getOrNull() ?: ""
|
||||
|
||||
// ---- account ---------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* The PKCE verifier and state for a sign-in that is out at the browser.
|
||||
*
|
||||
* Persisted rather than held in memory because handing control to a browser backgrounds this
|
||||
* process, and Android may kill it before the callback returns. An in-memory value works on a
|
||||
* developer's device and fails on a phone under memory pressure.
|
||||
*/
|
||||
var pendingVerifier: String
|
||||
get() = prefs.getString(PENDING_VERIFIER, "") ?: ""
|
||||
set(v) = prefs.edit().putString(PENDING_VERIFIER, v).apply()
|
||||
|
||||
var pendingState: String
|
||||
get() = prefs.getString(PENDING_STATE, "") ?: ""
|
||||
set(v) = prefs.edit().putString(PENDING_STATE, v).apply()
|
||||
|
||||
fun clearPendingAuth() = prefs.edit().remove(PENDING_VERIFIER).remove(PENDING_STATE).apply()
|
||||
|
||||
/** Display name of whoever is signed in on this device; empty when nobody is. */
|
||||
var accountName: String
|
||||
get() = prefs.getString(ACCOUNT_NAME, "") ?: ""
|
||||
set(v) = prefs.edit().putString(ACCOUNT_NAME, v).apply()
|
||||
|
||||
var accountId: String
|
||||
get() = prefs.getString(ACCOUNT_ID, "") ?: ""
|
||||
set(v) = prefs.edit().putString(ACCOUNT_ID, v).apply()
|
||||
|
||||
val signedIn: Boolean get() = accountName.isNotBlank()
|
||||
|
||||
private fun hex(s: String) = ByteArray(s.length / 2) {
|
||||
((Character.digit(s[it * 2], 16) shl 4) or Character.digit(s[it * 2 + 1], 16)).toByte()
|
||||
}
|
||||
@@ -120,5 +236,14 @@ class Settings(context: Context) {
|
||||
const val SERVER_URL = "server_url"
|
||||
const val SERVER_PIN = "server_pin"
|
||||
const val SERVER_CRED = "server_credential"
|
||||
const val SERVER_PUBLIC_URL = "server_public_url"
|
||||
const val SERVER_FACTS = "server_facts"
|
||||
const val SERVER_ADDRS = "server_addrs"
|
||||
const val CANARY_ZONE = "server_canary_zone"
|
||||
const val PENDING_VERIFIER = "pending_auth_verifier"
|
||||
const val PENDING_STATE = "pending_auth_state"
|
||||
const val ACCOUNT_NAME = "account_name"
|
||||
const val ACCOUNT_ID = "account_id"
|
||||
const val DURATION_PREFIX = "duration_ms."
|
||||
}
|
||||
}
|
||||
|
||||
@@ -10,7 +10,9 @@ import androidx.compose.foundation.layout.Row
|
||||
import androidx.compose.foundation.layout.Spacer
|
||||
import androidx.compose.foundation.layout.fillMaxWidth
|
||||
import androidx.compose.foundation.layout.height
|
||||
import androidx.compose.foundation.layout.width
|
||||
import androidx.compose.foundation.layout.padding
|
||||
import androidx.compose.foundation.shape.RoundedCornerShape
|
||||
import androidx.compose.foundation.rememberScrollState
|
||||
import androidx.compose.foundation.verticalScroll
|
||||
import androidx.compose.material3.Button
|
||||
@@ -19,6 +21,7 @@ import androidx.compose.material3.FilterChip
|
||||
import androidx.compose.material3.LocalContentColor
|
||||
import androidx.compose.material3.MaterialTheme
|
||||
import androidx.compose.material3.OutlinedTextField
|
||||
import androidx.compose.material3.Surface
|
||||
import androidx.compose.material3.Switch
|
||||
import androidx.compose.material3.Text
|
||||
import androidx.compose.material3.TextButton
|
||||
@@ -31,6 +34,7 @@ import androidx.compose.ui.Alignment
|
||||
import androidx.compose.ui.Modifier
|
||||
import androidx.compose.ui.text.font.FontFamily
|
||||
import androidx.compose.ui.unit.dp
|
||||
import androidx.compose.ui.unit.sp
|
||||
import app.echo_lot.privacy.PrivacyLevel
|
||||
|
||||
/**
|
||||
@@ -49,8 +53,12 @@ fun SettingsScreen(
|
||||
onDeleteAll: () -> Unit,
|
||||
onPreviewUpload: () -> Unit,
|
||||
onCheckServer: () -> Unit,
|
||||
accountName: String,
|
||||
onSignIn: () -> Unit,
|
||||
onSignOut: () -> Unit,
|
||||
onEnroll: (String) -> Unit,
|
||||
serverStatus: String?,
|
||||
enrollStatus: String?,
|
||||
onBack: () -> Unit,
|
||||
) {
|
||||
// SharedPreferences is not observable, so mirror each value into Compose state and write
|
||||
@@ -63,9 +71,22 @@ fun SettingsScreen(
|
||||
var privacy by remember { mutableStateOf(settings.privacyLevel) }
|
||||
var stableSalt by remember { mutableStateOf(settings.stableSalt) }
|
||||
var enrollLink by remember { mutableStateOf("") }
|
||||
var serverUrl by remember { mutableStateOf(settings.serverUrl) }
|
||||
// The public name, which is what the operator handed out and what a person recognises. The
|
||||
// endpoint actually dialled is shown beneath it when the two differ, rather than hidden — a
|
||||
// network engineer debugging a connection wants to see where it really goes.
|
||||
var serverUrl by remember { mutableStateOf(settings.serverPublicUrl) }
|
||||
var serverPin by remember { mutableStateOf(settings.serverPin) }
|
||||
var serverCred by remember { mutableStateOf(settings.serverCredential) }
|
||||
// Enrolling is asynchronous, so these are re-read when its result lands rather than when the
|
||||
// button is pressed — reading them immediately showed the previous server's values and looked
|
||||
// exactly like an enrollment that had silently done nothing.
|
||||
var serverFacts by remember { mutableStateOf(settings.serverFacts) }
|
||||
androidx.compose.runtime.LaunchedEffect(enrollStatus, serverStatus) {
|
||||
serverFacts = settings.serverFacts
|
||||
serverUrl = settings.serverPublicUrl
|
||||
serverPin = settings.serverPin
|
||||
serverCred = settings.serverCredential
|
||||
}
|
||||
|
||||
Column(
|
||||
Modifier.fillMaxWidth().safeDrawingPadding().verticalScroll(rememberScrollState()).padding(16.dp),
|
||||
@@ -152,6 +173,38 @@ fun SettingsScreen(
|
||||
}
|
||||
}
|
||||
|
||||
// ---- account ----
|
||||
Card(Modifier.fillMaxWidth()) {
|
||||
Column(Modifier.padding(14.dp), verticalArrangement = Arrangement.spacedBy(8.dp)) {
|
||||
Text("Account", style = MaterialTheme.typography.titleMedium)
|
||||
if (accountName.isNotBlank()) {
|
||||
Text("Signed in as $accountName", style = MaterialTheme.typography.bodyMedium)
|
||||
Text(
|
||||
"Runs from every device signed in to this account share one history.",
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
)
|
||||
TextButton(onClick = onSignOut) { Text("Sign out") }
|
||||
} else {
|
||||
Text(
|
||||
"Signing in is optional. It links this device to an account on your " +
|
||||
"server, so several devices share one history — and some servers only " +
|
||||
"accept uploads from a signed-in device.",
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
)
|
||||
Button(onClick = onSignIn, enabled = settings.serverConfigured) {
|
||||
Text("Sign in")
|
||||
}
|
||||
if (!settings.serverConfigured) {
|
||||
Text(
|
||||
"Enrol with a server first — the account belongs to the server, not " +
|
||||
"to the app.",
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---- upload ----
|
||||
Card(Modifier.fillMaxWidth()) {
|
||||
Column(Modifier.padding(14.dp), verticalArrangement = Arrangement.spacedBy(8.dp)) {
|
||||
@@ -183,17 +236,38 @@ fun SettingsScreen(
|
||||
onClick = {
|
||||
onEnroll(enrollLink)
|
||||
enrollLink = "" // spent either way; leaving it around invites a retry
|
||||
serverUrl = settings.serverUrl
|
||||
serverPin = settings.serverPin
|
||||
serverCred = settings.serverCredential
|
||||
},
|
||||
enabled = enrollLink.isNotBlank(),
|
||||
) { Text("Enroll") }
|
||||
// Beside the button that caused it. Enrolling is asynchronous, so without this the
|
||||
// only sign of success is three fields quietly changing further down the card.
|
||||
enrollStatus?.let {
|
||||
Text(it, style = MaterialTheme.typography.bodySmall)
|
||||
}
|
||||
|
||||
OutlinedTextField(
|
||||
value = serverUrl, onValueChange = { serverUrl = it; settings.serverUrl = it },
|
||||
value = serverUrl,
|
||||
onValueChange = {
|
||||
serverUrl = it
|
||||
// Typed by hand there is no discovery to consult, so what was entered is
|
||||
// both the public name and the endpoint. Setting only one of them would
|
||||
// leave the app dialling the previous server.
|
||||
settings.serverUrl = it
|
||||
settings.serverPublicUrl = it
|
||||
},
|
||||
label = { Text("Server URL") }, singleLine = true, modifier = Modifier.fillMaxWidth(),
|
||||
)
|
||||
// Directly under the field it explains. Anywhere else it reads as a stray sentence
|
||||
// about some other part of the screen.
|
||||
if (settings.serverUrl.isNotBlank() && settings.serverUrl != settings.serverPublicUrl) {
|
||||
Text(
|
||||
"Connects to ${settings.serverUrl} — this server publishes one name and " +
|
||||
"points devices at another, so its pinned certificate can share a port " +
|
||||
"with its web interface.",
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
color = LocalContentColor.current.copy(alpha = 0.7f),
|
||||
)
|
||||
}
|
||||
OutlinedTextField(
|
||||
value = serverPin, onValueChange = { serverPin = it; settings.serverPin = it },
|
||||
label = { Text("Certificate pin (SPKI, base64)") }, singleLine = true,
|
||||
@@ -220,6 +294,52 @@ fun SettingsScreen(
|
||||
serverStatus?.let {
|
||||
Text(it, style = MaterialTheme.typography.bodySmall)
|
||||
}
|
||||
// What the server reported, placed under the button that asks it rather than among
|
||||
// the fields above: these are facts to read, not settings to apply, and an
|
||||
// editable-looking box that changes nothing is worse than no box at all.
|
||||
//
|
||||
// Monospaced so the addresses line up under each other — column alignment is most
|
||||
// of what makes a list of IPs quicker to read than prose.
|
||||
if (serverFacts.isNotBlank()) {
|
||||
Surface(
|
||||
color = MaterialTheme.colorScheme.surfaceVariant,
|
||||
shape = RoundedCornerShape(8.dp),
|
||||
modifier = Modifier.fillMaxWidth(),
|
||||
) {
|
||||
Column(
|
||||
Modifier.padding(horizontal = 12.dp, vertical = 10.dp),
|
||||
verticalArrangement = Arrangement.spacedBy(2.dp),
|
||||
) {
|
||||
Text(
|
||||
"WHAT THIS SERVER REPORTS",
|
||||
style = MaterialTheme.typography.labelSmall,
|
||||
color = LocalContentColor.current.copy(alpha = 0.7f),
|
||||
)
|
||||
// Real columns rather than padded text: the label column has a fixed
|
||||
// width, so values line up whatever the font does, and a long value
|
||||
// wraps inside its own column instead of under the labels.
|
||||
for (line in serverFacts.lines()) {
|
||||
val label = line.substringBefore('|')
|
||||
val value = line.substringAfter('|', "")
|
||||
Row(Modifier.fillMaxWidth()) {
|
||||
Text(
|
||||
label,
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
color = LocalContentColor.current.copy(alpha = 0.7f),
|
||||
modifier = Modifier.width(72.dp),
|
||||
)
|
||||
Text(
|
||||
value,
|
||||
style = MaterialTheme.typography.bodySmall.copy(
|
||||
fontFamily = FontFamily.Monospace,
|
||||
),
|
||||
modifier = Modifier.weight(1f),
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Text(
|
||||
"This app is ${BuildConfig.APP_SEMVER} and speaks probe protocol " +
|
||||
"${app.echo_lot.protocol.Compat.PROTOCOL_VERSION}. It works with servers " +
|
||||
|
||||
+20
-3
@@ -168,6 +168,10 @@ class DownstreamMeasurement(private val ids: IdSource) {
|
||||
trainCount: Int = 100,
|
||||
trainSizeBytes: Int = 300,
|
||||
trainIntervalUs: Int = 3_000,
|
||||
// DSCP to mark the train with (0-63), or -1 to leave packets unmarked. Pairing a marked
|
||||
// downtrain with the server-observed DSCP of an upstream train is the two-direction
|
||||
// sec.dscp_ecn_survival measurement.
|
||||
trainDscp: Int = -1,
|
||||
): Pair<List<Test>, List<Finding>> {
|
||||
val tests = ArrayList<Test>()
|
||||
val findings = ArrayList<Finding>()
|
||||
@@ -175,7 +179,8 @@ class DownstreamMeasurement(private val ids: IdSource) {
|
||||
val df = bigSend(credential, sessionId, control, probe, sessionRef, sizes, df = true)
|
||||
val frag = bigSend(credential, sessionId, control, probe, sessionRef, sizes, df = false)
|
||||
val train = downTrain(
|
||||
credential, sessionId, control, probe, sessionRef, trainCount, trainSizeBytes, trainIntervalUs,
|
||||
credential, sessionId, control, probe, sessionRef, trainCount, trainSizeBytes,
|
||||
trainIntervalUs, trainDscp,
|
||||
)
|
||||
|
||||
tests.add(df.test); tests.add(frag.test); tests.add(train.test)
|
||||
@@ -338,15 +343,19 @@ class DownstreamMeasurement(private val ids: IdSource) {
|
||||
|
||||
private fun downTrain(
|
||||
credential: String, sessionId: String, control: ControlClient, probe: ProbeSession,
|
||||
sessionRef: String, count: Int, sizeBytes: Int, intervalUs: Int,
|
||||
sessionRef: String, count: Int, sizeBytes: Int, intervalUs: Int, dscp: Int = -1,
|
||||
): TrainResult {
|
||||
val testId = ids.uuid()
|
||||
val started = ids.monoNs()
|
||||
|
||||
// dscp is only sent when requested: an older server rejects unknown-value problems
|
||||
// louder than absent keys, and unmarked is the correct default for a plain loss train.
|
||||
val dscpField = if (dscp in 0..63) ""","dscp":$dscp""" else ""
|
||||
val reply = runCatching {
|
||||
control.action(
|
||||
credential, sessionId,
|
||||
"""{"action":"downtrain","count":$count,"size_bytes":$sizeBytes,"interval_us":$intervalUs}""",
|
||||
"""{"action":"downtrain","count":$count,"size_bytes":$sizeBytes,""" +
|
||||
""""interval_us":$intervalUs$dscpField}""",
|
||||
)
|
||||
}
|
||||
if (reply.isFailure) {
|
||||
@@ -394,6 +403,12 @@ class DownstreamMeasurement(private val ids: IdSource) {
|
||||
interArrivalMsAvg = interArrival.average().takeIf { interArrival.isNotEmpty() }?.let(::round1),
|
||||
interArrivalMsMax = interArrival.maxOrNull()?.let(::round1),
|
||||
sendIntervalUs = intervalUs,
|
||||
dscpRequested = dscp.takeIf { it in 0..63 },
|
||||
// The server says whether it could actually mark (dscp_applied); recorded so a
|
||||
// survival comparison never blames the path for a marking the sender skipped.
|
||||
dscpApplied = reply.getOrNull()?.let {
|
||||
Regex("\"dscp_applied\"\\s*:\\s*(true|false)").find(it)?.groupValues?.get(1)?.toBoolean()
|
||||
},
|
||||
),
|
||||
) as JsonObject
|
||||
|
||||
@@ -490,4 +505,6 @@ data class DownTrainMetrics(
|
||||
@SerialName("inter_arrival_ms_avg") val interArrivalMsAvg: Double? = null,
|
||||
@SerialName("inter_arrival_ms_max") val interArrivalMsMax: Double? = null,
|
||||
@SerialName("send_interval_us") val sendIntervalUs: Int,
|
||||
@SerialName("dscp_requested") val dscpRequested: Int? = null,
|
||||
@SerialName("dscp_applied") val dscpApplied: Boolean? = null,
|
||||
)
|
||||
|
||||
@@ -88,6 +88,14 @@ class ServerMeasurement(
|
||||
tests.add(test)
|
||||
allFindings.addAll(findings)
|
||||
|
||||
// The upstream train needs no grant and no capability beyond udp-probe itself; a
|
||||
// server that predates trains simply never answers the report request, which the
|
||||
// measurement reports as exactly that ambiguity rather than as network loss.
|
||||
val (utTest, utFindings) = UpstreamTrainMeasurement(ids)
|
||||
.run(ps, sessionRef = "sess-1")
|
||||
tests.add(utTest)
|
||||
allFindings.addAll(utFindings)
|
||||
|
||||
// Downstream needs a session the server has already seen traffic from — the echo
|
||||
// train just provided that — and a server that advertises the grants. Skipped
|
||||
// quietly against an older server rather than reported as a failure of the network.
|
||||
|
||||
+159
@@ -0,0 +1,159 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package app.echo_lot.engine
|
||||
|
||||
import app.echo_lot.measurement.*
|
||||
import app.echo_lot.protocol.ProbeSession
|
||||
import kotlinx.serialization.SerialName
|
||||
import kotlinx.serialization.Serializable
|
||||
import kotlinx.serialization.json.Json
|
||||
import kotlinx.serialization.json.JsonObject
|
||||
import kotlinx.serialization.json.encodeToJsonElement
|
||||
|
||||
/**
|
||||
* train.udp_updown — the client sends a paced train (types 0x03), then asks the server what
|
||||
* arrived (0x04 → 0x05) and lines both views up per sequence number.
|
||||
*
|
||||
* This is the measurement a round trip cannot make: an echo run only says "lost somewhere", the
|
||||
* train's two ledgers say lost on the way OUT, specifically, because the server's report names
|
||||
* exactly which sequence numbers reached it. The downstream direction has its own test
|
||||
* (train.udp_downstream) under a grant; this one needs none, since the client generates all the
|
||||
* traffic itself.
|
||||
*
|
||||
* The evidence is the schema's columnar TrainEvidence: one index per sent packet, with the
|
||||
* server-side columns null where a packet never arrived. Server timestamps are on the server's
|
||||
* own clock — only differences within that clock mean anything unless time.server_offset maps
|
||||
* them (two-clock rule).
|
||||
*/
|
||||
class UpstreamTrainMeasurement(private val ids: IdSource) {
|
||||
|
||||
private val json = Json { encodeDefaults = true; explicitNulls = true }
|
||||
|
||||
fun run(
|
||||
probe: ProbeSession,
|
||||
sessionRef: String,
|
||||
count: Int = 200,
|
||||
sizeBytes: Int = 200,
|
||||
interPacketMs: Long = 5,
|
||||
): Pair<Test, List<Finding>> {
|
||||
val testId = ids.uuid()
|
||||
val started = ids.monoNs()
|
||||
// The id only needs to be unique within this session; a clash across sessions is
|
||||
// meaningless because trains are buffered per session on the server.
|
||||
val trainId = (System.nanoTime() and 0x7FFFFFFF).toInt()
|
||||
|
||||
val sent = probe.sendTrain(trainId, count, sizeBytes, interPacketMs)
|
||||
// Let the tail arrive before asking for the ledger; packets still in flight when the
|
||||
// report is cut would read as upstream loss.
|
||||
Thread.sleep(300)
|
||||
val report = probe.trainReport(trainId)
|
||||
|
||||
if (report == null) {
|
||||
return Test(
|
||||
id = testId, type = TestType.TRAIN_UDP_UPDOWN, sessionRef = sessionRef,
|
||||
tier = Tier.APP, startedMonoNs = started, endedMonoNs = ids.monoNs(),
|
||||
status = TestStatus.FAILED,
|
||||
// Honest ambiguity: an old server drops 0x04 silently, and a lost report looks
|
||||
// identical from here. Neither says anything about the train itself.
|
||||
error = TestError(
|
||||
"no_report",
|
||||
"no train report arrived — the report was lost, or the server predates trains",
|
||||
),
|
||||
) to emptyList()
|
||||
}
|
||||
|
||||
val bySeq = report.rows.associateBy { it.seq }
|
||||
fun col255(v: Int): Int? = v.takeIf { it != 255 } // 255 = "not observed" on the wire
|
||||
|
||||
val evidence = TrainEvidence(
|
||||
epochMonoNs = started,
|
||||
seq = sent.map { it.seq },
|
||||
tTxNs = sent.map { it.tTxNs },
|
||||
tSrvRxNs = sent.map { bySeq[it.seq]?.tRxNs },
|
||||
tRxNs = sent.map { null }, // upstream only: nothing comes back per packet
|
||||
sizeBytes = sent.map { it.sizeBytes },
|
||||
ttlSeenByServer = sent.map { bySeq[it.seq]?.let { r -> col255(r.ttl) } },
|
||||
dscpSeenByServer = sent.map { bySeq[it.seq]?.let { r -> col255(r.dscp) } },
|
||||
ecnSeenByServer = sent.map { bySeq[it.seq]?.let { r -> col255(r.ecn) } },
|
||||
evidenceTruncated = report.truncated,
|
||||
).toEvidence()
|
||||
|
||||
// Loss against the server's total count, not its row list: rows past the server's buffer
|
||||
// cap are counted but not kept, and treating them as lost would invent loss exactly on
|
||||
// the biggest trains.
|
||||
val lossPct = if (sent.isEmpty()) 0.0 else {
|
||||
(sent.size - report.received).coerceAtLeast(0) * 100.0 / sent.size
|
||||
}
|
||||
val metrics = json.encodeToJsonElement(
|
||||
UpstreamTrainMetrics(
|
||||
sent = sent.size,
|
||||
receivedByServer = report.received,
|
||||
lossPct = round1(lossPct),
|
||||
reportPartsExpected = report.partsExpected,
|
||||
reportPartsReceived = report.partsReceived,
|
||||
truncated = report.truncated,
|
||||
),
|
||||
) as JsonObject
|
||||
|
||||
val findings = ArrayList<Finding>()
|
||||
if (sent.isNotEmpty() && report.received == 0) {
|
||||
findings.add(
|
||||
Finding(
|
||||
id = ids.uuid(),
|
||||
code = FindingRegistry.UDP_UNREACHABLE_UPSTREAM.code,
|
||||
category = FindingRegistry.UDP_UNREACHABLE_UPSTREAM.category,
|
||||
severity = FindingRegistry.UDP_UNREACHABLE_UPSTREAM.severity,
|
||||
confidence = Confidence.HIGH,
|
||||
title = "The server received none of ${sent.size} upstream packets",
|
||||
description = "Every train packet vanished on the way out, while the " +
|
||||
"report request's reply made it back — the outbound path drops this " +
|
||||
"traffic, the return path works.",
|
||||
evidenceRefs = listOf(EvidenceRef(testId)),
|
||||
),
|
||||
)
|
||||
} else if (lossPct >= 2.0) {
|
||||
findings.add(
|
||||
Finding(
|
||||
id = ids.uuid(),
|
||||
code = FindingRegistry.LOSS_UPSTREAM.code,
|
||||
category = FindingRegistry.LOSS_UPSTREAM.category,
|
||||
severity = FindingRegistry.LOSS_UPSTREAM.severity,
|
||||
confidence = Confidence.HIGH,
|
||||
title = "Upstream loss of ${round1(lossPct)} %",
|
||||
description = "The server received ${report.received} of the ${sent.size} " +
|
||||
"packets this device sent, and its per-sequence ledger names the " +
|
||||
"missing ones. This is outbound loss specifically; the return path " +
|
||||
"delivered the report.",
|
||||
evidenceRefs = listOf(EvidenceRef(testId)),
|
||||
),
|
||||
)
|
||||
}
|
||||
|
||||
val status = when {
|
||||
report.received == 0 && sent.isNotEmpty() -> TestStatus.FAILED
|
||||
report.partsReceived < report.partsExpected -> TestStatus.PARTIAL
|
||||
else -> TestStatus.OK
|
||||
}
|
||||
return Test(
|
||||
id = testId, type = TestType.TRAIN_UDP_UPDOWN, sessionRef = sessionRef, tier = Tier.APP,
|
||||
startedMonoNs = started, endedMonoNs = ids.monoNs(),
|
||||
status = status, evidence = evidence, metrics = metrics,
|
||||
) to findings
|
||||
}
|
||||
|
||||
private fun round1(v: Double) = Math.round(v * 10.0) / 10.0
|
||||
}
|
||||
|
||||
/** Metrics for train.udp_updown. */
|
||||
@Serializable
|
||||
data class UpstreamTrainMetrics(
|
||||
val sent: Int,
|
||||
/** The server's total count — includes packets past its row buffer (counted, not listed). */
|
||||
@SerialName("received_by_server") val receivedByServer: Int,
|
||||
@SerialName("loss_pct") val lossPct: Double,
|
||||
@SerialName("report_parts_expected") val reportPartsExpected: Int,
|
||||
@SerialName("report_parts_received") val reportPartsReceived: Int,
|
||||
/** The server's row buffer overflowed: rows are a sample, the count is still complete. */
|
||||
val truncated: Boolean,
|
||||
)
|
||||
@@ -0,0 +1,65 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package app.echo_lot.engine
|
||||
|
||||
import app.echo_lot.measurement.TestStatus
|
||||
import app.echo_lot.protocol.ControlClient
|
||||
import app.echo_lot.protocol.ProbeSession
|
||||
import kotlin.test.Test
|
||||
import kotlin.test.assertEquals
|
||||
import kotlin.test.assertNotNull
|
||||
import kotlin.test.assertTrue
|
||||
|
||||
/**
|
||||
* Upstream train (types 0x03-0x05) against a LIVE server. Self-skips without ECHOLOT_LIVE_*.
|
||||
*
|
||||
* What is asserted is the ledger property: the server's report must account for what was sent,
|
||||
* per sequence number, because directional loss attribution is the entire reason trains exist —
|
||||
* a test that only checked "a report came back" would pass against a server that counts nothing.
|
||||
*/
|
||||
class LiveUpstreamTrainTest {
|
||||
|
||||
private val url = System.getenv("ECHOLOT_LIVE_URL")
|
||||
private val pin = System.getenv("ECHOLOT_LIVE_PIN")
|
||||
private val cred = System.getenv("ECHOLOT_LIVE_CRED")
|
||||
private val udp = System.getenv("ECHOLOT_LIVE_UDP")
|
||||
private val target = System.getenv("ECHOLOT_LIVE_TARGET") ?: "fmr"
|
||||
|
||||
@Test
|
||||
fun serverLedgerAccountsForTheTrain() {
|
||||
if (url == null || pin == null || cred == null || udp == null) {
|
||||
println("LiveUpstreamTrainTest skipped (no ECHOLOT_LIVE_* env)"); return
|
||||
}
|
||||
val control = ControlClient(url, setOf(pin), "0.2.0")
|
||||
val session = control.createSession(cred, target)
|
||||
val (host, port) = udp.split(":").let { it[0] to it[1].toInt() }
|
||||
|
||||
val (test, findings) = ProbeSession(cred, session, host, port).use { ps ->
|
||||
ps.echo() // prime the session so its source is known
|
||||
UpstreamTrainMeasurement(SystemIdSource()).run(
|
||||
ps, sessionRef = "sess-1", count = 120, sizeBytes = 200, interPacketMs = 3,
|
||||
)
|
||||
}
|
||||
control.deleteSession(cred, session.sessionId)
|
||||
|
||||
val m = assertNotNull(test.metrics).toString()
|
||||
println("updown: ${test.status} $m")
|
||||
for (f in findings) println("finding ${f.code} [${f.severity}] ${f.title}")
|
||||
|
||||
assertEquals(TestStatus.OK, test.status, "train report incomplete or absent: $m")
|
||||
|
||||
val sent = Regex(""""sent":(\d+)""").find(m)?.groupValues?.get(1)?.toInt()
|
||||
val received = Regex(""""received_by_server":(\d+)""").find(m)?.groupValues?.get(1)?.toInt()
|
||||
assertNotNull(sent); assertNotNull(received)
|
||||
assertTrue(sent > 0, "nothing was sent: $m")
|
||||
// Over a working path the ledger must be near-complete; a lossy wifi may drop a few, but
|
||||
// a server that fails to count would show up as massive phantom loss here.
|
||||
assertTrue(received >= sent * 9 / 10, "server counted $received of $sent: $m")
|
||||
|
||||
// The columnar evidence must carry a server timestamp for arrived packets — that column
|
||||
// is what one-way delay math consumes after timesync.
|
||||
val ev = assertNotNull(test.evidence).toString()
|
||||
assertTrue(ev.contains("t_srv_rx_ns"), "no server rx column in evidence")
|
||||
}
|
||||
}
|
||||
@@ -35,9 +35,37 @@ data class Run(
|
||||
val device: DeviceInfo,
|
||||
val tiers: Tiers,
|
||||
@SerialName("profiles_used") val profilesUsed: List<String> = emptyList(),
|
||||
val constraints: Constraints = Constraints(),
|
||||
val notes: String? = null,
|
||||
)
|
||||
|
||||
/**
|
||||
* What limited this run — the counterpart to [Tiers], which records what was available.
|
||||
*
|
||||
* A constrained run is not a failed run, and it is not a normal one either. Without this, a run
|
||||
* taken through a VPN looks exactly like a clean run of a healthy network: the same shape, the
|
||||
* same green verdict, and no way for a reader — or a server aggregating thousands of these — to
|
||||
* know that almost nothing was actually measured.
|
||||
*/
|
||||
@Serializable
|
||||
data class Constraints(
|
||||
/** A VPN held the default route while this ran. */
|
||||
@SerialName("vpn_active") val vpnActive: Boolean = false,
|
||||
/**
|
||||
* Per-network probing was refused by the OS.
|
||||
*
|
||||
* Android blocks `Network.bindSocket()` on the underlying networks whenever a VPN is up, to
|
||||
* stop apps leaking around the tunnel. Every per-network test then measures nothing, so any
|
||||
* conclusion drawn about the wifi or cellular link underneath is unfounded.
|
||||
*/
|
||||
@SerialName("per_network_blocked") val perNetworkBlocked: Boolean = false,
|
||||
/** Networks that could not be measured, by id. */
|
||||
@SerialName("unmeasured_networks") val unmeasuredNetworks: List<String> = emptyList(),
|
||||
) {
|
||||
/** True when this run's results mean something different from an unconstrained one. */
|
||||
val constrained: Boolean get() = vpnActive || perNetworkBlocked
|
||||
}
|
||||
|
||||
@Serializable
|
||||
enum class Trigger {
|
||||
@SerialName("manual") MANUAL,
|
||||
|
||||
+120
-4
@@ -169,10 +169,125 @@ object FindingRegistry {
|
||||
// they silently rolled up under connectivity: the third instance of a prefix disagreeing with
|
||||
// its category and quietly moving a fault to a different verdict light.
|
||||
|
||||
/**
|
||||
* Renamed from `v6.broken`, which claimed more than the evidence supports.
|
||||
*
|
||||
* The only signal behind it is ICMPv6 echo getting no reply — and ICMPv6 echo is widely
|
||||
* filtered on networks where IPv6 otherwise works perfectly. A phone that reported this while
|
||||
* happily loading an IPv6-only site over TCP is what caught it. From here the two cases look
|
||||
* identical, so the finding now says what was observed and names both explanations rather than
|
||||
* picking one.
|
||||
*
|
||||
* It is worth reporting either way: filtered ICMPv6 breaks Path MTU Discovery, which is its
|
||||
* own fault even when IPv6 works.
|
||||
*/
|
||||
/**
|
||||
* A global IPv6 address with no default route.
|
||||
*
|
||||
* This is the structural version of the same complaint, and it is worth far more than the
|
||||
* ICMP one because it admits no other explanation: the device has an address it cannot route
|
||||
* with. Nothing is filtered, nothing is inferred — the routing table says so directly, and it
|
||||
* is already in the link snapshot.
|
||||
*
|
||||
* Not always a fault. A VPN that installs host routes to specific destinations produces
|
||||
* exactly this shape on purpose, and it works. What makes it worth reporting either way is
|
||||
* that applications cannot tell: having a global address, they will try IPv6 first and stall
|
||||
* for every destination the routes do not cover.
|
||||
*/
|
||||
/**
|
||||
* An IPv6 default route with no global address to use it from — the mirror of
|
||||
* [V6_NO_DEFAULT_ROUTE], and the more common misconfiguration of the two.
|
||||
*
|
||||
* The router is sending RAs that name it as a default gateway, but SLAAC produced no address:
|
||||
* no prefix information option, or a prefix without the autonomous flag, or DHCPv6-only
|
||||
* addressing the device did not complete. The network is announcing IPv6 service it does not
|
||||
* actually deliver.
|
||||
*
|
||||
* This is worth flagging above the ICMP signal because it is both certain and consequential.
|
||||
* Hosts see router advertisements, believe IPv6 is available, and pay a connection-attempt
|
||||
* timeout on every dual-stack destination before falling back to IPv4 — the classic "the
|
||||
* internet feels slow" complaint with no packet loss anywhere to explain it.
|
||||
*/
|
||||
val V6_ROUTE_WITHOUT_ADDRESS = FindingSpec(
|
||||
"v6.route_without_address", Category.IPV6, Severity.MEDIUM,
|
||||
"The network advertises an IPv6 default route but the device has no global IPv6 address.",
|
||||
rulesOut = "A working IPv6 setup: SLAAC did not produce a usable address on this link.",
|
||||
)
|
||||
|
||||
/**
|
||||
* A VPN prevented the underlying networks from being measured.
|
||||
*
|
||||
* Reported rather than worked around: Android refuses `Network.bindSocket()` on the networks
|
||||
* beneath a VPN precisely so apps cannot leak around the tunnel, and that is correct
|
||||
* behaviour. What is not acceptable is a run that quietly measures nothing and calls the
|
||||
* result healthy, so this says plainly which networks went unmeasured and why.
|
||||
*/
|
||||
/**
|
||||
* The network's DNS server answers, but this device cannot resolve through it.
|
||||
*
|
||||
* Worth separating from every other DNS failure because the remedy is somewhere else entirely.
|
||||
* A name that will not resolve looks identical to a user whatever the cause, and the two causes
|
||||
* pull in opposite directions: a server that does not answer means the network is broken and
|
||||
* the router is the thing to examine, while a server that answers a direct query on a device
|
||||
* that still cannot resolve means the platform resolver has wedged — fixed by toggling wifi,
|
||||
* and nothing to do with the network at all.
|
||||
*
|
||||
* Proven rather than inferred: the probe sends its own UDP query, bypassing the component under
|
||||
* suspicion, and compares that against what the platform returns for the same name.
|
||||
*/
|
||||
/**
|
||||
* The network hands out a search domain its DNS server will not answer for.
|
||||
*
|
||||
* A resolver appends search domains to lookups, so every name a client asks about can stall on
|
||||
* a domain the server ignores. The failure mode is silence rather than a negative answer, and
|
||||
* silence is indistinguishable from packet loss: clients retry instead of moving on, and some
|
||||
* give up on the lookup entirely. That makes it look like the device is broken when the
|
||||
* network is.
|
||||
*
|
||||
* Whether it bites depends on the resolver — some try the bare name first and never notice —
|
||||
* which is why two devices on the same network can disagree about whether DNS works.
|
||||
*/
|
||||
val DNS_SEARCH_DOMAIN_UNANSWERED = FindingSpec(
|
||||
"dns.search_domain_unanswered", Category.DNS, Severity.HIGH,
|
||||
"The network advertises a DNS search domain that its own server does not answer for.",
|
||||
rulesOut = "A fault on this device: the same server answers ordinary names normally.",
|
||||
)
|
||||
|
||||
val DNS_SYSTEM_RESOLVER_BROKEN = FindingSpec(
|
||||
"dns.system_resolver_broken", Category.DNS, Severity.HIGH,
|
||||
"The network's DNS server answers, but this device cannot resolve names through it.",
|
||||
rulesOut = "A network fault: the server replied to a query sent from this device.",
|
||||
)
|
||||
|
||||
val MEASUREMENT_VPN_CONSTRAINED = FindingSpec(
|
||||
"measurement.vpn_constrained", Category.CONNECTIVITY, Severity.INFO,
|
||||
"A VPN was active, so the networks underneath it could not be measured.",
|
||||
rulesOut = "Nothing — this run says little about the underlying network either way.",
|
||||
)
|
||||
|
||||
val V6_NO_DEFAULT_ROUTE = FindingSpec(
|
||||
"v6.no_default_route", Category.IPV6, Severity.MEDIUM,
|
||||
"The device has a global IPv6 address but no IPv6 default route.",
|
||||
rulesOut = "Guesswork: this is read from the routing table, not inferred from silence.",
|
||||
)
|
||||
|
||||
val V6_NO_ICMP_REPLY = FindingSpec(
|
||||
"v6.no_icmp_reply", Category.IPV6, Severity.LOW,
|
||||
"IPv6 is configured but ICMPv6 echo gets no reply.",
|
||||
rulesOut = "Nothing on its own: IPv6 may work fine with ICMP filtered.",
|
||||
)
|
||||
|
||||
/**
|
||||
* IPv6 is advertised and does not work — the claim `v6.broken` originally made on ICMP
|
||||
* silence alone, now reinstated because it can finally be backed: it is only emitted when a
|
||||
* real IPv6 TCP connection (v6.brokenness) failed on the same network whose ICMPv6 went
|
||||
* unanswered. Two independent transports failing on a network that advertises IPv6 is what
|
||||
* "broken" actually means; either signal alone still gets [V6_NO_ICMP_REPLY].
|
||||
*/
|
||||
val V6_BROKEN = FindingSpec(
|
||||
"v6.broken", Category.IPV6, Severity.MEDIUM,
|
||||
"IPv6 is configured on this network but does not work.",
|
||||
rulesOut = "Absence of IPv6: it is provisioned, it simply fails.",
|
||||
"v6.broken", Category.IPV6, Severity.HIGH,
|
||||
"IPv6 is advertised on this network but carries no traffic.",
|
||||
rulesOut = "ICMP filtering as the benign explanation: a TCP connection over IPv6 failed too.",
|
||||
)
|
||||
|
||||
/**
|
||||
@@ -196,7 +311,8 @@ object FindingRegistry {
|
||||
NAT_UDP_REBINDING, NAT_SYMMETRIC,
|
||||
THROUGHPUT_NO_DELIVERY, THROUGHPUT_BELOW_OFFERED,
|
||||
DNS_ANSWER_REWRITTEN, DNS_AUTHORITATIVE_UNREACHABLE,
|
||||
V6_BROKEN, V6_NOT_OFFERED,
|
||||
DNS_SEARCH_DOMAIN_UNANSWERED, DNS_SYSTEM_RESOLVER_BROKEN, MEASUREMENT_VPN_CONSTRAINED,
|
||||
V6_NO_DEFAULT_ROUTE, V6_ROUTE_WITHOUT_ADDRESS, V6_NO_ICMP_REPLY, V6_BROKEN, V6_NOT_OFFERED,
|
||||
)
|
||||
|
||||
private val byCode: Map<String, FindingSpec> = all.associateBy { it.code }
|
||||
|
||||
@@ -18,6 +18,28 @@ data class Network(
|
||||
val wifi: Wifi? = null,
|
||||
val cellular: Cellular? = null,
|
||||
val changes: List<NetworkChange> = emptyList(),
|
||||
@SerialName("system_verdict") val systemVerdict: SystemVerdict? = null,
|
||||
)
|
||||
|
||||
/**
|
||||
* What Android itself concluded about a network, as opposed to what we measured.
|
||||
*
|
||||
* Recorded because it is the verdict the user can see — the "no internet" warning in the status
|
||||
* bar — and because it is free: the platform has already done the work by the time a run starts.
|
||||
*
|
||||
* Its real value is disagreement. When Android says a network is unusable and our own probes reach
|
||||
* the internet regardless, the fault is in the device rather than the network, and that distinction
|
||||
* is the difference between "fix your router" and "toggle your wifi". Neither number alone can say
|
||||
* that; only the two together.
|
||||
*/
|
||||
@Serializable
|
||||
data class SystemVerdict(
|
||||
/** Android's own connectivity check passed. Null when the platform did not say. */
|
||||
val validated: Boolean? = null,
|
||||
/** Android believes a captive portal is intercepting this network. */
|
||||
@SerialName("captive_portal") val captivePortal: Boolean? = null,
|
||||
/** Some traffic works and some does not — Android's own hedge. */
|
||||
@SerialName("partial_connectivity") val partialConnectivity: Boolean? = null,
|
||||
)
|
||||
|
||||
@Serializable
|
||||
|
||||
@@ -35,6 +35,9 @@ data class CategorySummary(
|
||||
* (critical|high → red, medium|low → yellow, info/none → green).
|
||||
* - A category is `inconclusive` when > 50% of its tests are failed/unsupported.
|
||||
* - Overall = the worst category light; `inconclusive` only when ALL categories are.
|
||||
* - A run whose per-network probing was blocked is `inconclusive` outright, whatever the
|
||||
* categories say. The lights describe what the tests found; when the OS refused to let the
|
||||
* tests run, a green light would describe nothing at all.
|
||||
*
|
||||
* The mapping test-type → category comes from [TestType.category]. Only categories that have
|
||||
* findings or tests appear in the summary.
|
||||
@@ -44,7 +47,10 @@ object Verdicts {
|
||||
private fun isInconclusiveTest(s: TestStatus) =
|
||||
s == TestStatus.FAILED || s == TestStatus.UNSUPPORTED
|
||||
|
||||
fun derive(tests: List<Test>, findings: List<Finding>): Summary {
|
||||
fun derive(tests: List<Test>, findings: List<Finding>): Summary =
|
||||
derive(tests, findings, Constraints())
|
||||
|
||||
fun derive(tests: List<Test>, findings: List<Finding>, constraints: Constraints): Summary {
|
||||
val testsByCat = tests.groupBy { TestType.category(it.type) }
|
||||
val findingsByCat = findings.groupBy { it.category }
|
||||
val categories = (testsByCat.keys + findingsByCat.keys)
|
||||
@@ -72,7 +78,14 @@ object Verdicts {
|
||||
)
|
||||
}
|
||||
|
||||
val overall = deriveOverall(perCat.values)
|
||||
// A run that could not measure the networks it was asked about has not found them
|
||||
// healthy; it has found out nothing. Reporting that as green is the single most
|
||||
// misleading thing this function could do, so the constraint outranks the lights.
|
||||
val overall = if (constraints.perNetworkBlocked) {
|
||||
Verdict.INCONCLUSIVE
|
||||
} else {
|
||||
deriveOverall(perCat.values)
|
||||
}
|
||||
return Summary(overall = overall, categories = perCat)
|
||||
}
|
||||
|
||||
|
||||
@@ -89,6 +89,13 @@ object TestType {
|
||||
// dns
|
||||
const val DNS_RESOLVER_INVENTORY = "dns.resolver_inventory"
|
||||
const val DNS_CANARY = "dns.canary"
|
||||
/**
|
||||
* Does this device's own resolver work, as distinct from the network's DNS.
|
||||
*
|
||||
* Registry addition, v1.1. Kept apart from [DNS_CANARY], which asks whether answers are being
|
||||
* tampered with; this asks whether answers arrive at all, and where the failure sits.
|
||||
*/
|
||||
const val DNS_RESOLVER = "dns.resolver"
|
||||
const val DNS_INTERCEPTION = "dns.interception"
|
||||
const val DNS_TTL_INTEGRITY = "dns.ttl_integrity"
|
||||
const val DNS_ANSWER_INTEGRITY = "dns.answer_integrity"
|
||||
|
||||
@@ -0,0 +1,73 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package app.echo_lot.measurement
|
||||
|
||||
/**
|
||||
* The two ways a network can be half-configured for IPv6, read from the link snapshot.
|
||||
*
|
||||
* Pure model logic rather than something a ViewModel does, because "is this network's IPv6
|
||||
* broken, and in which direction" is exactly the kind of judgement that should be checkable
|
||||
* against a captured routing table without a phone in the loop.
|
||||
*/
|
||||
object V6Analysis {
|
||||
|
||||
/** Linux tunnel interfaces: WireGuard/Netbird (tun*, wg*), plus the usual VPN names. */
|
||||
private val TUNNEL_IFACE = Regex("""^(tun|tap|wg|ppp|ipsec|utun)\d*$""")
|
||||
|
||||
/** What one network's IPv6 configuration looks like. */
|
||||
data class Shape(
|
||||
val iface: String,
|
||||
/** A global address with no ::/0 route: an address the device cannot route with. */
|
||||
val addressWithoutRoute: Boolean,
|
||||
/** A ::/0 route with no global address: a route the device cannot source from. */
|
||||
val routeWithoutAddress: Boolean,
|
||||
/** The routes belong to a tunnel, so a partial view of IPv6 is likely deliberate. */
|
||||
val tunnel: Boolean,
|
||||
)
|
||||
|
||||
/**
|
||||
* Classifies each network's IPv6 configuration.
|
||||
*
|
||||
* Both shapes are read straight from the link snapshot rather than inferred from silence, so
|
||||
* unlike an ICMP signal there is no competing explanation for what was observed — and both
|
||||
* matter for the same reason: an application cannot tell in advance, so it tries IPv6 first
|
||||
* and waits.
|
||||
*
|
||||
* They differ in what they mean. An address with no route is what a VPN installing host routes
|
||||
* to specific destinations produces on purpose, and it works; calling that a fault would be the
|
||||
* "lack of IPv6 is a yellow condition" mistake in a new costume, so a tunnel downgrades it to
|
||||
* information. A route with no address is the opposite: the router advertised itself as a
|
||||
* default gateway but SLAAC produced nothing usable, so the network is announcing IPv6 service
|
||||
* it does not deliver. That one is a real misconfiguration however it arises.
|
||||
*/
|
||||
fun classify(networks: List<Network>): List<Shape> = networks.map { n ->
|
||||
val globalV6 = n.link.addresses.any { isGlobalV6(it.addr) }
|
||||
val v6Routes = n.link.routes.filter { it.dst.contains(':') }
|
||||
val hasDefault = v6Routes.any { it.dst == "::/0" }
|
||||
Shape(
|
||||
iface = n.iface ?: v6Routes.firstOrNull()?.iface.orEmpty(),
|
||||
addressWithoutRoute = globalV6 && !hasDefault,
|
||||
routeWithoutAddress = hasDefault && !globalV6,
|
||||
// Android labels the transport itself, which beats guessing from a name; the regex
|
||||
// stays as a backstop for tunnels Android does not own (a userspace WireGuard, say,
|
||||
// or anything seen through the shell tier).
|
||||
tunnel = n.transport == Transport.VPN ||
|
||||
v6Routes.any { TUNNEL_IFACE.containsMatchIn(it.iface.orEmpty()) },
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether an address is IPv6 and usable as a source for off-link traffic.
|
||||
*
|
||||
* ULAs count. A ULA is not globally routable, but it is a global-*scope* address the stack
|
||||
* will happily select as a source, which is the property that matters here — an overlay
|
||||
* network handing out fc00::/7 addresses is providing working IPv6 to the destinations it
|
||||
* carries, and treating that as "no address" would misreport every VPN as broken.
|
||||
*/
|
||||
private fun isGlobalV6(addr: String): Boolean {
|
||||
if (!addr.contains(':')) return false
|
||||
val a = addr.substringBefore('%').lowercase() // strip any zone index
|
||||
return !a.startsWith("fe80") && a != "::1" && a != "::"
|
||||
}
|
||||
}
|
||||
+125
@@ -0,0 +1,125 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package app.echo_lot.measurement
|
||||
|
||||
import kotlin.test.Test
|
||||
import kotlin.test.assertEquals
|
||||
import kotlin.test.assertFalse
|
||||
import kotlin.test.assertTrue
|
||||
|
||||
/**
|
||||
* The fixtures here are a real device's routing table, transcribed from `dumpsys connectivity`
|
||||
* on a OnePlus 15 with a Netbird tunnel up: wifi advertising a default route it cannot source
|
||||
* from, cellular working properly, and a VPN carrying host routes to two destinations.
|
||||
*
|
||||
* Using a captured table rather than invented ones matters, because the bug this guards against
|
||||
* is not "the boolean logic is wrong" — it is "the shapes I imagined are not the shapes real
|
||||
* networks produce".
|
||||
*/
|
||||
class V6AnalysisTest {
|
||||
|
||||
private fun net(
|
||||
id: String,
|
||||
transport: Transport,
|
||||
iface: String,
|
||||
addrs: List<String>,
|
||||
routes: List<Pair<String, String>>,
|
||||
) = Network(
|
||||
id = id,
|
||||
transport = transport,
|
||||
iface = iface,
|
||||
link = Link(
|
||||
addresses = addrs.map { Address(addr = it.substringBefore('/'), prefixLen = 64) },
|
||||
routes = routes.map { (dst, dev) -> Route(dst = dst, iface = dev) },
|
||||
),
|
||||
)
|
||||
|
||||
/** wlan0: an IPv6 default route via a link-local gateway, but SLAAC produced no address. */
|
||||
private val wifi = net(
|
||||
"w", Transport.WIFI, "wlan0",
|
||||
addrs = listOf("fe80::bcf6:edff:fe67:b139", "10.13.102.122"),
|
||||
routes = listOf(
|
||||
"fe80::/64" to "wlan0",
|
||||
"::/0" to "wlan0",
|
||||
"0.0.0.0/0" to "wlan0",
|
||||
),
|
||||
)
|
||||
|
||||
/** rmnet_data1: a properly configured cellular link — global address and a default route. */
|
||||
private val cellular = net(
|
||||
"c", Transport.CELLULAR, "rmnet_data1",
|
||||
addrs = listOf("2001:4bb8:46a:e724:289d:87ff:feb6:ebd3"),
|
||||
routes = listOf("::/0" to "rmnet_data1", "2001:4bb8:46a:e724::/64" to "rmnet_data1"),
|
||||
)
|
||||
|
||||
/** tun1: Netbird, with a ULA and host routes to exactly two destinations. */
|
||||
private val vpn = net(
|
||||
"v", Transport.VPN, "tun1",
|
||||
addrs = listOf("100.64.158.131", "fdfd:c4fe:c4fe:c4fe:1f3c:98a0:dd66:ac7"),
|
||||
routes = listOf(
|
||||
"2001:1ad0:c4fe:6767::2/128" to "tun1",
|
||||
"2001:1ad0:c4fe:a::136/128" to "tun1",
|
||||
"fdfd:c4fe:c4fe:c4fe::/64" to "tun1",
|
||||
),
|
||||
)
|
||||
|
||||
@Test
|
||||
fun `wifi advertising a route it cannot source from is reported`() {
|
||||
val s = V6Analysis.classify(listOf(wifi)).single()
|
||||
assertTrue(s.routeWithoutAddress, "::/0 with only a link-local address is the RA-without-SLAAC case")
|
||||
assertFalse(s.addressWithoutRoute)
|
||||
assertFalse(s.tunnel, "wifi is not a tunnel")
|
||||
assertEquals("wlan0", s.iface)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `a properly configured link produces no finding`() {
|
||||
val s = V6Analysis.classify(listOf(cellular)).single()
|
||||
assertFalse(s.routeWithoutAddress)
|
||||
assertFalse(s.addressWithoutRoute)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `a tunnel with host routes is deliberate, not broken`() {
|
||||
val s = V6Analysis.classify(listOf(vpn)).single()
|
||||
assertTrue(s.addressWithoutRoute, "a ULA and no ::/0 is an address with nothing to route it")
|
||||
assertTrue(s.tunnel, "so it must be reported as information, not as a fault")
|
||||
assertFalse(s.routeWithoutAddress)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `each network is judged on its own`() {
|
||||
// The whole point of per-network classification: "IPv6 is broken" is useless advice when
|
||||
// wifi is the broken one and cellular is fine.
|
||||
val shapes = V6Analysis.classify(listOf(wifi, cellular, vpn)).associateBy { it.iface }
|
||||
assertTrue(shapes.getValue("wlan0").routeWithoutAddress)
|
||||
assertFalse(shapes.getValue("rmnet_data1").routeWithoutAddress)
|
||||
assertFalse(shapes.getValue("rmnet_data1").addressWithoutRoute)
|
||||
assertTrue(shapes.getValue("tun1").addressWithoutRoute)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `a link-local-only network with no v6 route says nothing either way`() {
|
||||
// Plain IPv4-only wifi: no IPv6 offered at all. That is v6.not_offered's business, and
|
||||
// reporting it here as well would double up on a network that is merely legacy, not broken.
|
||||
val v4only = net(
|
||||
"4", Transport.WIFI, "wlan0",
|
||||
addrs = listOf("fe80::1", "192.168.1.5"),
|
||||
routes = listOf("0.0.0.0/0" to "wlan0"),
|
||||
)
|
||||
val s = V6Analysis.classify(listOf(v4only)).single()
|
||||
assertFalse(s.routeWithoutAddress)
|
||||
assertFalse(s.addressWithoutRoute)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `a zone index does not hide a link-local address`() {
|
||||
val zoned = net(
|
||||
"z", Transport.WIFI, "wlan0",
|
||||
addrs = listOf("fe80::1%wlan0"),
|
||||
routes = listOf("::/0" to "wlan0"),
|
||||
)
|
||||
assertTrue(V6Analysis.classify(listOf(zoned)).single().routeWithoutAddress)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,68 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package app.echo_lot.probe
|
||||
|
||||
import android.content.Context
|
||||
import android.net.ConnectivityManager
|
||||
import android.net.NetworkCapabilities
|
||||
import android.system.ErrnoException
|
||||
import android.system.OsConstants
|
||||
import app.echo_lot.measurement.Constraints
|
||||
import app.echo_lot.measurement.Transport
|
||||
import java.net.DatagramSocket
|
||||
|
||||
/**
|
||||
* Detects what will prevent this run from measuring (measurement-schema.md §3 `constraints`),
|
||||
* before any probe runs and independently of all of them.
|
||||
*
|
||||
* The known case: while a VPN holds the default route, Android refuses `Network.bindSocket()` on
|
||||
* the underlying networks (EPERM) so apps cannot leak around the tunnel. Every per-network test
|
||||
* then silently measures the tunnel or nothing, and the run comes out shaped exactly like a clean
|
||||
* run of a healthy network. Detecting that here — one throwaway bind per network — is what lets
|
||||
* the document say "these networks went unmeasured" instead of leaving the reader to infer it
|
||||
* from a pattern of `attempted: false` scattered across the tests.
|
||||
*/
|
||||
object ConstraintDetector {
|
||||
|
||||
fun detect(ctx: Context, entries: List<NetworkInventory.Entry>): Constraints {
|
||||
val cm = ctx.getSystemService(Context.CONNECTIVITY_SERVICE) as ConnectivityManager
|
||||
// "A VPN holds the default route" is judged from the ACTIVE network, not from a VPN
|
||||
// network merely existing in the list: a tunnel that was just disconnected lingers in
|
||||
// allNetworks while it tears down, and counting it kept the app claiming "measured
|
||||
// through a VPN" after the VPN was gone.
|
||||
val vpnActive = runCatching {
|
||||
cm.getNetworkCapabilities(cm.activeNetwork)
|
||||
?.hasTransport(NetworkCapabilities.TRANSPORT_VPN) == true
|
||||
}.getOrDefault(false)
|
||||
|
||||
val refused = ArrayList<String>()
|
||||
for (e in entries) {
|
||||
// The tunnel itself stays bindable — it is the underlying networks the OS walls off.
|
||||
if (e.model.transport == Transport.VPN) continue
|
||||
val err = try {
|
||||
DatagramSocket().use { s -> e.handle.bindSocket(s) }
|
||||
null
|
||||
} catch (t: Throwable) {
|
||||
t
|
||||
}
|
||||
// Only the OS *refusing* counts as blocked (EPERM: the VPN wall). A network that
|
||||
// happens to die mid-snapshot fails its bind too, but with a different errno, and
|
||||
// calling that "per-network probing blocked" would flip a whole healthy run to
|
||||
// INCONCLUSIVE over one network going away — the probes already record
|
||||
// attempted:false for that case.
|
||||
if (err != null && isPermissionRefusal(err)) refused.add(e.model.id)
|
||||
}
|
||||
return Constraints(
|
||||
vpnActive = vpnActive,
|
||||
perNetworkBlocked = refused.isNotEmpty(),
|
||||
unmeasuredNetworks = refused,
|
||||
)
|
||||
}
|
||||
|
||||
private fun isPermissionRefusal(t: Throwable): Boolean =
|
||||
generateSequence(t) { it.cause }.any {
|
||||
(it is ErrnoException && it.errno == OsConstants.EPERM) ||
|
||||
(it.message?.contains("EPERM") == true)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,227 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package app.echo_lot.probe
|
||||
|
||||
import android.content.Context
|
||||
import app.echo_lot.measurement.Test
|
||||
import app.echo_lot.measurement.TestStatus
|
||||
import app.echo_lot.measurement.TestType
|
||||
import app.echo_lot.measurement.Tier
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.withContext
|
||||
import kotlinx.serialization.json.JsonObject
|
||||
import kotlinx.serialization.json.buildJsonObject
|
||||
import kotlinx.serialization.json.put
|
||||
import java.net.DatagramPacket
|
||||
import java.net.DatagramSocket
|
||||
import java.net.InetAddress
|
||||
import java.net.InetSocketAddress
|
||||
import java.util.Random
|
||||
|
||||
/**
|
||||
* Asks the network's own DNS servers directly, then asks Android to resolve the same name, and
|
||||
* compares.
|
||||
*
|
||||
* The comparison is the point. A name that fails to resolve looks the same to a user whatever the
|
||||
* cause, but the causes want opposite responses: if the server does not answer, the network is
|
||||
* broken and the router is the thing to look at; if the server answers a raw query while the
|
||||
* platform still cannot resolve, the device's own resolver has wedged and toggling wifi fixes it in
|
||||
* seconds. Nothing else on a phone will tell you which of those you have.
|
||||
*
|
||||
* This is deliberately not a general DNS test — no recursion checks, no DNSSEC, no rewriting
|
||||
* detection; [DnsCanaryProbe] covers interception. This one answers a single question: is the
|
||||
* resolver on this device doing its job.
|
||||
*/
|
||||
class DnsResolverProbe(
|
||||
private val entries: List<NetworkInventory.Entry>,
|
||||
/** Resolved directly rather than through any cache; any name with a stable answer will do. */
|
||||
private val probeName: String = "one.one.one.one",
|
||||
) : Probe {
|
||||
override val type = TestType.DNS_RESOLVER
|
||||
override val tier = Tier.APP
|
||||
// A query per server with a 3s ceiling, plus one getaddrinfo that may sit out its own timeout.
|
||||
override val estimatedMs = 6_000L
|
||||
|
||||
override suspend fun run(ctx: Context, ids: ProbeIds): Test = withContext(Dispatchers.IO) {
|
||||
val b = TestBuilder(type, tier, ids)
|
||||
val perNetwork = LinkedHashMap<String, Pair<String, JsonObject>>()
|
||||
|
||||
for (e in entries) {
|
||||
val servers = e.model.link.dns?.servers.orEmpty()
|
||||
if (servers.isEmpty()) continue
|
||||
val label = "${e.model.transport.name.lowercase()}:${e.model.id}"
|
||||
|
||||
// Directly: does the configured server answer at all?
|
||||
var direct: Boolean? = null
|
||||
var directDetail = "no server answered"
|
||||
for (s in servers) {
|
||||
val r = try {
|
||||
queryDirect(s, probeName)
|
||||
} catch (e: DnsRefused) {
|
||||
// Distinguished deliberately: a server that replies with a failure is a
|
||||
// working server saying no, which points at the network rather than here.
|
||||
direct = false
|
||||
directDetail = "$s ${e.why}"
|
||||
break
|
||||
}
|
||||
if (r != null) {
|
||||
direct = true
|
||||
directDetail = "$s answered in ${r}ms"
|
||||
break
|
||||
}
|
||||
direct = false
|
||||
directDetail = "$s did not answer"
|
||||
}
|
||||
|
||||
// The search domains the network handed out, asked about separately.
|
||||
//
|
||||
// A resolver appends these to a lookup, so a search domain the server will not answer
|
||||
// for stalls every name a client asks about — and it fails as silence, which is
|
||||
// indistinguishable from packet loss, so clients retry rather than moving on. Asking
|
||||
// about a name that cannot exist is deliberate: the answer wanted here is NXDOMAIN,
|
||||
// and what matters is only whether anything comes back at all.
|
||||
val searchDomains = e.model.link.dns?.searchDomains.orEmpty()
|
||||
var searchAnswered: Boolean? = null
|
||||
var searchDetail = ""
|
||||
for (d in searchDomains) {
|
||||
val nonce = "echolot-probe-" + java.util.UUID.randomUUID().toString().take(8)
|
||||
val answered = servers.any { srv ->
|
||||
runCatching { queryDirect(srv, "$nonce.$d") != null }
|
||||
.getOrElse { it is DnsRefused } // a refusal is still an answer
|
||||
}
|
||||
if (!answered) {
|
||||
searchAnswered = false
|
||||
searchDetail = "$d is not answered at all — queries under it vanish"
|
||||
break
|
||||
}
|
||||
searchAnswered = true
|
||||
searchDetail = "$d answers"
|
||||
}
|
||||
|
||||
// Through the platform: what an app actually gets.
|
||||
val viaSystem = runCatching {
|
||||
e.handle.getAllByName(probeName).isNotEmpty()
|
||||
}.getOrElse { false }
|
||||
|
||||
perNetwork[label] = e.model.id to buildJsonObject {
|
||||
put("network_ref", e.model.id)
|
||||
put("servers", servers.joinToString(","))
|
||||
direct?.let { put("direct_answer", it) }
|
||||
put("direct_detail", directDetail)
|
||||
if (searchDomains.isNotEmpty()) {
|
||||
put("search_domains", searchDomains.joinToString(","))
|
||||
searchAnswered?.let { put("search_answered", it) }
|
||||
put("search_detail", searchDetail)
|
||||
}
|
||||
put("system_resolves", viaSystem)
|
||||
// Named here rather than left for a finding to infer, because the pairing is the
|
||||
// whole observation and splitting it across two places invites reading one alone.
|
||||
put(
|
||||
"verdict",
|
||||
when {
|
||||
// Ordered by which component is at fault, most specific first. A search
|
||||
// domain that swallows queries explains a failure that would otherwise be
|
||||
// blamed on the device, so it has to be tested before that conclusion.
|
||||
searchAnswered == false -> "search domain swallows queries"
|
||||
viaSystem -> "resolver working"
|
||||
direct == true -> "server answers, device resolver does not"
|
||||
direct == false -> "server does not answer"
|
||||
else -> "not determined"
|
||||
},
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
if (perNetwork.isEmpty()) {
|
||||
return@withContext b.build(
|
||||
TestStatus.SKIPPED,
|
||||
evidence = buildJsonObject { put("reason", "no network advertised a DNS server") },
|
||||
)
|
||||
}
|
||||
val evidence = buildJsonObject {
|
||||
put("name", probeName)
|
||||
for ((label, v) in perNetwork) put(label, v.second)
|
||||
}
|
||||
// OK means the measurement ran, not that DNS is healthy — the finding says that.
|
||||
b.build(TestStatus.OK, evidence = evidence)
|
||||
}
|
||||
|
||||
/**
|
||||
* Sends one A query straight to [server] over UDP. Returns the round trip in ms, or null.
|
||||
*
|
||||
* Hand-rolled rather than via any resolver API on purpose: the entire point is to bypass the
|
||||
* component under suspicion. Anything that goes through the platform resolver would inherit
|
||||
* exactly the fault this is trying to detect.
|
||||
*/
|
||||
private fun queryDirect(server: String, name: String): Long? {
|
||||
return try {
|
||||
queryDirectOrThrow(server, name)
|
||||
} catch (e: DnsRefused) {
|
||||
throw e
|
||||
} catch (t: Throwable) {
|
||||
null
|
||||
}
|
||||
}
|
||||
|
||||
private fun queryDirectOrThrow(server: String, name: String): Long? = run {
|
||||
val id = Random().nextInt(0xFFFF)
|
||||
val query = buildQuery(id, name)
|
||||
DatagramSocket().use { sock ->
|
||||
sock.soTimeout = 3000
|
||||
val addr = InetAddress.getByName(server) // a literal from DHCP; no lookup happens
|
||||
val t0 = System.nanoTime()
|
||||
sock.send(DatagramPacket(query, query.size, InetSocketAddress(addr, 53)))
|
||||
val buf = ByteArray(512)
|
||||
val reply = DatagramPacket(buf, buf.size)
|
||||
sock.receive(reply)
|
||||
val ms = (System.nanoTime() - t0) / 1_000_000
|
||||
// A reply is not an answer. Counting any packet as success would let a REFUSED or
|
||||
// SERVFAIL — both perfectly well-formed responses — be reported as "the server
|
||||
// answers", and this probe's whole output is the claim that the server is fine and
|
||||
// the device is not. That would be an accusation pointed at the wrong component,
|
||||
// stated with confidence.
|
||||
val replyId = ((buf[0].toInt() and 0xFF) shl 8) or (buf[1].toInt() and 0xFF)
|
||||
val rcode = if (reply.length >= 4) buf[3].toInt() and 0x0F else -1
|
||||
val answers = if (reply.length >= 8) {
|
||||
((buf[6].toInt() and 0xFF) shl 8) or (buf[7].toInt() and 0xFF)
|
||||
} else 0
|
||||
when {
|
||||
replyId != id || reply.length < 12 -> null
|
||||
rcode != 0 -> throw DnsRefused(rcodeName(rcode))
|
||||
answers == 0 -> throw DnsRefused("answered with no records")
|
||||
else -> ms
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** The server replied, but with a failure — which is a network fault, not a device one. */
|
||||
private class DnsRefused(val why: String) : Exception(why)
|
||||
|
||||
private fun rcodeName(rcode: Int): String = when (rcode) {
|
||||
1 -> "rejected the query as malformed"
|
||||
2 -> "reported its own failure (SERVFAIL)"
|
||||
3 -> "said the name does not exist (NXDOMAIN)"
|
||||
4 -> "does not implement this query"
|
||||
5 -> "refused the query (REFUSED)"
|
||||
else -> "returned rcode $rcode"
|
||||
}
|
||||
|
||||
/** A minimal DNS query: one question, class IN, type A, recursion desired. */
|
||||
private fun buildQuery(id: Int, name: String): ByteArray {
|
||||
val labels = name.split('.').filter { it.isNotEmpty() }
|
||||
val out = ArrayList<Byte>(32)
|
||||
out.add((id shr 8).toByte()); out.add(id.toByte())
|
||||
out.add(0x01); out.add(0x00) // recursion desired
|
||||
out.add(0x00); out.add(0x01) // one question
|
||||
repeat(6) { out.add(0x00) } // no answers, authority or additional
|
||||
for (l in labels) {
|
||||
out.add(l.length.toByte())
|
||||
for (c in l.toByteArray(Charsets.US_ASCII)) out.add(c)
|
||||
}
|
||||
out.add(0x00) // root label
|
||||
out.add(0x00); out.add(0x01) // type A
|
||||
out.add(0x00); out.add(0x01) // class IN
|
||||
return out.toByteArray()
|
||||
}
|
||||
}
|
||||
@@ -40,24 +40,37 @@ class IcmpProbe(
|
||||
|
||||
override suspend fun run(ctx: Context, ids: ProbeIds): Test = withContext(Dispatchers.IO) {
|
||||
val b = TestBuilder(type, tier, ids)
|
||||
val perNetwork = LinkedHashMap<String, String>()
|
||||
val perNetwork = LinkedHashMap<String, Pair<String?, Attempt>>()
|
||||
var anyOk = false
|
||||
val rtts = ArrayList<Double>()
|
||||
|
||||
// Default network first, then each active network explicitly.
|
||||
attempt(null).let { (ok, detail, rtt) ->
|
||||
perNetwork["default"] = detail; if (ok) { anyOk = true; rtt?.let(rtts::add) }
|
||||
attempt(null).let { a ->
|
||||
perNetwork["default"] = null to a
|
||||
if (a.ok) { anyOk = true; a.rttMs?.let(rtts::add) }
|
||||
}
|
||||
for (e in entries) {
|
||||
val label = "${e.model.transport.name.lowercase()}:${e.model.id}"
|
||||
val (ok, detail, rtt) = attempt(e.handle)
|
||||
perNetwork[label] = detail
|
||||
if (ok) { anyOk = true; rtt?.let(rtts::add) }
|
||||
val a = attempt(e.handle)
|
||||
perNetwork[label] = e.model.id to a
|
||||
if (a.ok) { anyOk = true; a.rttMs?.let(rtts::add) }
|
||||
}
|
||||
|
||||
// Per-network results are recorded structurally, not just as prose. The aggregate status
|
||||
// can only say "some network answered"; a finding needs to know *which* network failed,
|
||||
// and recovering that by parsing a human-readable detail string would be a trap waiting to
|
||||
// spring the first time the wording changes.
|
||||
val evidence: JsonObject = buildJsonObject {
|
||||
put("target", target)
|
||||
for ((k, v) in perNetwork) put(k, v)
|
||||
for ((label, r) in perNetwork) {
|
||||
val (netId, a) = r
|
||||
put(label, buildJsonObject {
|
||||
netId?.let { put("network_ref", it) }
|
||||
put("ok", a.ok)
|
||||
put("attempted", a.attempted)
|
||||
put("detail", a.detail)
|
||||
})
|
||||
}
|
||||
}
|
||||
val metrics: JsonObject = buildJsonObject {
|
||||
put("networks_ok", rtts.size)
|
||||
@@ -69,14 +82,31 @@ class IcmpProbe(
|
||||
b.build(status, evidence = evidence, metrics = metrics)
|
||||
}
|
||||
|
||||
private data class Attempt(val ok: Boolean, val detail: String, val rttMs: Double?)
|
||||
/**
|
||||
* One network's result.
|
||||
*
|
||||
* [attempted] separates "we sent an echo request and heard nothing" from "we never got as far
|
||||
* as sending one". Both leave [ok] false, and collapsing them is how a probe ends up asserting
|
||||
* something about a network it never touched: binding to a non-default network can fail with
|
||||
* EPERM, and reporting that as ICMPv6 silence blames the carrier for the app's own inability
|
||||
* to use the interface.
|
||||
*/
|
||||
private data class Attempt(
|
||||
val ok: Boolean,
|
||||
val attempted: Boolean,
|
||||
val detail: String,
|
||||
val rttMs: Double?,
|
||||
)
|
||||
|
||||
private fun attempt(network: Network?): Attempt {
|
||||
var fd: FileDescriptor? = null
|
||||
var sent = false
|
||||
return try {
|
||||
val proto = if (v6) OsConstants.IPPROTO_ICMPV6 else OsConstants.IPPROTO_ICMP
|
||||
val family = if (v6) OsConstants.AF_INET6 else OsConstants.AF_INET
|
||||
fd = Os.socket(family, OsConstants.SOCK_DGRAM, proto)
|
||||
// Everything up to and including sendto is setup. A failure here means the test did
|
||||
// not run on this network — not that the network stayed silent.
|
||||
network?.bindSocket(fd)
|
||||
Os.setsockoptTimeval(fd, OsConstants.SOL_SOCKET, OsConstants.SO_RCVTIMEO, StructTimeval.fromMillis(3000))
|
||||
val addr = network?.getByName(target) ?: InetAddress.getByName(target)
|
||||
@@ -85,14 +115,16 @@ class IcmpProbe(
|
||||
val packet = buildEchoRequest(v6, ident.toShort(), 1)
|
||||
val t0 = System.nanoTime()
|
||||
Os.sendto(fd, packet, 0, packet.size, 0, addr, 0)
|
||||
sent = true
|
||||
val buf = ByteBuffer.allocate(1500)
|
||||
val received = Os.recvfrom(fd, buf, 0, null)
|
||||
val rttMs = (System.nanoTime() - t0) / 1_000_000.0
|
||||
val replyType = if (received > 0) buf.get(0).toInt() and 0xFF else -1
|
||||
val ok = replyType == (if (v6) 129 else 0)
|
||||
Attempt(ok, "reply type=$replyType rtt_ms=${"%.1f".format(Locale.ROOT, rttMs)} bytes=$received", if (ok) rttMs else null)
|
||||
Attempt(ok, true, "reply type=$replyType rtt_ms=${"%.1f".format(Locale.ROOT, rttMs)} bytes=$received", if (ok) rttMs else null)
|
||||
} catch (e: Throwable) {
|
||||
Attempt(false, "error: ${e.message ?: e.javaClass.simpleName}", null)
|
||||
// A timeout after a successful send is a real "no reply"; anything before it is not.
|
||||
Attempt(false, sent, "error: ${e.message ?: e.javaClass.simpleName}", null)
|
||||
} finally {
|
||||
fd?.let { runCatching { Os.close(it) } }
|
||||
}
|
||||
|
||||
@@ -0,0 +1,107 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package app.echo_lot.probe
|
||||
|
||||
import android.content.Context
|
||||
import android.net.nsd.NsdManager
|
||||
import android.net.nsd.NsdServiceInfo
|
||||
import android.net.wifi.WifiManager
|
||||
import app.echo_lot.measurement.Test
|
||||
import app.echo_lot.measurement.TestStatus
|
||||
import app.echo_lot.measurement.TestType
|
||||
import app.echo_lot.measurement.Tier
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.delay
|
||||
import kotlinx.coroutines.withContext
|
||||
import kotlinx.serialization.json.buildJsonObject
|
||||
import kotlinx.serialization.json.put
|
||||
import kotlinx.serialization.json.putJsonObject
|
||||
import java.util.Collections
|
||||
|
||||
/**
|
||||
* local.mdns_inventory — what answers mDNS on this network (MulticastLock + NSD discovery).
|
||||
* The service inventory doubles as the VLAN-leakage detector: a chromecast answering on the
|
||||
* guest wifi is a segmentation fault made visible. Folded from the prober, validated on both
|
||||
* known devices (4 services each).
|
||||
*
|
||||
* Two hardware-bought lessons are load-bearing here:
|
||||
* - The `_services._dns-sd._udp.` meta-query returned 0 on BOTH devices while concrete types
|
||||
* found live services — NsdManager's meta-query support is unreliable across builds, so the
|
||||
* concrete types are the measurement and the meta-query result is itself evidence.
|
||||
* - 4 s of listening missed services that 10 s catches; mDNS answers straggle.
|
||||
*/
|
||||
class MdnsInventoryProbe : Probe {
|
||||
override val type = TestType.LOCAL_MDNS_INVENTORY
|
||||
override val tier = Tier.APP
|
||||
override val estimatedMs = 10_500L
|
||||
|
||||
/** Meta-query + common concrete types (HTTP covers HA/printers/NAS; googlecast is ubiquitous). */
|
||||
private val queries = listOf(
|
||||
"meta" to "_services._dns-sd._udp.",
|
||||
"http" to "_http._tcp.",
|
||||
"googlecast" to "_googlecast._tcp.",
|
||||
)
|
||||
|
||||
private class Recorder : NsdManager.DiscoveryListener {
|
||||
val names: MutableList<String> = Collections.synchronizedList(mutableListOf())
|
||||
@Volatile var started = false
|
||||
@Volatile var startFailCode: Int? = null
|
||||
override fun onStartDiscoveryFailed(t: String?, code: Int) { startFailCode = code }
|
||||
override fun onStopDiscoveryFailed(t: String?, code: Int) {}
|
||||
override fun onDiscoveryStarted(t: String?) { started = true }
|
||||
override fun onDiscoveryStopped(t: String?) {}
|
||||
override fun onServiceFound(s: NsdServiceInfo?) { s?.serviceName?.let { names.add(it) } }
|
||||
override fun onServiceLost(s: NsdServiceInfo?) {}
|
||||
}
|
||||
|
||||
override suspend fun run(ctx: Context, ids: ProbeIds): Test = withContext(Dispatchers.IO) {
|
||||
val b = TestBuilder(type, tier, ids)
|
||||
val wifi = ctx.getSystemService(WifiManager::class.java)
|
||||
val lock = wifi?.createMulticastLock("echolot")?.apply {
|
||||
setReferenceCounted(false)
|
||||
runCatching { acquire() }
|
||||
}
|
||||
val nsd = ctx.getSystemService(NsdManager::class.java)
|
||||
?: return@withContext b.build(
|
||||
TestStatus.UNSUPPORTED,
|
||||
evidence = buildJsonObject { put("reason", "NsdManager unavailable") },
|
||||
)
|
||||
|
||||
val recorders = queries.map { (label, type) ->
|
||||
val r = Recorder()
|
||||
runCatching { nsd.discoverServices(type, NsdManager.PROTOCOL_DNS_SD, r) }
|
||||
.onFailure { r.startFailCode = -1 }
|
||||
Triple(label, type, r)
|
||||
}
|
||||
try {
|
||||
delay(10_000)
|
||||
var total = 0
|
||||
var anyStarted = false
|
||||
val evidence = buildJsonObject {
|
||||
put("multicast_lock", lock?.isHeld == true)
|
||||
for ((label, type, r) in recorders) {
|
||||
runCatching { nsd.stopServiceDiscovery(r) }
|
||||
anyStarted = anyStarted || r.started
|
||||
val names = r.names.distinct()
|
||||
total += names.size
|
||||
putJsonObject(label) {
|
||||
put("query", type)
|
||||
put("started", r.started)
|
||||
r.startFailCode?.let { put("start_fail_code", it) }
|
||||
put("found", names.size)
|
||||
if (names.isNotEmpty()) put("names", names.joinToString(", ").take(300))
|
||||
}
|
||||
}
|
||||
}
|
||||
val metrics = buildJsonObject { put("services_found", total) }
|
||||
// Zero services on a started discovery is a legitimate result (an empty or properly
|
||||
// isolated network), not a failure — only discovery refusing to start is one.
|
||||
b.build(if (anyStarted) TestStatus.OK else TestStatus.FAILED,
|
||||
evidence = evidence, metrics = metrics)
|
||||
} finally {
|
||||
recorders.forEach { (_, _, r) -> runCatching { nsd.stopServiceDiscovery(r) } }
|
||||
runCatching { lock?.release() }
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -39,6 +39,32 @@ object NetworkInventory {
|
||||
return out
|
||||
}
|
||||
|
||||
/**
|
||||
* Android's own verdict on the network, read straight from the capabilities it already has.
|
||||
*
|
||||
* PARTIAL_CONNECTIVITY only exists from API 28 and CAPTIVE_PORTAL from 23, so both are read
|
||||
* defensively: an older platform that cannot answer should leave the field null rather than
|
||||
* assert a false.
|
||||
*/
|
||||
private fun systemVerdict(caps: NetworkCapabilities): app.echo_lot.measurement.SystemVerdict =
|
||||
app.echo_lot.measurement.SystemVerdict(
|
||||
validated = caps.hasCapability(NetworkCapabilities.NET_CAPABILITY_VALIDATED),
|
||||
captivePortal = runCatching {
|
||||
caps.hasCapability(NetworkCapabilities.NET_CAPABILITY_CAPTIVE_PORTAL)
|
||||
}.getOrNull(),
|
||||
// NET_CAPABILITY_PARTIAL_CONNECTIVITY is @SystemApi, so the constant is not in the
|
||||
// public SDK even though the platform sets it from API 28. The number is stable —
|
||||
// changing it would break every system app that reads it — but this is a value the
|
||||
// SDK does not promise us, so it is asked for defensively and reported as unknown
|
||||
// rather than as false if anything about it is not as expected.
|
||||
partialConnectivity = runCatching {
|
||||
caps.hasCapability(NET_CAPABILITY_PARTIAL_CONNECTIVITY)
|
||||
}.getOrNull(),
|
||||
)
|
||||
|
||||
/** @SystemApi NetworkCapabilities.NET_CAPABILITY_PARTIAL_CONNECTIVITY, API 28+. */
|
||||
private const val NET_CAPABILITY_PARTIAL_CONNECTIVITY = 24
|
||||
|
||||
private fun toModel(id: String, caps: NetworkCapabilities, lp: LinkProperties): MNetwork {
|
||||
val transport = when {
|
||||
caps.hasTransport(NetworkCapabilities.TRANSPORT_WIFI) -> Transport.WIFI
|
||||
@@ -72,6 +98,7 @@ object NetworkInventory {
|
||||
return MNetwork(
|
||||
id = id, transport = transport, iface = lp.interfaceName,
|
||||
link = Link(mtu = lp.mtu.takeIf { it > 0 }, addresses = addresses, routes = routes, dns = dns),
|
||||
systemVerdict = systemVerdict(caps),
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,70 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package app.echo_lot.probe
|
||||
|
||||
import android.system.Os
|
||||
import java.io.FileDescriptor
|
||||
|
||||
/**
|
||||
* Linux socket-option ABI numbers that android.system.OsConstants does NOT reliably expose.
|
||||
* Stable across Android's supported ABIs at the IP/IPv6 protocol levels, which is why they can
|
||||
* be hardcoded: if setsockoptInt with one of these succeeds, the kernel accepted the option; if
|
||||
* it throws ErrnoException, it did not. Either outcome is data. Do not "fix" these to
|
||||
* OsConstants names — they don't exist there (validated in the prober; see its OsAbi.kt).
|
||||
*
|
||||
* Measured fact worth keeping: `Os.getsockoptInt` is absent on both known devices (OnePlus 15
|
||||
* A16, Lenovo TB330FU A15), so path-MTU values must be read from the errqueue (`ee_info`), never
|
||||
* from getsockopt(IP_MTU).
|
||||
*/
|
||||
object OsAbi {
|
||||
// IP level
|
||||
const val IP_TTL = 2
|
||||
const val IP_MTU_DISCOVER = 10
|
||||
const val IP_MTU = 14
|
||||
const val IP_RECVERR = 11
|
||||
const val IP_PMTUDISC_DO = 2 // set DF, honor PMTU
|
||||
const val IP_PMTUDISC_PROBE = 3 // set DF, ignore PMTU (for probing)
|
||||
|
||||
// IPv6 level
|
||||
const val IPV6_MTU_DISCOVER = 23
|
||||
const val IPV6_MTU = 24
|
||||
const val IPV6_RECVERR = 25
|
||||
const val IPV6_UNICAST_HOPS = 16
|
||||
const val IPV6_PMTUDISC_PROBE = 3
|
||||
|
||||
// recv flags — not in OsConstants on any current API level
|
||||
const val MSG_ERRQUEUE = 0x2000
|
||||
const val MSG_DONTWAIT = 0x40
|
||||
|
||||
// struct sock_extended_err (uapi/linux/errqueue.h), fixed layout on all Android ABIs:
|
||||
// u32 ee_errno; u8 ee_origin; u8 ee_type; u8 ee_code; u8 ee_pad; u32 ee_info; u32 ee_data;
|
||||
// followed directly by the offender sockaddr (SO_EE_OFFENDER).
|
||||
const val SOCK_EE_SIZE = 16
|
||||
const val SO_EE_ORIGIN_ICMP = 2
|
||||
const val ICMP_TIME_EXCEEDED = 11
|
||||
const val ICMP_DEST_UNREACH = 3
|
||||
|
||||
/** Try setsockoptInt; return null on success, or the errno name on failure. */
|
||||
fun trySetIntOpt(fd: FileDescriptor, level: Int, opt: Int, value: Int): String? =
|
||||
try {
|
||||
Os.setsockoptInt(fd, level, opt, value)
|
||||
null
|
||||
} catch (e: Throwable) {
|
||||
e.message ?: e.javaClass.simpleName
|
||||
}
|
||||
|
||||
/**
|
||||
* getsockoptInt is not part of the stable public Os surface on every API level, so it is
|
||||
* reached via reflection; callers must treat failure as "unreadable", not as an error.
|
||||
*/
|
||||
fun tryGetIntOpt(fd: FileDescriptor, level: Int, opt: Int): Result<Int> = runCatching {
|
||||
val m = Os::class.java.getMethod(
|
||||
"getsockoptInt",
|
||||
FileDescriptor::class.java,
|
||||
Int::class.javaPrimitiveType,
|
||||
Int::class.javaPrimitiveType,
|
||||
)
|
||||
m.invoke(null, fd, level, opt) as Int
|
||||
}
|
||||
}
|
||||
@@ -55,9 +55,10 @@ class TestBuilder(
|
||||
evidence: JsonObject? = null,
|
||||
metrics: JsonObject? = null,
|
||||
error: TestError? = null,
|
||||
params: JsonObject? = null,
|
||||
): Test = Test(
|
||||
id = id, type = type, networkRef = networkRef, sessionRef = sessionRef, tier = tier,
|
||||
startedMonoNs = startedMonoNs, endedMonoNs = ids.monoNs(),
|
||||
status = status, error = error, evidence = evidence, metrics = metrics,
|
||||
status = status, error = error, params = params, evidence = evidence, metrics = metrics,
|
||||
)
|
||||
}
|
||||
|
||||
@@ -60,6 +60,15 @@ class StunProbe(
|
||||
|
||||
override suspend fun run(ctx: Context, ids: ProbeIds): Test = withContext(Dispatchers.IO) {
|
||||
val b = TestBuilder(type, tier, ids)
|
||||
// Without a server there is nothing to ask. Skipped rather than failed: "the STUN test
|
||||
// failed" reads as a finding about the network, when the truth is that this device is
|
||||
// not enrolled anywhere and no packet was ever sent.
|
||||
if (serverHost.isBlank()) {
|
||||
return@withContext b.build(
|
||||
TestStatus.SKIPPED,
|
||||
evidence = buildJsonObject { put("reason", "no server configured to ask") },
|
||||
)
|
||||
}
|
||||
DatagramSocket().use { sock ->
|
||||
sock.soTimeout = 3000
|
||||
val localPort = sock.localPort
|
||||
|
||||
@@ -0,0 +1,255 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package app.echo_lot.probe
|
||||
|
||||
import android.content.Context
|
||||
import android.system.Os
|
||||
import android.system.OsConstants
|
||||
import app.echo_lot.measurement.Flow
|
||||
import app.echo_lot.measurement.Hop
|
||||
import app.echo_lot.measurement.HopProbe
|
||||
import app.echo_lot.measurement.Test
|
||||
import app.echo_lot.measurement.TestError
|
||||
import app.echo_lot.measurement.TestStatus
|
||||
import app.echo_lot.measurement.TestType
|
||||
import app.echo_lot.measurement.Tier
|
||||
import app.echo_lot.measurement.TracerouteEvidence
|
||||
import app.echo_lot.measurement.toEvidence
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.delay
|
||||
import kotlinx.coroutines.withContext
|
||||
import kotlinx.serialization.json.buildJsonObject
|
||||
import kotlinx.serialization.json.put
|
||||
import java.io.FileDescriptor
|
||||
import java.net.InetAddress
|
||||
import java.net.InetSocketAddress
|
||||
import java.nio.ByteBuffer
|
||||
import java.nio.ByteOrder
|
||||
|
||||
/**
|
||||
* traceroute.udp4 — UDP traceroute reading ICMP time-exceeded off the socket error queue via
|
||||
* Os.recvmsg(MSG_ERRQUEUE): no root, no raw socket, no native code. Folded from the prober,
|
||||
* which validated real hop addresses on both known devices (6 hops on the OnePlus 15, 5 on the
|
||||
* Lenovo) and thereby retired the planned C-over-JNI errqueue shim.
|
||||
*
|
||||
* StructMsghdr/StructCmsghdr/recvmsg are reached via reflection (repo convention for uncertain
|
||||
* OS paths): present since roughly API 34, absent before, and the probe must run — and report —
|
||||
* on both. An absent API is UNSUPPORTED with the reason, never a crash.
|
||||
*/
|
||||
class TracerouteProbe(
|
||||
private val targetHost: String = "1.1.1.1",
|
||||
private val maxHops: Int = 6,
|
||||
) : Probe {
|
||||
override val type = TestType.TRACEROUTE_UDP4
|
||||
override val tier = Tier.APP
|
||||
// Validated wall clock is ~250 ms on a healthy path; the ceiling is maxHops silent hops at
|
||||
// 900 ms each, which only a blackholing path produces.
|
||||
override val estimatedMs = 1_500L
|
||||
|
||||
private companion object {
|
||||
const val BASE_PORT = 33434
|
||||
/** ICMP errors take one RTT to surface on the errqueue; poll briefly, never block. */
|
||||
const val HOP_DEADLINE_NS = 900_000_000L
|
||||
const val POLL_INTERVAL_MS = 40L
|
||||
}
|
||||
|
||||
override suspend fun run(ctx: Context, ids: ProbeIds): Test = withContext(Dispatchers.IO) {
|
||||
val b = TestBuilder(type, tier, ids)
|
||||
val params = buildJsonObject {
|
||||
put("target", targetHost); put("max_hops", maxHops); put("base_port", BASE_PORT)
|
||||
}
|
||||
|
||||
val api = ErrqueueApi.resolve()
|
||||
?: return@withContext b.build(
|
||||
TestStatus.UNSUPPORTED,
|
||||
params = params,
|
||||
error = TestError(
|
||||
"no_recvmsg",
|
||||
"StructMsghdr/Os.recvmsg not on this API level — errqueue unreadable",
|
||||
),
|
||||
)
|
||||
|
||||
var fd: FileDescriptor? = null
|
||||
try {
|
||||
fd = Os.socket(OsConstants.AF_INET, OsConstants.SOCK_DGRAM, OsConstants.IPPROTO_UDP)
|
||||
OsAbi.trySetIntOpt(fd, OsConstants.IPPROTO_IP, OsAbi.IP_RECVERR, 1)?.let {
|
||||
return@withContext b.build(
|
||||
TestStatus.UNSUPPORTED,
|
||||
params = params,
|
||||
error = TestError("ip_recverr_rejected", it),
|
||||
)
|
||||
}
|
||||
val target = InetAddress.getByName(targetHost)
|
||||
|
||||
val hops = ArrayList<Hop>(maxHops)
|
||||
var hopsSeen = 0
|
||||
var reachedTarget = false
|
||||
var srcPort = 0
|
||||
for (ttl in 1..maxHops) {
|
||||
OsAbi.trySetIntOpt(fd, OsConstants.IPPROTO_IP, OsAbi.IP_TTL, ttl)
|
||||
val t0 = System.nanoTime()
|
||||
val sent = runCatching {
|
||||
Os.sendto(fd, ByteArray(32), 0, 32, 0, target, BASE_PORT + ttl)
|
||||
}
|
||||
if (sent.isFailure) {
|
||||
hops.add(Hop(ttl, listOf(HopProbe(icmp = "sendto failed: " +
|
||||
(sent.exceptionOrNull()?.message ?: "?")))))
|
||||
continue
|
||||
}
|
||||
if (srcPort == 0) {
|
||||
// Only readable after the implicit bind the first send performs.
|
||||
srcPort = runCatching {
|
||||
(Os.getsockname(fd) as? InetSocketAddress)?.port ?: 0
|
||||
}.getOrDefault(0)
|
||||
}
|
||||
|
||||
var hop: ErrqueueApi.ErrEvent? = null
|
||||
val deadline = System.nanoTime() + HOP_DEADLINE_NS
|
||||
while (hop == null && System.nanoTime() < deadline) {
|
||||
hop = api.pollErrqueue(fd)
|
||||
if (hop == null) delay(POLL_INTERVAL_MS)
|
||||
}
|
||||
val rttNs = System.nanoTime() - t0
|
||||
when {
|
||||
hop == null -> hops.add(Hop(ttl, listOf(HopProbe()))) // silent hop: all null
|
||||
hop.parseError != null ->
|
||||
hops.add(Hop(ttl, listOf(HopProbe(icmp = "unparsed: ${hop.parseError}"))))
|
||||
else -> {
|
||||
hopsSeen++
|
||||
hops.add(Hop(ttl, listOf(HopProbe(
|
||||
replyFrom = hop.offender,
|
||||
rttNs = rttNs,
|
||||
icmp = when (hop.icmpType) {
|
||||
OsAbi.ICMP_TIME_EXCEEDED -> "time_exceeded"
|
||||
OsAbi.ICMP_DEST_UNREACH -> "dest_unreachable"
|
||||
else -> "type_${hop.icmpType}"
|
||||
},
|
||||
))))
|
||||
if (hop.icmpType == OsAbi.ICMP_DEST_UNREACH) reachedTarget = true
|
||||
}
|
||||
}
|
||||
if (reachedTarget) break
|
||||
}
|
||||
|
||||
// dst_port varies per TTL (classic traceroute, and what was validated on hardware),
|
||||
// so this flow is explicitly NOT fixed-tuple; base_port is in params.
|
||||
val evidence = TracerouteEvidence(
|
||||
flow = Flow(srcPort = srcPort, dstPort = BASE_PORT, fixedTuple = false),
|
||||
hops = hops,
|
||||
).toEvidence()
|
||||
val metrics = buildJsonObject {
|
||||
put("hops_seen", hopsSeen)
|
||||
put("reached_target", reachedTarget)
|
||||
}
|
||||
val status = when {
|
||||
hopsSeen > 0 -> TestStatus.OK
|
||||
// API present, sends succeeded, nothing surfaced: a fact about this path or
|
||||
// kernel, not proof the mechanism is missing.
|
||||
else -> TestStatus.PARTIAL
|
||||
}
|
||||
b.build(status, params = params, evidence = evidence, metrics = metrics)
|
||||
} catch (e: Throwable) {
|
||||
b.build(
|
||||
TestStatus.FAILED,
|
||||
params = params,
|
||||
error = TestError("uncaught", e.message ?: e.javaClass.simpleName),
|
||||
)
|
||||
} finally {
|
||||
fd?.let { runCatching { Os.close(it) } }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Reflection facade over android.system.{StructMsghdr, StructCmsghdr, Os.recvmsg}.
|
||||
* Resolved once; null if any piece is missing on this API level.
|
||||
*/
|
||||
internal class ErrqueueApi private constructor(
|
||||
private val msghdrCtor: java.lang.reflect.Constructor<*>,
|
||||
private val recvmsg: java.lang.reflect.Method,
|
||||
private val cmsgLevel: java.lang.reflect.Field,
|
||||
private val cmsgType: java.lang.reflect.Field,
|
||||
private val cmsgData: java.lang.reflect.Field,
|
||||
private val msgControl: java.lang.reflect.Field,
|
||||
) {
|
||||
class ErrEvent(
|
||||
val offender: String?,
|
||||
val icmpType: Int,
|
||||
val origin: Int,
|
||||
val parseError: String? = null,
|
||||
)
|
||||
|
||||
/** One non-blocking MSG_ERRQUEUE read; null when the queue is empty. */
|
||||
fun pollErrqueue(fd: FileDescriptor): ErrEvent? {
|
||||
return try {
|
||||
val iov = arrayOf(ByteBuffer.allocate(512))
|
||||
// (SocketAddress msg_name, ByteBuffer[] msg_iov, StructCmsghdr[] msg_control, flags)
|
||||
val msghdr = msghdrCtor.newInstance(
|
||||
InetSocketAddress(0), iov, null, 0,
|
||||
)
|
||||
recvmsg.invoke(null, fd, msghdr, OsAbi.MSG_ERRQUEUE or OsAbi.MSG_DONTWAIT)
|
||||
val control = msgControl.get(msghdr) as? Array<*>
|
||||
?: return ErrEvent(null, -1, -1, "msg_control empty after recvmsg")
|
||||
for (cmsg in control.filterNotNull()) {
|
||||
val level = cmsgLevel.getInt(cmsg)
|
||||
val type = cmsgType.getInt(cmsg)
|
||||
if (level == OsConstants.IPPROTO_IP && type == OsAbi.IP_RECVERR) {
|
||||
return parseSockExtendedErr(cmsgData.get(cmsg))
|
||||
}
|
||||
}
|
||||
ErrEvent(null, -1, -1, "no IP_RECVERR cmsg among ${control.size}")
|
||||
} catch (e: Throwable) {
|
||||
// The single most load-bearing line: reflection wraps errno in
|
||||
// InvocationTargetException, and EAGAIN there means "queue empty", not failure.
|
||||
val cause = (e as? java.lang.reflect.InvocationTargetException)?.cause ?: e
|
||||
val msg = cause.message ?: cause.javaClass.simpleName
|
||||
if ("EAGAIN" in msg || "EWOULDBLOCK" in msg) null
|
||||
else ErrEvent(null, -1, -1, msg)
|
||||
}
|
||||
}
|
||||
|
||||
/** cmsg_data = struct sock_extended_err + offender sockaddr_in (see OsAbi). */
|
||||
private fun parseSockExtendedErr(data: Any?): ErrEvent {
|
||||
val bytes: ByteArray = when (data) {
|
||||
is ByteArray -> data
|
||||
is ByteBuffer -> ByteArray(data.remaining()).also { data.duplicate().get(it) }
|
||||
else -> return ErrEvent(null, -1, -1, "cmsg_data is ${data?.javaClass?.name}")
|
||||
}
|
||||
if (bytes.size < OsAbi.SOCK_EE_SIZE) {
|
||||
return ErrEvent(null, -1, -1, "cmsg_data too short: ${bytes.size}")
|
||||
}
|
||||
val origin = bytes[4].toInt() and 0xFF
|
||||
val icmpType = bytes[5].toInt() and 0xFF
|
||||
// SO_EE_OFFENDER: sockaddr_in directly after the fixed struct; family is in native
|
||||
// byte order, sin_addr at offset +4 within the sockaddr.
|
||||
val offender = if (bytes.size >= OsAbi.SOCK_EE_SIZE + 8) {
|
||||
val family = ByteBuffer.wrap(bytes, OsAbi.SOCK_EE_SIZE, 2)
|
||||
.order(ByteOrder.nativeOrder()).short.toInt()
|
||||
if (family == OsConstants.AF_INET) {
|
||||
val a = bytes.copyOfRange(OsAbi.SOCK_EE_SIZE + 4, OsAbi.SOCK_EE_SIZE + 8)
|
||||
InetAddress.getByAddress(a).hostAddress
|
||||
} else null
|
||||
} else null
|
||||
return ErrEvent(offender, icmpType, origin)
|
||||
}
|
||||
|
||||
companion object {
|
||||
fun resolve(): ErrqueueApi? = runCatching {
|
||||
val msghdrCls = Class.forName("android.system.StructMsghdr")
|
||||
val cmsghdrCls = Class.forName("android.system.StructCmsghdr")
|
||||
ErrqueueApi(
|
||||
// Picked by shape, not by position: the 4-arg form is
|
||||
// (SocketAddress, ByteBuffer[], StructCmsghdr[], int) on every level that has it.
|
||||
msghdrCtor = msghdrCls.constructors.first { it.parameterCount == 4 },
|
||||
recvmsg = Os::class.java.getMethod(
|
||||
"recvmsg", FileDescriptor::class.java, msghdrCls, Int::class.javaPrimitiveType,
|
||||
),
|
||||
cmsgLevel = cmsghdrCls.getField("cmsg_level"),
|
||||
cmsgType = cmsghdrCls.getField("cmsg_type"),
|
||||
cmsgData = cmsghdrCls.getField("cmsg_data"),
|
||||
msgControl = msghdrCls.getField("msg_control"),
|
||||
)
|
||||
}.getOrNull()
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,131 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package app.echo_lot.probe
|
||||
|
||||
import android.content.Context
|
||||
import app.echo_lot.measurement.Test
|
||||
import app.echo_lot.measurement.TestStatus
|
||||
import app.echo_lot.measurement.TestType
|
||||
import app.echo_lot.measurement.Tier
|
||||
import app.echo_lot.measurement.Network as MNetwork
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.withContext
|
||||
import kotlinx.serialization.json.JsonObject
|
||||
import kotlinx.serialization.json.buildJsonObject
|
||||
import kotlinx.serialization.json.put
|
||||
import java.net.Inet6Address
|
||||
import java.net.InetSocketAddress
|
||||
import java.util.Locale
|
||||
|
||||
/**
|
||||
* v6.brokenness — does IPv6 actually carry traffic, asked with a real TCP connection.
|
||||
*
|
||||
* This exists to corroborate (or refute) the ICMPv6 silence that icmp.ping6 observes. ICMPv6 echo
|
||||
* is widely filtered on networks where IPv6 works fine, so silence alone cannot distinguish
|
||||
* "IPv6 is broken" from "ping is filtered" — a phone that reported v6.broken while happily
|
||||
* loading IPv6-only sites is what proved the point. A TCP connect over IPv6 to the configured
|
||||
* server settles it: if it succeeds, IPv6 works and the ICMP silence is filtering; if it fails
|
||||
* too, on a network that advertises IPv6, the brokenness claim finally has evidence behind it.
|
||||
*
|
||||
* Only networks that claim to offer IPv6 (a global address or a v6 default route) are attempted:
|
||||
* connecting over v6 on an IPv4-only network fails by design, and recording that as evidence
|
||||
* would manufacture the exact false positive this probe exists to kill.
|
||||
*/
|
||||
class V6ConnectProbe(
|
||||
private val entries: List<NetworkInventory.Entry>,
|
||||
private val serverHost: String,
|
||||
private val port: Int = 443,
|
||||
) : Probe {
|
||||
override val type = TestType.V6_BROKENNESS
|
||||
override val tier = Tier.APP
|
||||
// One 3s connect timeout per v6-provisioned network, at most.
|
||||
override val estimatedMs = 4_000L
|
||||
|
||||
override suspend fun run(ctx: Context, ids: ProbeIds): Test = withContext(Dispatchers.IO) {
|
||||
val b = TestBuilder(type, tier, ids)
|
||||
// Same rule as the STUN and canary probes: with no server there is no target, and
|
||||
// borrowing someone else's infrastructure to get one is not this app's call to make.
|
||||
if (serverHost.isBlank()) {
|
||||
return@withContext b.build(
|
||||
TestStatus.SKIPPED,
|
||||
evidence = buildJsonObject { put("reason", "no server configured to connect to") },
|
||||
)
|
||||
}
|
||||
val candidates = entries.filter { ipv6Provisioned(it.model) }
|
||||
if (candidates.isEmpty()) {
|
||||
return@withContext b.build(
|
||||
TestStatus.SKIPPED,
|
||||
evidence = buildJsonObject {
|
||||
put("reason", "no active network claims to offer IPv6")
|
||||
},
|
||||
)
|
||||
}
|
||||
|
||||
var okCount = 0
|
||||
var attemptedCount = 0
|
||||
val evidence: JsonObject = buildJsonObject {
|
||||
put("target", "$serverHost:$port")
|
||||
for (e in candidates) {
|
||||
val label = "${e.model.transport.name.lowercase()}:${e.model.id}"
|
||||
val a = attempt(e)
|
||||
if (a.attempted) attemptedCount++
|
||||
if (a.ok) okCount++
|
||||
put(label, buildJsonObject {
|
||||
put("network_ref", e.model.id)
|
||||
put("ok", a.ok)
|
||||
put("attempted", a.attempted)
|
||||
put("detail", a.detail)
|
||||
})
|
||||
}
|
||||
}
|
||||
val status = when {
|
||||
attemptedCount == 0 -> TestStatus.SKIPPED // resolution/binding never got that far
|
||||
okCount == attemptedCount -> TestStatus.OK
|
||||
okCount > 0 -> TestStatus.PARTIAL
|
||||
else -> TestStatus.FAILED
|
||||
}
|
||||
b.build(status, evidence = evidence)
|
||||
}
|
||||
|
||||
/** Same attempted/ok separation as IcmpProbe: a connect we never sent proves nothing. */
|
||||
private data class Attempt(val ok: Boolean, val attempted: Boolean, val detail: String)
|
||||
|
||||
private fun attempt(e: NetworkInventory.Entry): Attempt {
|
||||
// Resolved through this network's own resolver; a v6 address obtained over another
|
||||
// network would still be connected to over this one, which is what matters.
|
||||
val addr = runCatching {
|
||||
e.handle.getAllByName(serverHost).filterIsInstance<Inet6Address>().firstOrNull()
|
||||
}.getOrNull()
|
||||
?: return Attempt(false, false, "no AAAA answer for $serverHost via this network")
|
||||
|
||||
// createSocket() binds to the network at creation; failing here means the app could not
|
||||
// use the interface at all (e.g. EPERM under a VPN) — nothing was sent, nothing is known.
|
||||
val socket = try {
|
||||
e.handle.socketFactory.createSocket()
|
||||
} catch (t: Throwable) {
|
||||
return Attempt(false, false, "socket unavailable: ${t.message ?: t.javaClass.simpleName}")
|
||||
}
|
||||
return try {
|
||||
val t0 = System.nanoTime()
|
||||
socket.connect(InetSocketAddress(addr, port), 3000)
|
||||
val rttMs = (System.nanoTime() - t0) / 1_000_000.0
|
||||
Attempt(true, true, "connected to [${addr.hostAddress}]:$port " +
|
||||
"rtt_ms=${"%.1f".format(Locale.ROOT, rttMs)}")
|
||||
} catch (t: Throwable) {
|
||||
// A refused connection would still prove the path forwards IPv6, but against our own
|
||||
// server's 443 the realistic failures are timeout and unreachable — both silence.
|
||||
Attempt(false, true, "error: ${t.message ?: t.javaClass.simpleName}")
|
||||
} finally {
|
||||
runCatching { socket.close() }
|
||||
}
|
||||
}
|
||||
|
||||
/** The network claims IPv6: a global (non-link-local) address or a v6 default route. */
|
||||
private fun ipv6Provisioned(n: MNetwork): Boolean =
|
||||
n.link.addresses.any { a ->
|
||||
a.addr.contains(':') &&
|
||||
!a.addr.startsWith("fe80", ignoreCase = true) &&
|
||||
!a.addr.startsWith("::1")
|
||||
} || n.link.routes.any { it.dst == "::/0" }
|
||||
}
|
||||
@@ -35,13 +35,57 @@ class ControlClient(
|
||||
private val controlUrl: String,
|
||||
pins: Set<String>,
|
||||
private val appVersion: String = "",
|
||||
/**
|
||||
* Addresses to fall back to when the server's name will not resolve, learned from its profile.
|
||||
*
|
||||
* A measurement tool that cannot report from a broken network is useless exactly when it
|
||||
* matters, and a wedged resolver is one of the faults it is built to find — it should not also
|
||||
* be the thing that stops the finding being delivered.
|
||||
*
|
||||
* Safe because the pin is the trust and the name is not part of it: the server presents the
|
||||
* same certificate whether it was reached by name or by address, and a wrong address fails the
|
||||
* pin like anything else would.
|
||||
*/
|
||||
private val fallbackAddrs: List<String> = emptyList(),
|
||||
) {
|
||||
|
||||
private val json = Json { ignoreUnknownKeys = true }
|
||||
private val socketFactory = Pinning.sslContext(pins).socketFactory
|
||||
|
||||
/**
|
||||
* The base URL to use, substituting a cached address only when the name genuinely fails.
|
||||
*
|
||||
* Resolved once per client and only on failure, so a working network pays nothing and never
|
||||
* silently drifts onto an address that may be stale.
|
||||
*/
|
||||
private val base: String by lazy { resolveBase() }
|
||||
|
||||
private fun resolveBase(): String {
|
||||
if (fallbackAddrs.isEmpty()) return controlUrl
|
||||
val uri = runCatching { java.net.URI(controlUrl) }.getOrNull() ?: return controlUrl
|
||||
val host = uri.host ?: return controlUrl
|
||||
if (runCatching { java.net.InetAddress.getByName(host) }.isSuccess) return controlUrl
|
||||
|
||||
val port = if (uri.port > 0) uri.port else 443
|
||||
for (ip in fallbackAddrs) {
|
||||
// Checked rather than assumed: on a v4-only network a v6 address would otherwise be
|
||||
// chosen and fail slowly, which is the wrong answer delivered late.
|
||||
val reachable = runCatching {
|
||||
java.net.Socket().use { sock ->
|
||||
sock.connect(java.net.InetSocketAddress(ip, port), 4000)
|
||||
true
|
||||
}
|
||||
}.getOrDefault(false)
|
||||
if (reachable) {
|
||||
val literal = if (ip.contains(':')) "[$ip]" else ip
|
||||
return uri.scheme + "://" + literal + ":" + port
|
||||
}
|
||||
}
|
||||
return controlUrl
|
||||
}
|
||||
|
||||
private fun open(path: String, method: String, credential: String?): HttpsURLConnection {
|
||||
val conn = URL(controlUrl.trimEnd('/') + path).openConnection() as HttpsURLConnection
|
||||
val conn = URL(base.trimEnd('/') + path).openConnection() as HttpsURLConnection
|
||||
conn.sslSocketFactory = socketFactory
|
||||
conn.setHostnameVerifier { _, _ -> true } // pin is the trust, not the name
|
||||
conn.requestMethod = method
|
||||
@@ -184,6 +228,33 @@ class ControlClient(
|
||||
open("/v1/runs/$runId", "DELETE", credential).responseCode
|
||||
}
|
||||
|
||||
/**
|
||||
* Ties this device to the person the ID token identifies.
|
||||
*
|
||||
* The device credential proves *which device*, the token proves *which person*; the server
|
||||
* requires both. Returns the raw JSON reply (account id and display name).
|
||||
*/
|
||||
fun linkAccount(credential: String, idToken: String): String {
|
||||
val conn = open("/v1/account/link", "POST", credential)
|
||||
writeJson(conn, """{"id_token":${jstr(idToken)}}""")
|
||||
val text = body(conn)
|
||||
check(conn.responseCode in 200..299) { "sign-in failed: ${conn.responseCode} $text" }
|
||||
return text
|
||||
}
|
||||
|
||||
/** Signs out on this device. The device stays enrolled. */
|
||||
fun unlinkAccount(credential: String) {
|
||||
open("/v1/account/link", "DELETE", credential).responseCode
|
||||
}
|
||||
|
||||
/** Whether anyone is signed in on this device, and who. */
|
||||
fun accountStatus(credential: String): String {
|
||||
val conn = open("/v1/account", "GET", credential)
|
||||
val text = body(conn)
|
||||
check(conn.responseCode == 200) { "account status failed: ${conn.responseCode} $text" }
|
||||
return text
|
||||
}
|
||||
|
||||
fun observations(credential: String, sessionId: String): String {
|
||||
val conn = open("/v1/sessions/$sessionId/observations", "GET", credential)
|
||||
val text = body(conn)
|
||||
|
||||
@@ -45,11 +45,22 @@ data class EnrollmentLink(
|
||||
* the reason the pin travels in the link at all.
|
||||
*/
|
||||
fun redeem(deviceName: String? = null, appVersion: String = ""): Enrolled {
|
||||
val client = ControlClient(controlUrl, setOf(pin), appVersion)
|
||||
// The link may name the server's public address rather than its control endpoint, so that
|
||||
// a person is handed a name they recognise. Ask where to actually connect.
|
||||
//
|
||||
// Only the address comes from here. The pin still comes from the link, because a pin
|
||||
// fetched over an ordinary TLS connection would be worth exactly what the certificate
|
||||
// authorities are worth — and pinning exists to survive one the operator does not
|
||||
// control, such as a root injected by corporate device management. An intercepted
|
||||
// discovery can therefore send this device to the wrong host, where the pin will not
|
||||
// match: an outage, not a compromise.
|
||||
val endpoint = discover(controlUrl) ?: controlUrl
|
||||
val client = ControlClient(endpoint, setOf(pin), appVersion)
|
||||
val response = client.enroll(token, deviceName)
|
||||
val profile = client.profile(response.credential)
|
||||
return Enrolled(
|
||||
controlUrl = controlUrl,
|
||||
controlUrl = endpoint,
|
||||
publicUrl = controlUrl,
|
||||
pin = pin,
|
||||
credential = response.credential,
|
||||
deviceId = response.deviceId,
|
||||
@@ -57,6 +68,29 @@ data class EnrollmentLink(
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* Asks a server where its control plane lives. Null when it does not say, or cannot be asked.
|
||||
*
|
||||
* Deliberately forgiving: a server that predates this, or one whose link already names the
|
||||
* control endpoint directly, simply answers nothing and the link's own URL is used. Enrollment
|
||||
* must not start failing because an optional lookup did.
|
||||
*/
|
||||
private fun discover(publicUrl: String): String? = runCatching {
|
||||
val conn = (java.net.URL(publicUrl.trimEnd('/') + "/v1/discover").openConnection()
|
||||
as java.net.HttpURLConnection).apply {
|
||||
connectTimeout = 8_000
|
||||
readTimeout = 8_000
|
||||
setRequestProperty("Accept", "application/json")
|
||||
}
|
||||
if (conn.responseCode !in 200..299) return null
|
||||
val body = conn.inputStream.bufferedReader().use { it.readText() }
|
||||
kotlinx.serialization.json.Json { ignoreUnknownKeys = true }
|
||||
.parseToJsonElement(body)
|
||||
.let { (it as kotlinx.serialization.json.JsonObject)["control_url"] }
|
||||
?.let { (it as kotlinx.serialization.json.JsonPrimitive).content }
|
||||
?.takeIf { it.isNotBlank() }
|
||||
}.getOrNull()
|
||||
|
||||
companion object {
|
||||
const val SCHEME = "echolot"
|
||||
const val HOST = "enroll"
|
||||
@@ -111,7 +145,15 @@ data class EnrollmentLink(
|
||||
|
||||
/** A server this device is now enrolled with, ready to be stored in settings. */
|
||||
data class Enrolled(
|
||||
/** Where this device connects: the endpoint whose certificate the pin matches. */
|
||||
val controlUrl: String,
|
||||
/**
|
||||
* The address a person was given, kept for display.
|
||||
*
|
||||
* Shown instead of [controlUrl] because the endpoint is plumbing — it exists to select a
|
||||
* certificate — while this is the name the operator handed out and would recognise.
|
||||
*/
|
||||
val publicUrl: String,
|
||||
val pin: String,
|
||||
val credential: String,
|
||||
val deviceId: String,
|
||||
|
||||
@@ -29,6 +29,15 @@ data class Target(
|
||||
val id: String,
|
||||
val ip4: String? = null,
|
||||
val ip6: String? = null,
|
||||
/**
|
||||
* The second address, which RFC 5780 behaviour discovery redirects to.
|
||||
*
|
||||
* Worth surfacing rather than treating as an implementation detail: a report that says "the
|
||||
* server did not answer" means something different depending on which of its addresses was
|
||||
* asked, and an operator reading one needs to be able to tell.
|
||||
*/
|
||||
@SerialName("ip4_alt") val ip4Alt: String? = null,
|
||||
@SerialName("ip6_alt") val ip6Alt: String? = null,
|
||||
@SerialName("udp_port") val udpPort: Int = 0,
|
||||
@SerialName("tcp_port") val tcpPort: Int = 0,
|
||||
@SerialName("stun_port") val stunPort: Int = 0,
|
||||
@@ -74,6 +83,25 @@ data class CompatInfo(
|
||||
@SerialName("app_max") val appMax: String = "",
|
||||
)
|
||||
|
||||
/**
|
||||
* How to sign in to this server's identity provider, advertised so the app can offer the button
|
||||
* only when there is something behind it — and drive the flow without anyone typing an issuer URL.
|
||||
*/
|
||||
@Serializable
|
||||
data class AuthInfo(
|
||||
val enabled: Boolean = false,
|
||||
val issuer: String = "",
|
||||
@SerialName("client_id") val clientId: String = "",
|
||||
val flow: String = "",
|
||||
@SerialName("redirect_uri") val redirectUri: String = "",
|
||||
val scopes: String = "openid profile email",
|
||||
@SerialName("authorization_endpoint") val authorizationEndpoint: String = "",
|
||||
@SerialName("token_endpoint") val tokenEndpoint: String = "",
|
||||
@SerialName("end_session_endpoint") val endSessionEndpoint: String = "",
|
||||
/** Present when the server has an issuer configured but could not reach it. */
|
||||
@SerialName("discovery_error") val discoveryError: String? = null,
|
||||
)
|
||||
|
||||
@Serializable
|
||||
data class Profile(
|
||||
@SerialName("profile_version") val profileVersion: Int = 0,
|
||||
@@ -86,6 +114,7 @@ data class Profile(
|
||||
val pins: List<String> = emptyList(),
|
||||
val uploads: UploadPolicy = UploadPolicy(),
|
||||
val compat: CompatInfo = CompatInfo(),
|
||||
val auth: AuthInfo = AuthInfo(),
|
||||
) {
|
||||
fun supports(capability: String) = capability in capabilities
|
||||
}
|
||||
|
||||
@@ -0,0 +1,145 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package app.echo_lot.protocol
|
||||
|
||||
import kotlinx.serialization.json.Json
|
||||
import kotlinx.serialization.json.jsonObject
|
||||
import kotlinx.serialization.json.jsonPrimitive
|
||||
import java.io.IOException
|
||||
import java.net.HttpURLConnection
|
||||
import java.net.URL
|
||||
import java.net.URLEncoder
|
||||
import java.security.MessageDigest
|
||||
import java.security.SecureRandom
|
||||
import java.util.Base64
|
||||
|
||||
/**
|
||||
* Sign-in for the app: authorization code with PKCE (RFC 7636).
|
||||
*
|
||||
* The app is a *public* client — it ships to devices, so any secret compiled into it can be read
|
||||
* out of the APK with `unzip` and `strings`. PKCE is what replaces the client secret, and it
|
||||
* defends a specific attack that matters here more than most places: the redirect comes back
|
||||
* through a custom URI scheme, and on Android *any* app may register `echolot://`. A malicious one
|
||||
* could intercept the callback and take the authorization code. Because the code can only be
|
||||
* exchanged by presenting the verifier — which never left this process and cannot be derived from
|
||||
* the challenge that did — a stolen code is worth nothing.
|
||||
*
|
||||
* Nothing from the IdP is kept afterwards. The ID token is used once, to prove to the server who
|
||||
* is signing in, and then discarded: the device credential is what authenticates every later
|
||||
* request. So there are no access tokens to store, no refresh tokens to rotate, and no token
|
||||
* lifetime for the app to manage.
|
||||
*/
|
||||
object OidcLogin {
|
||||
|
||||
/** A started sign-in. [verifier] and [state] must survive until the callback returns. */
|
||||
data class Pending(val authorizationUrl: String, val verifier: String, val state: String)
|
||||
|
||||
/**
|
||||
* Builds the authorization URL and the secrets that must be held until the callback.
|
||||
*
|
||||
* Everything comes from the server's profile rather than being compiled in, so pointing the
|
||||
* app at a different server with a different IdP is configuration, not a rebuild.
|
||||
*/
|
||||
fun begin(auth: AuthInfo, random: SecureRandom = SecureRandom()): Pending {
|
||||
require(auth.enabled && auth.authorizationEndpoint.isNotBlank()) {
|
||||
"this server has no identity provider configured"
|
||||
}
|
||||
val verifier = randomUrlSafe(random)
|
||||
val state = randomUrlSafe(random)
|
||||
val challenge = b64(MessageDigest.getInstance("SHA-256").digest(verifier.toByteArray()))
|
||||
|
||||
val q = buildString {
|
||||
append("response_type=code")
|
||||
append("&client_id=").append(enc(auth.clientId))
|
||||
append("&redirect_uri=").append(enc(auth.redirectUri))
|
||||
append("&scope=").append(enc(auth.scopes))
|
||||
append("&state=").append(enc(state))
|
||||
append("&code_challenge=").append(enc(challenge))
|
||||
append("&code_challenge_method=S256")
|
||||
}
|
||||
val sep = if (auth.authorizationEndpoint.contains('?')) "&" else "?"
|
||||
return Pending(auth.authorizationEndpoint + sep + q, verifier, state)
|
||||
}
|
||||
|
||||
/** What came back on the `echolot://auth` redirect. */
|
||||
data class Callback(val code: String?, val state: String?, val error: String?)
|
||||
|
||||
/** Parses the redirect URI the browser handed back to the app. */
|
||||
fun parseCallback(uri: String): Callback {
|
||||
val q = uri.substringAfter('?', "")
|
||||
var code: String? = null
|
||||
var state: String? = null
|
||||
var error: String? = null
|
||||
for (pair in q.split('&')) {
|
||||
val k = pair.substringBefore('=')
|
||||
val v = dec(pair.substringAfter('=', ""))
|
||||
when (k) {
|
||||
"code" -> code = v
|
||||
"state" -> state = v
|
||||
"error" -> error = v
|
||||
"error_description" -> if (error != null) error = "$error: $v"
|
||||
}
|
||||
}
|
||||
return Callback(code, state, error)
|
||||
}
|
||||
|
||||
/** The sign-in failed in a way worth showing someone, rather than a transport error. */
|
||||
class LoginFailed(message: String) : Exception(message)
|
||||
|
||||
/**
|
||||
* Exchanges the code for an ID token.
|
||||
*
|
||||
* The state is compared before anything else happens. A callback whose state does not match
|
||||
* the one this process generated did not come from a flow this process started — which is
|
||||
* precisely how an attacker gets a victim to complete *their* login — so it is refused before
|
||||
* the code is spent.
|
||||
*/
|
||||
fun complete(auth: AuthInfo, pending: Pending, callbackUri: String): String {
|
||||
val cb = parseCallback(callbackUri)
|
||||
if (cb.error != null) throw LoginFailed(cb.error)
|
||||
if (cb.state.isNullOrEmpty() || cb.state != pending.state) {
|
||||
throw LoginFailed("this sign-in did not start on this device — start again")
|
||||
}
|
||||
val code = cb.code ?: throw LoginFailed("the identity provider returned no authorization code")
|
||||
|
||||
val body = buildString {
|
||||
append("grant_type=authorization_code")
|
||||
append("&code=").append(enc(code))
|
||||
append("&redirect_uri=").append(enc(auth.redirectUri))
|
||||
append("&client_id=").append(enc(auth.clientId))
|
||||
append("&code_verifier=").append(enc(pending.verifier))
|
||||
}
|
||||
val conn = (URL(auth.tokenEndpoint).openConnection() as HttpURLConnection).apply {
|
||||
requestMethod = "POST"
|
||||
doOutput = true
|
||||
connectTimeout = 15_000
|
||||
readTimeout = 15_000
|
||||
setRequestProperty("Content-Type", "application/x-www-form-urlencoded")
|
||||
setRequestProperty("Accept", "application/json")
|
||||
}
|
||||
conn.outputStream.use { it.write(body.toByteArray()) }
|
||||
val text = try {
|
||||
val stream = if (conn.responseCode in 200..299) conn.inputStream else conn.errorStream
|
||||
stream?.bufferedReader()?.use { it.readText() } ?: ""
|
||||
} catch (e: IOException) {
|
||||
throw LoginFailed("could not reach the identity provider: ${e.message}")
|
||||
}
|
||||
if (conn.responseCode !in 200..299) {
|
||||
throw LoginFailed("the identity provider refused the sign-in (${conn.responseCode})")
|
||||
}
|
||||
val idToken = runCatching {
|
||||
Json.parseToJsonElement(text).jsonObject["id_token"]?.jsonPrimitive?.content
|
||||
}.getOrNull()
|
||||
return idToken?.takeIf { it.isNotBlank() }
|
||||
?: throw LoginFailed("the identity provider returned no id_token")
|
||||
}
|
||||
|
||||
private fun randomUrlSafe(random: SecureRandom): String =
|
||||
ByteArray(32).also(random::nextBytes).let(::b64)
|
||||
|
||||
private fun b64(b: ByteArray): String = Base64.getUrlEncoder().withoutPadding().encodeToString(b)
|
||||
private fun enc(s: String): String = URLEncoder.encode(s, "UTF-8")
|
||||
private fun dec(s: String): String =
|
||||
runCatching { java.net.URLDecoder.decode(s, "UTF-8") }.getOrDefault(s)
|
||||
}
|
||||
@@ -152,6 +152,149 @@ class ProbeSession(
|
||||
/** What one upstream run put on the wire locally. */
|
||||
data class Sent(val packets: Int, val bytes: Long, val durationMs: Long, val kbps: Int)
|
||||
|
||||
// ---- upstream trains (spec §3.2, types 0x03-0x05) --------------------------------------
|
||||
|
||||
/** One TRAIN_DATA packet as sent: its wire seq, local tx time and size. */
|
||||
data class TrainPacket(val seq: Int, val tTxNs: Long, val sizeBytes: Int)
|
||||
|
||||
/** One row of the server's received view. 255 in ttl/dscp/ecn means "not observed". */
|
||||
data class TrainRow(
|
||||
val seq: Int, val tRxNs: Long, val sizeBytes: Int,
|
||||
val ttl: Int, val dscp: Int, val ecn: Int,
|
||||
)
|
||||
|
||||
/**
|
||||
* The server's account of one train. [received] counts every packet that arrived, buffered
|
||||
* or not; [truncated] mirrors the wire flag (rows beyond the server's cap were counted but
|
||||
* not kept). [partsExpected]/[partsReceived] make a lossy report path visible instead of
|
||||
* letting missing rows masquerade as train loss.
|
||||
*/
|
||||
data class TrainReport(
|
||||
val trainId: Int,
|
||||
val received: Int,
|
||||
val truncated: Boolean,
|
||||
val rows: List<TrainRow>,
|
||||
val partsExpected: Int,
|
||||
val partsReceived: Int,
|
||||
)
|
||||
|
||||
/**
|
||||
* Sends one paced upstream train. Nothing comes back per packet by design; pair with
|
||||
* [trainReport] to learn what arrived. The absolute schedule (not sleep-per-packet) is the
|
||||
* same anti-drift choice as [sendThroughput].
|
||||
*/
|
||||
fun sendTrain(trainId: Int, count: Int, sizeBytes: Int = 200, interPacketMs: Long = 5): List<TrainPacket> {
|
||||
val size = sizeBytes.coerceIn(Wire.HEADER_SIZE + 4, 1472)
|
||||
val out = ArrayList<TrainPacket>(count)
|
||||
val start = System.nanoTime()
|
||||
var next = start
|
||||
for (i in 0 until count) {
|
||||
val payload = ByteArray(size - Wire.HEADER_SIZE)
|
||||
payload[0] = (trainId ushr 24).toByte(); payload[1] = (trainId ushr 16).toByte()
|
||||
payload[2] = (trainId ushr 8).toByte(); payload[3] = trainId.toByte()
|
||||
val tTx = nowNs()
|
||||
val pkt = Wire.build(Wire.TYPE_TRAIN_DATA, prefix, ++seq, tTx, key, payload)
|
||||
try {
|
||||
socket.send(DatagramPacket(pkt, pkt.size, server))
|
||||
} catch (e: java.io.IOException) {
|
||||
// A local send failure is our condition, not the path's: report what actually
|
||||
// left rather than letting the server's report read as loss.
|
||||
break
|
||||
}
|
||||
out.add(TrainPacket(seq, tTx, pkt.size))
|
||||
next += interPacketMs * 1_000_000
|
||||
val sleepNs = next - System.nanoTime()
|
||||
if (sleepNs > 0) Thread.sleep(sleepNs / 1_000_000, (sleepNs % 1_000_000).toInt())
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
/**
|
||||
* Fetches the server's received view of a train (one REPORT_REQ, N REPORT datagrams).
|
||||
*
|
||||
* Returns null when no report arrives at all — indistinguishable between "report lost" and
|
||||
* "server predates trains", and the caller must say so rather than choose. Missing parts of
|
||||
* a multi-part report are tolerated and visible via partsReceived < partsExpected.
|
||||
*/
|
||||
fun trainReport(trainId: Int, timeoutMs: Long = 3_000): TrainReport? {
|
||||
val req = ByteArray(4)
|
||||
req[0] = (trainId ushr 24).toByte(); req[1] = (trainId ushr 16).toByte()
|
||||
req[2] = (trainId ushr 8).toByte(); req[3] = trainId.toByte()
|
||||
val pkt = Wire.build(Wire.TYPE_TRAIN_REPORT_REQ, prefix, ++seq, nowNs(), key, req)
|
||||
socket.send(DatagramPacket(pkt, pkt.size, server))
|
||||
|
||||
var received = 0
|
||||
var truncated = false
|
||||
var partsExpected = -1
|
||||
val seenParts = HashSet<Int>()
|
||||
val rows = ArrayList<TrainRow>()
|
||||
val deadline = System.nanoTime() + timeoutMs * 1_000_000
|
||||
val buf = ByteArray(2048)
|
||||
val prevTimeout = socket.soTimeout
|
||||
try {
|
||||
while (partsExpected < 0 || seenParts.size < partsExpected) {
|
||||
val remainMs = ((deadline - System.nanoTime()) / 1_000_000).toInt()
|
||||
if (remainMs <= 0) break
|
||||
socket.soTimeout = remainMs.coerceAtMost(1000)
|
||||
val dp = DatagramPacket(buf, buf.size)
|
||||
try {
|
||||
socket.receive(dp)
|
||||
} catch (e: java.net.SocketTimeoutException) {
|
||||
continue
|
||||
}
|
||||
val p = Wire.parseVerified(buf, dp.length, key) ?: continue
|
||||
if (p.type != Wire.TYPE_TRAIN_REPORT) continue
|
||||
val part = parseReportPart(p.payload, trainId) ?: continue
|
||||
if (!seenParts.add(part.part)) continue
|
||||
received = part.received
|
||||
truncated = truncated || part.truncated
|
||||
partsExpected = part.parts
|
||||
rows.addAll(part.rows)
|
||||
}
|
||||
} finally {
|
||||
socket.soTimeout = prevTimeout
|
||||
}
|
||||
if (seenParts.isEmpty()) return null
|
||||
rows.sortBy { it.seq }
|
||||
return TrainReport(trainId, received, truncated, rows, partsExpected, seenParts.size)
|
||||
}
|
||||
|
||||
private class ReportPart(
|
||||
val part: Int, val parts: Int, val received: Int,
|
||||
val truncated: Boolean, val rows: List<TrainRow>,
|
||||
)
|
||||
|
||||
/** Mirrors the server's columnar layout (dataplane/train.go buildTrainReport). */
|
||||
private fun parseReportPart(b: ByteArray, wantId: Int): ReportPart? {
|
||||
if (b.size < 16) return null
|
||||
fun u16(off: Int) = ((b[off].toInt() and 0xFF) shl 8) or (b[off + 1].toInt() and 0xFF)
|
||||
fun u32(off: Int) = ((b[off].toLong() and 0xFF) shl 24) or ((b[off + 1].toLong() and 0xFF) shl 16) or
|
||||
((b[off + 2].toLong() and 0xFF) shl 8) or (b[off + 3].toLong() and 0xFF)
|
||||
if (u32(0).toInt() != wantId) return null
|
||||
val received = u32(4).toInt()
|
||||
val part = u16(8)
|
||||
val parts = u16(10)
|
||||
val truncated = (b[12].toInt() and 0x01) != 0
|
||||
val n = u16(14)
|
||||
if (b.size < 16 + n * 17) return null
|
||||
val rows = ArrayList<TrainRow>(n)
|
||||
var off = 16
|
||||
val seqs = IntArray(n) { u32(off + it * 4).toInt() }; off += n * 4
|
||||
val tRx = LongArray(n) {
|
||||
var v = 0L
|
||||
for (j in 0 until 8) v = (v shl 8) or (b[off + it * 8 + j].toLong() and 0xFF)
|
||||
v
|
||||
}; off += n * 8
|
||||
val sizes = IntArray(n) { u16(off + it * 2) }; off += n * 2
|
||||
val ttls = IntArray(n) { b[off + it].toInt() and 0xFF }; off += n
|
||||
val dscps = IntArray(n) { b[off + it].toInt() and 0xFF }; off += n
|
||||
val ecns = IntArray(n) { b[off + it].toInt() and 0xFF }
|
||||
for (i in 0 until n) {
|
||||
rows.add(TrainRow(seqs[i], tRx[i], sizes[i], ttls[i], dscps[i], ecns[i]))
|
||||
}
|
||||
return ReportPart(part, parts, received, truncated, rows)
|
||||
}
|
||||
|
||||
/** One packet received from the server, with the wire size actually delivered. */
|
||||
data class Received(val type: Int, val seq: Int, val sizeBytes: Int, val tRxNs: Long)
|
||||
|
||||
|
||||
@@ -23,6 +23,14 @@ object Wire {
|
||||
|
||||
const val TYPE_ECHO_REQ: Int = 0x01
|
||||
const val TYPE_ECHO_RESP: Int = 0x02
|
||||
/**
|
||||
* Upstream train (spec §3.2): DATA is deliberately unanswered — a per-packet reply would
|
||||
* double the traffic and drag the return path into a measurement of the outbound one. The
|
||||
* server's received view comes back afterwards via REPORT_REQ → one or more REPORTs.
|
||||
*/
|
||||
const val TYPE_TRAIN_DATA: Int = 0x03
|
||||
const val TYPE_TRAIN_REPORT_REQ: Int = 0x04
|
||||
const val TYPE_TRAIN_REPORT: Int = 0x05
|
||||
const val TYPE_TIMESYNC_REQ: Int = 0x07
|
||||
const val TYPE_TIMESYNC_RSP: Int = 0x08
|
||||
const val TYPE_MTU_PROBE: Int = 0x09
|
||||
|
||||
@@ -0,0 +1,100 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package app.echo_lot.protocol
|
||||
|
||||
import java.security.SecureRandom
|
||||
import kotlin.test.Test
|
||||
import kotlin.test.assertEquals
|
||||
import kotlin.test.assertFailsWith
|
||||
import kotlin.test.assertNotEquals
|
||||
import kotlin.test.assertTrue
|
||||
|
||||
class OidcLoginTest {
|
||||
|
||||
private val auth = AuthInfo(
|
||||
enabled = true,
|
||||
issuer = "https://id.example.net/application/o/echolot-app/",
|
||||
clientId = "the-client",
|
||||
redirectUri = "echolot://auth",
|
||||
scopes = "openid profile email",
|
||||
authorizationEndpoint = "https://id.example.net/application/o/authorize/",
|
||||
tokenEndpoint = "https://id.example.net/application/o/token/",
|
||||
)
|
||||
|
||||
@Test
|
||||
fun theAuthorizationUrlCarriesEverythingTheIdPNeeds() {
|
||||
val p = OidcLogin.begin(auth)
|
||||
val url = p.authorizationUrl
|
||||
assertTrue(url.startsWith(auth.authorizationEndpoint + "?"), url)
|
||||
for (part in listOf(
|
||||
"response_type=code",
|
||||
"client_id=the-client",
|
||||
"redirect_uri=echolot%3A%2F%2Fauth",
|
||||
"code_challenge_method=S256",
|
||||
"scope=openid+profile+email",
|
||||
)) {
|
||||
assertTrue(url.contains(part), "missing $part in $url")
|
||||
}
|
||||
assertTrue(url.contains("code_challenge="), url)
|
||||
// The verifier itself must never appear in the URL — that is the entire point of PKCE.
|
||||
assertTrue(!url.contains(p.verifier), "the code verifier leaked into the authorize URL")
|
||||
}
|
||||
|
||||
// Two sign-ins must not share a verifier or state, or one intercepted flow compromises the next.
|
||||
@Test
|
||||
fun everySignInGetsFreshSecrets() {
|
||||
val a = OidcLogin.begin(auth, SecureRandom())
|
||||
val b = OidcLogin.begin(auth, SecureRandom())
|
||||
assertNotEquals(a.verifier, b.verifier)
|
||||
assertNotEquals(a.state, b.state)
|
||||
assertTrue(a.verifier.length >= 43, "verifier is shorter than RFC 7636 allows")
|
||||
}
|
||||
|
||||
@Test
|
||||
fun parsesTheRedirectTheBrowserHandsBack() {
|
||||
val cb = OidcLogin.parseCallback("echolot://auth?code=abc123&state=xyz")
|
||||
assertEquals("abc123", cb.code)
|
||||
assertEquals("xyz", cb.state)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun parsesAnErrorRedirect() {
|
||||
val cb = OidcLogin.parseCallback("echolot://auth?error=access_denied&error_description=User%20said%20no")
|
||||
assertEquals("access_denied", cb.error?.substringBefore(":"))
|
||||
assertTrue(cb.code == null)
|
||||
}
|
||||
|
||||
// A callback whose state does not match is how an attacker gets someone to complete *their*
|
||||
// sign-in. It must be refused before the code is spent, without any network call.
|
||||
@Test
|
||||
fun aMismatchedStateIsRefusedBeforeTheCodeIsSpent() {
|
||||
val p = OidcLogin.begin(auth)
|
||||
val e = assertFailsWith<OidcLogin.LoginFailed> {
|
||||
OidcLogin.complete(auth, p, "echolot://auth?code=stolen&state=not-ours")
|
||||
}
|
||||
assertTrue(e.message!!.contains("did not start on this device"), e.message!!)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun aMissingStateIsRefused() {
|
||||
val p = OidcLogin.begin(auth)
|
||||
assertFailsWith<OidcLogin.LoginFailed> {
|
||||
OidcLogin.complete(auth, p, "echolot://auth?code=abc")
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
fun anErrorRedirectSurfacesTheReason() {
|
||||
val p = OidcLogin.begin(auth)
|
||||
val e = assertFailsWith<OidcLogin.LoginFailed> {
|
||||
OidcLogin.complete(auth, p, "echolot://auth?error=access_denied&state=${p.state}")
|
||||
}
|
||||
assertTrue(e.message!!.contains("access_denied"))
|
||||
}
|
||||
|
||||
@Test
|
||||
fun refusesToStartWhenTheServerHasNoIdentityProvider() {
|
||||
assertFailsWith<IllegalArgumentException> { OidcLogin.begin(AuthInfo(enabled = false)) }
|
||||
}
|
||||
}
|
||||
@@ -32,4 +32,8 @@ dependencies {
|
||||
implementation(libs.shizuku.provider)
|
||||
implementation(libs.kotlinx.coroutines.android)
|
||||
implementation(libs.kotlinx.serialization.json)
|
||||
// JVM unit tests for the pure dump parsers (DumpParsers.kt) against the archived
|
||||
// vendor fixtures — no device, no Android runtime.
|
||||
testImplementation(libs.kotlin.test.junit)
|
||||
testImplementation(libs.junit4)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,174 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package app.echo_lot.shizuku
|
||||
|
||||
/**
|
||||
* Parsed view of one IPv6 default route from `ip -6 route show table all`.
|
||||
*
|
||||
* [table] stays a string: Android's per-network route tables use ids past Int range (the Lenovo
|
||||
* TB330FU prints `table 1000000015`), and `table local` is not a number at all — parsing to a
|
||||
* numeric type either overflows or silently drops rows, and the id is only ever compared, never
|
||||
* computed with.
|
||||
*/
|
||||
data class V6DefaultRoute(
|
||||
/** Link-local address of the advertising router; null for gateway-less defaults (dummy0). */
|
||||
val gateway: String?,
|
||||
val dev: String,
|
||||
val table: String?, // null = main table (`ip` omits the token there)
|
||||
val proto: String?, // "ra" marks a route installed from a Router Advertisement
|
||||
val metric: Long?,
|
||||
/** Remaining RA route lifetime (`expires NNNsec`); null when the route does not age out. */
|
||||
val expiresSec: Long?,
|
||||
)
|
||||
|
||||
/** One `ip neigh show` row. [lladdr] is null for FAILED/INCOMPLETE entries — the kernel tried to
|
||||
* resolve and has nothing, which is itself signal. */
|
||||
data class NeighborEntry(
|
||||
val ip: String,
|
||||
val dev: String?,
|
||||
val lladdr: String?,
|
||||
val state: String?, // REACHABLE/STALE/FAILED/... — kept verbatim, the kernel's vocabulary
|
||||
val router: Boolean,
|
||||
)
|
||||
|
||||
/** A NEIGH transition seen inside the `ip monitor` window. */
|
||||
data class NeighborEvent(val entry: NeighborEntry, val deleted: Boolean)
|
||||
|
||||
/**
|
||||
* Pure-string parsers for the shell battery's `ip` command outputs. No Android imports on
|
||||
* purpose: these run (and are unit-tested) on the JVM against the real vendor dumps archived
|
||||
* from the prober, which is the only way to catch a vendor format drift before it ships.
|
||||
*
|
||||
* All parsers degrade to an empty result on missing or unrecognized input — the battery's
|
||||
* captures are best-effort (the Lenovo's `ip monitor` times out under newProcess, the
|
||||
* UserService path prepends a stray `uid=2000` line, evidence strings are trimmed mid-line
|
||||
* at 1200 chars), so an exception here would turn a degraded capture into a lost test.
|
||||
*/
|
||||
object DumpParsers {
|
||||
|
||||
/** True when [raw] is real command output rather than an executor error sentinel. */
|
||||
fun captureUsable(raw: String?): Boolean {
|
||||
if (raw.isNullOrBlank()) return false
|
||||
val t = raw.trimStart()
|
||||
return !t.startsWith("SHIZUKU_") && !t.startsWith("EXEC_") && !t.startsWith("NEWPROCESS_")
|
||||
}
|
||||
|
||||
/** Extracts every `default …` route from `ip -6 route show table all` output. */
|
||||
fun parseV6DefaultRoutes(raw: String?): List<V6DefaultRoute> {
|
||||
if (!captureUsable(raw)) return emptyList()
|
||||
val routes = ArrayList<V6DefaultRoute>()
|
||||
for (line in raw!!.lineSequence()) {
|
||||
val tok = line.trim().split(WS)
|
||||
if (tok.firstOrNull() != "default") continue
|
||||
var gateway: String? = null; var dev: String? = null; var table: String? = null
|
||||
var proto: String? = null; var metric: Long? = null; var expires: Long? = null
|
||||
var i = 1
|
||||
while (i < tok.size - 1) {
|
||||
when (tok[i]) {
|
||||
"via" -> gateway = tok[i + 1]
|
||||
"dev" -> dev = tok[i + 1]
|
||||
"table" -> table = tok[i + 1]
|
||||
"proto" -> proto = tok[i + 1]
|
||||
"metric" -> metric = tok[i + 1].toLongOrNull()
|
||||
// `expires 1269sec` — the unit is glued to the number.
|
||||
"expires" -> expires = tok[i + 1].removeSuffix("sec").toLongOrNull()
|
||||
}
|
||||
i++
|
||||
}
|
||||
// A default route without a device is not something `ip` prints; treat it as a
|
||||
// truncated/garbled line rather than fabricating a partial route.
|
||||
if (dev != null) routes.add(V6DefaultRoute(gateway, dev, table, proto, metric, expires))
|
||||
}
|
||||
return routes
|
||||
}
|
||||
|
||||
/**
|
||||
* Maps interface name → link-layer address from `ip addr show`. Only `link/ether` counts:
|
||||
* loopback/ipip/gre pseudo-addresses are not identities, and the RA-source cross-reference
|
||||
* this feeds compares Ethernet MACs.
|
||||
*/
|
||||
fun parseInterfaceMacs(raw: String?): Map<String, String> {
|
||||
if (!captureUsable(raw)) return emptyMap()
|
||||
val macs = LinkedHashMap<String, String>()
|
||||
var current: String? = null
|
||||
for (line in raw!!.lineSequence()) {
|
||||
val header = STANZA_HEADER.find(line)
|
||||
if (header != null) {
|
||||
// "5: tunl0@NONE:" — the name is the part before an optional @suffix.
|
||||
current = header.groupValues[1].substringBefore('@')
|
||||
continue
|
||||
}
|
||||
val dev = current ?: continue
|
||||
val tok = line.trim().split(WS)
|
||||
if (tok.size >= 2 && tok[0] == "link/ether" && MAC.matches(tok[1])) {
|
||||
macs.putIfAbsent(dev, tok[1])
|
||||
}
|
||||
}
|
||||
return macs
|
||||
}
|
||||
|
||||
/** Parses `ip neigh show` output into entries; non-neighbor lines (uid noise) are skipped. */
|
||||
fun parseNeighbors(raw: String?): List<NeighborEntry> {
|
||||
if (!captureUsable(raw)) return emptyList()
|
||||
return raw!!.lineSequence()
|
||||
.mapNotNull { parseNeighborTokens(it.trim().split(WS)) }
|
||||
.toList()
|
||||
}
|
||||
|
||||
/**
|
||||
* Extracts NEIGH transitions from an `ip monitor all` capture. Each event line carries a
|
||||
* `[NEIGH]` label (other families — ROUTE, ADDR, LINK — are ignored) and deletions are
|
||||
* printed as `Deleted <entry>`. An empty result is normal: a quiet 5 s window sees nothing.
|
||||
*/
|
||||
fun parseNeighborEvents(raw: String?): List<NeighborEvent> {
|
||||
if (!captureUsable(raw)) return emptyList()
|
||||
val events = ArrayList<NeighborEvent>()
|
||||
for (line in raw!!.lineSequence()) {
|
||||
val m = MONITOR_LABEL.find(line.trim()) ?: continue
|
||||
if (!m.groupValues[1].equals("NEIGH", ignoreCase = true)) continue
|
||||
var rest = line.trim().removeRange(m.range).trim()
|
||||
val deleted = rest.startsWith("Deleted ", ignoreCase = true)
|
||||
if (deleted) rest = rest.substring("Deleted ".length)
|
||||
parseNeighborTokens(rest.split(WS))?.let { events.add(NeighborEvent(it, deleted)) }
|
||||
}
|
||||
return events
|
||||
}
|
||||
|
||||
/**
|
||||
* (ip → lladdr) for every neighbor that has one. This is the comparison surface the future
|
||||
* gateway-MAC-change finding diffs across runs, so it is computed here — in the tested,
|
||||
* pure layer — rather than re-derived from JSON by each consumer.
|
||||
*/
|
||||
fun lladdrByIp(neighbors: List<NeighborEntry>): Map<String, String> =
|
||||
neighbors.mapNotNull { n -> n.lladdr?.let { n.ip to it } }.toMap()
|
||||
|
||||
/** One neighbor row: `<ip> dev <if> [lladdr <mac>] [router] [proxy] <STATE>`. */
|
||||
private fun parseNeighborTokens(tok: List<String>): NeighborEntry? {
|
||||
val ip = tok.firstOrNull() ?: return null
|
||||
// The first token must look like an address — this is what drops the UserService path's
|
||||
// stray "uid=2000" line and any grep noise without needing to know every noise shape.
|
||||
if (!IP_LIKE.matches(ip) || (!ip.contains('.') && !ip.contains(':'))) return null
|
||||
var dev: String? = null; var lladdr: String? = null; var state: String? = null
|
||||
var router = false
|
||||
var i = 1
|
||||
while (i < tok.size) {
|
||||
when (tok[i]) {
|
||||
"dev" -> { dev = tok.getOrNull(i + 1); i++ }
|
||||
"lladdr" -> { lladdr = tok.getOrNull(i + 1); i++ }
|
||||
"router" -> router = true
|
||||
"proxy" -> {} // recorded nowhere: proxy entries have no bearing on ARP watching
|
||||
else -> if (STATE.matches(tok[i])) state = tok[i]
|
||||
}
|
||||
i++
|
||||
}
|
||||
return NeighborEntry(ip, dev, lladdr, state, router)
|
||||
}
|
||||
|
||||
private val WS = Regex("\\s+")
|
||||
private val STANZA_HEADER = Regex("^\\d+:\\s+([^:\\s]+):")
|
||||
private val MAC = Regex("^[0-9a-fA-F]{2}(:[0-9a-fA-F]{2}){5}$")
|
||||
private val IP_LIKE = Regex("^[0-9a-fA-F:.]+(%[\\w-]+)?$")
|
||||
private val STATE = Regex("^(REACHABLE|STALE|DELAY|PROBE|FAILED|INCOMPLETE|PERMANENT|NOARP|NONE)$")
|
||||
private val MONITOR_LABEL = Regex("^\\[(\\w+)]")
|
||||
}
|
||||
+14
-2
@@ -98,20 +98,32 @@ object ShizukuAvailability {
|
||||
.addFlags(android.content.Intent.FLAG_ACTIVITY_NEW_TASK)
|
||||
|
||||
/**
|
||||
* Reports the state now and on every binder transition. Returns a function that removes the
|
||||
* listeners again (call it from onCleared).
|
||||
* Reports the state now, on every binder transition, and when a permission request is
|
||||
* answered. Returns a function that removes the listeners again (call it from onCleared).
|
||||
*
|
||||
* The permission listener matters as much as the binder ones: granting permission does not
|
||||
* make the binder arrive or die, so without it the banner still read "running but not
|
||||
* authorised" after the user had just authorised it — the one moment they are looking for
|
||||
* confirmation that it worked.
|
||||
*
|
||||
* It is still not sufficient on its own. Permission can be granted inside Shizuku's own app,
|
||||
* where nothing calls back into this process at all, so callers should re-check on resume as
|
||||
* well; see [current].
|
||||
*/
|
||||
fun observe(context: Context, onChange: (State) -> Unit): () -> Unit {
|
||||
val app = context.applicationContext
|
||||
val received = Shizuku.OnBinderReceivedListener { onChange(current(app)) }
|
||||
val dead = Shizuku.OnBinderDeadListener { onChange(current(app)) }
|
||||
val permission = Shizuku.OnRequestPermissionResultListener { _, _ -> onChange(current(app)) }
|
||||
// "Sticky" fires immediately if the binder already arrived before we registered.
|
||||
runCatching { Shizuku.addBinderReceivedListenerSticky(received) }
|
||||
runCatching { Shizuku.addBinderDeadListener(dead) }
|
||||
runCatching { Shizuku.addRequestPermissionResultListener(permission) }
|
||||
onChange(current(app))
|
||||
return {
|
||||
runCatching { Shizuku.removeBinderReceivedListener(received) }
|
||||
runCatching { Shizuku.removeBinderDeadListener(dead) }
|
||||
runCatching { Shizuku.removeRequestPermissionResultListener(permission) }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -10,15 +10,24 @@ import app.echo_lot.measurement.TestType
|
||||
import app.echo_lot.measurement.Tier
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.withContext
|
||||
import kotlinx.serialization.json.JsonObject
|
||||
import kotlinx.serialization.json.buildJsonObject
|
||||
import kotlinx.serialization.json.put
|
||||
import kotlinx.serialization.json.putJsonArray
|
||||
import kotlinx.serialization.json.putJsonObject
|
||||
import kotlinx.serialization.json.addJsonObject
|
||||
|
||||
/**
|
||||
* The Shizuku shell-tier probe: runs the privileged command battery (neighbor table, RA routes
|
||||
* with lifetimes, netlink monitor, IpClient DHCP logs, wifi dump) that the app UID cannot, and
|
||||
* captures the real per-device dump formats the production parsers must handle. Emitted as a
|
||||
* shizuku-tier `link.ip_monitor` test (the representative shell-tier link test); `exec_path`
|
||||
* records whether the UserService or the newProcess fallback carried it.
|
||||
* captures the real per-device dump formats the production parsers must handle. Emits three
|
||||
* shizuku-tier tests from the one battery:
|
||||
* - `link.ip_monitor` — the raw captures (the shell tier's ground truth), `exec_path` records
|
||||
* whether the UserService or the newProcess fallback carried it;
|
||||
* - `link.ra_source` — parsed from the v6 route table + `ip addr`: who advertises IPv6 here;
|
||||
* - `sec.arp_watch` — parsed from the neighbor table + monitor window: (ip → lladdr) pairs for
|
||||
* gateway-MAC-change detection.
|
||||
* The battery runs once; the derived tests parse its captures, so they share its time window.
|
||||
*/
|
||||
class ShizukuProbe {
|
||||
val type = TestType.LINK_IP_MONITOR
|
||||
@@ -34,24 +43,34 @@ class ShizukuProbe {
|
||||
"wifi_dump" to "dumpsys wifi 2>/dev/null | grep -iA1 -m 20 -e 'mDhcpResults' -e 'Gateway' -e 'DNS' || true",
|
||||
)
|
||||
|
||||
/** Runs the battery and returns a Test. [uuid]/[monoNs] come from the run's id/clock source. */
|
||||
suspend fun run(context: Context, uuid: () -> String, monoNs: () -> Long): Test = withContext(Dispatchers.IO) {
|
||||
val id = uuid()
|
||||
/**
|
||||
* Runs the battery and returns the three tests, battery first. [uuid]/[monoNs] come from the
|
||||
* run's id/clock source. When the shell tier is unavailable all three come back UNSUPPORTED —
|
||||
* one silent test would leave the other two types missing from the document, which reads as
|
||||
* "never attempted" rather than "tier absent".
|
||||
*/
|
||||
suspend fun run(context: Context, uuid: () -> String, monoNs: () -> Long): List<Test> = withContext(Dispatchers.IO) {
|
||||
val started = monoNs()
|
||||
val runner = ShizukuRunner(context)
|
||||
val st = runner.status()
|
||||
|
||||
fun envelope(status: TestStatus, evidence: kotlinx.serialization.json.JsonObject, metrics: kotlinx.serialization.json.JsonObject? = null) =
|
||||
Test(id = id, type = type, tier = tier, startedMonoNs = started, endedMonoNs = monoNs(),
|
||||
fun envelope(type: String, status: TestStatus, evidence: JsonObject, metrics: JsonObject? = null) =
|
||||
Test(id = uuid(), type = type, tier = tier, startedMonoNs = started, endedMonoNs = monoNs(),
|
||||
status = status, evidence = evidence, metrics = metrics)
|
||||
|
||||
fun allUnsupported(evidence: JsonObject) = listOf(
|
||||
envelope(TestType.LINK_IP_MONITOR, TestStatus.UNSUPPORTED, evidence),
|
||||
envelope(TestType.LINK_RA_SOURCE, TestStatus.UNSUPPORTED, evidence),
|
||||
envelope(TestType.SEC_ARP_WATCH, TestStatus.UNSUPPORTED, evidence),
|
||||
)
|
||||
|
||||
if (!st.binderAlive) {
|
||||
return@withContext envelope(TestStatus.UNSUPPORTED, buildJsonObject {
|
||||
return@withContext allUnsupported(buildJsonObject {
|
||||
put("binder_alive", false); put("detail", "Shizuku not running")
|
||||
})
|
||||
}
|
||||
if (!st.permissionGranted && !runner.requestPermission()) {
|
||||
return@withContext envelope(TestStatus.UNSUPPORTED, buildJsonObject {
|
||||
return@withContext allUnsupported(buildJsonObject {
|
||||
put("binder_alive", true); put("permission", false)
|
||||
})
|
||||
}
|
||||
@@ -77,6 +96,125 @@ class ShizukuProbe {
|
||||
ok >= 1 -> TestStatus.PARTIAL
|
||||
else -> TestStatus.FAILED
|
||||
}
|
||||
envelope(status, evidence, metrics)
|
||||
listOf(
|
||||
envelope(type, status, evidence, metrics),
|
||||
raSourceTest(batch, ::envelope),
|
||||
arpWatchTest(batch, ::envelope),
|
||||
)
|
||||
}
|
||||
|
||||
/** `link.ra_source` from the battery's `ip -6 route` / `ip addr` / `ip neigh` captures. */
|
||||
private fun raSourceTest(
|
||||
batch: ShizukuRunner.BatchResult,
|
||||
envelope: (String, TestStatus, JsonObject, JsonObject?) -> Test,
|
||||
): Test {
|
||||
val routeRaw = batch.results["ip6_route"]
|
||||
if (!DumpParsers.captureUsable(routeRaw)) {
|
||||
// The source command failed (executor sentinel or empty) — say so instead of
|
||||
// presenting "no default routes" as a measurement of the network.
|
||||
return envelope(TestType.LINK_RA_SOURCE, TestStatus.SKIPPED, buildJsonObject {
|
||||
put("exec_path", batch.execPath)
|
||||
put("reason", "ip -6 route capture unavailable: ${(routeRaw ?: "absent").take(80)}")
|
||||
}, null)
|
||||
}
|
||||
val routes = DumpParsers.parseV6DefaultRoutes(routeRaw)
|
||||
val macs = DumpParsers.parseInterfaceMacs(batch.results["ip_addr"])
|
||||
// The RA sender's own identity: its link-local gateway address resolved through the
|
||||
// neighbor table gives the router's MAC, which is what survives address renumbering.
|
||||
val neighMacs = DumpParsers.lladdrByIp(DumpParsers.parseNeighbors(batch.results["ip_neigh"]))
|
||||
|
||||
val evidence = buildJsonObject {
|
||||
put("exec_path", batch.execPath)
|
||||
putJsonArray("default_routes") {
|
||||
for (r in routes) addJsonObject {
|
||||
r.gateway?.let { put("gateway", it) }
|
||||
put("dev", r.dev)
|
||||
r.table?.let { put("table", it) }
|
||||
r.proto?.let { put("proto", it) }
|
||||
r.metric?.let { put("metric", it) }
|
||||
r.expiresSec?.let { put("expires_sec", it) }
|
||||
r.gateway?.let { gw -> neighMacs[gw]?.let { put("gateway_lladdr", it) } }
|
||||
}
|
||||
}
|
||||
putJsonObject("interface_mac") {
|
||||
// Only interfaces that actually carry a default route: the full MAC inventory
|
||||
// belongs to the raw capture, not to this test's claim.
|
||||
for (dev in routes.map { it.dev }.distinct()) macs[dev]?.let { put(dev, it) }
|
||||
}
|
||||
}
|
||||
val metrics = buildJsonObject {
|
||||
put("routes_total", routes.size)
|
||||
put("routes_ra", routes.count { it.proto == "ra" })
|
||||
}
|
||||
val status = when {
|
||||
routes.any { it.proto == "ra" } -> TestStatus.OK
|
||||
// Routes parsed but none RA-installed, or a capture we couldn't parse a single
|
||||
// default from: could be a genuinely RA-less link, could be vendor format drift —
|
||||
// PARTIAL keeps it visible either way instead of quietly claiming success.
|
||||
else -> TestStatus.PARTIAL
|
||||
}
|
||||
return envelope(TestType.LINK_RA_SOURCE, status, evidence, metrics)
|
||||
}
|
||||
|
||||
/** `sec.arp_watch` from the battery's `ip neigh` snapshot + `ip monitor` window. */
|
||||
private fun arpWatchTest(
|
||||
batch: ShizukuRunner.BatchResult,
|
||||
envelope: (String, TestStatus, JsonObject, JsonObject?) -> Test,
|
||||
): Test {
|
||||
val neighRaw = batch.results["ip_neigh"]
|
||||
val monitorRaw = batch.results["ip_monitor"]
|
||||
val neighUsable = DumpParsers.captureUsable(neighRaw)
|
||||
// The monitor window is best-effort (EXEC_TIMEOUT under newProcess on the Lenovo); the
|
||||
// snapshot alone still yields the (ip → lladdr) pairs the MAC-change finding diffs.
|
||||
val monitorRan = DumpParsers.captureUsable(monitorRaw)
|
||||
if (!neighUsable && !monitorRan) {
|
||||
return envelope(TestType.SEC_ARP_WATCH, TestStatus.SKIPPED, buildJsonObject {
|
||||
put("exec_path", batch.execPath)
|
||||
put("reason", "ip neigh capture unavailable: ${(neighRaw ?: "absent").take(80)}")
|
||||
}, null)
|
||||
}
|
||||
val neighbors = DumpParsers.parseNeighbors(neighRaw)
|
||||
val events = DumpParsers.parseNeighborEvents(monitorRaw)
|
||||
|
||||
val evidence = buildJsonObject {
|
||||
put("exec_path", batch.execPath)
|
||||
putJsonArray("neighbors") {
|
||||
for (n in neighbors) addJsonObject {
|
||||
put("ip", n.ip)
|
||||
n.dev?.let { put("dev", it) }
|
||||
n.lladdr?.let { put("lladdr", it) }
|
||||
n.state?.let { put("state", it) }
|
||||
if (n.router) put("router", true)
|
||||
}
|
||||
}
|
||||
// The comparison surface, precomputed: a MAC-change finding diffs this map between
|
||||
// runs without re-walking the neighbor array.
|
||||
putJsonObject("lladdr_by_ip") {
|
||||
for ((ip, mac) in DumpParsers.lladdrByIp(neighbors)) put(ip, mac)
|
||||
}
|
||||
put("monitor_ran", monitorRan)
|
||||
if (!monitorRan) put("monitor_reason", (monitorRaw ?: "absent").take(80))
|
||||
putJsonArray("monitor_events") {
|
||||
for (e in events) addJsonObject {
|
||||
put("ip", e.entry.ip)
|
||||
e.entry.dev?.let { put("dev", it) }
|
||||
e.entry.lladdr?.let { put("lladdr", it) }
|
||||
e.entry.state?.let { put("state", it) }
|
||||
if (e.deleted) put("deleted", true)
|
||||
}
|
||||
}
|
||||
}
|
||||
val metrics = buildJsonObject {
|
||||
put("neighbors_total", neighbors.size)
|
||||
put("neighbors_with_lladdr", neighbors.count { it.lladdr != null })
|
||||
put("monitor_events", events.size)
|
||||
}
|
||||
val status = when {
|
||||
neighbors.isNotEmpty() -> TestStatus.OK
|
||||
// A snapshot that parsed to nothing (or a monitor-only capture) is thin evidence:
|
||||
// usable command output with zero entries is unusual enough to flag, not to fail.
|
||||
else -> TestStatus.PARTIAL
|
||||
}
|
||||
return envelope(TestType.SEC_ARP_WATCH, status, evidence, metrics)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,303 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package app.echo_lot.shizuku
|
||||
|
||||
import kotlin.test.Test
|
||||
import kotlin.test.assertEquals
|
||||
import kotlin.test.assertFalse
|
||||
import kotlin.test.assertNull
|
||||
import kotlin.test.assertTrue
|
||||
|
||||
/**
|
||||
* The fixtures below are the REAL shell-battery captures from the two archived prober reports
|
||||
* (echolot-prober/reports/CPH2747-android16-sdk36-build5.json — OnePlus 15, UserService path;
|
||||
* TB330FU-android15-sdk35-build5.json — Lenovo TB330FU, newProcess fallback), trimmed to the
|
||||
* relevant lines but otherwise verbatim. That includes their warts on purpose: the UserService
|
||||
* path's stray `uid=2000` first line, the 1200-char evidence trim cutting the last line mid-word,
|
||||
* the Lenovo's 10-digit route table ids and its `EXEC_TIMEOUT(newProcess)` monitor sentinel.
|
||||
* A parser that only survives clean textbook output has not been tested.
|
||||
*/
|
||||
class DumpParsersTest {
|
||||
|
||||
// ---- OnePlus 15 (CPH2747, Android 16) — UserService exec path ----
|
||||
|
||||
private val onePlusIp6Route = """
|
||||
uid=2000
|
||||
fe80::/64 dev wlan0 table 1028 proto kernel metric 256 pref medium
|
||||
fe80::/64 dev wlan0 table 1028 proto static metric 1024 pref medium
|
||||
default via fe80::7a9a:18ff:fe54:b8f9 dev wlan0 table 1028 proto ra metric 1024 expires 1269sec pref medium
|
||||
fe80::/64 dev vgate0 table 1031 proto kernel metric 256 pref medium
|
||||
2001:4bb8:417:bd78::/64 dev rmnet_data4 table 1032 proto kernel metric 256 pref medium
|
||||
2001:4bb8:417:bd78::/64 dev rmnet_data4 table 1032 proto static metric 1024 pref medium
|
||||
fe80::/64 dev rmnet_data4 table 1032 proto kernel metric 256 pref medium
|
||||
default via fe80::246f:12be:21ef:1b54 dev rmnet_data4 table 1032 proto ra metric 1024 expires 64373sec hoplimit 255 pref medium
|
||||
2001:4bb8:2fb:fe4c::/64 dev rmnet_data2 table 1000000022 proto static metric 1024 pref medium
|
||||
fe80::/64 dev wlan0 table 1000000028 proto static metric 1024 pref medium
|
||||
2001:4bb8:417:bd78::/64 dev rmnet_data4 table 1000000032 proto static metric 1024 pref medium
|
||||
fe80::/64 dev dummy0 table 1002 proto kernel metric 256 pref medium
|
||||
default dev dummy0 table 1002 proto static metric 1024 pref medium
|
||||
fe80::/64 dev ifb0 table 1003 proto kernel metric 256 pref medium
|
||||
fe80::/64 dev ifb1 table 1004 proto kerne
|
||||
""".trimIndent()
|
||||
|
||||
private val onePlusIpNeigh = """
|
||||
uid=2000
|
||||
10.13.102.111 dev wlan0 FAILED
|
||||
10.13.102.50 dev wlan0 lladdr 50:57:9c:4f:7a:3c STALE
|
||||
10.13.102.31 dev wlan0 lladdr 98:5f:d3:f6:f1:75 STALE
|
||||
10.13.102.116 dev wlan0 lladdr 0c:08:b4:03:68:0e STALE
|
||||
10.13.102.120 dev wlan0 lladdr 0e:d8:14:58:6c:8b STALE
|
||||
10.13.102.5 dev wlan0 lladdr 90:09:d0:1a:83:e4 STALE
|
||||
10.13.102.1 dev wlan0 lladdr 78:9a:18:54:b8:f9 REACHABLE
|
||||
10.13.102.21 dev wlan0 lladdr c8:7f:54:01:94:7c STALE
|
||||
fe80::babe:f4ff:febc:caf9 dev wlan0 lladdr b8:be:f4:bc:ca:f9 REACHABLE
|
||||
fe80::7a9a:18ff:fe54:b8f9 dev wlan0 lladdr 78:9a:18:54:b8:f9 router STALE
|
||||
fe80::babe:f4ff:febc:cacf dev wlan0 lladdr b8:be:f4:bc:ca:cf REACHABLE
|
||||
fe80::babe:f4ff:fec2:bf14 dev wlan0 lladdr b8:be:f4:c2:bf:14 REACHABLE
|
||||
""".trimIndent()
|
||||
|
||||
// The 1200-char trim cut this capture off before wlan0's stanza — so the real archived
|
||||
// evidence has NO MAC for the interface that carries the default route. The parser must
|
||||
// yield what is there and nothing else; the probe records the gap instead of inventing one.
|
||||
private val onePlusIpAddr = """
|
||||
uid=2000
|
||||
1: lo: <LOOPBACK,UP,LOWER_UP> mtu 65536 qdisc noqueue state UNKNOWN group default qlen 1000
|
||||
link/loopback 00:00:00:00:00:00 brd 00:00:00:00:00:00
|
||||
inet 127.0.0.1/8 scope host lo
|
||||
valid_lft forever preferred_lft forever
|
||||
inet6 ::1/128 scope host
|
||||
valid_lft forever preferred_lft forever
|
||||
2: dummy0: <BROADCAST,NOARP,UP,LOWER_UP> mtu 1500 qdisc noqueue state UNKNOWN group default qlen 1000
|
||||
link/ether be:3d:e2:93:78:b9 brd ff:ff:ff:ff:ff:ff
|
||||
inet6 fe80::bc3d:e2ff:fe93:78b9/64 scope link
|
||||
valid_lft forever preferred_lft forever
|
||||
3: ifb0: <BROADCAST,NOARP,UP,LOWER_UP> mtu 1500 qdisc htb state UNKNOWN group default qlen 1000
|
||||
link/ether ba:6e:46:b5:3d:bb brd ff:ff:ff:ff:ff:ff
|
||||
4: ifb1: <BROADCAST,NOARP,UP,LOWER_UP> mtu 1500 qdisc htb state UNKNOWN group default qlen 1000
|
||||
link/ether d6:2a:e2:f5:93:8f brd ff:ff:ff:ff:ff:ff
|
||||
5: tunl0@NONE: <NOARP> mtu 1480 qdisc noop state DOWN group default qlen 1000
|
||||
link/ipip 0.0.0.0 brd 0.0.0.0
|
||||
6: gre0@NONE: <NO
|
||||
""".trimIndent()
|
||||
|
||||
// A monitor window that ran but saw nothing: the capture is "usable", just empty of events.
|
||||
private val onePlusIpMonitor = "uid=2000"
|
||||
|
||||
// ---- Lenovo TB330FU (Android 15) — newProcess fallback ----
|
||||
|
||||
private val lenovoIp6Route = """
|
||||
fe80::/64 dev wlan0 table 1000000015 proto static metric 1024 pref medium
|
||||
fe80::/64 dev dummy0 table 1002 proto kernel metric 256 pref medium
|
||||
default dev dummy0 table 1002 proto static metric 1024 pref medium
|
||||
fe80::/64 dev wlan0 table 1015 proto kernel metric 256 pref medium
|
||||
fe80::/64 dev wlan0 table 1015 proto static metric 1024 pref medium
|
||||
default via fe80::7a9a:18ff:fe54:b8f9 dev wlan0 table 1015 proto ra metric 1024 expires 1622sec pref medium
|
||||
local ::1 dev lo table local proto kernel metric 0 pref medium
|
||||
local fe80::416:b9ff:feac:5b65 dev wlan0 table local proto kernel metric 0 pref medium
|
||||
local fe80::1450:43ff:feec:93c4 dev dummy0 table local proto kernel metric 0 pref medium
|
||||
multicast ff00::/8 dev dummy0 table local proto kernel metric 256 pref medium
|
||||
multicast ff00::/8 dev wlan0 table local proto kernel metric 256 pref medium
|
||||
""".trimIndent()
|
||||
|
||||
private val lenovoIpNeigh = """
|
||||
10.13.102.21 dev wlan0 lladdr c8:7f:54:01:94:7c STALE
|
||||
10.13.102.64 dev wlan0 lladdr b8:be:f4:c2:bf:14 STALE
|
||||
10.13.102.5 dev wlan0 lladdr 90:09:d0:1a:83:e4 STALE
|
||||
10.13.102.1 dev wlan0 lladdr 78:9a:18:54:b8:f9 STALE
|
||||
10.13.102.79 dev wlan0 lladdr 02:11:32:25:63:bb STALE
|
||||
fe80::babe:f4ff:febc:cacf dev wlan0 lladdr b8:be:f4:bc:ca:cf STALE
|
||||
fe80::7a9a:18ff:fe54:b8f9 dev wlan0 lladdr 78:9a:18:54:b8:f9 router STALE
|
||||
fe80::babe:f4ff:fec2:bf14 dev wlan0 lladdr b8:be:f4:c2:bf:14 STALE
|
||||
fe80::babe:f4ff:febc:caf9 dev wlan0 lladdr b8:be:f4:bc:ca:f9 STALE
|
||||
""".trimIndent()
|
||||
|
||||
private val lenovoIpAddr = """
|
||||
1: lo: <LOOPBACK,UP,LOWER_UP> mtu 65536 qdisc noqueue state UNKNOWN group default qlen 1000
|
||||
link/loopback 00:00:00:00:00:00 brd 00:00:00:00:00:00
|
||||
inet 127.0.0.1/8 scope host lo
|
||||
valid_lft forever preferred_lft forever
|
||||
2: dummy0: <BROADCAST,NOARP,UP,LOWER_UP> mtu 1500 qdisc noqueue state UNKNOWN group default qlen 1000
|
||||
link/ether 16:50:43:ec:93:c4 brd ff:ff:ff:ff:ff:ff
|
||||
inet6 fe80::1450:43ff:feec:93c4/64 scope link
|
||||
valid_lft forever preferred_lft forever
|
||||
3: ifb0: <BROADCAST,NOARP> mtu 1500 qdisc noop state DOWN group default qlen 32
|
||||
link/ether f6:d4:d4:9b:51:9c brd ff:ff:ff:ff:ff:ff
|
||||
4: ifb1: <BROADCAST,NOARP> mtu 1500 qdisc noop state DOWN group default qlen 32
|
||||
link/ether fe:16:ea:60:a2:d1 brd ff:ff:ff:ff:ff:ff
|
||||
5: tunl0@NONE: <NOARP> mtu 1480 qdisc noop state DOWN group default qlen 1000
|
||||
link/ipip 0.0.0.0 brd 0.0.0.0
|
||||
7: gretap0@NONE: <BROADCAST,MULTICAST> mtu 1462 qdisc noop state DOWN group default qlen 1000
|
||||
link/ether 00:00:00:00:00:00 brd ff:ff:ff:ff:f
|
||||
""".trimIndent()
|
||||
|
||||
// On the Lenovo the 5 s monitor window exceeds the newProcess exec timeout — the executor's
|
||||
// sentinel is all we get, and the arp_watch test must still stand on the snapshot alone.
|
||||
private val lenovoIpMonitor = "EXEC_TIMEOUT(newProcess)"
|
||||
|
||||
// ---- link.ra_source: v6 default routes ----
|
||||
|
||||
@Test
|
||||
fun onePlusDefaultRoutesParsed() {
|
||||
val routes = DumpParsers.parseV6DefaultRoutes(onePlusIp6Route)
|
||||
assertEquals(3, routes.size)
|
||||
|
||||
val wlan = routes.single { it.dev == "wlan0" }
|
||||
assertEquals("fe80::7a9a:18ff:fe54:b8f9", wlan.gateway)
|
||||
assertEquals("1028", wlan.table)
|
||||
assertEquals("ra", wlan.proto)
|
||||
assertEquals(1024L, wlan.metric)
|
||||
assertEquals(1269L, wlan.expiresSec)
|
||||
|
||||
// The cellular default: `hoplimit 255` sits between expires and pref and must not derail
|
||||
// the token walk.
|
||||
val rmnet = routes.single { it.dev == "rmnet_data4" }
|
||||
assertEquals("fe80::246f:12be:21ef:1b54", rmnet.gateway)
|
||||
assertEquals("1032", rmnet.table)
|
||||
assertEquals(64373L, rmnet.expiresSec)
|
||||
|
||||
// Android's gateway-less dummy0 default is a real route; it is the proto that tells a
|
||||
// consumer it is not an RA.
|
||||
val dummy = routes.single { it.dev == "dummy0" }
|
||||
assertNull(dummy.gateway)
|
||||
assertEquals("static", dummy.proto)
|
||||
assertNull(dummy.expiresSec)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun lenovoNumberedTablesAllCaptured() {
|
||||
val routes = DumpParsers.parseV6DefaultRoutes(lenovoIp6Route)
|
||||
// Two default routes: the RA one in table 1015 and the dummy0 one in 1002. The 10-digit
|
||||
// table 1000000015 and the `table local` rows carry no default and must neither appear
|
||||
// nor break parsing.
|
||||
assertEquals(setOf("1002", "1015"), routes.map { it.table }.toSet())
|
||||
|
||||
val ra = routes.single { it.proto == "ra" }
|
||||
assertEquals("fe80::7a9a:18ff:fe54:b8f9", ra.gateway)
|
||||
assertEquals("wlan0", ra.dev)
|
||||
assertEquals("1015", ra.table)
|
||||
assertEquals(1622L, ra.expiresSec)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun tenDigitTableIdOnADefaultRouteSurvives() {
|
||||
// Not seen on a default route in the wild yet, but the Lenovo proves vendors put routes
|
||||
// in tables past Int range — the day one holds a default, it must not overflow away.
|
||||
val routes = DumpParsers.parseV6DefaultRoutes(
|
||||
"default via fe80::1 dev wlan0 table 1000000015 proto ra metric 1024 expires 100sec pref medium"
|
||||
)
|
||||
assertEquals(1, routes.size)
|
||||
assertEquals("1000000015", routes[0].table)
|
||||
assertEquals(100L, routes[0].expiresSec)
|
||||
}
|
||||
|
||||
// ---- link.ra_source: interface MACs ----
|
||||
|
||||
@Test
|
||||
fun onePlusInterfaceMacsParsed() {
|
||||
val macs = DumpParsers.parseInterfaceMacs(onePlusIpAddr)
|
||||
assertEquals("be:3d:e2:93:78:b9", macs["dummy0"])
|
||||
assertEquals("ba:6e:46:b5:3d:bb", macs["ifb0"])
|
||||
// link/loopback and link/ipip are not identities.
|
||||
assertFalse("lo" in macs)
|
||||
assertFalse("tunl0" in macs)
|
||||
// The capture is cut mid-stanza-header ("6: gre0@NONE: <NO") — no exception, no entry.
|
||||
assertNull(macs["gre0"])
|
||||
}
|
||||
|
||||
@Test
|
||||
fun lenovoInterfaceMacsParsed() {
|
||||
val macs = DumpParsers.parseInterfaceMacs(lenovoIpAddr)
|
||||
assertEquals("16:50:43:ec:93:c4", macs["dummy0"])
|
||||
assertEquals("fe:16:ea:60:a2:d1", macs["ifb1"])
|
||||
// The @-suffixed stanza name resolves to the bare interface name.
|
||||
assertEquals("00:00:00:00:00:00", macs["gretap0"])
|
||||
}
|
||||
|
||||
// ---- sec.arp_watch: neighbor snapshot ----
|
||||
|
||||
@Test
|
||||
fun onePlusNeighborsParsed() {
|
||||
val n = DumpParsers.parseNeighbors(onePlusIpNeigh)
|
||||
assertEquals(12, n.size) // the `uid=2000` noise line is not a neighbor
|
||||
|
||||
val failed = n.single { it.ip == "10.13.102.111" }
|
||||
assertNull(failed.lladdr)
|
||||
assertEquals("FAILED", failed.state)
|
||||
assertEquals("wlan0", failed.dev)
|
||||
|
||||
val gw = n.single { it.ip == "10.13.102.1" }
|
||||
assertEquals("78:9a:18:54:b8:f9", gw.lladdr)
|
||||
assertEquals("REACHABLE", gw.state)
|
||||
|
||||
val v6gw = n.single { it.ip == "fe80::7a9a:18ff:fe54:b8f9" }
|
||||
assertTrue(v6gw.router)
|
||||
assertEquals("78:9a:18:54:b8:f9", v6gw.lladdr)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun lenovoNeighborsParsed() {
|
||||
val n = DumpParsers.parseNeighbors(lenovoIpNeigh)
|
||||
assertEquals(9, n.size)
|
||||
assertTrue(n.all { it.lladdr != null && it.state == "STALE" })
|
||||
assertEquals(1, n.count { it.router })
|
||||
}
|
||||
|
||||
@Test
|
||||
fun lladdrByIpIsTheComparisonSurface() {
|
||||
val pairs = DumpParsers.lladdrByIp(DumpParsers.parseNeighbors(onePlusIpNeigh))
|
||||
// 12 neighbors, 11 with a MAC — the FAILED entry must drop out, or a diff against a
|
||||
// later run would flag "null → MAC" as a gateway change.
|
||||
assertEquals(11, pairs.size)
|
||||
assertEquals("78:9a:18:54:b8:f9", pairs["10.13.102.1"])
|
||||
assertFalse("10.13.102.111" in pairs)
|
||||
}
|
||||
|
||||
// ---- sec.arp_watch: monitor window ----
|
||||
|
||||
@Test
|
||||
fun monitorSentinelIsUnusableAndYieldsNoEvents() {
|
||||
assertFalse(DumpParsers.captureUsable(lenovoIpMonitor))
|
||||
assertTrue(DumpParsers.parseNeighborEvents(lenovoIpMonitor).isEmpty())
|
||||
}
|
||||
|
||||
@Test
|
||||
fun quietMonitorWindowIsUsableButEmpty() {
|
||||
// OnePlus: the monitor ran (only the uid noise line came back) — "ran and saw nothing"
|
||||
// must stay distinguishable from "never ran".
|
||||
assertTrue(DumpParsers.captureUsable(onePlusIpMonitor))
|
||||
assertTrue(DumpParsers.parseNeighborEvents(onePlusIpMonitor).isEmpty())
|
||||
}
|
||||
|
||||
@Test
|
||||
fun monitorNeighEventsParsedFromLabeledLines() {
|
||||
// Synthetic, in `ip monitor all` label format — neither archived run caught a live
|
||||
// transition, but the format is fixed by iproute2's print_neigh/print_headers.
|
||||
val sample = """
|
||||
[NEIGH]10.13.102.1 dev wlan0 lladdr 78:9a:18:54:b8:f9 REACHABLE
|
||||
[NEIGH]Deleted 10.13.102.5 dev wlan0 lladdr 90:09:d0:1a:83:e4 STALE
|
||||
[ROUTE]default via 10.13.102.1 dev wlan0 table 1015
|
||||
[NEIGH]fe80::7a9a:18ff:fe54:b8f9 dev wlan0 lladdr 78:9a:18:54:b8:f9 router STALE
|
||||
""".trimIndent()
|
||||
val events = DumpParsers.parseNeighborEvents(sample)
|
||||
assertEquals(3, events.size) // the ROUTE line belongs to a different family
|
||||
assertEquals("10.13.102.1", events[0].entry.ip)
|
||||
assertFalse(events[0].deleted)
|
||||
assertTrue(events[1].deleted)
|
||||
assertEquals("90:09:d0:1a:83:e4", events[1].entry.lladdr)
|
||||
assertTrue(events[2].entry.router)
|
||||
}
|
||||
|
||||
// ---- degradation: missing or garbage input ----
|
||||
|
||||
@Test
|
||||
fun missingAndGarbageInputYieldsEmptyResultsNotExceptions() {
|
||||
for (bad in listOf(null, "", " \n ", "EXEC_TIMEOUT(newProcess)", "SHIZUKU_BINDER_DEAD",
|
||||
"NEWPROCESS_UNAVAILABLE", "total garbage\nno routes here at all\ndefault", "default")) {
|
||||
assertTrue(DumpParsers.parseV6DefaultRoutes(bad).isEmpty(), "routes from: $bad")
|
||||
assertTrue(DumpParsers.parseNeighbors(bad).isEmpty(), "neighbors from: $bad")
|
||||
assertTrue(DumpParsers.parseInterfaceMacs(bad).isEmpty(), "macs from: $bad")
|
||||
assertTrue(DumpParsers.parseNeighborEvents(bad).isEmpty(), "events from: $bad")
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -8,6 +8,7 @@ lifecycle = "2.8.7"
|
||||
activityCompose = "1.9.3"
|
||||
composeBom = "2024.10.01"
|
||||
shizuku = "13.1.5"
|
||||
junit4 = "4.13.2"
|
||||
|
||||
[libraries]
|
||||
kotlinx-serialization-json = { group = "org.jetbrains.kotlinx", name = "kotlinx-serialization-json", version.ref = "kotlinxSerialization" }
|
||||
@@ -23,6 +24,10 @@ androidx-ui-tooling = { group = "androidx.compose.ui", name = "ui-tooling" }
|
||||
androidx-ui-tooling-preview = { group = "androidx.compose.ui", name = "ui-tooling-preview" }
|
||||
androidx-material3 = { group = "androidx.compose.material3", name = "material3" }
|
||||
shizuku-api = { group = "dev.rikka.shizuku", name = "api", version.ref = "shizuku" }
|
||||
# Android-module unit tests run on JUnit 4 (AGP's default); the JVM modules use kotlin("test")
|
||||
# with the JUnit Platform instead — that helper isn't available under AGP 9's built-in Kotlin.
|
||||
kotlin-test-junit = { group = "org.jetbrains.kotlin", name = "kotlin-test-junit", version.ref = "kotlin" }
|
||||
junit4 = { group = "junit", name = "junit", version.ref = "junit4" }
|
||||
shizuku-provider = { group = "dev.rikka.shizuku", name = "provider", version.ref = "shizuku" }
|
||||
|
||||
[plugins]
|
||||
|
||||
@@ -5,8 +5,13 @@
|
||||
# Mints an enrollment link on the probe server and prints it — as text, as a QR code if
|
||||
# `qrencode` is around, and as an adb command if a device is attached.
|
||||
#
|
||||
# The admin listener is localhost-only by design, so this goes over SSH. The link carries a
|
||||
# single-use bearer token: treat it like a password until it is redeemed.
|
||||
# The link is minted by the server binary on the host rather than over HTTP. The admin API this
|
||||
# used to call is gone: the admin UI that replaced it is authenticated, as it should be, and
|
||||
# adding a second unauthenticated door on loopback is what briefly exposed the old one to the
|
||||
# network. A root shell on the host needs no authentication anyway — whoever has one already has
|
||||
# every privilege the server has.
|
||||
#
|
||||
# The link carries a single-use bearer token: treat it like a password until it is redeemed.
|
||||
#
|
||||
# Usage: echolot-app/scripts/enroll-link.sh [note]
|
||||
set -euo pipefail
|
||||
@@ -14,13 +19,18 @@ set -euo pipefail
|
||||
SSH_HOST="${ECHOLOT_SSH:-claude-echolot}"
|
||||
NOTE="${1:-manual}"
|
||||
|
||||
MINTED=$(ssh -o BatchMode=yes "$SSH_HOST" \
|
||||
"curl -s -X POST 'http://127.0.0.1:8444/admin/enroll-tokens?note=$NOTE'")
|
||||
# The env file is sourced rather than assumed: the state directory and the public URL live there,
|
||||
# and minting against the wrong state directory would produce a token the running server has
|
||||
# never heard of.
|
||||
REMOTE='set -a; . /etc/echolot/server.env; set +a;
|
||||
exec /usr/local/bin/echolot-server --mint-enroll-token'
|
||||
RAW=$(ssh -o BatchMode=yes "$SSH_HOST" "sudo sh -c \"$REMOTE '$NOTE'\"" 2>/dev/null || true)
|
||||
URI=$(printf '%s' "$RAW" | tr -d '\r' | grep -m1 '^echolot://enroll' || true)
|
||||
|
||||
URI=$(printf '%s' "$MINTED" | python -c 'import json,sys;print(json.load(sys.stdin).get("enroll_uri",""))')
|
||||
if [ -z "$URI" ]; then
|
||||
echo "server returned no enroll_uri (needs server-v0.5.4+):" >&2
|
||||
echo "$MINTED" >&2
|
||||
echo "could not mint a link — needs a server with --mint-enroll-token (v0.9.7+)." >&2
|
||||
echo "raw response:" >&2
|
||||
printf '%s\n' "$RAW" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
|
||||
+16
-8
@@ -106,9 +106,7 @@ ECHOLOT_UDP_LISTEN=203.0.113.10:8442,203.0.113.11:8442,[2001:db8::10]:8442,[2001
|
||||
|
||||
Passing `--self-update-api` to `--install-systemd` additionally installs a daily randomized
|
||||
self-update timer (`echolot-server-update.timer`) that restarts the service after a successful
|
||||
update. Updates are checksum-verified against the release's `SHA256SUMS` (integrity, not
|
||||
authenticity — signature verification remains TODO before treating the update source as
|
||||
untrusted).
|
||||
update.
|
||||
|
||||
### Self-update (opt-in, native only)
|
||||
|
||||
@@ -119,14 +117,24 @@ echolot-server --self-update \
|
||||
|
||||
Fetches the newest `server-v*` release asset for this OS/arch and atomically replaces the
|
||||
binary; systemd's `Restart=` brings up the new version. Run it from a systemd timer for
|
||||
unattended updates. TODO before enabling anywhere untrusted: signature verification of the
|
||||
downloaded asset.
|
||||
unattended updates.
|
||||
|
||||
Releases are trusted by signature, not by host: CI signs `SHA256SUMS` with an ed25519 key that
|
||||
exists only in its secret store (`RELEASE_SIGNING_KEY`), and the updater verifies
|
||||
`SHA256SUMS.sig` against the public key baked into the binary before believing any checksum —
|
||||
an unsigned or re-signed release is refused, so a compromised Gitea can withhold updates but not
|
||||
inject one. Running your own release pipeline? Mint a keypair with
|
||||
`go run ./cmd/release-sign -gen`, set the secret, and point `ECHOLOT_SELF_UPDATE_PUBKEY` (or
|
||||
`--self-update-pubkey`) at your public key.
|
||||
|
||||
## First contact
|
||||
|
||||
```sh
|
||||
# 1. mint an enrollment token (admin listener is loopback-only)
|
||||
curl -s -X POST 'http://127.0.0.1:8444/admin/enroll-tokens?note=phone'
|
||||
# 1. mint an enrollment token (admin listener is loopback-only; authenticates as the
|
||||
# break-glass admin — set that once with --set-admin-password)
|
||||
curl -s -u admin:<password> -H 'Accept: application/json' \
|
||||
-X POST 'http://127.0.0.1:8444/admin/enroll-tokens?note=phone'
|
||||
# → { "token": "…", "expires_in_s": 86400, "enroll_uri": "echolot://enroll?…" }
|
||||
# 2. device enrolls with it (normally via the echolot:// QR code)
|
||||
curl -sk -X POST https://<host>:8443/v1/enroll -H 'Authorization: Bearer <token>'
|
||||
# 3. device fetches its profile
|
||||
@@ -144,7 +152,7 @@ go vet ./...
|
||||
|
||||
CI (`.gitea/workflows/build-server.yml`): tests on every push touching `server/`;
|
||||
tagging `server-v1.2.3` builds + pushes the container image to the Gitea registry and
|
||||
attaches static linux amd64/arm64 binaries (+ SHA256SUMS) to a release — the same
|
||||
attaches static linux amd64/arm64 binaries (+ signed SHA256SUMS) to a release — the same
|
||||
artifacts `--self-update` consumes.
|
||||
|
||||
## TLS for the admin UI
|
||||
|
||||
@@ -19,7 +19,6 @@ import (
|
||||
"crypto/tls"
|
||||
"crypto/x509"
|
||||
"crypto/x509/pkix"
|
||||
"encoding/json"
|
||||
"encoding/pem"
|
||||
"errors"
|
||||
"fmt"
|
||||
@@ -39,6 +38,7 @@ import (
|
||||
|
||||
"echo-lot.app/server/internal/acmehttp"
|
||||
"echo-lot.app/server/internal/adminauth"
|
||||
"echo-lot.app/server/internal/adminui"
|
||||
"echo-lot.app/server/internal/canarydns"
|
||||
"echo-lot.app/server/internal/certreload"
|
||||
"echo-lot.app/server/internal/compat"
|
||||
@@ -46,6 +46,7 @@ import (
|
||||
"echo-lot.app/server/internal/control"
|
||||
"echo-lot.app/server/internal/dataplane"
|
||||
"echo-lot.app/server/internal/oidc"
|
||||
"echo-lot.app/server/internal/ratelimit"
|
||||
"echo-lot.app/server/internal/runs"
|
||||
"echo-lot.app/server/internal/selftest"
|
||||
"echo-lot.app/server/internal/selfupdate"
|
||||
@@ -110,17 +111,85 @@ func run() error {
|
||||
return system.UninstallSystemd()
|
||||
case actions.SetAdminPassword:
|
||||
return setAdminPassword(cfg)
|
||||
case actions.MintEnrollToken != "":
|
||||
return mintEnrollToken(cfg, actions.MintEnrollToken)
|
||||
case actions.SelfUpdate:
|
||||
return selfupdate.Run(cfg.SelfUpdateAPI, Version)
|
||||
return selfupdate.Run(cfg.SelfUpdateAPI, cfg.SelfUpdatePubKey, Version)
|
||||
}
|
||||
return serve(cfg)
|
||||
}
|
||||
|
||||
// mintEnrollToken prints a §2.1 bootstrap link for a new device.
|
||||
//
|
||||
// The link is assembled here rather than by hand because it has to carry the public URL and the
|
||||
// base64 SPKI pin percent-encoded correctly, and a pin wrong by one character fails later as an
|
||||
// inscrutable TLS error rather than as a bad pin.
|
||||
func mintEnrollToken(cfg *config.Config, note string) error {
|
||||
st, err := store.Open(cfg.StateDir)
|
||||
if err != nil {
|
||||
return fmt.Errorf("state store: %w", err)
|
||||
}
|
||||
cert, err := loadOrCreateCert(cfg)
|
||||
if err != nil {
|
||||
return fmt.Errorf("tls: %w", err)
|
||||
}
|
||||
pin, err := control.SpkiPinB64(cert)
|
||||
if err != nil {
|
||||
return fmt.Errorf("pin: %w", err)
|
||||
}
|
||||
tok, err := st.NewEnrollToken(24*time.Hour, note)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
base := cfg.PublicControlURL
|
||||
if base == "" {
|
||||
return fmt.Errorf("set ECHOLOT_PUBLIC_URL so the link can say where to connect")
|
||||
}
|
||||
fmt.Println(control.EnrollmentURI(base, pin, tok))
|
||||
// stderr, so piping the command somewhere yields the link alone.
|
||||
fmt.Fprintln(os.Stderr, "\nSingle use, valid 24 hours. Treat it like a password until spent.")
|
||||
return nil
|
||||
}
|
||||
|
||||
// controlURL is the address devices connect to: the hostname that selects the pinned certificate.
|
||||
//
|
||||
// Falls back to the public URL when no separate control hostname is configured, so a server that
|
||||
// does not share the admin port keeps answering discovery with something usable.
|
||||
func controlURL(cfg *config.Config) string {
|
||||
if cfg.ControlHostname == "" {
|
||||
return cfg.PublicControlURL
|
||||
}
|
||||
return "https://" + cfg.ControlHostname
|
||||
}
|
||||
|
||||
func serve(cfg *config.Config) error {
|
||||
slog.Info("echolot-server starting", "version", Version, "mode",
|
||||
map[bool]string{true: "container", false: "native"}[cfg.Docker],
|
||||
"state_dir", cfg.StateDir)
|
||||
|
||||
// The reserved addresses' proof is only as good as 80/443 actually being free there.
|
||||
// CheckReserved already keeps OUR listeners away, but a process outside this config pollutes
|
||||
// them just as silently — the adb-beacon receiver on 0.0.0.0:443 did exactly that. So ask
|
||||
// the OS, not the config. A hard stop for the same reason CheckReserved is one: the failure
|
||||
// is invisible, and its first symptom is a measurement calling an intercepted network clean.
|
||||
if reserved := cfg.ReservedIPs(); len(reserved) > 0 {
|
||||
occupied, unverifiable := selftest.ReservedWebPortsFree(reserved)
|
||||
if len(occupied) > 0 {
|
||||
return fmt.Errorf(
|
||||
"refusing to start: something outside this server is listening on reserved "+
|
||||
"measurement address(es) %s\n"+
|
||||
"The interception proof those addresses exist for is void while anything "+
|
||||
"answers there.\nFind it with `ss -tlnp | grep -E ':(80|443) '`, stop it, "+
|
||||
"or remove the address from ECHOLOT_RESERVED_ADDRS if it is no longer reserved",
|
||||
strings.Join(occupied, ", "))
|
||||
}
|
||||
for _, u := range unverifiable {
|
||||
// Not fatal: an address with a typo, or one this host no longer carries, is a
|
||||
// config problem — refusing to serve over it would take the whole instrument down.
|
||||
slog.Warn("could not verify a reserved web port is free", "addr", u)
|
||||
}
|
||||
}
|
||||
|
||||
st, err := store.Open(cfg.StateDir)
|
||||
if err != nil {
|
||||
return fmt.Errorf("state store: %w", err)
|
||||
@@ -137,6 +206,16 @@ func serve(cfg *config.Config) error {
|
||||
|
||||
sessions := session.NewManager(15 * time.Minute)
|
||||
dp := &dataplane.Server{Sessions: sessions}
|
||||
// Spec §2.5 ceilings. Control plane answers 429; the data plane drops silently. 0 = off.
|
||||
if cfg.RateUDPPps > 0 {
|
||||
pps := float64(cfg.RateUDPPps)
|
||||
// Burst of two seconds' worth: a 5000-packet train arrives as one burst by design.
|
||||
dp.PacketRate = ratelimit.New(pps, 2*pps)
|
||||
}
|
||||
if cfg.RateUDPKbps > 0 {
|
||||
bytesPerSec := float64(cfg.RateUDPKbps) * 125 // kbps -> bytes/s
|
||||
dp.ByteRate = ratelimit.New(bytesPerSec, bytesPerSec)
|
||||
}
|
||||
// TCP echo shares the control cert for its elt-echo TLS variant.
|
||||
tcpSrv := &tcpecho.Server{
|
||||
TLSConfig: &tls.Config{Certificates: []tls.Certificate{cert}, MinVersion: tls.VersionTLS12},
|
||||
@@ -184,27 +263,40 @@ func serve(cfg *config.Config) error {
|
||||
// Identity is optional. Without an issuer the server simply has no sign-in, and
|
||||
// uploads=account can never be satisfied — which is the honest outcome, not a silent
|
||||
// downgrade to anonymous.
|
||||
var idp *oidc.Verifier
|
||||
if cfg.OIDCIssuer != "" && (cfg.OIDCClientID != "" || cfg.OIDCAppClientID != "") {
|
||||
// One verifier per issuer. An IdP may mint a distinct issuer per application — Authentik
|
||||
// derives it from the application slug — and a token's `iss` must match whoever signed it.
|
||||
// Each verifier accepts only the client belonging to its own issuer, so a token minted for
|
||||
// the phone cannot be replayed at the admin login and vice versa.
|
||||
var idp, adminIdP *oidc.Verifier
|
||||
appIssuer := cfg.OIDCAppIssuer
|
||||
if appIssuer == "" {
|
||||
appIssuer = cfg.OIDCIssuer // IdPs with one global issuer
|
||||
}
|
||||
if appIssuer != "" && cfg.OIDCAppClientID != "" {
|
||||
idp = oidc.New(oidc.Config{
|
||||
Issuer: cfg.OIDCIssuer,
|
||||
ClientID: cfg.OIDCClientID,
|
||||
AppClientID: cfg.OIDCAppClientID,
|
||||
AdminGroup: cfg.OIDCAdminGroup,
|
||||
Issuer: appIssuer, AppClientID: cfg.OIDCAppClientID, AdminGroup: cfg.OIDCAdminGroup,
|
||||
}, nil)
|
||||
slog.Info("identity provider configured", "issuer", cfg.OIDCIssuer,
|
||||
"admin_client_id", cfg.OIDCClientID, "app_client_id", cfg.OIDCAppClientID,
|
||||
"admin_group", cfg.OIDCAdminGroup)
|
||||
slog.Info("identity: app client", "issuer", appIssuer, "client_id", cfg.OIDCAppClientID)
|
||||
}
|
||||
if cfg.OIDCIssuer != "" && cfg.OIDCClientID != "" {
|
||||
adminIdP = oidc.New(oidc.Config{
|
||||
Issuer: cfg.OIDCIssuer, ClientID: cfg.OIDCClientID, AdminGroup: cfg.OIDCAdminGroup,
|
||||
}, nil)
|
||||
slog.Info("identity: admin client", "issuer", cfg.OIDCIssuer,
|
||||
"client_id", cfg.OIDCClientID, "admin_group", cfg.OIDCAdminGroup)
|
||||
if cfg.OIDCAdminGroup == "" {
|
||||
slog.Warn("no admin group set: nobody will be an admin via OIDC " +
|
||||
"(set ECHOLOT_OIDC_ADMIN_GROUP)")
|
||||
}
|
||||
} else if cfg.UploadsMode == string(runs.ModeAccount) {
|
||||
}
|
||||
if idp == nil && adminIdP == nil && cfg.UploadsMode == string(runs.ModeAccount) {
|
||||
slog.Warn("uploads=account but no identity provider is configured — " +
|
||||
"every upload will be refused")
|
||||
}
|
||||
|
||||
ip4, ip6, ip4Alt, ip6Alt := cfg.MeasurementAddrs()
|
||||
ctl := &control.Server{
|
||||
IP4: ip4, IP6: ip6, IP4Alt: ip4Alt, IP6Alt: ip6Alt,
|
||||
Store: st, Sessions: sessions, Name: cfg.Name,
|
||||
UDPPort: mustPort(firstAddr(cfg.UDPListen)), TCPPort: mustPort(firstAddr(cfg.TCPListen)),
|
||||
StunPort: mustPort(firstAddr(cfg.StunListen)), PinB64: pin, CertChain: cert.Certificate,
|
||||
@@ -216,6 +308,7 @@ func serve(cfg *config.Config) error {
|
||||
AppRange: appRange,
|
||||
PublicControlURL: publicControlURL(cfg),
|
||||
OIDC: idp,
|
||||
AdminOIDC: adminIdP,
|
||||
}
|
||||
// Left nil when there is no raw socket, so the handler answers "not implemented" with a
|
||||
// reason rather than failing somewhere deeper.
|
||||
@@ -223,6 +316,12 @@ func serve(cfg *config.Config) error {
|
||||
ctl.FragSend = dp.FragSend
|
||||
}
|
||||
ctl.DownThroughput = dp.DownThroughput
|
||||
if cfg.RateSessionsPerMin > 0 {
|
||||
ctl.RateSessions = ratelimit.New(float64(cfg.RateSessionsPerMin)/60, float64(cfg.RateSessionsPerMin))
|
||||
}
|
||||
if cfg.RateActionsPerMin > 0 {
|
||||
ctl.RateActions = ratelimit.New(float64(cfg.RateActionsPerMin)/60, float64(cfg.RateActionsPerMin))
|
||||
}
|
||||
|
||||
ctx, stop := signal.NotifyContext(context.Background(), syscall.SIGINT, syscall.SIGTERM)
|
||||
defer stop()
|
||||
@@ -285,37 +384,49 @@ func serve(cfg *config.Config) error {
|
||||
return best
|
||||
}
|
||||
|
||||
// Admin/health (plain HTTP, localhost by default; spec §7)
|
||||
admin := http.NewServeMux()
|
||||
admin.HandleFunc("GET /healthz", func(w http.ResponseWriter, _ *http.Request) {
|
||||
fmt.Fprintf(w, `{"ok":true,"version":%q}`, Version)
|
||||
})
|
||||
admin.HandleFunc("GET /admin/selftest", func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_ = json.NewEncoder(w).Encode(selftestPtr.Load())
|
||||
})
|
||||
// TODO(spec §7): enrollment token management + device list. Until the
|
||||
// admin UI exists, mint tokens with: echolot-admin (or curl on this
|
||||
// listener once the endpoint lands).
|
||||
admin.HandleFunc("POST /admin/enroll-tokens", func(w http.ResponseWriter, r *http.Request) {
|
||||
tok, err := st.NewEnrollToken(24*time.Hour, r.URL.Query().Get("note"))
|
||||
if err != nil {
|
||||
http.Error(w, err.Error(), 500)
|
||||
return
|
||||
}
|
||||
// The whole bootstrap, not just the token: this is what gets pasted or turned into a
|
||||
// QR code, and assembling it here is what keeps an operator from transcribing a pin by
|
||||
// hand — a pin wrong by one character fails as an inscrutable TLS error days later.
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
enc := json.NewEncoder(w)
|
||||
enc.SetEscapeHTML(false) // the link is full of / and =; escaping them helps nobody
|
||||
_ = enc.Encode(map[string]any{
|
||||
"token": tok,
|
||||
"expires_in_s": 86400,
|
||||
"enroll_uri": ctl.EnrollmentLink(tok),
|
||||
})
|
||||
})
|
||||
adminSrv := &http.Server{Addr: cfg.AdminListen, Handler: admin, ReadHeaderTimeout: 10 * time.Second}
|
||||
// The admin interface. Every route except /healthz requires a session — the old arrangement
|
||||
// (no auth, kept safe by binding to loopback) failed the moment the address changed, and a
|
||||
// binding address is a deployment detail rather than an access control.
|
||||
secret, err := st.SessionSecret()
|
||||
if err != nil {
|
||||
return fmt.Errorf("admin session secret: %w", err)
|
||||
}
|
||||
adminSecure := cfg.AdminTLSCert != ""
|
||||
ui := &adminui.Server{
|
||||
Store: st,
|
||||
Runs: runStore,
|
||||
OIDC: adminIdP,
|
||||
Sessions: adminauth.NewSessions(secret, 12*time.Hour),
|
||||
Throttle: adminauth.NewThrottle(),
|
||||
AdminUser: cfg.AdminUser,
|
||||
BaseURL: cfg.AdminBaseURL,
|
||||
ClientSecret: cfg.OIDCClientSecret,
|
||||
Secure: adminSecure,
|
||||
EnrollLink: ctl.EnrollmentLink,
|
||||
// Where devices should connect, for /v1/discover. Derived from the control hostname so it
|
||||
// cannot drift from the name that actually selects the pinned certificate.
|
||||
ControlURL: controlURL(cfg),
|
||||
ServerName: cfg.Name,
|
||||
SelfTest: func() any { return selftestPtr.Load() },
|
||||
Version: Version,
|
||||
}
|
||||
if st.LocalAdmin() == nil && adminIdP == nil {
|
||||
slog.Warn("nobody can sign in to the admin UI: no break-glass password is set " +
|
||||
"(--set-admin-password) and no identity provider is configured")
|
||||
}
|
||||
admin := ui.Handler()
|
||||
|
||||
// One listener per configured address, all serving the same handler.
|
||||
//
|
||||
// Multi-address rather than a wildcard because this host reserves addresses for measurement:
|
||||
// binding 0.0.0.0 would put the admin UI on port 443 of the reserved pair, and their value
|
||||
// comes precisely from nothing answering there. Explicit addresses are also what let the
|
||||
// service and management addresses differ without a second process.
|
||||
adminAddrs := config.Addrs(cfg.AdminListen)
|
||||
if len(adminAddrs) == 0 {
|
||||
return fmt.Errorf("admin: no listen address configured")
|
||||
}
|
||||
var adminTLS *tls.Config
|
||||
if cfg.AdminTLSCert != "" {
|
||||
// Terminated here rather than behind a reverse proxy: this binary already serves TLS for
|
||||
// the control plane, so it is reuse rather than new machinery, and one process with one
|
||||
@@ -325,16 +436,92 @@ func serve(cfg *config.Config) error {
|
||||
if err != nil {
|
||||
return fmt.Errorf("admin TLS: %w", err)
|
||||
}
|
||||
adminSrv.TLSConfig = reloader.TLSConfig()
|
||||
adminTLS = reloader.TLSConfig()
|
||||
if exp := reloader.NotAfter(); !exp.IsZero() {
|
||||
slog.Info("admin UI TLS", "listen", cfg.AdminListen, "cert_expires", exp.Format(time.RFC3339))
|
||||
slog.Info("admin UI TLS", "listen", adminAddrs, "cert_expires", exp.Format(time.RFC3339))
|
||||
if time.Until(exp) < 14*24*time.Hour {
|
||||
slog.Warn("admin certificate expires soon", "expires", exp.Format(time.RFC3339))
|
||||
}
|
||||
}
|
||||
go func() { errCh <- fmt.Errorf("admin: %w", adminSrv.ListenAndServeTLS("", "")) }()
|
||||
} else {
|
||||
go func() { errCh <- fmt.Errorf("admin: %w", adminSrv.ListenAndServe()) }()
|
||||
}
|
||||
// Kept for shutdown: each listener gets its own server, and a graceful stop has to reach all
|
||||
// of them or an in-flight admin request is cut off mid-response on every address but one.
|
||||
var adminSrvs []*http.Server
|
||||
// Sharing port 443 between two services that cannot share a certificate. The name in the TLS
|
||||
// handshake picks the certificate, and the name in the request picks the handler; both have to
|
||||
// agree or a client would get the pinned certificate and the admin UI behind it.
|
||||
//
|
||||
// The control plane keeps its own listener as well. Devices enrolled before this carry the old
|
||||
// URL in their settings, and taking that away would strand every one of them for the sake of a
|
||||
// port number.
|
||||
ctlHandler := ctl.Handler()
|
||||
sharedCert := cert
|
||||
// Which side of the port a request belongs to.
|
||||
//
|
||||
// The control hostname is the obvious case. A bare IP is the other one, and it matters: a
|
||||
// client whose DNS has failed can still reach the server by an address it cached from the
|
||||
// profile, and a measurement tool that cannot report from a broken network is useless
|
||||
// precisely when it is needed. That client authenticates by pin, so the name it used to get
|
||||
// here is not part of the trust decision.
|
||||
//
|
||||
// Safe to route that way because the admin UI is only ever reached by name: browsers always
|
||||
// send SNI and nobody bookmarks an IP for a site with a Let's Encrypt certificate. Anything
|
||||
// addressing this server numerically is a pinned client.
|
||||
isControl := func(host string) bool {
|
||||
if h, _, err := net.SplitHostPort(host); err == nil {
|
||||
host = h
|
||||
}
|
||||
host = strings.Trim(host, "[]")
|
||||
if cfg.ControlHostname != "" && strings.EqualFold(host, cfg.ControlHostname) {
|
||||
return true
|
||||
}
|
||||
return net.ParseIP(host) != nil
|
||||
}
|
||||
pickCert := func(hi *tls.ClientHelloInfo) (*tls.Certificate, error) {
|
||||
// No SNI at all also means a numeric client: every browser sends it.
|
||||
if hi.ServerName == "" || isControl(hi.ServerName) {
|
||||
return &sharedCert, nil
|
||||
}
|
||||
if adminTLS != nil && adminTLS.GetCertificate != nil {
|
||||
return adminTLS.GetCertificate(hi)
|
||||
}
|
||||
return &sharedCert, nil
|
||||
}
|
||||
route := func(w http.ResponseWriter, r *http.Request) {
|
||||
if isControl(r.Host) {
|
||||
ctlHandler.ServeHTTP(w, r)
|
||||
return
|
||||
}
|
||||
admin.ServeHTTP(w, r)
|
||||
}
|
||||
sharedTLS := &tls.Config{GetCertificate: pickCert, MinVersion: tls.VersionTLS12}
|
||||
if cfg.ControlHostname != "" {
|
||||
slog.Info("control plane shares the admin port",
|
||||
"hostname", cfg.ControlHostname, "listen", adminAddrs)
|
||||
}
|
||||
|
||||
for _, addr := range adminAddrs {
|
||||
// Bound before the goroutine starts, so a bad address fails startup rather than being
|
||||
// reported asynchronously after the process has already declared itself healthy.
|
||||
ln, err := net.Listen("tcp", addr)
|
||||
if err != nil {
|
||||
return fmt.Errorf("admin listen %s: %w", addr, err)
|
||||
}
|
||||
srv := &http.Server{
|
||||
Handler: http.HandlerFunc(route),
|
||||
ReadHeaderTimeout: 10 * time.Second,
|
||||
TLSConfig: sharedTLS,
|
||||
}
|
||||
adminSrvs = append(adminSrvs, srv)
|
||||
go func(ln net.Listener, addr string) {
|
||||
// Plaintext only where there is no certificate at all — checkAdminExposure has
|
||||
// already refused that anywhere but loopback.
|
||||
if adminTLS == nil && cfg.ControlHostname == "" {
|
||||
errCh <- fmt.Errorf("admin %s: %w", addr, srv.Serve(ln))
|
||||
return
|
||||
}
|
||||
errCh <- fmt.Errorf("admin %s: %w", addr, srv.ServeTLS(ln, "", ""))
|
||||
}(ln, addr)
|
||||
}
|
||||
|
||||
// ACME HTTP-01 responder. Permanent rather than started per renewal: nothing binds and
|
||||
@@ -348,14 +535,21 @@ func serve(cfg *config.Config) error {
|
||||
if err := acmehttp.EnsureWebroot(webroot); err != nil {
|
||||
return fmt.Errorf("acme webroot: %w", err)
|
||||
}
|
||||
acmeSrv := &http.Server{
|
||||
Addr: cfg.ACMEHTTPListen,
|
||||
Handler: acmehttp.Handler(webroot, cfg.AdminBaseURL),
|
||||
ReadHeaderTimeout: 10 * time.Second,
|
||||
acmeHandler := acmehttp.Handler(webroot, cfg.AdminBaseURL)
|
||||
for _, addr := range config.Addrs(cfg.ACMEHTTPListen) {
|
||||
ln, err := net.Listen("tcp", addr)
|
||||
if err != nil {
|
||||
return fmt.Errorf("acme-http listen %s: %w", addr, err)
|
||||
}
|
||||
srv := &http.Server{Handler: acmeHandler, ReadHeaderTimeout: 10 * time.Second}
|
||||
go func(ln net.Listener, addr string) {
|
||||
errCh <- fmt.Errorf("acme-http %s: %w", addr, srv.Serve(ln))
|
||||
}(ln, addr)
|
||||
}
|
||||
slog.Info("acme http-01 responder", "listen", cfg.ACMEHTTPListen, "webroot", webroot,
|
||||
"redirects_to", cfg.AdminBaseURL)
|
||||
go func() { errCh <- fmt.Errorf("acme-http: %w", acmeSrv.ListenAndServe()) }()
|
||||
// Every address the name may resolve to needs the responder: the CA picks one, and a
|
||||
// challenge that lands on an unbound address fails a renewal rather than a request.
|
||||
slog.Info("acme http-01 responder", "listen", config.Addrs(cfg.ACMEHTTPListen),
|
||||
"webroot", webroot, "redirects_to", cfg.AdminBaseURL)
|
||||
}
|
||||
|
||||
// UDP data plane — one socket per configured address. Distinct sockets
|
||||
@@ -421,7 +615,8 @@ func serve(cfg *config.Config) error {
|
||||
var dnsTCP []net.Listener
|
||||
if dnsAddrs := config.Addrs(cfg.DNSListen); len(dnsAddrs) > 0 && cfg.CanaryZone != "" {
|
||||
v4, v6 := firstByFamily(dnsAddrs)
|
||||
cd := canarydns.New(cfg.CanaryZone, cfg.Name, v4, v6)
|
||||
cd := canarydns.New(cfg.CanaryZone, cfg.Name, v4, v6,
|
||||
time.Duration(cfg.DNSLogRetentionH)*time.Hour)
|
||||
for _, addr := range dnsAddrs {
|
||||
ua, err := net.ResolveUDPAddr("udp", addr)
|
||||
if err != nil {
|
||||
@@ -447,7 +642,7 @@ func serve(cfg *config.Config) error {
|
||||
}
|
||||
|
||||
slog.Info("listening",
|
||||
"control", ctlAddrs, "admin", cfg.AdminListen, "udp", udpAddrs,
|
||||
"control", ctlAddrs, "admin", adminAddrs, "udp", udpAddrs,
|
||||
"tcp", config.Addrs(cfg.TCPListen), "stun", config.Addrs(cfg.StunListen),
|
||||
"dns", config.Addrs(cfg.DNSListen), "capabilities", ctl.Capabilities)
|
||||
|
||||
@@ -457,7 +652,9 @@ func serve(cfg *config.Config) error {
|
||||
shutCtx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
|
||||
defer cancel()
|
||||
_ = ctlSrv.Shutdown(shutCtx)
|
||||
_ = adminSrv.Shutdown(shutCtx)
|
||||
for _, srv := range adminSrvs {
|
||||
_ = srv.Shutdown(shutCtx)
|
||||
}
|
||||
for _, c := range udpConns {
|
||||
_ = c.Close()
|
||||
}
|
||||
|
||||
@@ -0,0 +1,84 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// release-sign signs a release manifest (SHA256SUMS) with the project's ed25519 key, producing
|
||||
// the detached <file>.sig that self-updating servers verify before trusting the checksums.
|
||||
//
|
||||
// release-sign -gen mint a keypair (seed on stdout — store it as the CI
|
||||
// secret RELEASE_SIGNING_KEY; publish the public key)
|
||||
// release-sign <file> sign; key read from $RELEASE_SIGNING_KEY, writes <file>.sig
|
||||
// release-sign -verify -pub <b64> <f> check <f> against <f>.sig — what the updater will do
|
||||
//
|
||||
// Run from CI (build-server.yml); the private key exists only in the Actions secret store, never
|
||||
// on the release host, which is the property that makes the signature worth having.
|
||||
package main
|
||||
|
||||
import (
|
||||
"flag"
|
||||
"fmt"
|
||||
"os"
|
||||
|
||||
"echo-lot.app/server/internal/relsign"
|
||||
)
|
||||
|
||||
func main() {
|
||||
gen := flag.Bool("gen", false, "generate a keypair and exit")
|
||||
verify := flag.Bool("verify", false, "verify <file> against <file>.sig instead of signing")
|
||||
pub := flag.String("pub", "", "public key (base64) for -verify")
|
||||
flag.Parse()
|
||||
|
||||
if err := run(*gen, *verify, *pub, flag.Args()); err != nil {
|
||||
fmt.Fprintln(os.Stderr, "release-sign:", err)
|
||||
os.Exit(1)
|
||||
}
|
||||
}
|
||||
|
||||
func run(gen, verify bool, pub string, args []string) error {
|
||||
if gen {
|
||||
pubB64, seedB64, err := relsign.GenerateKey()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Printf("public key (embed / ECHOLOT_SELF_UPDATE_PUBKEY):\n%s\n\n"+
|
||||
"private key (CI secret RELEASE_SIGNING_KEY — this is the only copy):\n%s\n",
|
||||
pubB64, seedB64)
|
||||
return nil
|
||||
}
|
||||
if len(args) != 1 {
|
||||
return fmt.Errorf("usage: release-sign [-gen | -verify -pub <b64>] <file>")
|
||||
}
|
||||
file := args[0]
|
||||
data, err := os.ReadFile(file)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if verify {
|
||||
if pub == "" {
|
||||
return fmt.Errorf("-verify needs -pub")
|
||||
}
|
||||
sig, err := os.ReadFile(file + ".sig")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := relsign.Verify(pub, data, string(sig)); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Printf("%s: signature OK\n", file)
|
||||
return nil
|
||||
}
|
||||
|
||||
seed := os.Getenv("RELEASE_SIGNING_KEY")
|
||||
if seed == "" {
|
||||
return fmt.Errorf("RELEASE_SIGNING_KEY is not set — refusing to produce an unsigned release")
|
||||
}
|
||||
sig, err := relsign.Sign(seed, data)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(file+".sig", []byte(sig+"\n"), 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Printf("wrote %s.sig\n", file)
|
||||
return nil
|
||||
}
|
||||
@@ -1,3 +1,5 @@
|
||||
module echo-lot.app/server
|
||||
|
||||
go 1.24
|
||||
|
||||
require github.com/skip2/go-qrcode v0.0.0-20200617195104-da1b6568686e // indirect
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
github.com/skip2/go-qrcode v0.0.0-20200617195104-da1b6568686e h1:MRM5ITcdelLK2j1vwZ3Je0FKVCfqOLp5zO6trqMLYs0=
|
||||
github.com/skip2/go-qrcode v0.0.0-20200617195104-da1b6568686e/go.mod h1:XV66xRDqSt+GTGFMVlhk3ULuV0y9ZmzeVGR4mloJI3M=
|
||||
@@ -166,12 +166,21 @@ func (t *Throttle) Succeeded() {
|
||||
|
||||
// ---- sessions ---------------------------------------------------------------------------
|
||||
|
||||
// Session is an authenticated admin, however they proved it.
|
||||
// Session is an authenticated account, however it proved itself. Not necessarily an admin:
|
||||
// signing in and being allowed to administer the server are separate questions, and a plain user
|
||||
// gets a session so they can manage their own uploads.
|
||||
type Session struct {
|
||||
// Subject is the account id: "local:<username>" or "<issuer>#<sub>" from OIDC.
|
||||
Subject string
|
||||
// Display is what the UI shows.
|
||||
Display string
|
||||
// Admin is authorisation, decided at sign-in and carried inside the signed payload.
|
||||
//
|
||||
// Inside, specifically — not derived later from the subject, and not stored beside the MAC.
|
||||
// A flag outside the signature is a privilege escalation anyone can perform with a text
|
||||
// editor, and re-deriving it per request would mean re-reading group membership from the IdP
|
||||
// on a path that has no token to do it with.
|
||||
Admin bool
|
||||
Expires time.Time
|
||||
}
|
||||
|
||||
@@ -203,12 +212,16 @@ func NewSecret() ([]byte, error) {
|
||||
|
||||
var ErrSession = errors.New("session is not valid")
|
||||
|
||||
// Issue returns the cookie value for a newly authenticated admin.
|
||||
func (s *Sessions) Issue(subject, display string) string {
|
||||
// Issue returns the cookie value for a newly authenticated account.
|
||||
func (s *Sessions) Issue(subject, display string, admin bool) string {
|
||||
exp := time.Now().Add(s.ttl).Unix()
|
||||
role := "u"
|
||||
if admin {
|
||||
role = "a"
|
||||
}
|
||||
payload := base64.RawURLEncoding.EncodeToString([]byte(subject)) + "." +
|
||||
base64.RawURLEncoding.EncodeToString([]byte(display)) + "." +
|
||||
strconv.FormatInt(exp, 10)
|
||||
strconv.FormatInt(exp, 10) + "." + role
|
||||
return payload + "." + s.mac(payload)
|
||||
}
|
||||
|
||||
@@ -225,7 +238,7 @@ func (s *Sessions) Parse(value string) (*Session, error) {
|
||||
return nil, ErrSession
|
||||
}
|
||||
parts := strings.Split(payload, ".")
|
||||
if len(parts) != 3 {
|
||||
if len(parts) != 4 {
|
||||
return nil, ErrSession
|
||||
}
|
||||
subject, err := base64.RawURLEncoding.DecodeString(parts[0])
|
||||
@@ -243,7 +256,13 @@ func (s *Sessions) Parse(value string) (*Session, error) {
|
||||
if time.Now().After(time.Unix(exp, 0)) {
|
||||
return nil, fmt.Errorf("%w: expired", ErrSession)
|
||||
}
|
||||
return &Session{Subject: string(subject), Display: string(display), Expires: time.Unix(exp, 0)}, nil
|
||||
// Anything that is not exactly the admin marker is a user. A malformed role must fail closed:
|
||||
// the safe reading of an unparseable privilege claim is the smaller privilege.
|
||||
admin := parts[3] == "a"
|
||||
return &Session{
|
||||
Subject: string(subject), Display: string(display),
|
||||
Admin: admin, Expires: time.Unix(exp, 0),
|
||||
}, nil
|
||||
}
|
||||
|
||||
func (s *Sessions) mac(payload string) string {
|
||||
|
||||
@@ -4,6 +4,8 @@
|
||||
package adminauth
|
||||
|
||||
import (
|
||||
"encoding/base64"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
@@ -96,7 +98,7 @@ func TestAnEmptyCredentialNeverVerifies(t *testing.T) {
|
||||
func TestSessionRoundTrip(t *testing.T) {
|
||||
secret, _ := NewSecret()
|
||||
s := NewSessions(secret, time.Hour)
|
||||
got, err := s.Parse(s.Issue("local:admin", "Admin"))
|
||||
got, err := s.Parse(s.Issue("local:admin", "Admin", true))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -110,7 +112,7 @@ func TestSessionRoundTrip(t *testing.T) {
|
||||
func TestTamperedSessionsAreRejected(t *testing.T) {
|
||||
secret, _ := NewSecret()
|
||||
s := NewSessions(secret, time.Hour)
|
||||
good := s.Issue("local:admin", "Admin")
|
||||
good := s.Issue("local:admin", "Admin", true)
|
||||
|
||||
parts := strings.Split(good, ".")
|
||||
tampered := []string{
|
||||
@@ -131,7 +133,7 @@ func TestTamperedSessionsAreRejected(t *testing.T) {
|
||||
func TestSessionsFromAnotherSecretAreRejected(t *testing.T) {
|
||||
a, _ := NewSecret()
|
||||
b, _ := NewSecret()
|
||||
issued := NewSessions(a, time.Hour).Issue("local:admin", "Admin")
|
||||
issued := NewSessions(a, time.Hour).Issue("local:admin", "Admin", true)
|
||||
if _, err := NewSessions(b, time.Hour).Parse(issued); err == nil {
|
||||
t.Fatal("a session signed with a different secret was accepted — rotating the secret " +
|
||||
"must invalidate every existing session")
|
||||
@@ -143,7 +145,7 @@ func TestExpiredSessionsAreRejected(t *testing.T) {
|
||||
// A negative TTL is not reachable through NewSessions, so issue with a real one and check
|
||||
// the boundary via a session that has already run out.
|
||||
s := NewSessions(secret, time.Millisecond)
|
||||
v := s.Issue("local:admin", "Admin")
|
||||
v := s.Issue("local:admin", "Admin", true)
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
if _, err := s.Parse(v); err == nil {
|
||||
t.Fatal("an expired session was accepted")
|
||||
@@ -201,3 +203,46 @@ func TestThrottleForgivesAfterAQuietPeriod(t *testing.T) {
|
||||
t.Fatalf("an operator returning later was still throttled: %v", d)
|
||||
}
|
||||
}
|
||||
|
||||
// The admin flag is an authorisation decision carried in a cookie the client holds, so the
|
||||
// interesting cases are all about what happens when the client lies about it.
|
||||
func TestSessionAdminFlag(t *testing.T) {
|
||||
s := NewSessions([]byte("secret"), time.Hour)
|
||||
|
||||
t.Run("round trips both ways", func(t *testing.T) {
|
||||
admin, err := s.Parse(s.Issue("local:admin", "Admin", true))
|
||||
if err != nil || !admin.Admin {
|
||||
t.Fatalf("admin session did not survive: %+v err=%v", admin, err)
|
||||
}
|
||||
user, err := s.Parse(s.Issue("oidc#1", "Markus", false))
|
||||
if err != nil || user.Admin {
|
||||
t.Fatalf("user session came back as admin: %+v err=%v", user, err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("promoting yourself invalidates the cookie", func(t *testing.T) {
|
||||
// The whole point of putting the flag inside the MAC: editing it must break the signature
|
||||
// rather than produce a valid admin session.
|
||||
v := s.Issue("oidc#1", "Markus", false)
|
||||
i := strings.LastIndex(v, ".")
|
||||
tampered := strings.TrimSuffix(v[:i], ".u") + ".a" + v[i:]
|
||||
if got, err := s.Parse(tampered); err == nil {
|
||||
t.Fatalf("a self-promoted cookie was accepted as %+v", got)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("an unparseable role is not an admin", func(t *testing.T) {
|
||||
// Fail closed: whatever a malformed privilege claim means, it does not mean "more access".
|
||||
// Signed by us, so it passes the MAC — only the role parsing stands between it and admin.
|
||||
exp := strconv.FormatInt(time.Now().Add(time.Hour).Unix(), 10)
|
||||
payload := base64.RawURLEncoding.EncodeToString([]byte("oidc#1")) + "." +
|
||||
base64.RawURLEncoding.EncodeToString([]byte("Markus")) + "." + exp + ".ADMIN"
|
||||
sess, err := s.Parse(payload + "." + s.mac(payload))
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected parse error: %v", err)
|
||||
}
|
||||
if sess.Admin {
|
||||
t.Fatal("a role of \"ADMIN\" was treated as the admin marker")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
@@ -0,0 +1,426 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// Package adminui serves the operator's web interface.
|
||||
//
|
||||
// Everything here is behind authentication, without exception. The previous arrangement — an
|
||||
// unauthenticated listener kept safe by binding to loopback — worked exactly until the address
|
||||
// changed, and then failed silently and publicly. Binding address is a deployment detail; it is
|
||||
// not an access control, and this package does not treat it as one.
|
||||
//
|
||||
// Rendered server-side with html/template and no JavaScript. The pages are lists and forms; a
|
||||
// framework would add a build step, a dependency tree and an update treadmill to a program that
|
||||
// currently has none of those.
|
||||
package adminui
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"crypto/sha256"
|
||||
"encoding/base64"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"log/slog"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"echo-lot.app/server/internal/adminauth"
|
||||
"echo-lot.app/server/internal/oidc"
|
||||
"echo-lot.app/server/internal/runs"
|
||||
"echo-lot.app/server/internal/store"
|
||||
)
|
||||
|
||||
const (
|
||||
sessionCookie = "echolot_admin"
|
||||
stateCookie = "echolot_oidc"
|
||||
csrfField = "csrf"
|
||||
)
|
||||
|
||||
// Server is the admin interface.
|
||||
type Server struct {
|
||||
Store *store.Store
|
||||
Runs *runs.Store
|
||||
OIDC *oidc.Verifier // admin client; nil when no IdP is configured
|
||||
Sessions *adminauth.Sessions
|
||||
Throttle *adminauth.Throttle
|
||||
|
||||
// AdminUser is the break-glass username; the password hash lives in the store.
|
||||
AdminUser string
|
||||
// BaseURL is where this UI is reachable, for building the OIDC redirect. Must match the URI
|
||||
// registered at the IdP exactly.
|
||||
BaseURL string
|
||||
// ClientSecret authenticates the confidential admin client at the token endpoint.
|
||||
ClientSecret string
|
||||
// Secure marks cookies Secure. Off only for loopback HTTP, where there is no network to
|
||||
// intercept and browsers refuse Secure cookies over plaintext anyway.
|
||||
Secure bool
|
||||
|
||||
// EnrollLink builds the §2.1 bootstrap link for a token. Injected rather than rebuilt here,
|
||||
// so the SPKI pin and public URL stay owned by the control server that actually knows them.
|
||||
EnrollLink func(token string) string
|
||||
// ControlURL is where devices should actually connect, handed out by /v1/discover so the
|
||||
// enrollment link can show the public name instead. ServerName is for display.
|
||||
ControlURL string
|
||||
ServerName string
|
||||
// SelfTest and Version render on the dashboard.
|
||||
SelfTest func() any
|
||||
Version string
|
||||
}
|
||||
|
||||
// Handler builds the routes. Only /healthz is reachable without a session.
|
||||
func (s *Server) Handler() http.Handler {
|
||||
mux := http.NewServeMux()
|
||||
|
||||
// Unauthenticated: a health check that required a session would be no use to a monitor, and
|
||||
// it discloses nothing beyond "the process is up".
|
||||
mux.HandleFunc("GET /healthz", func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
fmt.Fprintf(w, `{"ok":true,"version":%q}`+"\n", s.Version)
|
||||
})
|
||||
|
||||
// Unauthenticated on purpose, and deliberately says almost nothing: where the control plane
|
||||
// is, and nothing about who may talk to it.
|
||||
//
|
||||
// This exists so an enrollment link can carry the name a person recognises while the app
|
||||
// still connects to the name that selects the pinned certificate. It hands out an address,
|
||||
// never a pin — the pin travels in the link itself. Serving the pin here would collapse
|
||||
// pinning to whatever the CA system says, and pinning exists precisely to survive a
|
||||
// certificate authority the operator does not control.
|
||||
//
|
||||
// So the worst an intercepted discovery can do is send a device to the wrong host, where the
|
||||
// pin check fails. That is a denial of service, not a compromise.
|
||||
mux.HandleFunc("GET /v1/discover", func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_ = json.NewEncoder(w).Encode(map[string]string{
|
||||
"control_url": s.ControlURL,
|
||||
"name": s.ServerName,
|
||||
})
|
||||
})
|
||||
|
||||
mux.HandleFunc("GET /login", s.loginForm)
|
||||
mux.HandleFunc("POST /login", s.loginSubmit)
|
||||
mux.HandleFunc("GET /auth/start", s.oidcStart)
|
||||
mux.HandleFunc("GET /admin/callback", s.oidcCallback)
|
||||
mux.HandleFunc("POST /logout", s.logout)
|
||||
|
||||
// Any signed-in account. These handlers scope what they show to the session themselves —
|
||||
// an admin sees everything, a user sees their own devices and runs.
|
||||
mux.HandleFunc("GET /", s.guard(s.dashboard))
|
||||
mux.HandleFunc("GET /devices", s.guard(s.devices))
|
||||
mux.HandleFunc("GET /runs", s.guard(s.runsList))
|
||||
mux.HandleFunc("GET /runs/{device}/{id}", s.guard(s.runView))
|
||||
mux.HandleFunc("POST /runs/{device}/{id}/delete", s.guard(s.runDelete))
|
||||
// Deleting your own upload is yours to do; revoking a device or minting an enrolment token
|
||||
// affects the whole server, so those stay with the admin.
|
||||
mux.HandleFunc("POST /devices/{id}/revoke", s.guard(s.adminOnly(s.revokeDevice)))
|
||||
mux.HandleFunc("POST /enroll-tokens", s.guard(s.adminOnly(s.mintToken)))
|
||||
// The spec-shaped mint endpoint (§2.1: {token, expires_in_s, enroll_uri}), for curl and
|
||||
// scripts. Authenticates its own way — see apiAdmin — because guard's redirect-to-login is
|
||||
// useless to a caller without a browser.
|
||||
mux.HandleFunc("POST /admin/enroll-tokens", s.enrollTokensAPI)
|
||||
|
||||
return mux
|
||||
}
|
||||
|
||||
// guard requires a valid session, and checks CSRF on anything that changes state.
|
||||
func (s *Server) guard(h func(http.ResponseWriter, *http.Request, *adminauth.Session)) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
sess := s.session(r)
|
||||
if sess == nil {
|
||||
http.Redirect(w, r, "/login", http.StatusSeeOther)
|
||||
return
|
||||
}
|
||||
if r.Method != http.MethodGet && r.Method != http.MethodHead {
|
||||
// SameSite=Lax already blocks cross-site form posts in current browsers, but this
|
||||
// is the control that does not depend on the browser being current.
|
||||
if !s.csrfOK(r, sess) {
|
||||
http.Error(w, "stale form — reload the page and try again", http.StatusForbidden)
|
||||
return
|
||||
}
|
||||
}
|
||||
h(w, r, sess)
|
||||
}
|
||||
}
|
||||
|
||||
func (s *Server) session(r *http.Request) *adminauth.Session {
|
||||
c, err := r.Cookie(sessionCookie)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
sess, err := s.Sessions.Parse(c.Value)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
return sess
|
||||
}
|
||||
|
||||
// csrfToken derives a per-session token. Derived rather than stored so it needs no server-side
|
||||
// state and cannot drift out of sync with the session it belongs to.
|
||||
func (s *Server) csrfToken(sess *adminauth.Session) string {
|
||||
sum := sha256.Sum256([]byte("csrf|" + sess.Subject + "|" + sess.Expires.String()))
|
||||
return base64.RawURLEncoding.EncodeToString(sum[:16])
|
||||
}
|
||||
|
||||
func (s *Server) csrfOK(r *http.Request, sess *adminauth.Session) bool {
|
||||
if err := r.ParseForm(); err != nil {
|
||||
return false
|
||||
}
|
||||
return r.PostFormValue(csrfField) == s.csrfToken(sess)
|
||||
}
|
||||
|
||||
// adminOnly refuses a handler to a signed-in account that is not an administrator.
|
||||
//
|
||||
// A separate wrapper rather than a check inside each handler: an authorisation rule that has to be
|
||||
// remembered in every handler is one that will eventually be forgotten in a new one, and the route
|
||||
// table is where someone looks to find out who may do what.
|
||||
func (s *Server) adminOnly(
|
||||
h func(http.ResponseWriter, *http.Request, *adminauth.Session),
|
||||
) func(http.ResponseWriter, *http.Request, *adminauth.Session) {
|
||||
return func(w http.ResponseWriter, r *http.Request, sess *adminauth.Session) {
|
||||
if !sess.Admin {
|
||||
slog.Info("admin action refused", "account", sess.Subject, "path", r.URL.Path)
|
||||
http.Error(w, "that action needs an administrator account", http.StatusForbidden)
|
||||
return
|
||||
}
|
||||
h(w, r, sess)
|
||||
}
|
||||
}
|
||||
|
||||
func (s *Server) setSession(w http.ResponseWriter, subject, display string, admin bool) {
|
||||
http.SetCookie(w, &http.Cookie{
|
||||
Name: sessionCookie,
|
||||
Value: s.Sessions.Issue(subject, display, admin),
|
||||
Path: "/",
|
||||
HttpOnly: true, // the cookie is a bearer credential; script has no business reading it
|
||||
Secure: s.Secure,
|
||||
SameSite: http.SameSiteLaxMode,
|
||||
})
|
||||
}
|
||||
|
||||
func (s *Server) logout(w http.ResponseWriter, r *http.Request) {
|
||||
http.SetCookie(w, &http.Cookie{
|
||||
Name: sessionCookie, Value: "", Path: "/", MaxAge: -1,
|
||||
HttpOnly: true, Secure: s.Secure, SameSite: http.SameSiteLaxMode,
|
||||
})
|
||||
http.Redirect(w, r, "/login", http.StatusSeeOther)
|
||||
}
|
||||
|
||||
// ---- local password ---------------------------------------------------------------------
|
||||
|
||||
func (s *Server) loginSubmit(w http.ResponseWriter, r *http.Request) {
|
||||
if err := r.ParseForm(); err != nil {
|
||||
http.Error(w, "bad form", http.StatusBadRequest)
|
||||
return
|
||||
}
|
||||
// The delay is applied before the answer, so a wrong guess costs time whether or not the
|
||||
// username exists — the timing carries no information either way.
|
||||
if d := s.Throttle.Delay(); d > 0 {
|
||||
time.Sleep(d)
|
||||
}
|
||||
user := r.PostFormValue("username")
|
||||
pass := r.PostFormValue("password")
|
||||
|
||||
cred := s.Store.LocalAdmin()
|
||||
if cred == nil || !cred.Verify(user, pass) {
|
||||
s.Throttle.Failed()
|
||||
slog.Info("admin login failed", "user", user, "from", clientIP(r))
|
||||
s.render(w, r, "login", map[string]any{
|
||||
"Error": "Incorrect username or password.",
|
||||
"OIDC": s.oidcAvailable(),
|
||||
})
|
||||
return
|
||||
}
|
||||
s.Throttle.Succeeded()
|
||||
slog.Info("admin login", "user", user, "method", "local", "from", clientIP(r))
|
||||
s.setSession(w, "local:"+cred.Username, cred.Username, true)
|
||||
http.Redirect(w, r, "/", http.StatusSeeOther)
|
||||
}
|
||||
|
||||
// apiAdmin authenticates a programmatic admin request: the normal session cookie, or HTTP Basic
|
||||
// against the break-glass credential for callers without a cookie jar (the README's curl).
|
||||
//
|
||||
// The cookie path keeps CSRF, exactly like guard: a cookie is an ambient credential and this
|
||||
// endpoint changes state. Basic auth is exempt — the password is supplied explicitly per
|
||||
// request, so there is nothing for a cross-site form to ride on — and a wrong guess pays the
|
||||
// same throttle as the login form, so this is no better a password oracle than that is.
|
||||
func (s *Server) apiAdmin(w http.ResponseWriter, r *http.Request) (subject string, ok bool) {
|
||||
if sess := s.session(r); sess != nil {
|
||||
if !sess.Admin {
|
||||
http.Error(w, "that action needs an administrator account", http.StatusForbidden)
|
||||
return "", false
|
||||
}
|
||||
if !s.csrfOK(r, sess) {
|
||||
http.Error(w, "stale form — reload the page and try again", http.StatusForbidden)
|
||||
return "", false
|
||||
}
|
||||
return sess.Subject, true
|
||||
}
|
||||
if user, pass, hasBasic := r.BasicAuth(); hasBasic {
|
||||
if d := s.Throttle.Delay(); d > 0 {
|
||||
time.Sleep(d)
|
||||
}
|
||||
if cred := s.Store.LocalAdmin(); cred != nil && cred.Verify(user, pass) {
|
||||
s.Throttle.Succeeded()
|
||||
return "local:" + user, true
|
||||
}
|
||||
s.Throttle.Failed()
|
||||
slog.Info("admin api auth failed", "user", user, "from", clientIP(r))
|
||||
}
|
||||
w.Header().Set("WWW-Authenticate", `Basic realm="echolot-admin"`)
|
||||
http.Error(w, "authentication required", http.StatusUnauthorized)
|
||||
return "", false
|
||||
}
|
||||
|
||||
// ---- OIDC -------------------------------------------------------------------------------
|
||||
|
||||
func (s *Server) oidcAvailable() bool {
|
||||
return s.OIDC != nil && s.OIDC.Config().Enabled() && s.BaseURL != ""
|
||||
}
|
||||
|
||||
// oidcStart redirects to the IdP with state and PKCE.
|
||||
//
|
||||
// PKCE even though this is a confidential client: it costs one hash and closes code interception
|
||||
// independently of the secret, which is worth having when the redirect crosses a browser.
|
||||
func (s *Server) oidcStart(w http.ResponseWriter, r *http.Request) {
|
||||
if !s.oidcAvailable() {
|
||||
http.Error(w, "no identity provider is configured on this server", http.StatusNotImplemented)
|
||||
return
|
||||
}
|
||||
d, err := s.OIDC.Discover(r.Context())
|
||||
if err != nil {
|
||||
http.Error(w, "identity provider unreachable: "+err.Error(), http.StatusBadGateway)
|
||||
return
|
||||
}
|
||||
state, verifier := randomToken(), randomToken()
|
||||
challenge := sha256.Sum256([]byte(verifier))
|
||||
|
||||
// state and the PKCE verifier ride in one short-lived cookie: the callback must prove it
|
||||
// belongs to the browser that started the flow, or an attacker can feed us their own code.
|
||||
http.SetCookie(w, &http.Cookie{
|
||||
Name: stateCookie, Value: state + "." + verifier, Path: "/",
|
||||
HttpOnly: true, Secure: s.Secure, SameSite: http.SameSiteLaxMode, MaxAge: 600,
|
||||
})
|
||||
|
||||
q := url.Values{
|
||||
"response_type": {"code"},
|
||||
"client_id": {s.OIDC.Config().ClientID},
|
||||
"redirect_uri": {s.redirectURI()},
|
||||
"scope": {"openid profile email"},
|
||||
"state": {state},
|
||||
"code_challenge": {base64.RawURLEncoding.EncodeToString(challenge[:])},
|
||||
"code_challenge_method": {"S256"},
|
||||
}
|
||||
http.Redirect(w, r, d.AuthorizationEndpoint+"?"+q.Encode(), http.StatusSeeOther)
|
||||
}
|
||||
|
||||
func (s *Server) redirectURI() string {
|
||||
return strings.TrimRight(s.BaseURL, "/") + "/admin/callback"
|
||||
}
|
||||
|
||||
func (s *Server) oidcCallback(w http.ResponseWriter, r *http.Request) {
|
||||
if !s.oidcAvailable() {
|
||||
http.Error(w, "no identity provider configured", http.StatusNotImplemented)
|
||||
return
|
||||
}
|
||||
c, err := r.Cookie(stateCookie)
|
||||
if err != nil {
|
||||
http.Error(w, "sign-in did not start here — try again from the login page", http.StatusBadRequest)
|
||||
return
|
||||
}
|
||||
http.SetCookie(w, &http.Cookie{Name: stateCookie, Value: "", Path: "/", MaxAge: -1})
|
||||
|
||||
state, verifier, ok := strings.Cut(c.Value, ".")
|
||||
if !ok || state == "" || r.URL.Query().Get("state") != state {
|
||||
http.Error(w, "sign-in state did not match — start again", http.StatusBadRequest)
|
||||
return
|
||||
}
|
||||
code := r.URL.Query().Get("code")
|
||||
if code == "" {
|
||||
http.Error(w, "no authorization code returned: "+r.URL.Query().Get("error"), http.StatusBadRequest)
|
||||
return
|
||||
}
|
||||
|
||||
idToken, err := s.exchange(r.Context(), code, verifier)
|
||||
if err != nil {
|
||||
slog.Info("admin oidc exchange failed", "err", err, "from", clientIP(r))
|
||||
http.Error(w, "could not complete sign-in", http.StatusBadGateway)
|
||||
return
|
||||
}
|
||||
claims, err := s.OIDC.Verify(r.Context(), idToken)
|
||||
if err != nil {
|
||||
slog.Info("admin oidc token rejected", "err", err, "from", clientIP(r))
|
||||
http.Error(w, "the identity token was not accepted", http.StatusForbidden)
|
||||
return
|
||||
}
|
||||
// Authentication and authorisation are answered separately here. Someone who is not in the
|
||||
// admin group has still proved who they are, and their own uploads are their business to
|
||||
// manage — refusing them a session outright, as this used to, left a legitimate account with
|
||||
// no way to see or delete the data it had sent.
|
||||
admin := s.OIDC.IsAdmin(claims)
|
||||
slog.Info("login", "account", claims.AccountID(), "method", "oidc", "admin", admin,
|
||||
"from", clientIP(r))
|
||||
s.setSession(w, claims.AccountID(), claims.Display(), admin)
|
||||
http.Redirect(w, r, "/", http.StatusSeeOther)
|
||||
}
|
||||
|
||||
// exchange trades the authorization code for tokens at the IdP.
|
||||
func (s *Server) exchange(ctx context.Context, code, verifier string) (string, error) {
|
||||
d, err := s.OIDC.Discover(ctx)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
form := url.Values{
|
||||
"grant_type": {"authorization_code"},
|
||||
"code": {code},
|
||||
"redirect_uri": {s.redirectURI()},
|
||||
"client_id": {s.OIDC.Config().ClientID},
|
||||
"code_verifier": {verifier},
|
||||
}
|
||||
if s.ClientSecret != "" {
|
||||
form.Set("client_secret", s.ClientSecret)
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodPost, d.TokenEndpoint,
|
||||
strings.NewReader(form.Encode()))
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
req.Header.Set("Content-Type", "application/x-www-form-urlencoded")
|
||||
|
||||
resp, err := (&http.Client{Timeout: 15 * time.Second}).Do(req)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
body, _ := io.ReadAll(io.LimitReader(resp.Body, 1<<20))
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return "", fmt.Errorf("token endpoint: %s: %s", resp.Status, strings.TrimSpace(string(body)))
|
||||
}
|
||||
var tok struct {
|
||||
IDToken string `json:"id_token"`
|
||||
}
|
||||
if err := json.Unmarshal(body, &tok); err != nil {
|
||||
return "", err
|
||||
}
|
||||
if tok.IDToken == "" {
|
||||
return "", fmt.Errorf("token endpoint returned no id_token")
|
||||
}
|
||||
return tok.IDToken, nil
|
||||
}
|
||||
|
||||
func randomToken() string {
|
||||
b := make([]byte, 32)
|
||||
_, _ = rand.Read(b)
|
||||
return base64.RawURLEncoding.EncodeToString(b)
|
||||
}
|
||||
|
||||
// clientIP is for logs only. X-Forwarded-For is deliberately ignored: nothing is meant to sit in
|
||||
// front of this listener, so a header claiming otherwise is a caller's assertion about itself.
|
||||
func clientIP(r *http.Request) string {
|
||||
if i := strings.LastIndex(r.RemoteAddr, ":"); i > 0 {
|
||||
return r.RemoteAddr[:i]
|
||||
}
|
||||
return r.RemoteAddr
|
||||
}
|
||||
@@ -0,0 +1,88 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package adminui
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"echo-lot.app/server/internal/adminauth"
|
||||
)
|
||||
|
||||
// tokenFixture wires just enough of the Server for the mint endpoint: a break-glass admin and
|
||||
// a stand-in EnrollLink (the real one belongs to the control server, injected the same way).
|
||||
func tokenFixture(t *testing.T) *Server {
|
||||
t.Helper()
|
||||
s, _, _, _ := fixture(t)
|
||||
secret, err := s.Store.SessionSecret()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
s.Sessions = adminauth.NewSessions(secret, time.Hour)
|
||||
s.Throttle = adminauth.NewThrottle()
|
||||
cred, err := adminauth.NewCredential("admin", "a-long-test-password")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := s.Store.SetLocalAdmin(cred); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
s.EnrollLink = func(tok string) string { return "echolot://enroll?v=1&t=" + tok }
|
||||
return s
|
||||
}
|
||||
|
||||
func TestEnrollTokensAPISpecShape(t *testing.T) {
|
||||
h := tokenFixture(t).Handler()
|
||||
|
||||
// No credentials → 401 with a challenge, never a token.
|
||||
req := httptest.NewRequest("POST", "/admin/enroll-tokens?note=phone", nil)
|
||||
rec := httptest.NewRecorder()
|
||||
h.ServeHTTP(rec, req)
|
||||
if rec.Code != http.StatusUnauthorized || rec.Header().Get("WWW-Authenticate") == "" {
|
||||
t.Fatalf("unauthenticated: code=%d", rec.Code)
|
||||
}
|
||||
|
||||
// Wrong password → still 401.
|
||||
req = httptest.NewRequest("POST", "/admin/enroll-tokens", nil)
|
||||
req.SetBasicAuth("admin", "wrong")
|
||||
rec = httptest.NewRecorder()
|
||||
h.ServeHTTP(rec, req)
|
||||
if rec.Code != http.StatusUnauthorized {
|
||||
t.Fatalf("bad password: code=%d, want 401", rec.Code)
|
||||
}
|
||||
|
||||
// Basic + Accept: application/json → the §2.1 shape.
|
||||
req = httptest.NewRequest("POST", "/admin/enroll-tokens?note=phone", nil)
|
||||
req.SetBasicAuth("admin", "a-long-test-password")
|
||||
req.Header.Set("Accept", "application/json")
|
||||
rec = httptest.NewRecorder()
|
||||
h.ServeHTTP(rec, req)
|
||||
if rec.Code != http.StatusOK {
|
||||
t.Fatalf("mint: code=%d body=%s", rec.Code, rec.Body.String())
|
||||
}
|
||||
var body struct {
|
||||
Token string `json:"token"`
|
||||
ExpiresS int `json:"expires_in_s"`
|
||||
EnrollURI string `json:"enroll_uri"`
|
||||
}
|
||||
if err := json.Unmarshal(rec.Body.Bytes(), &body); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if body.Token == "" || body.ExpiresS != 86400 || !strings.HasPrefix(body.EnrollURI, "echolot://enroll?") {
|
||||
t.Fatalf("spec shape violated: %+v", body)
|
||||
}
|
||||
|
||||
// Without Accept: the browser flow — redirect to the QR page, link in the query.
|
||||
req = httptest.NewRequest("POST", "/admin/enroll-tokens", nil)
|
||||
req.SetBasicAuth("admin", "a-long-test-password")
|
||||
rec = httptest.NewRecorder()
|
||||
h.ServeHTTP(rec, req)
|
||||
if rec.Code != http.StatusSeeOther || !strings.HasPrefix(rec.Header().Get("Location"), "/devices?link=") {
|
||||
t.Fatalf("html flow: code=%d location=%q", rec.Code, rec.Header().Get("Location"))
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,273 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package adminui
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"html/template"
|
||||
"log/slog"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"echo-lot.app/server/internal/adminauth"
|
||||
"echo-lot.app/server/internal/runs"
|
||||
"echo-lot.app/server/internal/store"
|
||||
)
|
||||
|
||||
// visibleDevices returns the devices a session may see: everything for an administrator, and for
|
||||
// anyone else the devices linked to their own account.
|
||||
//
|
||||
// Every page goes through this rather than filtering for itself. Scoping applied per-page is
|
||||
// scoping that will be missing from the next page someone adds, and the failure is silent — a
|
||||
// listing that quietly shows other people's uploads looks exactly like one that does not.
|
||||
func (s *Server) visibleDevices(sess *adminauth.Session) []store.Device {
|
||||
all := s.Store.Devices()
|
||||
if sess.Admin {
|
||||
return all
|
||||
}
|
||||
owned := make(map[string]bool)
|
||||
for _, id := range s.Store.DeviceIDsForAccount(sess.Subject) {
|
||||
owned[id] = true
|
||||
}
|
||||
out := make([]store.Device, 0, len(owned))
|
||||
for _, d := range all {
|
||||
if owned[d.ID] {
|
||||
out = append(out, d)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// mayTouchRun reports whether this session may read or delete a given run.
|
||||
//
|
||||
// Checked against the device list rather than against the run's own metadata, so an unlinked or
|
||||
// revoked device stops granting access the moment the link is gone.
|
||||
func (s *Server) mayTouchRun(sess *adminauth.Session, device string) bool {
|
||||
if sess.Admin {
|
||||
return true
|
||||
}
|
||||
for _, d := range s.visibleDevices(sess) {
|
||||
if d.ID == device {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func (s *Server) loginForm(w http.ResponseWriter, r *http.Request) {
|
||||
if s.session(r) != nil {
|
||||
http.Redirect(w, r, "/", http.StatusSeeOther)
|
||||
return
|
||||
}
|
||||
s.render(w, r, "login", map[string]any{
|
||||
"OIDC": s.oidcAvailable(),
|
||||
"LocalSet": s.Store.LocalAdmin() != nil,
|
||||
"AdminUser": s.AdminUser,
|
||||
})
|
||||
}
|
||||
|
||||
func (s *Server) dashboard(w http.ResponseWriter, r *http.Request, sess *adminauth.Session) {
|
||||
devices := s.visibleDevices(sess)
|
||||
linked := 0
|
||||
for _, d := range devices {
|
||||
if d.LinkedToAccount() {
|
||||
linked++
|
||||
}
|
||||
}
|
||||
var selftest any
|
||||
// The self-test describes the server's own health, which is an operator's concern; a user
|
||||
// looking at their uploads has no use for it and no ability to act on it.
|
||||
if s.SelfTest != nil && sess.Admin {
|
||||
selftest = s.SelfTest()
|
||||
}
|
||||
s.render(w, r, "dashboard", map[string]any{
|
||||
"Session": sess,
|
||||
"CSRF": s.csrfToken(sess),
|
||||
"Devices": len(devices),
|
||||
"Linked": linked,
|
||||
"Runs": s.totalRuns(devices),
|
||||
"SelfTest": selftest,
|
||||
"Version": s.Version,
|
||||
"Admin": sess.Admin,
|
||||
})
|
||||
}
|
||||
|
||||
func (s *Server) totalRuns(devices []store.Device) int {
|
||||
if s.Runs == nil {
|
||||
return 0
|
||||
}
|
||||
n := 0
|
||||
for _, d := range devices {
|
||||
n += len(s.Runs.List(d.ID))
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
func (s *Server) devices(w http.ResponseWriter, r *http.Request, sess *adminauth.Session) {
|
||||
devices := s.visibleDevices(sess)
|
||||
// Newest first: the device someone is looking for is almost always the one just enrolled.
|
||||
sort.Slice(devices, func(i, j int) bool { return devices[i].Enrolled.After(devices[j].Enrolled) })
|
||||
|
||||
type row struct {
|
||||
store.Device
|
||||
Runs int
|
||||
}
|
||||
rows := make([]row, 0, len(devices))
|
||||
for _, d := range devices {
|
||||
n := 0
|
||||
if s.Runs != nil {
|
||||
n = len(s.Runs.List(d.ID))
|
||||
}
|
||||
rows = append(rows, row{Device: d, Runs: n})
|
||||
}
|
||||
// html/template rewrites an href whose scheme it does not recognise to "#ZgotmplZ", so the
|
||||
// enrollment link rendered as a dead anchor that did nothing when tapped — silently, since the
|
||||
// markup looks fine and only the sanitised attribute gives it away.
|
||||
//
|
||||
// Marking it template.URL opts out of that sanitising, which is only safe because the shape is
|
||||
// checked first: this value arrives in a query parameter, so without the check a crafted
|
||||
// /devices?link=javascript:… would put a script URL straight into the page.
|
||||
link := r.URL.Query().Get("link")
|
||||
var href template.URL
|
||||
if strings.HasPrefix(link, "echolot://enroll?") {
|
||||
href = template.URL(link)
|
||||
}
|
||||
s.render(w, r, "devices", map[string]any{
|
||||
"Session": sess, "CSRF": s.csrfToken(sess), "Rows": rows,
|
||||
// Rendered from the same validated value as the href, so a rejected link produces neither.
|
||||
"Link": link, "LinkHref": href, "LinkQR": qrSVG(string(href)), "Admin": sess.Admin,
|
||||
})
|
||||
}
|
||||
|
||||
func (s *Server) revokeDevice(w http.ResponseWriter, r *http.Request, sess *adminauth.Session) {
|
||||
id := r.PathValue("id")
|
||||
if err := s.Store.DeleteDevice(id); err != nil {
|
||||
http.Error(w, err.Error(), http.StatusBadRequest)
|
||||
return
|
||||
}
|
||||
// Worth a log line: revoking a device is destructive, immediate, and someone will eventually
|
||||
// want to know who did it and when.
|
||||
slog.Info("device revoked", "device", id, "by", sess.Subject)
|
||||
http.Redirect(w, r, "/devices", http.StatusSeeOther)
|
||||
}
|
||||
|
||||
func (s *Server) mintToken(w http.ResponseWriter, r *http.Request, sess *adminauth.Session) {
|
||||
tok, err := s.Store.NewEnrollToken(24*time.Hour, "admin-ui")
|
||||
if err != nil {
|
||||
http.Error(w, err.Error(), http.StatusInternalServerError)
|
||||
return
|
||||
}
|
||||
slog.Info("enrolment token minted", "by", sess.Subject)
|
||||
// The whole link, not the bare token: it carries the URL and the pin as well, and assembling
|
||||
// those by hand is where an operator gets a pin wrong by one character.
|
||||
http.Redirect(w, r, "/devices?link="+url.QueryEscape(s.EnrollLink(tok)), http.StatusSeeOther)
|
||||
}
|
||||
|
||||
// enrollTokensAPI is POST /admin/enroll-tokens, the endpoint the spec's §2.1 example names.
|
||||
// Content-negotiated: Accept: application/json gets the spec shape {token, expires_in_s,
|
||||
// enroll_uri}; anything else (a browser) gets the same redirect-to-QR flow as the form above,
|
||||
// so the one path serves both audiences.
|
||||
func (s *Server) enrollTokensAPI(w http.ResponseWriter, r *http.Request) {
|
||||
subject, ok := s.apiAdmin(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
note := r.URL.Query().Get("note")
|
||||
if note == "" {
|
||||
note = "admin-api"
|
||||
}
|
||||
const ttl = 24 * time.Hour
|
||||
tok, err := s.Store.NewEnrollToken(ttl, note)
|
||||
if err != nil {
|
||||
http.Error(w, err.Error(), http.StatusInternalServerError)
|
||||
return
|
||||
}
|
||||
slog.Info("enrolment token minted", "by", subject, "note", note)
|
||||
if !strings.Contains(r.Header.Get("Accept"), "application/json") {
|
||||
http.Redirect(w, r, "/devices?link="+url.QueryEscape(s.EnrollLink(tok)), http.StatusSeeOther)
|
||||
return
|
||||
}
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
// The whole link, not the bare token (§2.1): the server is the only party holding URL, pin
|
||||
// and token at once, and a hand-assembled pin wrong by one character fails as an inscrutable
|
||||
// TLS error later rather than loudly here.
|
||||
_ = json.NewEncoder(w).Encode(map[string]any{
|
||||
"token": tok,
|
||||
"expires_in_s": int(ttl.Seconds()),
|
||||
"enroll_uri": s.EnrollLink(tok),
|
||||
})
|
||||
}
|
||||
|
||||
// EnrollLink is supplied by the caller so this package does not need the control server's pin.
|
||||
var _ = 0
|
||||
|
||||
func (s *Server) runsList(w http.ResponseWriter, r *http.Request, sess *adminauth.Session) {
|
||||
type row struct {
|
||||
runs.Meta
|
||||
DeviceName string
|
||||
}
|
||||
var rows []row
|
||||
for _, d := range s.visibleDevices(sess) {
|
||||
if s.Runs == nil {
|
||||
break
|
||||
}
|
||||
name := d.Name
|
||||
if name == "" {
|
||||
name = d.ID
|
||||
}
|
||||
for _, m := range s.Runs.List(d.ID) {
|
||||
rows = append(rows, row{Meta: m, DeviceName: name})
|
||||
}
|
||||
}
|
||||
sort.Slice(rows, func(i, j int) bool { return rows[i].UploadedAt.After(rows[j].UploadedAt) })
|
||||
if len(rows) > 200 {
|
||||
rows = rows[:200] // a page, not the archive; the count is on the dashboard
|
||||
}
|
||||
s.render(w, r, "runs", map[string]any{
|
||||
"Session": sess, "CSRF": s.csrfToken(sess), "Rows": rows, "Admin": sess.Admin,
|
||||
})
|
||||
}
|
||||
|
||||
func (s *Server) runView(w http.ResponseWriter, r *http.Request, sess *adminauth.Session) {
|
||||
// 404 rather than 403 for someone else's run: a distinguishable "you may not see this" tells
|
||||
// an unauthorised caller that the run exists, which is itself something they should not learn.
|
||||
if !s.mayTouchRun(sess, r.PathValue("device")) {
|
||||
http.NotFound(w, r)
|
||||
return
|
||||
}
|
||||
body, err := s.Runs.Get(r.PathValue("device"), r.PathValue("id"))
|
||||
if err != nil {
|
||||
http.NotFound(w, r)
|
||||
return
|
||||
}
|
||||
// Re-indented for reading, but otherwise exactly what was stored. An admin sees the document
|
||||
// at the privacy level its uploader chose — there is nothing here that can un-redact it.
|
||||
var pretty json.RawMessage = body
|
||||
out, err := json.MarshalIndent(json.RawMessage(pretty), "", " ")
|
||||
if err != nil {
|
||||
out = body
|
||||
}
|
||||
s.render(w, r, "run", map[string]any{
|
||||
"Session": sess, "CSRF": s.csrfToken(sess), "Admin": sess.Admin,
|
||||
"Device": r.PathValue("device"), "ID": r.PathValue("id"),
|
||||
"JSON": string(out),
|
||||
})
|
||||
}
|
||||
|
||||
func (s *Server) runDelete(w http.ResponseWriter, r *http.Request, sess *adminauth.Session) {
|
||||
device, id := r.PathValue("device"), r.PathValue("id")
|
||||
if !s.mayTouchRun(sess, device) {
|
||||
http.NotFound(w, r)
|
||||
return
|
||||
}
|
||||
if err := s.Runs.Delete(device, id); err != nil {
|
||||
http.Error(w, err.Error(), http.StatusInternalServerError)
|
||||
return
|
||||
}
|
||||
slog.Info("run deleted", "device", device, "run", id, "by", sess.Subject)
|
||||
http.Redirect(w, r, "/runs", http.StatusSeeOther)
|
||||
}
|
||||
@@ -0,0 +1,58 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package adminui
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"html/template"
|
||||
"strings"
|
||||
|
||||
qrcode "github.com/skip2/go-qrcode"
|
||||
)
|
||||
|
||||
// qrSVG renders text as an inline SVG QR code, or empty if it will not encode.
|
||||
//
|
||||
// Inline SVG rather than a PNG data: URI because the page's CSP is `default-src 'none'` and means
|
||||
// it. A data: image would need img-src opened up; markup needs nothing, and the QR is generated
|
||||
// here from a boolean matrix, so nothing a user supplied reaches the output.
|
||||
//
|
||||
// Drawn as one path rather than a rect per module: a link of this length encodes to roughly 60x60
|
||||
// modules, and two thousand elements is a lot of DOM for a picture of a square.
|
||||
func qrSVG(text string) template.HTML {
|
||||
if text == "" {
|
||||
return ""
|
||||
}
|
||||
// Medium recovery: a phone camera reading a screen has no dirt or creases to survive, and
|
||||
// lower recovery keeps the module count down, which keeps it scannable on a small display.
|
||||
q, err := qrcode.New(text, qrcode.Medium)
|
||||
if err != nil {
|
||||
return "" // too long to encode; the link text below it still works
|
||||
}
|
||||
bitmap := q.Bitmap()
|
||||
n := len(bitmap)
|
||||
if n == 0 {
|
||||
return ""
|
||||
}
|
||||
|
||||
var path strings.Builder
|
||||
for y, row := range bitmap {
|
||||
for x, dark := range row {
|
||||
if dark {
|
||||
fmt.Fprintf(&path, "M%d %dh1v1h-1z", x, y)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A quiet zone is part of the spec, not decoration: without it a scanner cannot find the
|
||||
// symbol's edges against whatever is next to it on the page.
|
||||
var out strings.Builder
|
||||
fmt.Fprintf(&out,
|
||||
`<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 %d %d" `+
|
||||
`width="240" height="240" shape-rendering="crispEdges" role="img" `+
|
||||
`aria-label="Enrolment link as a QR code">`+
|
||||
`<rect width="%d" height="%d" fill="#fff"/>`+
|
||||
`<path d="%s" fill="#000"/></svg>`,
|
||||
n, n, n, n, path.String())
|
||||
return template.HTML(out.String())
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
package adminui
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestQrSVGEncodesAnEnrolmentLink(t *testing.T) {
|
||||
link := "echolot://enroll?v=1&u=https%3A%2F%2Ffmr.echo-lot.app&p=pin-sha256%3AzRV9qkiLnRexAeh4RrSfJzbPWO%2BU%2F2Oj2%2FNVM%2FKfXlg%3D&t=20e6ccaa2a028dc0aab16442c258d1b8eadb5794682905fe"
|
||||
out := string(qrSVG(link))
|
||||
if !strings.HasPrefix(out, "<svg") || !strings.Contains(out, "<path d=\"M") {
|
||||
t.Fatalf("expected an svg with a path, got %.80q", out)
|
||||
}
|
||||
// A quiet zone is part of the symbol; without it scanners cannot find its edges.
|
||||
if !strings.Contains(out, `fill="#fff"`) {
|
||||
t.Error("no light background rendered")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQrSVGEmptyForNoLink(t *testing.T) {
|
||||
if qrSVG("") != "" {
|
||||
t.Error("no link should render no code")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,420 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package adminui
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"html/template"
|
||||
"log/slog"
|
||||
"net/http"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Templates are parsed once at start. html/template escapes by context, which is what makes it
|
||||
// safe to render device names and finding text that ultimately arrived over a network.
|
||||
var tpl = template.Must(template.New("base").Funcs(template.FuncMap{
|
||||
"kb": func(n int64) int64 { return n / 1024 },
|
||||
// verdictClass keeps an uploaded string out of the class attribute. The verdict arrives inside
|
||||
// a document a device sent us, so interpolating it into markup would be trusting a stranger's
|
||||
// text with a place in the stylesheet; mapping through a fixed set costs nothing and closes it.
|
||||
"verdictClass": func(v string) string {
|
||||
switch strings.ToLower(v) {
|
||||
case "green", "yellow", "red", "inconclusive":
|
||||
return "v-" + strings.ToLower(v)
|
||||
default:
|
||||
return "v-unknown"
|
||||
}
|
||||
},
|
||||
// verdictLabel says what the light means rather than what it is called. "yellow" is a colour;
|
||||
// "worth a look" is a finding, and the reader is here to act on it.
|
||||
"verdictLabel": func(v string) string {
|
||||
switch strings.ToLower(v) {
|
||||
case "green":
|
||||
return "clean"
|
||||
case "yellow":
|
||||
return "worth a look"
|
||||
case "red":
|
||||
return "faults found"
|
||||
case "inconclusive":
|
||||
return "inconclusive"
|
||||
default:
|
||||
return "not recorded"
|
||||
}
|
||||
},
|
||||
}).Parse(baseHTML))
|
||||
|
||||
func (s *Server) render(w http.ResponseWriter, r *http.Request, page string, data map[string]any) {
|
||||
data["Page"] = page
|
||||
var buf bytes.Buffer
|
||||
if err := tpl.Execute(&buf, data); err != nil {
|
||||
slog.Error("admin template", "page", page, "err", err)
|
||||
http.Error(w, "template error", http.StatusInternalServerError)
|
||||
return
|
||||
}
|
||||
w.Header().Set("Content-Type", "text/html; charset=utf-8")
|
||||
// There is no script here and nothing loaded from anywhere else, so a strict policy costs
|
||||
// nothing and closes injected-script attacks even if an escaping bug ever slips through.
|
||||
w.Header().Set("Content-Security-Policy", "default-src 'none'; style-src 'unsafe-inline'; form-action 'self'")
|
||||
w.Header().Set("Referrer-Policy", "no-referrer")
|
||||
w.Header().Set("X-Content-Type-Options", "nosniff")
|
||||
_, _ = buf.WriteTo(w)
|
||||
}
|
||||
|
||||
// The visual language is an echo sounder's, which is what the name means: an instrument that emits
|
||||
// a ping and reads what comes back. That gives the palette (the colours of a water column rather
|
||||
// than a neutral near-black), the type (machine-set, because an instrument's readings are), and
|
||||
// the one piece of real ornament — a trace of returns across time on the runs page.
|
||||
//
|
||||
// No web fonts: the CSP forbids loading anything, and shipping font files with a single Go binary
|
||||
// would trade the property that makes this server pleasant to run for a typeface. So the character
|
||||
// has to come from treatment — tracking, case, scale, rules — rather than from novel letterforms.
|
||||
//
|
||||
// Tables become stacked records below 46rem rather than scrolling sideways. That is not a fallback:
|
||||
// a sounding log prints as label-and-value pairs, and on a phone that form is easier to read than
|
||||
// any table, so the mobile layout is the more faithful one of the two.
|
||||
const baseHTML = `<!doctype html>
|
||||
<html lang="en"><head><meta charset="utf-8">
|
||||
<meta name="viewport" content="width=device-width,initial-scale=1">
|
||||
<title>Echolot — {{.Page}}</title>
|
||||
<style>
|
||||
:root{
|
||||
--abyss:#071419; --hull:#0d2028; --raise:#122a34; --rule:#17323d;
|
||||
--ink:#dce8ea; --dim:#7d97a1; --trace:#6fc9b4;
|
||||
--green:#57ad82; --amber:#cf9b3c; --red:#c25757; --slate:#62767f;
|
||||
--mono:ui-monospace,"SF Mono","IBM Plex Mono","JetBrains Mono",Menlo,Consolas,monospace;
|
||||
--prose:system-ui,-apple-system,"Segoe UI",sans-serif;
|
||||
color-scheme:dark;
|
||||
}
|
||||
*{box-sizing:border-box}
|
||||
body{margin:0;background:var(--abyss);color:var(--ink);
|
||||
font:400 15px/1.55 var(--prose);-webkit-text-size-adjust:100%}
|
||||
|
||||
/* ---- masthead ------------------------------------------------------------------------ */
|
||||
/* Wraps rather than overflows: a rigid row pushes the account and its sign-out button past
|
||||
the edge of a phone screen, where they cannot be reached at all. */
|
||||
.top{display:flex;flex-wrap:wrap;align-items:center;gap:.5rem 1.4rem;
|
||||
padding:.85rem 1.1rem;background:var(--hull);border-bottom:1px solid var(--rule)}
|
||||
.mark{font:600 .95rem/1 var(--mono);letter-spacing:.02em;margin:0;color:var(--ink)}
|
||||
.mark span{color:var(--trace)}
|
||||
.top nav{display:flex;flex-wrap:wrap;gap:.15rem .9rem}
|
||||
.top nav a{font:500 .82rem/1 var(--mono);letter-spacing:.06em;color:var(--dim);
|
||||
text-decoration:none;padding:.35rem 0;border-bottom:1px solid transparent}
|
||||
.top nav a:hover{color:var(--ink)}
|
||||
.top nav a[aria-current]{color:var(--trace);border-bottom-color:var(--trace)}
|
||||
.who{margin-left:auto;display:flex;align-items:center;gap:.7rem;flex-wrap:wrap;
|
||||
font:.78rem/1.3 var(--mono);color:var(--dim)}
|
||||
|
||||
main{padding:1.1rem;max-width:64rem}
|
||||
|
||||
/* ---- headings: a graduation mark, like a depth scale ------------------------------- */
|
||||
h2{font:500 1.05rem/1.2 var(--mono);letter-spacing:-.01em;margin:1.4rem 0 .2rem;
|
||||
padding-left:.7rem;border-left:2px solid var(--trace)}
|
||||
h2:first-child{margin-top:0}
|
||||
h3{font:500 .8rem/1 var(--mono);letter-spacing:.14em;text-transform:uppercase;
|
||||
color:var(--dim);margin:0 0 .7rem}
|
||||
.lede{color:var(--dim);font-size:.9rem;margin:.5rem 0 1rem;max-width:46rem}
|
||||
|
||||
/* ---- readout: how an instrument prints a value ------------------------------------- */
|
||||
.readout{list-style:none;margin:0;padding:0}
|
||||
.readout li{display:flex;align-items:baseline;gap:.5rem;padding:.3rem 0;
|
||||
font:.85rem/1.4 var(--mono)}
|
||||
.readout .k{color:var(--dim);white-space:nowrap}
|
||||
/* The dotted leader is how a sounding log runs a label out to its value. It is also the thing
|
||||
that lets a label and a number sit on one line at any width without a table. */
|
||||
.readout .lead{flex:1 1 auto;min-width:1.5rem;align-self:center;height:1px;
|
||||
background:repeating-linear-gradient(90deg,var(--rule) 0 2px,transparent 2px 5px)}
|
||||
.readout .v{color:var(--ink);text-align:right;overflow-wrap:anywhere}
|
||||
|
||||
/* ---- the trace: one bar per run, oldest to newest ---------------------------------- */
|
||||
/* The signature, and the only ornament here: an echo sounder draws returns against time, and
|
||||
so does this. Rows arrive newest-first, so the strip is reversed in CSS rather than in Go. */
|
||||
.trace{display:flex;flex-direction:row-reverse;justify-content:flex-end;align-items:flex-end;gap:2px;
|
||||
height:3rem;padding:.7rem .8rem;background:var(--hull);
|
||||
border:1px solid var(--rule);border-radius:3px;overflow:hidden}
|
||||
.trace i{flex:1 1 3px;min-width:2px;max-width:9px;border-radius:1px;opacity:.9}
|
||||
.trace .v-green{height:35%;background:var(--green)}
|
||||
.trace .v-yellow{height:65%;background:var(--amber)}
|
||||
.trace .v-red{height:100%;background:var(--red)}
|
||||
.trace .v-inconclusive{height:22%;background:var(--slate)}
|
||||
.trace .v-unknown{height:12%;background:var(--rule)}
|
||||
.trace-key{display:flex;flex-wrap:wrap;gap:.3rem .9rem;margin:.45rem 0 0;
|
||||
font:.72rem/1 var(--mono);letter-spacing:.05em;color:var(--dim)}
|
||||
.trace-key b{font-weight:400;color:var(--dim)}
|
||||
.trace-key em{font-style:normal;display:inline-block;width:.5rem;height:.5rem;
|
||||
border-radius:1px;margin-right:.35rem;vertical-align:baseline;background:var(--rule)}
|
||||
.trace-key em.v-green{background:var(--green)}
|
||||
.trace-key em.v-yellow{background:var(--amber)}
|
||||
.trace-key em.v-red{background:var(--red)}
|
||||
.trace-key em.v-inconclusive{background:var(--slate)}
|
||||
|
||||
/* ---- records: tables that stack on a phone ----------------------------------------- */
|
||||
.rec{border:1px solid var(--rule);border-radius:3px;background:var(--hull);
|
||||
padding:.75rem .85rem;margin:.5rem 0}
|
||||
.rec-head{display:flex;flex-wrap:wrap;align-items:baseline;gap:.5rem;
|
||||
font:.85rem/1.3 var(--mono);margin-bottom:.35rem}
|
||||
.rec-head .id{overflow-wrap:anywhere;color:var(--ink)}
|
||||
.rec form{margin-top:.6rem}
|
||||
/* Why a check matters is a sentence, so it is set as one — full width under the row rather
|
||||
than squeezed into a column, where it would wrap to a ribbon two words wide. */
|
||||
.why{font:.85rem/1.5 var(--prose);color:var(--dim);margin-top:.45rem;max-width:52rem}
|
||||
.tag{font:.68rem/1 var(--mono);letter-spacing:.1em;text-transform:uppercase;
|
||||
padding:.24rem .45rem;border-radius:2px;border:1px solid currentColor;white-space:nowrap}
|
||||
.v-green{color:var(--green)} .v-yellow{color:var(--amber)}
|
||||
.v-red{color:var(--red)} .v-inconclusive{color:var(--slate)} .v-unknown{color:var(--dim)}
|
||||
|
||||
/* ---- panels, controls, states ------------------------------------------------------ */
|
||||
.narrow{max-width:27rem}
|
||||
.panel{background:var(--hull);border:1px solid var(--rule);border-radius:3px;
|
||||
padding:.95rem 1rem;margin:.9rem 0;min-width:0}
|
||||
/* White plate behind the code: a QR needs the light modules to actually be light, and this
|
||||
page is dark. */
|
||||
.qr{display:inline-block;background:#fff;padding:8px;border-radius:4px;margin:.2rem 0;line-height:0}
|
||||
.qr svg{display:block;width:min(240px,60vw);height:auto}
|
||||
.empty{border:1px dashed var(--rule);border-radius:3px;padding:1.4rem 1rem;
|
||||
color:var(--dim);font-size:.9rem}
|
||||
code,pre,.mono{font-family:var(--mono);font-size:.82rem}
|
||||
code{overflow-wrap:anywhere;color:var(--trace)}
|
||||
pre{background:#040d11;border:1px solid var(--rule);border-radius:3px;padding:.8rem;
|
||||
overflow:auto;max-height:32rem;max-width:100%;color:var(--ink)}
|
||||
a{color:var(--trace)}
|
||||
button{font:500 .82rem/1 var(--mono);letter-spacing:.05em;background:var(--trace);
|
||||
color:#04181a;border:0;border-radius:3px;padding:.55rem .9rem;cursor:pointer}
|
||||
.btn{display:inline-block;font:500 .82rem/1 var(--mono);letter-spacing:.05em;
|
||||
background:var(--trace);color:#04181a;border-radius:3px;padding:.55rem .9rem;
|
||||
text-decoration:none}
|
||||
button.plain{background:transparent;color:var(--dim);border:1px solid var(--rule)}
|
||||
button.danger{background:transparent;color:var(--red);border:1px solid var(--red)}
|
||||
button:hover{filter:brightness(1.08)}
|
||||
input{font:.9rem var(--mono);background:#040d11;color:var(--ink);border:1px solid var(--rule);
|
||||
border-radius:3px;padding:.55rem .6rem;max-width:100%;width:100%}
|
||||
label{display:block;font:.72rem/1 var(--mono);letter-spacing:.12em;text-transform:uppercase;
|
||||
color:var(--dim);margin:.9rem 0 .3rem}
|
||||
.err{border:1px solid var(--red);color:var(--ink);background:rgba(194,87,87,.09);
|
||||
padding:.6rem .75rem;border-radius:3px;font-size:.9rem}
|
||||
.muted{color:var(--dim)}
|
||||
form.inline{display:inline}
|
||||
:focus-visible{outline:2px solid var(--trace);outline-offset:2px}
|
||||
@media (prefers-reduced-motion:reduce){*{transition:none!important;animation:none!important}}
|
||||
|
||||
/* ---- above 46rem the records line up in columns ------------------------------------ */
|
||||
@media (min-width:46rem){
|
||||
.top{padding:.85rem 1.6rem}
|
||||
main{padding:1.6rem}
|
||||
.recs{margin:.8rem 0}
|
||||
/* Every row shares one grid, so the columns agree across rows without a header or a table. */
|
||||
.rec{display:grid;grid-template-columns:minmax(12.5rem,18rem) minmax(0,1fr) auto;gap:.35rem 1.4rem;
|
||||
align-items:baseline;background:none;border:0;border-bottom:1px solid var(--rule);
|
||||
border-radius:0;padding:.6rem 0;margin:0}
|
||||
.rec-head{margin:0;flex-direction:column;align-items:flex-start;gap:.3rem}
|
||||
.rec form{margin:0}
|
||||
/* Widths follow the content: a device name needs room, a finding count does not. */
|
||||
.rec .readout{display:grid;grid-template-columns:1.7fr .9fr .9fr 1.1fr;gap:.15rem 1.2rem}
|
||||
.rec .readout li{padding:0}
|
||||
.rec .readout .lead{display:none}
|
||||
.rec .readout .v{text-align:left}
|
||||
.open{white-space:nowrap}
|
||||
/* Spans the full row: the sentence is the useful part, not a fourth column. */
|
||||
.why{grid-column:1/-1;margin-top:.1rem}
|
||||
}
|
||||
</style></head><body>
|
||||
<header class="top">
|
||||
<h1 class="mark">echo<span>lot</span></h1>
|
||||
{{if ne .Page "login"}}
|
||||
<nav>
|
||||
<a href="/"{{if eq .Page "dashboard"}} aria-current="page"{{end}}>overview</a>
|
||||
<a href="/devices"{{if eq .Page "devices"}} aria-current="page"{{end}}>devices</a>
|
||||
<a href="/runs"{{if eq .Page "runs"}} aria-current="page"{{end}}>runs</a>
|
||||
</nav>
|
||||
<span class="who">{{.Session.Display}}{{if not .Session.Admin}} · your account{{end}}
|
||||
<form method="post" action="/logout" class="inline"><button class="plain">Sign out</button></form>
|
||||
</span>
|
||||
{{end}}
|
||||
</header>
|
||||
<main>
|
||||
|
||||
{{if eq .Page "login"}}
|
||||
<h2>Sign in</h2>
|
||||
<p class="lede">This server keeps the measurements your devices have uploaded.</p>
|
||||
{{if .OIDC}}
|
||||
<p><a class="btn" href="/auth/start">Sign in with your identity provider</a></p>
|
||||
{{end}}
|
||||
{{if .LocalSet}}
|
||||
<form method="post" action="/login" class="panel narrow">
|
||||
<h3>Break-glass account</h3>
|
||||
<label for="u">Username</label>
|
||||
<input id="u" name="username" autocomplete="username" value="{{.AdminUser}}">
|
||||
<label for="p">Password</label>
|
||||
<input id="p" name="password" type="password" autocomplete="current-password">
|
||||
<p><button>Sign in</button></p>
|
||||
</form>
|
||||
{{else}}
|
||||
<p class="err">No break-glass account is set. Run
|
||||
<code>echolot-server --set-admin-password</code> on the host to create one.</p>
|
||||
{{end}}
|
||||
|
||||
{{else if eq .Page "dashboard"}}
|
||||
<h2>{{if .Admin}}This server{{else}}Your account{{end}}</h2>
|
||||
<ul class="readout panel">
|
||||
<li><span class="k">{{if .Admin}}devices enrolled{{else}}your devices{{end}}</span>
|
||||
<span class="lead"></span><span class="v">{{.Devices}}</span></li>
|
||||
{{if .Admin}}
|
||||
<li><span class="k">linked to an account</span>
|
||||
<span class="lead"></span><span class="v">{{.Linked}}</span></li>
|
||||
{{end}}
|
||||
<li><span class="k">{{if .Admin}}runs stored{{else}}your runs{{end}}</span>
|
||||
<span class="lead"></span><span class="v">{{.Runs}}</span></li>
|
||||
{{if .Admin}}
|
||||
<li><span class="k">server version</span>
|
||||
<span class="lead"></span><span class="v">{{.Version}}</span></li>
|
||||
{{end}}
|
||||
</ul>
|
||||
{{if not .Admin}}
|
||||
<div class="panel">
|
||||
<p>You can see every device you have signed in on, read everything they have uploaded, and
|
||||
delete any of it.</p>
|
||||
<p class="muted">Enrolling devices, revoking them, and reading other people's uploads need an
|
||||
administrator account.</p>
|
||||
</div>
|
||||
{{end}}
|
||||
{{with .SelfTest}}
|
||||
<h2>Self-test</h2>
|
||||
<p class="lede">What this server can measure from where it stands, checked at startup. A
|
||||
capability missing here is missing from every run this server takes part in — so a
|
||||
client asking for that measurement gets nothing, rather than a wrong answer.</p>
|
||||
<ul class="readout panel">
|
||||
<li><span class="k">kernel settings</span><span class="lead"></span>
|
||||
<span class="v {{if .SysctlOK}}v-green{{else}}v-yellow{{end}}">{{if .SysctlOK}}as needed{{else}}need attention{{end}}</span></li>
|
||||
<li><span class="k">egress path MTU</span><span class="lead"></span>
|
||||
<span class="v {{if .MTUOK}}v-green{{else}}v-yellow{{end}}">{{if .MTUOK}}full 1500{{else}}reduced{{end}}</span></li>
|
||||
</ul>
|
||||
{{if .Sysctls}}
|
||||
<h3>Kernel settings</h3>
|
||||
<div class="recs">
|
||||
{{range .Sysctls}}
|
||||
<div class="rec">
|
||||
<div class="rec-head"><span class="id">{{.Name}}</span>
|
||||
<span class="tag {{if eq .Severity "ok"}}v-green{{else}}v-yellow{{end}}">{{.Severity}}</span></div>
|
||||
<ul class="readout">
|
||||
<li><span class="k">found</span><span class="lead"></span><span class="v">{{.Got}}</span></li>
|
||||
<li><span class="k">wanted</span><span class="lead"></span><span class="v">{{.Want}}</span></li>
|
||||
</ul>
|
||||
<div class="why">{{.Why}}</div>
|
||||
</div>
|
||||
{{end}}
|
||||
</div>
|
||||
{{end}}
|
||||
{{if .EgressMTU}}
|
||||
<h3>Egress path MTU</h3>
|
||||
<div class="recs">
|
||||
{{range .EgressMTU}}
|
||||
<div class="rec">
|
||||
<div class="rec-head"><span class="id">{{.Target}}</span>
|
||||
<span class="tag {{if .FullMTU}}v-green{{else}}v-yellow{{end}}">{{if .FullMTU}}full{{else}}reduced{{end}}</span></div>
|
||||
<ul class="readout">
|
||||
<li><span class="k">discovered</span><span class="lead"></span>
|
||||
<span class="v">{{if .DiscoveredMTU}}{{.DiscoveredMTU}} bytes{{else}}not measured{{end}}</span></li>
|
||||
</ul>
|
||||
{{with .Err}}<div class="why">{{.}}</div>{{end}}
|
||||
</div>
|
||||
{{end}}
|
||||
</div>
|
||||
{{end}}
|
||||
{{end}}
|
||||
|
||||
{{else if eq .Page "devices"}}
|
||||
<h2>{{if .Admin}}Devices{{else}}Your devices{{end}}</h2>
|
||||
{{with .Link}}
|
||||
<div class="panel">
|
||||
<h3>Enrolment link</h3>
|
||||
<p class="lede">Single use, valid 24 hours. Treat it like a password until it is spent.</p>
|
||||
<!-- On the phone being enrolled this is the whole procedure: the scheme is registered by the
|
||||
app, so following the link hands it the token directly. Copying a 200-character string
|
||||
between two devices is the step that goes wrong, and it does not have to happen at all. -->
|
||||
{{with $.LinkHref}}<p><a class="btn" href="{{.}}">Open in the Echolot app</a></p>{{end}}
|
||||
<p class="muted">Works on the phone you are enrolling. From another device, scan this:</p>
|
||||
{{with $.LinkQR}}<div class="qr">{{.}}</div>{{end}}
|
||||
<p class="muted">Or copy the link into the app's enrolment field, or deliver it over adb.</p>
|
||||
<p><code>{{.}}</code></p>
|
||||
<p class="muted mono">adb shell am start -a android.intent.action.VIEW -d "{{.}}"</p>
|
||||
</div>
|
||||
{{end}}
|
||||
{{if .Admin}}
|
||||
<form method="post" action="/enroll-tokens">
|
||||
<input type="hidden" name="csrf" value="{{.CSRF}}">
|
||||
<button>Create enrolment link</button>
|
||||
</form>
|
||||
{{end}}
|
||||
{{if .Rows}}<div class="recs">
|
||||
{{range .Rows}}
|
||||
<div class="rec">
|
||||
<div class="rec-head">
|
||||
<span class="id">{{if .Name}}{{.Name}}{{else}}{{.ID}}{{end}}</span>
|
||||
{{if .LinkedToAccount}}<span class="tag v-green">{{.AccountName}}</span>
|
||||
{{else}}<span class="tag v-unknown">no account</span>{{end}}
|
||||
</div>
|
||||
<ul class="readout">
|
||||
<li><span class="k">device</span><span class="lead"></span><span class="v">{{.ID}}</span></li>
|
||||
<li><span class="k">enrolled</span><span class="lead"></span>
|
||||
<span class="v">{{.Enrolled.Format "2006-01-02 15:04"}}</span></li>
|
||||
<li><span class="k">runs</span><span class="lead"></span><span class="v">{{.Runs}}</span></li>
|
||||
</ul>
|
||||
{{if $.Admin}}
|
||||
<form method="post" action="/devices/{{.ID}}/revoke">
|
||||
<input type="hidden" name="csrf" value="{{$.CSRF}}">
|
||||
<button class="danger">Revoke</button>
|
||||
</form>
|
||||
{{end}}
|
||||
</div>
|
||||
{{end}}
|
||||
</div>{{else}}
|
||||
<p class="empty">{{if .Admin}}No devices yet. Create an enrolment link and open it on the phone
|
||||
you want to measure from.{{else}}No devices yet. Sign in from the Echolot app on your phone to
|
||||
link one to this account.{{end}}</p>
|
||||
{{end}}
|
||||
|
||||
{{else if eq .Page "runs"}}
|
||||
<h2>{{if .Admin}}Uploaded runs{{else}}Your uploaded runs{{end}}</h2>
|
||||
<p class="lede">Each run is shown exactly as it arrived, at the privacy level its uploader chose.
|
||||
Nothing here can un-redact one.</p>
|
||||
{{if .Rows}}
|
||||
<div class="trace">{{range .Rows}}<i class="{{verdictClass .Verdict}}"></i>{{end}}</div>
|
||||
<p class="trace-key"><b>oldest → newest</b>
|
||||
<b><em class="v-green"></em>clean</b>
|
||||
<b><em class="v-yellow"></em>worth a look</b>
|
||||
<b><em class="v-red"></em>faults</b>
|
||||
<b><em class="v-inconclusive"></em>inconclusive</b></p>
|
||||
{{end}}
|
||||
{{if .Rows}}<div class="recs">
|
||||
{{range .Rows}}
|
||||
<div class="rec">
|
||||
<div class="rec-head">
|
||||
<span class="id">{{.UploadedAt.Format "2006-01-02 15:04"}}</span>
|
||||
<span class="tag {{verdictClass .Verdict}}">{{verdictLabel .Verdict}}</span>
|
||||
</div>
|
||||
<ul class="readout">
|
||||
<li><span class="k">device</span><span class="lead"></span><span class="v">{{.DeviceName}}</span></li>
|
||||
<li><span class="k">findings</span><span class="lead"></span><span class="v">{{.FindingCount}}</span></li>
|
||||
<li><span class="k">size</span><span class="lead"></span><span class="v">{{kb .SizeBytes}} kB</span></li>
|
||||
<li><span class="k">privacy</span><span class="lead"></span><span class="v">{{.Anonymization}}</span></li>
|
||||
</ul>
|
||||
<div class="open"><a href="/runs/{{.DeviceID}}/{{.ID}}">Open run</a></div>
|
||||
</div>
|
||||
{{end}}
|
||||
</div>{{else}}
|
||||
<p class="empty">Nothing uploaded yet. Take a measurement in the app and upload it; it will
|
||||
appear here.</p>
|
||||
{{end}}
|
||||
|
||||
{{else if eq .Page "run"}}
|
||||
<h2>Run {{.ID}}</h2>
|
||||
<p class="lede">The document as stored, indented for reading. Nothing has been added or removed.</p>
|
||||
<form method="post" action="/runs/{{.Device}}/{{.ID}}/delete">
|
||||
<input type="hidden" name="csrf" value="{{.CSRF}}">
|
||||
<button class="danger">Delete this run</button>
|
||||
</form>
|
||||
<pre>{{.JSON}}</pre>
|
||||
{{end}}
|
||||
|
||||
</main></body></html>
|
||||
`
|
||||
@@ -0,0 +1,119 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package adminui
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"echo-lot.app/server/internal/adminauth"
|
||||
"echo-lot.app/server/internal/runs"
|
||||
"echo-lot.app/server/internal/store"
|
||||
)
|
||||
|
||||
// Two accounts, one device each, plus an unlinked device nobody owns.
|
||||
func fixture(t *testing.T) (*Server, string, string, string) {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
st, err := store.Open(dir)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
rs, err := runs.Open(dir, runs.DefaultPolicy())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
enroll := func(name string) string {
|
||||
tok, err := st.NewEnrollToken(time.Hour, "test")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
d, err := st.Redeem(tok, name)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return d.ID
|
||||
}
|
||||
mine, theirs, orphan := enroll("mine"), enroll("theirs"), enroll("orphan")
|
||||
if err := st.LinkAccount(mine, "oidc#me", "Me"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := st.LinkAccount(theirs, "oidc#you", "You"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, d := range []string{mine, theirs, orphan} {
|
||||
if _, err := rs.Put(d, []byte(`{"run":{"id":"r"}}`), true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
return &Server{Store: st, Runs: rs}, mine, theirs, orphan
|
||||
}
|
||||
|
||||
func user() *adminauth.Session { return &adminauth.Session{Subject: "oidc#me", Display: "Me"} }
|
||||
func admin() *adminauth.Session { return &adminauth.Session{Subject: "local:a", Admin: true} }
|
||||
|
||||
func TestVisibleDevicesScopesToAccount(t *testing.T) {
|
||||
s, mine, theirs, orphan := fixture(t)
|
||||
|
||||
got := s.visibleDevices(user())
|
||||
if len(got) != 1 || got[0].ID != mine {
|
||||
t.Fatalf("a user should see only their own device, got %+v", got)
|
||||
}
|
||||
|
||||
all := s.visibleDevices(admin())
|
||||
if len(all) != 3 {
|
||||
t.Fatalf("an admin should see every device, got %d", len(all))
|
||||
}
|
||||
_ = theirs
|
||||
_ = orphan
|
||||
}
|
||||
|
||||
func TestUnlinkedDevicesBelongToNobody(t *testing.T) {
|
||||
// An enrolled but never-signed-in device is not "everyone's" — a user must not inherit it
|
||||
// just because no account claimed it.
|
||||
s, _, _, orphan := fixture(t)
|
||||
if s.mayTouchRun(user(), orphan) {
|
||||
t.Fatal("an unlinked device was treated as the user's own")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunAccessFollowsDeviceOwnership(t *testing.T) {
|
||||
s, mine, theirs, _ := fixture(t)
|
||||
|
||||
if !s.mayTouchRun(user(), mine) {
|
||||
t.Fatal("a user cannot reach their own run")
|
||||
}
|
||||
if s.mayTouchRun(user(), theirs) {
|
||||
t.Fatal("a user reached someone else's run")
|
||||
}
|
||||
if !s.mayTouchRun(admin(), theirs) {
|
||||
t.Fatal("an admin should reach any run")
|
||||
}
|
||||
}
|
||||
|
||||
func TestAccessEndsWhenTheLinkDoes(t *testing.T) {
|
||||
// Ownership is read from the device list on every request rather than captured at sign-in,
|
||||
// so unlinking takes effect immediately — a session issued while linked must not keep working.
|
||||
s, mine, _, _ := fixture(t)
|
||||
sess := user()
|
||||
if !s.mayTouchRun(sess, mine) {
|
||||
t.Fatal("precondition: the device should start out owned")
|
||||
}
|
||||
if err := s.Store.LinkAccount(mine, "", ""); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if s.mayTouchRun(sess, mine) {
|
||||
t.Fatal("access survived the account link being removed")
|
||||
}
|
||||
}
|
||||
|
||||
func TestEmptySubjectMatchesNothing(t *testing.T) {
|
||||
// The dangerous degenerate case: a session with no subject must own nothing, not everything
|
||||
// that happens to have an empty account id.
|
||||
s, _, _, _ := fixture(t)
|
||||
anon := &adminauth.Session{Subject: "", Display: ""}
|
||||
if got := s.visibleDevices(anon); len(got) != 0 {
|
||||
t.Fatalf("an empty subject matched %d devices", len(got))
|
||||
}
|
||||
}
|
||||
@@ -63,20 +63,24 @@ type Server struct {
|
||||
nsName string // this server's own name for NS/authority answers
|
||||
primaryV4 netip.Addr
|
||||
primaryV6 netip.Addr
|
||||
retention time.Duration // query-log age limit; <= 0 means only the ring cap bounds it
|
||||
|
||||
mu sync.Mutex
|
||||
log []Query // ring, newest last
|
||||
retainTo time.Time
|
||||
mu sync.Mutex
|
||||
log []Query // ring, newest last
|
||||
}
|
||||
|
||||
const logCap = 8192
|
||||
|
||||
// New creates a server for zone (with or without trailing dot). nsName is the
|
||||
// server's own hostname (for the zone's NS record); primary v4/v6 are this
|
||||
// host's addresses used to answer the zone apex / NS glue.
|
||||
func New(zone, nsName string, v4, v6 netip.Addr) *Server {
|
||||
// host's addresses used to answer the zone apex / NS glue. retention is how
|
||||
// long logged queries are kept (spec §6; the privacy default is 24 h).
|
||||
func New(zone, nsName string, v4, v6 netip.Addr, retention time.Duration) *Server {
|
||||
z := strings.ToLower(strings.TrimSuffix(zone, ".")) + "."
|
||||
return &Server{zone: z, nsName: strings.TrimSuffix(nsName, ".") + ".", primaryV4: v4, primaryV6: v6}
|
||||
return &Server{
|
||||
zone: z, nsName: strings.TrimSuffix(nsName, ".") + ".",
|
||||
primaryV4: v4, primaryV6: v6, retention: retention,
|
||||
}
|
||||
}
|
||||
|
||||
// RecentForPrefix returns logged queries whose qname contains ".<prefix>."
|
||||
@@ -84,6 +88,7 @@ func New(zone, nsName string, v4, v6 netip.Addr) *Server {
|
||||
func (s *Server) RecentForPrefix(prefix string) []Query {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
s.dropExpiredLocked(time.Now().UTC())
|
||||
needle := "." + strings.ToLower(prefix) + "."
|
||||
var out []Query
|
||||
for _, q := range s.log {
|
||||
@@ -97,12 +102,34 @@ func (s *Server) RecentForPrefix(prefix string) []Query {
|
||||
func (s *Server) record(q Query) {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
s.dropExpiredLocked(q.At)
|
||||
if len(s.log) >= logCap {
|
||||
s.log = s.log[1:]
|
||||
}
|
||||
s.log = append(s.log, q)
|
||||
}
|
||||
|
||||
// dropExpiredLocked enforces the retention window on the query log.
|
||||
//
|
||||
// The 24-hour retention was advertised as the privacy default (spec §6/§7) and then not
|
||||
// enforced: the ring only bounded *count*, so on a quiet server a resolver's queries could sit
|
||||
// in memory for weeks. Aged out on every write and every read — whichever comes first — so an
|
||||
// idle log still forgets on schedule the moment anyone looks. Entries are appended in time
|
||||
// order, so expiry is always a prefix of the slice.
|
||||
func (s *Server) dropExpiredLocked(now time.Time) {
|
||||
if s.retention <= 0 {
|
||||
return
|
||||
}
|
||||
cutoff := now.Add(-s.retention)
|
||||
i := 0
|
||||
for i < len(s.log) && s.log[i].At.Before(cutoff) {
|
||||
i++
|
||||
}
|
||||
if i > 0 {
|
||||
s.log = append([]Query(nil), s.log[i:]...) // reallocate so the old backing array frees
|
||||
}
|
||||
}
|
||||
|
||||
// ServeUDP / ServeTCP run read loops; call one per bound address.
|
||||
func (s *Server) ServeUDP(conn *net.UDPConn) error {
|
||||
buf := make([]byte, 1500)
|
||||
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
"net"
|
||||
"net/netip"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// buildQuery makes a single-question DNS query, optionally with an EDNS OPT.
|
||||
@@ -65,7 +66,7 @@ func parseResponse(t *testing.T, resp []byte) (flags uint16, answers []ans) {
|
||||
}
|
||||
|
||||
func newTestServer() *Server {
|
||||
return New("c.echo-lot.app", "fmr", netip.MustParseAddr("192.0.2.1"), netip.MustParseAddr("2001:db8::1"))
|
||||
return New("c.echo-lot.app", "fmr", netip.MustParseAddr("192.0.2.1"), netip.MustParseAddr("2001:db8::1"), 24*time.Hour)
|
||||
}
|
||||
|
||||
func TestReferenceRecords(t *testing.T) {
|
||||
@@ -159,3 +160,29 @@ func TestOutOfZoneNXDomain(t *testing.T) {
|
||||
t.Fatalf("out-of-zone should be NXDOMAIN, flags=%#x", flags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestQueryLogRetentionForgetsOldEntries(t *testing.T) {
|
||||
s := newTestServer() // 24 h retention
|
||||
now := time.Now().UTC()
|
||||
s.record(Query{QName: "old.sess1.c.echo-lot.app", At: now.Add(-25 * time.Hour)})
|
||||
s.record(Query{QName: "fresh.sess1.c.echo-lot.app", At: now})
|
||||
|
||||
got := s.RecentForPrefix("sess1")
|
||||
if len(got) != 1 || got[0].QName != "fresh.sess1.c.echo-lot.app" {
|
||||
t.Fatalf("retention not enforced: %+v", got)
|
||||
}
|
||||
|
||||
// Reads must age the log too: an idle server still has to forget on schedule.
|
||||
s.log[0].At = now.Add(-25 * time.Hour)
|
||||
if got := s.RecentForPrefix("sess1"); len(got) != 0 {
|
||||
t.Fatalf("read path did not expire entries: %+v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestZeroRetentionKeepsEverything(t *testing.T) {
|
||||
s := New("c.echo-lot.app", "fmr", netip.MustParseAddr("192.0.2.1"), netip.MustParseAddr("2001:db8::1"), 0)
|
||||
s.record(Query{QName: "ancient.sess1.c.echo-lot.app", At: time.Now().UTC().Add(-1000 * time.Hour)})
|
||||
if got := s.RecentForPrefix("sess1"); len(got) != 1 {
|
||||
t.Fatal("retention 0 must mean 'ring cap only', not 'keep nothing'")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package config
|
||||
|
||||
import "testing"
|
||||
|
||||
// The fixture is fmr's real UDP listen spec, because the point of deriving these from the bound
|
||||
// listeners is that they cannot disagree with what the server actually answers on.
|
||||
const fmrUDP = "89.185.109.150:8442,89.185.109.151:8442," +
|
||||
"[2001:1ad0:c4fe:6767::150]:8442,[2001:1ad0:c4fe:6767::151]:8442"
|
||||
|
||||
func TestMeasurementAddrsSplitsPrimaryFromReserved(t *testing.T) {
|
||||
c := &Config{
|
||||
UDPListen: fmrUDP,
|
||||
ReservedAddrs: "89.185.109.151,2001:1ad0:c4fe:6767::151",
|
||||
}
|
||||
ip4, ip6, ip4Alt, ip6Alt := c.MeasurementAddrs()
|
||||
for _, tc := range []struct{ got, want, name string }{
|
||||
{ip4, "89.185.109.150", "ip4"},
|
||||
{ip6, "2001:1ad0:c4fe:6767::150", "ip6"},
|
||||
{ip4Alt, "89.185.109.151", "ip4_alt"},
|
||||
{ip6Alt, "2001:1ad0:c4fe:6767::151", "ip6_alt"},
|
||||
} {
|
||||
if tc.got != tc.want {
|
||||
t.Errorf("%s = %q, want %q", tc.name, tc.got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestMeasurementAddrsWithNothingReserved(t *testing.T) {
|
||||
// No reservation means no alternate: reporting a second address as the RFC 5780 alternate
|
||||
// when it was never set aside for that would tell a client to expect a redirect that the
|
||||
// server has no intention of sending.
|
||||
c := &Config{UDPListen: fmrUDP}
|
||||
ip4, ip6, ip4Alt, ip6Alt := c.MeasurementAddrs()
|
||||
if ip4 == "" || ip6 == "" {
|
||||
t.Fatalf("primaries should still be found: ip4=%q ip6=%q", ip4, ip6)
|
||||
}
|
||||
if ip4Alt != "" || ip6Alt != "" {
|
||||
t.Errorf("no address is reserved, so there is no alternate; got %q / %q", ip4Alt, ip6Alt)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMeasurementAddrsIgnoresWhatItCannotRead(t *testing.T) {
|
||||
// A wildcard bind names no address, and a hostname is not resolved here. Either would be a
|
||||
// guess presented to clients as fact.
|
||||
c := &Config{UDPListen: ":8442,probe.example.net:8442,89.185.109.150:8442"}
|
||||
ip4, ip6, _, _ := c.MeasurementAddrs()
|
||||
if ip4 != "89.185.109.150" {
|
||||
t.Errorf("ip4 = %q, want the one address that was actually spelled out", ip4)
|
||||
}
|
||||
if ip6 != "" {
|
||||
t.Errorf("ip6 = %q, want empty — none was configured", ip6)
|
||||
}
|
||||
}
|
||||
@@ -41,6 +41,24 @@ type Config struct {
|
||||
// Admin UI / health listener (spec §7: localhost-only by default)
|
||||
AdminListen string // ECHOLOT_ADMIN_LISTEN / --admin-listen
|
||||
|
||||
// ReservedAddrs are IPs reserved for measurement: addresses whose listening state must stay
|
||||
// known, so that "nothing answered on port 443" is a fact about the network rather than a
|
||||
// fact about this server's configuration. Enforced by CheckReserved.
|
||||
ReservedAddrs string // ECHOLOT_RESERVED_ADDRS / --reserved-addrs
|
||||
|
||||
// ControlHostname lets the control plane share port 443 with the admin UI.
|
||||
//
|
||||
// They cannot share a certificate: the control plane is trusted by SPKI pin and so uses a
|
||||
// long-lived self-signed certificate, while a browser needs one a CA vouches for. One name on
|
||||
// one port means one certificate, so sharing the port requires two names — this one selects
|
||||
// the pinned certificate and the control-plane routes by SNI, everything else gets the admin
|
||||
// UI. Empty leaves the control plane on its own listener only.
|
||||
//
|
||||
// Why bother: captive portals and corporate firewalls routinely permit only 80 and 443, which
|
||||
// are exactly the networks this tool exists to diagnose. A control plane on 8443 is
|
||||
// unreachable precisely when it matters most.
|
||||
ControlHostname string // ECHOLOT_CONTROL_HOSTNAME / --control-hostname
|
||||
|
||||
// State directory: device store, generated TLS material.
|
||||
StateDir string // ECHOLOT_STATE_DIR / --state-dir
|
||||
|
||||
@@ -52,6 +70,11 @@ type Config struct {
|
||||
// e.g. https://git.example.net/api/v1/repos/owner/repo
|
||||
SelfUpdateAPI string // ECHOLOT_SELF_UPDATE_API / --self-update-api
|
||||
|
||||
// SelfUpdatePubKey overrides the release-signing public key baked into the binary
|
||||
// (selfupdate.DefaultPublicKeyB64) — for operators running their own release pipeline
|
||||
// against their own Gitea. Empty = the built-in project key.
|
||||
SelfUpdatePubKey string // ECHOLOT_SELF_UPDATE_PUBKEY / --self-update-pubkey
|
||||
|
||||
// Uploaded-run storage. The default is "anonymous": any enrolled device may upload,
|
||||
// which is what a self-hosted server wants. Operators of shared servers turn it down.
|
||||
UploadsMode string // ECHOLOT_UPLOADS / --uploads (off|anonymous|account)
|
||||
@@ -60,6 +83,18 @@ type Config struct {
|
||||
UploadMaxRuns int // ECHOLOT_UPLOAD_MAX_RUNS / --upload-max-runs (per device)
|
||||
UploadMinAnon string // ECHOLOT_UPLOAD_MIN_ANONYMIZATION / --upload-min-anonymization
|
||||
|
||||
// Rate limits (spec §2.5), each applied per credential and per source IP. 0 disables a
|
||||
// ceiling. The UDP ceilings are deliberately above the largest legitimate run (a 200 Mbps
|
||||
// upstream throughput test is ~21k pps of 1472-byte packets), so they only ever catch abuse
|
||||
// — a rate limit that clips a real measurement produces a confidently wrong number.
|
||||
RateSessionsPerMin int // ECHOLOT_RATE_SESSIONS_PER_MIN / --rate-sessions-per-min
|
||||
RateActionsPerMin int // ECHOLOT_RATE_ACTIONS_PER_MIN / --rate-actions-per-min
|
||||
RateUDPPps int // ECHOLOT_RATE_UDP_PPS / --rate-udp-pps
|
||||
RateUDPKbps int // ECHOLOT_RATE_UDP_KBPS / --rate-udp-kbps
|
||||
|
||||
// How long canary DNS query logs are kept, in hours (spec §6; privacy default 24).
|
||||
DNSLogRetentionH int // ECHOLOT_DNS_LOG_RETENTION_H / --dns-log-retention-h
|
||||
|
||||
// Client compatibility window. Bounds are SemVer; an empty maximum means unbounded. The
|
||||
// defaults sit at breaking boundaries, so shipping a patch never requires changing them.
|
||||
MinAppVersion string // ECHOLOT_MIN_APP_VERSION / --min-app-version
|
||||
@@ -74,7 +109,13 @@ type Config struct {
|
||||
OIDCIssuer string // ECHOLOT_OIDC_ISSUER / --oidc-issuer
|
||||
OIDCClientID string // ECHOLOT_OIDC_CLIENT_ID / --oidc-client-id (confidential, admin UI)
|
||||
OIDCAppClientID string // ECHOLOT_OIDC_APP_CLIENT_ID / --oidc-app-client-id (public, the phone app)
|
||||
OIDCAdminGroup string // ECHOLOT_OIDC_ADMIN_GROUP / --oidc-admin-group
|
||||
// Issuer for the app's client, when the IdP gives each application its own.
|
||||
//
|
||||
// Authentik derives the issuer from the application slug, so two applications mean two
|
||||
// issuers — and a token's `iss` must match the one that minted it. Empty means both clients
|
||||
// share ECHOLOT_OIDC_ISSUER, which is what IdPs with a single global issuer do.
|
||||
OIDCAppIssuer string // ECHOLOT_OIDC_APP_ISSUER / --oidc-app-issuer
|
||||
OIDCAdminGroup string // ECHOLOT_OIDC_ADMIN_GROUP / --oidc-admin-group
|
||||
|
||||
// Break-glass admin username; the password lives hashed in the state store.
|
||||
AdminUser string // ECHOLOT_ADMIN_USER / --admin-user
|
||||
@@ -156,9 +197,12 @@ func Load(args []string) (*Config, *Actions, error) {
|
||||
fs.StringVar(&c.HTTPEchoListen, "http-echo-listen", envOr("HTTP_ECHO_LISTEN", ""), "optional CLEARTEXT http-echo listen address(es); empty disables (spec §4)")
|
||||
fs.StringVar(&c.MTUProbeTargets, "mtu-probe-targets", envOr("MTU_PROBE_TARGETS", "1.1.1.1,2606:4700:4700::1111"), "egress-MTU self-proof anchors, comma-separated")
|
||||
fs.StringVar(&c.AdminListen, "admin-listen", envOr("ADMIN_LISTEN", "127.0.0.1:8444"), "admin/health listen address (keep localhost)")
|
||||
fs.StringVar(&c.ControlHostname, "control-hostname", envOr("CONTROL_HOSTNAME", ""), "hostname that selects the pinned control-plane certificate when sharing the admin UI's port")
|
||||
fs.StringVar(&c.ReservedAddrs, "reserved-addrs", envOr("RESERVED_ADDRS", ""), "comma-separated IPs reserved for measurement; no listener but the STUN alternate may bind them")
|
||||
fs.StringVar(&c.StateDir, "state-dir", envOr("STATE_DIR", defaultStateDir()), "state directory (device store, generated TLS)")
|
||||
fs.StringVar(&c.Name, "name", envOr("NAME", "echolot"), "server profile name")
|
||||
fs.StringVar(&c.SelfUpdateAPI, "self-update-api", envOr("SELF_UPDATE_API", ""), "Gitea repo API base for self-update; empty disables")
|
||||
fs.StringVar(&c.SelfUpdatePubKey, "self-update-pubkey", envOr("SELF_UPDATE_PUBKEY", ""), "release-signing public key (base64 ed25519) self-update verifies against; empty uses the built-in project key")
|
||||
fs.StringVar(&c.UploadsMode, "uploads", envOr("UPLOADS", "anonymous"), "who may upload measurement runs: off|anonymous|account")
|
||||
fs.Int64Var(&c.UploadMaxBytes, "upload-max-bytes", int64(envInt("UPLOAD_MAX_BYTES", 4<<20)), "largest accepted uploaded run, bytes")
|
||||
fs.IntVar(&c.UploadRetentionDays, "upload-retention-days", envInt("UPLOAD_RETENTION_DAYS", 90), "delete uploaded runs older than this; 0 disables")
|
||||
@@ -167,6 +211,7 @@ func Load(args []string) (*Config, *Actions, error) {
|
||||
fs.StringVar(&c.OIDCIssuer, "oidc-issuer", envOr("OIDC_ISSUER", ""), "OpenID Connect issuer URL; empty disables sign-in")
|
||||
fs.StringVar(&c.OIDCClientID, "oidc-client-id", envOr("OIDC_CLIENT_ID", ""), "confidential OIDC client id for the admin UI")
|
||||
fs.StringVar(&c.OIDCAppClientID, "oidc-app-client-id", envOr("OIDC_APP_CLIENT_ID", ""), "public OIDC client id used by the Android app (PKCE)")
|
||||
fs.StringVar(&c.OIDCAppIssuer, "oidc-app-issuer", envOr("OIDC_APP_ISSUER", ""), "issuer for the app client when the IdP uses per-application issuers; empty = same as --oidc-issuer")
|
||||
fs.StringVar(&c.OIDCAdminGroup, "oidc-admin-group", envOr("OIDC_ADMIN_GROUP", ""), "group claim required for admin access; empty means nobody is an admin via OIDC")
|
||||
fs.StringVar(&c.OIDCClientSecret, "oidc-client-secret", secretOr("OIDC_CLIENT_SECRET", ""), "secret for the confidential admin client; prefer ECHOLOT_OIDC_CLIENT_SECRET_FILE")
|
||||
fs.StringVar(&c.AdminBaseURL, "admin-base-url", envOr("ADMIN_BASE_URL", ""), "public URL of the admin UI, for the OIDC redirect (e.g. https://admin.example.net)")
|
||||
@@ -176,6 +221,11 @@ func Load(args []string) (*Config, *Actions, error) {
|
||||
fs.StringVar(&c.ACMEHTTPListen, "acme-http-listen", envOr("ACME_HTTP_LISTEN", ""), "port-80 listener for ACME HTTP-01 challenges and http->https redirects")
|
||||
fs.StringVar(&c.ACMEWebroot, "acme-webroot", envOr("ACME_WEBROOT", ""), "directory an ACME client writes challenges into (default <state-dir>/acme)")
|
||||
fs.StringVar(&c.PublicControlURL, "public-url", envOr("PUBLIC_URL", ""), "public control-plane URL for enrollment links, e.g. https://probe.example.net:8443")
|
||||
fs.IntVar(&c.RateSessionsPerMin, "rate-sessions-per-min", envInt("RATE_SESSIONS_PER_MIN", 10), "per-credential and per-IP ceiling on session creation (spec §2.5); 0 disables")
|
||||
fs.IntVar(&c.RateActionsPerMin, "rate-actions-per-min", envInt("RATE_ACTIONS_PER_MIN", 60), "per-credential and per-IP ceiling on §5 actions; 0 disables")
|
||||
fs.IntVar(&c.RateUDPPps, "rate-udp-pps", envInt("RATE_UDP_PPS", 25_000), "per-credential and per-IP data-plane packet ceiling, packets/s, silent drop; 0 disables")
|
||||
fs.IntVar(&c.RateUDPKbps, "rate-udp-kbps", envInt("RATE_UDP_KBPS", 250_000), "per-credential and per-IP data-plane byte ceiling, kbit/s, silent drop; 0 disables")
|
||||
fs.IntVar(&c.DNSLogRetentionH, "dns-log-retention-h", envInt("DNS_LOG_RETENTION_H", 24), "hours canary DNS query logs are kept (spec §6 privacy default 24); 0 keeps until the ring overwrites")
|
||||
fs.StringVar(&c.MinAppVersion, "min-app-version", envOr("MIN_APP_VERSION", "0.2.0"), "oldest app version this server will serve (SemVer, inclusive)")
|
||||
fs.StringVar(&c.MaxAppVersion, "max-app-version", envOr("MAX_APP_VERSION", "1.0.0"), "first app version this server will refuse (SemVer, exclusive); empty = unbounded")
|
||||
fs.BoolVar(&c.Docker, "docker", envOr("DOCKER", "") == "1", "force container mode (config from env, no systemd/self-update)")
|
||||
@@ -188,6 +238,7 @@ func Load(args []string) (*Config, *Actions, error) {
|
||||
fs.BoolVar(&a.SetAdminPassword, "set-admin-password", false,
|
||||
"set the break-glass admin password (username as --admin-user, password read from stdin) and exit")
|
||||
fs.StringVar(&c.AdminUser, "admin-user", envOr("ADMIN_USER", "admin"), "username for the break-glass admin")
|
||||
fs.StringVar(&a.MintEnrollToken, "mint-enroll-token", "", "mint a single-use enrollment link (argument is a note for the audit log) and exit")
|
||||
fs.BoolVar(&a.SelfUpdate, "self-update", false, "check for a newer release, replace this binary, and exit")
|
||||
fs.BoolVar(&a.Version, "version", false, "print version and exit")
|
||||
|
||||
@@ -202,13 +253,16 @@ func Load(args []string) (*Config, *Actions, error) {
|
||||
// is a usage error rather than success — otherwise a service manager sees a clean exit and
|
||||
// concludes the server ran and finished.
|
||||
if !a.Serve && !a.InstallSystemd && !a.UninstallSystemd && !a.SelfUpdate &&
|
||||
!a.SetAdminPassword && !a.Version {
|
||||
!a.SetAdminPassword && !a.Version && a.MintEnrollToken == "" {
|
||||
a.Help = true
|
||||
}
|
||||
if a.Serve {
|
||||
if err := c.checkAdminExposure(); err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
if err := c.CheckReserved(c.Listeners()); err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
}
|
||||
if c.Docker && (a.InstallSystemd || a.UninstallSystemd || a.SelfUpdate) {
|
||||
return nil, nil, fmt.Errorf("systemd/self-update actions are native-mode only (container detected; override with ECHOLOT_DOCKER=0 if this is wrong)")
|
||||
@@ -269,6 +323,66 @@ func (c *Config) checkAdminExposure() error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// Listeners enumerates every configured listen spec, for CheckReserved.
|
||||
//
|
||||
// Kept as one list here rather than checked at each call site, so a listener added later is
|
||||
// caught by the compiler when this function is updated — and, more to the point, so that the
|
||||
// person adding one sees the reserved-address rule exists at all.
|
||||
func (c *Config) Listeners() []Listener {
|
||||
return []Listener{
|
||||
// The instrument: these belong on the reserved addresses as much as anywhere.
|
||||
{Name: "control-listen", Spec: c.ControlListen, Measurement: true},
|
||||
{Name: "udp-listen", Spec: c.UDPListen, Measurement: true},
|
||||
{Name: "tcp-listen", Spec: c.TCPListen, Measurement: true},
|
||||
{Name: "dns-listen", Spec: c.DNSListen, Measurement: true},
|
||||
{Name: "stun-listen", Spec: c.StunListen, Measurement: true},
|
||||
// http-echo is deliberately not marked as measurement: it is cleartext HTTP, so on a
|
||||
// reserved address it would be the very listener that ruins the port-80 test.
|
||||
{Name: "http-echo-listen", Spec: c.HTTPEchoListen},
|
||||
// Services. These have no business on an address kept for measuring.
|
||||
{Name: "admin-listen", Spec: c.AdminListen},
|
||||
{Name: "acme-http-listen", Spec: c.ACMEHTTPListen},
|
||||
}
|
||||
}
|
||||
|
||||
// MeasurementAddrs picks out the addresses this server can be measured on, by family, splitting
|
||||
// primaries from the reserved alternates.
|
||||
//
|
||||
// Derived from what is actually bound rather than configured separately: a second list of the
|
||||
// server's own addresses is a second thing to keep in step, and the copy that drifts is the one
|
||||
// clients are told about.
|
||||
func (c *Config) MeasurementAddrs() (ip4, ip6, ip4Alt, ip6Alt string) {
|
||||
reserved := map[string]bool{}
|
||||
for _, ip := range c.ReservedIPs() {
|
||||
reserved[ip.String()] = true
|
||||
}
|
||||
// The UDP data plane binds every address a client may be pointed at, which makes it the
|
||||
// honest source for this.
|
||||
for _, a := range Addrs(c.UDPListen) {
|
||||
host, _, err := net.SplitHostPort(a)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
host = strings.Trim(host, "[]")
|
||||
ip := net.ParseIP(host)
|
||||
if ip == nil {
|
||||
continue
|
||||
}
|
||||
alt := reserved[ip.String()]
|
||||
switch {
|
||||
case ip.To4() != nil && alt && ip4Alt == "":
|
||||
ip4Alt = host
|
||||
case ip.To4() != nil && !alt && ip4 == "":
|
||||
ip4 = host
|
||||
case ip.To4() == nil && alt && ip6Alt == "":
|
||||
ip6Alt = host
|
||||
case ip.To4() == nil && !alt && ip6 == "":
|
||||
ip6 = host
|
||||
}
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
// Addrs splits a comma-separated listen spec into individual addresses.
|
||||
// Explicit per-address binds matter on multi-IP hosts: a wildcard bind
|
||||
// (":8443") would also claim addresses reserved for other purposes (e.g. an
|
||||
@@ -296,6 +410,13 @@ type Actions struct {
|
||||
UninstallSystemd bool
|
||||
SelfUpdate bool
|
||||
SetAdminPassword bool
|
||||
// MintEnrollToken is the note to record against a freshly minted enrollment link.
|
||||
//
|
||||
// A local action rather than an HTTP endpoint: whoever can run this binary against the state
|
||||
// directory already has every privilege the server has, so authenticating them to themselves
|
||||
// would be theatre — and an unauthenticated endpoint on loopback is how the admin API was
|
||||
// briefly reachable from the network by accident.
|
||||
MintEnrollToken string
|
||||
Version bool
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,123 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package config
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"net"
|
||||
"sort"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Listener is one configured listen spec, named for the error message.
|
||||
type Listener struct {
|
||||
Name string // the flag/env this came from, e.g. "control-listen"
|
||||
Spec string // comma-separated listen addresses
|
||||
// Measurement marks a listener that is part of the instrument rather than a service on the
|
||||
// host. Those belong on the reserved addresses — STUN's RFC 5780 alternate, the UDP data
|
||||
// plane, the canary DNS — and reserving an address only to forbid the measurements that need
|
||||
// it would defeat the purpose.
|
||||
Measurement bool
|
||||
}
|
||||
|
||||
// webPorts are the ports whose closed state on a reserved address is itself the measurement.
|
||||
//
|
||||
// A TLS handshake that completes on a port known not to be listening proves interception, with no
|
||||
// competing explanation. That proof is the whole reason for reserving an address, and it survives
|
||||
// exactly as long as nothing binds these two ports there.
|
||||
var webPorts = map[string]bool{"80": true, "443": true}
|
||||
|
||||
// CheckReserved refuses to start when a listener would occupy an address reserved for measurement.
|
||||
//
|
||||
// The reserved addresses are the instrument, not the service. Their diagnostic value comes from
|
||||
// their listening state being *known*: if nothing listens on port 443 there, then a TLS handshake
|
||||
// that completes proves something on the path intercepted it, with no other explanation available.
|
||||
// One stray listener silently converts that proof into an ambiguity.
|
||||
//
|
||||
// This is a hard stop rather than a warning for the same reason as [Config.checkAdminExposure]: the
|
||||
// failure is invisible. A polluted reserved address does not crash, log, or behave oddly — it just
|
||||
// quietly turns a conclusive test into an inconclusive one, and the first symptom is a measurement
|
||||
// that says the network is clean when it is not. Nobody reads a warning for that.
|
||||
//
|
||||
// Wildcard binds are the realistic way this happens. Every listener defaults to ":port", and the
|
||||
// next one added will be copied from an existing default; that binds every address on the host,
|
||||
// reserved ones included, without anyone deciding to.
|
||||
func (c *Config) CheckReserved(listeners []Listener) error {
|
||||
reserved := c.ReservedIPs()
|
||||
if len(reserved) == 0 {
|
||||
return nil
|
||||
}
|
||||
var problems []string
|
||||
for _, l := range listeners {
|
||||
for _, addr := range Addrs(l.Spec) {
|
||||
host, port, err := net.SplitHostPort(addr)
|
||||
if err != nil {
|
||||
// Not host:port — a bare port or something malformed. Leave it to the listener
|
||||
// itself to complain; guessing here would produce a confusing error about the
|
||||
// wrong problem.
|
||||
continue
|
||||
}
|
||||
host = strings.Trim(host, "[]")
|
||||
if host == "" || host == "0.0.0.0" || host == "::" {
|
||||
problems = append(problems, fmt.Sprintf(
|
||||
" --%s=%q binds every address on this host, including the reserved ones",
|
||||
l.Name, addr))
|
||||
continue
|
||||
}
|
||||
ip := net.ParseIP(host)
|
||||
if ip == nil {
|
||||
continue // a hostname; cannot resolve it here without lying about what we checked
|
||||
}
|
||||
for _, r := range reserved {
|
||||
if !ip.Equal(r) {
|
||||
continue
|
||||
}
|
||||
switch {
|
||||
case webPorts[port]:
|
||||
problems = append(problems, fmt.Sprintf(
|
||||
" --%s=%q puts port %s on reserved address %s, which is the one thing "+
|
||||
"that address exists to keep closed", l.Name, addr, port, r))
|
||||
case !l.Measurement:
|
||||
problems = append(problems, fmt.Sprintf(
|
||||
" --%s=%q binds reserved address %s; only measurement listeners belong there",
|
||||
l.Name, addr, r))
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if len(problems) == 0 {
|
||||
return nil
|
||||
}
|
||||
sort.Strings(problems)
|
||||
return fmt.Errorf(
|
||||
"refusing to start: these listeners would occupy addresses reserved for measurement\n%s\n"+
|
||||
"\nReserved: %s\n"+
|
||||
"Those addresses are the instrument. A test can only prove interception on a port that\n"+
|
||||
"is known not to be listening, so anything bound there destroys the conclusion rather\n"+
|
||||
"than merely sharing the address.\n"+
|
||||
" Fix it one of three ways:\n"+
|
||||
" - bind each listener to explicit service addresses instead of a wildcard\n"+
|
||||
" - remove the address from ECHOLOT_RESERVED_ADDRS if it is no longer reserved\n"+
|
||||
" - unset ECHOLOT_RESERVED_ADDRS if this host has no reserved addresses",
|
||||
strings.Join(problems, "\n"), joinIPs(reserved))
|
||||
}
|
||||
|
||||
// ReservedIPs parses the configured reserved addresses, ignoring anything unparseable.
|
||||
func (c *Config) ReservedIPs() []net.IP {
|
||||
var out []net.IP
|
||||
for _, s := range Addrs(c.ReservedAddrs) {
|
||||
if ip := net.ParseIP(strings.Trim(s, "[]")); ip != nil {
|
||||
out = append(out, ip)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func joinIPs(ips []net.IP) string {
|
||||
s := make([]string, 0, len(ips))
|
||||
for _, ip := range ips {
|
||||
s = append(s, ip.String())
|
||||
}
|
||||
return strings.Join(s, ", ")
|
||||
}
|
||||
@@ -0,0 +1,151 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package config
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
const (
|
||||
svc4 = "89.185.109.150"
|
||||
res4 = "89.185.109.151"
|
||||
res6 = "2001:1ad0:c4fe:6767::151"
|
||||
)
|
||||
|
||||
func withReserved(l ...Listener) error {
|
||||
c := &Config{ReservedAddrs: res4 + "," + res6}
|
||||
return c.CheckReserved(l)
|
||||
}
|
||||
|
||||
func TestWildcardBindIsRefused(t *testing.T) {
|
||||
// The realistic failure: every listener defaults to ":port", and the next one added gets
|
||||
// copied from an existing default. Nobody decides to claim the reserved address; it just
|
||||
// happens, and nothing looks wrong afterwards.
|
||||
err := withReserved(Listener{Name: "control-listen", Spec: ":8443"})
|
||||
if err == nil {
|
||||
t.Fatal("a wildcard bind was allowed while addresses were reserved")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "binds every address") {
|
||||
t.Fatalf("the error should say why a wildcard is the problem, got: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWebPortsOnReservedAreRefusedEvenForMeasurement(t *testing.T) {
|
||||
// The strictest rule, and the one carrying the diagnostic value: 80 and 443 must stay closed
|
||||
// on a reserved address whatever wants them, because their closed state *is* the measurement.
|
||||
for _, spec := range []string{res4 + ":443", "[" + res6 + "]:80"} {
|
||||
err := withReserved(Listener{Name: "control-listen", Spec: spec, Measurement: true})
|
||||
if err == nil {
|
||||
t.Fatalf("port 80/443 on a reserved address was allowed: %q", spec)
|
||||
}
|
||||
if !strings.Contains(err.Error(), "keep closed") {
|
||||
t.Errorf("the error should explain what is lost, got: %v", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestServiceOnReservedIsRefused(t *testing.T) {
|
||||
if err := withReserved(Listener{Name: "admin-listen", Spec: res4 + ":8444"}); err == nil {
|
||||
t.Fatal("a service was allowed onto a reserved address")
|
||||
}
|
||||
}
|
||||
|
||||
func TestMeasurementListenersBelongOnReserved(t *testing.T) {
|
||||
// The live fmr config: the UDP data plane, canary DNS and STUN all bind the reserved pair on
|
||||
// purpose. A guard that refused this would be describing a rule nobody wants.
|
||||
err := withReserved(
|
||||
Listener{Name: "udp-listen", Spec: res4 + ":8442,[" + res6 + "]:8442", Measurement: true},
|
||||
Listener{Name: "dns-listen", Spec: res4 + ":53,[" + res6 + "]:53", Measurement: true},
|
||||
Listener{Name: "stun-listen", Spec: res4 + ":3478", Measurement: true},
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("measurement listeners must be allowed on reserved addresses: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestServiceAddressesAreFine(t *testing.T) {
|
||||
err := withReserved(
|
||||
Listener{Name: "control-listen", Spec: svc4 + ":8443,[2001:1ad0:c4fe:6767::150]:8443"},
|
||||
Listener{Name: "admin-listen", Spec: "127.0.0.1:8444"},
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("service addresses should be allowed: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHttpEchoIsNotTreatedAsMeasurement(t *testing.T) {
|
||||
// http-echo is cleartext HTTP. On a reserved address it is precisely the listener that would
|
||||
// ruin the port-80 test, so it does not get the measurement exemption.
|
||||
if err := withReserved(Listener{Name: "http-echo-listen", Spec: res4 + ":8080"}); err == nil {
|
||||
t.Fatal("http-echo was allowed onto a reserved address")
|
||||
}
|
||||
}
|
||||
|
||||
func TestNoReservationMeansNoOpinion(t *testing.T) {
|
||||
// A host with nothing reserved must keep working exactly as before, wildcards included.
|
||||
c := &Config{}
|
||||
if err := c.CheckReserved([]Listener{{Name: "control-listen", Spec: ":8443"}}); err != nil {
|
||||
t.Fatalf("with no reserved addresses this must not interfere: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEveryOffenderIsNamed(t *testing.T) {
|
||||
// Reporting one problem at a time turns a config fix into several restart cycles, and on a
|
||||
// remote host each cycle is a chance to lock yourself out.
|
||||
err := withReserved(
|
||||
Listener{Name: "control-listen", Spec: ":8443"},
|
||||
Listener{Name: "tcp-listen", Spec: res4 + ":8441"},
|
||||
)
|
||||
if err == nil {
|
||||
t.Fatal("expected a refusal")
|
||||
}
|
||||
for _, want := range []string{"control-listen", "tcp-listen"} {
|
||||
if !strings.Contains(err.Error(), want) {
|
||||
t.Errorf("the error should name %s; got: %v", want, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestHostnamesAreNotGuessedAt(t *testing.T) {
|
||||
// Resolving here would check a name against whatever DNS says at startup, which is not
|
||||
// necessarily what it will say later — and a guard that is sometimes right is worse than one
|
||||
// with a stated limit.
|
||||
if err := withReserved(Listener{Name: "control-listen", Spec: "fmr-1.echo-lot.app:8443"}); err != nil {
|
||||
t.Fatalf("a hostname must be left alone, not resolved: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestListenersCoversEverySpec(t *testing.T) {
|
||||
// A listener missing from Listeners() is invisible to the guard, which is the one way this
|
||||
// protection fails silently. Fill every spec with the reserved address: each one that is
|
||||
// actually enumerated produces a complaint naming it.
|
||||
// Every spec on port 443 of the reserved address: the web-port rule applies to measurement
|
||||
// listeners too, so each one that is genuinely enumerated must produce a complaint.
|
||||
c := &Config{
|
||||
ReservedAddrs: res4,
|
||||
ControlListen: res4 + ":443",
|
||||
UDPListen: res4 + ":443",
|
||||
TCPListen: res4 + ":443",
|
||||
DNSListen: res4 + ":443",
|
||||
HTTPEchoListen: res4 + ":443",
|
||||
AdminListen: res4 + ":443",
|
||||
ACMEHTTPListen: res4 + ":443",
|
||||
StunListen: res4 + ":443",
|
||||
}
|
||||
err := c.CheckReserved(c.Listeners())
|
||||
if err == nil {
|
||||
t.Fatal("expected a refusal")
|
||||
}
|
||||
// Every spec is on the reserved address; the web-port rule catches even the measurement ones,
|
||||
// so anything missing from Listeners() is invisible here and that is what this asserts.
|
||||
for _, want := range []string{
|
||||
"control-listen", "udp-listen", "tcp-listen", "dns-listen",
|
||||
"http-echo-listen", "admin-listen", "acme-http-listen", "stun-listen",
|
||||
} {
|
||||
if !strings.Contains(err.Error(), want) {
|
||||
t.Errorf("%s is not enumerated in Listeners(), so the guard cannot see it", want)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -16,6 +16,7 @@ import (
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"log/slog"
|
||||
"net"
|
||||
@@ -29,6 +30,7 @@ import (
|
||||
"echo-lot.app/server/internal/compat"
|
||||
"echo-lot.app/server/internal/dataplane"
|
||||
"echo-lot.app/server/internal/oidc"
|
||||
"echo-lot.app/server/internal/ratelimit"
|
||||
"echo-lot.app/server/internal/runs"
|
||||
"echo-lot.app/server/internal/session"
|
||||
"echo-lot.app/server/internal/store"
|
||||
@@ -57,10 +59,15 @@ type Server struct {
|
||||
// DelayedEcho schedules/sends a DELAYED_ECHO for a session (may be nil).
|
||||
DelayedEcho func(sess *session.Session, actionID string) error
|
||||
// Granted server->client sends (spec §5). Both consume an asymmetric grant.
|
||||
DownTrain func(sess *session.Session, g *session.Grant, count, sizeBytes, intervalUs int) (int, error)
|
||||
// DownTrain's dscp is -1 for "leave the socket's default marking alone".
|
||||
DownTrain func(sess *session.Session, g *session.Grant, count, sizeBytes, intervalUs, dscp int) (int, error)
|
||||
BigSend func(sess *session.Session, g *session.Grant, sizes []int, df bool) ([]dataplane.BigSendResult, error)
|
||||
// OIDC verifies ID tokens when the operator has configured an issuer (may be nil).
|
||||
// OIDC verifies ID tokens presented by the *app* (may be nil).
|
||||
OIDC *oidc.Verifier
|
||||
// AdminOIDC verifies tokens from the admin UI's own client. Separate because an IdP may
|
||||
// give each application its own issuer — Authentik derives it from the application slug —
|
||||
// and a verifier pins exactly one issuer and the clients belonging to it.
|
||||
AdminOIDC *oidc.Verifier
|
||||
// Runs stores uploaded measurement documents (may be nil: uploads unsupported).
|
||||
Runs *runs.Store
|
||||
// FragSend emits one datagram as hand-built IP fragments in a chosen order (may be nil:
|
||||
@@ -76,6 +83,10 @@ type Server struct {
|
||||
CanaryQueries func(sessionPrefix string) any
|
||||
// CanaryZone is surfaced in the profile so the app knows what to query.
|
||||
CanaryZone string
|
||||
// The addresses this server can be measured on. The "_alt" pair is the second address
|
||||
// RFC 5780 behaviour discovery redirects to, and the one reserved from services so that
|
||||
// nothing answering there is itself a measurement.
|
||||
IP4, IP6, IP4Alt, IP6Alt string
|
||||
// ProvenGood reports the server's self-test signal (may be nil). Surfaced
|
||||
// in the profile so a client can trust — or skip — MTU tests: if the
|
||||
// server's own egress isn't full-MTU, client MTU results measure the
|
||||
@@ -89,6 +100,11 @@ type Server struct {
|
||||
// AppRange is the app-version window this server will serve. Zero value means the built-in
|
||||
// default (see DefaultAppRange).
|
||||
AppRange compat.Range
|
||||
|
||||
// Spec §2.5 token buckets, keyed per credential and per source IP inside the limiter.
|
||||
// Nil disables a ceiling (config value 0).
|
||||
RateSessions *ratelimit.Limiter // POST /v1/sessions
|
||||
RateActions *ratelimit.Limiter // POST /v1/sessions/{id}/actions
|
||||
}
|
||||
|
||||
// AppVersionHeader is how a client states its version. A client too old to send it is treated as
|
||||
@@ -98,7 +114,12 @@ const AppVersionHeader = "X-Echolot-App-Version"
|
||||
|
||||
// ProtocolVersion is the wire contract (probe-protocol.md) this build implements. It is what the
|
||||
// version window is really about; the release version is only a proxy for it.
|
||||
const ProtocolVersion = "1.0.0"
|
||||
//
|
||||
// 1.0.1: upstream trains (§3.2 types 0x03/0x04/0x05) and the action-id bytes at payload[8:16]
|
||||
// of granted packets. Patch, not minor: both are additive — a client that never sends
|
||||
// TRAIN_REPORT_REQ and never reads granted payloads (today's client reads only header fields)
|
||||
// sees no difference, so the fleet must not be split over it (§8.1).
|
||||
const ProtocolVersion = "1.0.1"
|
||||
|
||||
// SchemaVersion is the measurement-document format this server can store.
|
||||
const SchemaVersion = "1.0.0"
|
||||
@@ -156,10 +177,12 @@ func (s *Server) Handler() http.Handler {
|
||||
|
||||
gate := s.requireCompatibleApp
|
||||
mux.HandleFunc("POST /v1/enroll", gate(s.enroll))
|
||||
mux.HandleFunc("POST /v1/sessions", gate(s.newSession))
|
||||
// §2.5 buckets sit on the two endpoints that make the server DO things — create state,
|
||||
// send traffic. GET /v1/profile stays ungated on every axis (see above).
|
||||
mux.HandleFunc("POST /v1/sessions", gate(s.rateLimited(s.RateSessions, s.newSession)))
|
||||
mux.HandleFunc("DELETE /v1/sessions/{id}", gate(s.deleteSession))
|
||||
mux.HandleFunc("GET /v1/sessions/{id}/observations", gate(s.observations))
|
||||
mux.HandleFunc("POST /v1/sessions/{id}/actions", gate(s.actions))
|
||||
mux.HandleFunc("POST /v1/sessions/{id}/actions", gate(s.rateLimited(s.RateActions, s.actions)))
|
||||
mux.HandleFunc("POST /v1/echo", gate(s.httpEcho))
|
||||
mux.HandleFunc("GET /v1/tls-reference", gate(s.tlsReference))
|
||||
mux.HandleFunc("POST /v1/runs", gate(s.uploadRun))
|
||||
@@ -169,7 +192,6 @@ func (s *Server) Handler() http.Handler {
|
||||
mux.HandleFunc("POST /v1/account/link", gate(s.linkAccount))
|
||||
mux.HandleFunc("DELETE /v1/account/link", gate(s.unlinkAccount))
|
||||
mux.HandleFunc("GET /v1/account", gate(s.accountStatus))
|
||||
// TODO(spec §5): frag_send, throughput (both build on the same grant machinery)
|
||||
return mux
|
||||
}
|
||||
|
||||
@@ -191,6 +213,37 @@ func selftestSignal(f func() (bool, bool)) map[string]any {
|
||||
return map[string]any{"mtu_ok": mtuOK, "sysctl_ok": sysctlOK}
|
||||
}
|
||||
|
||||
// rateLimited enforces one §2.5 bucket policy on an endpoint: a token per credential AND one per
|
||||
// source IP. Two keys because each closes the other's hole — keyed only by credential, one
|
||||
// address cycles through credentials; keyed only by address, one credential rides many
|
||||
// addresses. The refusal is 429 with Retry-After, which is the whole point of a token bucket
|
||||
// over a hard drop here: a well-behaved client is told when to come back.
|
||||
func (s *Server) rateLimited(l *ratelimit.Limiter, next http.HandlerFunc) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
okCred, waitCred := l.Allow("cred:" + bearer(r))
|
||||
okIP, waitIP := l.Allow("ip:" + remoteIP(r))
|
||||
if !okCred || !okIP {
|
||||
wait := max(waitCred, waitIP)
|
||||
secs := int(wait/time.Second) + 1 // Retry-After is whole seconds, rounded up
|
||||
w.Header().Set("Retry-After", strconv.Itoa(secs))
|
||||
writeJSON(w, http.StatusTooManyRequests, map[string]any{
|
||||
"error": "rate limited", "retry_after_s": secs,
|
||||
})
|
||||
return
|
||||
}
|
||||
next(w, r)
|
||||
}
|
||||
}
|
||||
|
||||
// remoteIP is the request's source address without the port, for rate-limit keys.
|
||||
func remoteIP(r *http.Request) string {
|
||||
host, _, err := net.SplitHostPort(r.RemoteAddr)
|
||||
if err != nil {
|
||||
return r.RemoteAddr
|
||||
}
|
||||
return strings.Trim(host, "[]")
|
||||
}
|
||||
|
||||
// sessionAuth resolves {id} and requires the bearer to be the owning device.
|
||||
func (s *Server) sessionAuth(w http.ResponseWriter, r *http.Request) *session.Session {
|
||||
dev := s.Store.DeviceByCredential(bearer(r))
|
||||
@@ -227,8 +280,14 @@ func (s *Server) observations(w http.ResponseWriter, r *http.Request) {
|
||||
dnsCanary = s.CanaryQueries(sess.ID[:16]) // the session's wire prefix
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{
|
||||
"udp": map[string]any{"packets_seen": packetsSeen, "packets": udp},
|
||||
"tcp": tcp,
|
||||
"udp": map[string]any{
|
||||
"packets_seen": packetsSeen,
|
||||
"packets": udp,
|
||||
// The per-train received view (spec §6 "trains"), columnar like the wire report —
|
||||
// the flat packet list above stays for clients that predate trains.
|
||||
"trains": trainsJSON(sess.Trains()),
|
||||
},
|
||||
"tcp": tcp,
|
||||
"connect_back": cb,
|
||||
// The sender's own count, which is what makes the receiver's count mean something.
|
||||
"throughput": sess.ThroughputReports(),
|
||||
@@ -256,6 +315,7 @@ func (s *Server) actions(w http.ResponseWriter, r *http.Request) {
|
||||
SizeBytes int `json:"size_bytes"`
|
||||
IntervalUs int `json:"interval_us"`
|
||||
SizesBytes []int `json:"sizes_bytes"`
|
||||
DSCP *int `json:"dscp"`
|
||||
DF *bool `json:"df"`
|
||||
Mode string `json:"mode"`
|
||||
FragBytes int `json:"frag_bytes"`
|
||||
@@ -318,6 +378,11 @@ func (s *Server) actions(w http.ResponseWriter, r *http.Request) {
|
||||
writeJSON(w, http.StatusNotImplemented, map[string]string{"error": "downtrain not wired"})
|
||||
return
|
||||
}
|
||||
dscp, err := dscpArg(req.DSCP)
|
||||
if err != nil {
|
||||
writeJSON(w, http.StatusBadRequest, map[string]string{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
// A downstream train sends far more than it receives, so it needs a grant (§3.4).
|
||||
count := clamp(req.Count, 1, 5000)
|
||||
size := clamp(req.SizeBytes, dataMinPacket, 1500)
|
||||
@@ -328,13 +393,20 @@ func (s *Server) actions(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
go func() {
|
||||
sent, err := s.DownTrain(sess, g, count, size, interval)
|
||||
sent, err := s.DownTrain(sess, g, count, size, interval, dscp)
|
||||
slog.Info("downtrain finished", "action", actionID, "sent", sent, "bytes", g.Sent(), "err", err)
|
||||
}()
|
||||
writeJSON(w, http.StatusAccepted, map[string]any{
|
||||
resp := map[string]any{
|
||||
"action_id": actionID, "count": count, "size_bytes": size, "interval_us": interval,
|
||||
"grant": map[string]any{"max_bytes": g.MaxBytes, "max_kbps": g.MaxKbps},
|
||||
})
|
||||
}
|
||||
if dscp >= 0 {
|
||||
resp["dscp"] = dscp
|
||||
// Told up front, not discovered: a client measuring DSCP survival on a burst the
|
||||
// server could not mark would conclude the network stripped it.
|
||||
resp["dscp_applied"] = dataplane.TOSSupported
|
||||
}
|
||||
writeJSON(w, http.StatusAccepted, resp)
|
||||
|
||||
case "big_send":
|
||||
if s.BigSend == nil {
|
||||
@@ -517,8 +589,54 @@ func (s *Server) maxDFPayload(sess *session.Session) int {
|
||||
return mtu - overhead
|
||||
}
|
||||
|
||||
// dataMinPacket is the smallest datagram that still carries a header + a little payload.
|
||||
const dataMinPacket = 40
|
||||
// dscpArg validates the optional downtrain dscp parameter (spec §5). Absent means -1: leave the
|
||||
// socket's default marking alone, which is different from asking for DSCP 0 (explicitly
|
||||
// best-effort). Out-of-range values are refused rather than clamped — a clamped 46→63 would mark
|
||||
// the burst with a class the client never asked for and silently change what the test measures.
|
||||
func dscpArg(v *int) (int, error) {
|
||||
if v == nil {
|
||||
return -1, nil
|
||||
}
|
||||
if *v < 0 || *v > 63 {
|
||||
return 0, fmt.Errorf("dscp %d is out of range: the field is 6 bits (0..63)", *v)
|
||||
}
|
||||
return *v, nil
|
||||
}
|
||||
|
||||
// trainsJSON renders the per-train received view columnar — one array per field, matching the
|
||||
// wire report and the schema's train-evidence shape — with []int for the byte-wide columns
|
||||
// because encoding/json would base64 a []uint8.
|
||||
func trainsJSON(trains []session.Train) []map[string]any {
|
||||
out := make([]map[string]any, 0, len(trains))
|
||||
for _, t := range trains {
|
||||
n := len(t.Entries)
|
||||
seq := make([]uint32, n)
|
||||
trx := make([]int64, n)
|
||||
size := make([]int, n)
|
||||
ttl := make([]int, n)
|
||||
dscp := make([]int, n)
|
||||
ecn := make([]int, n)
|
||||
for i, e := range t.Entries {
|
||||
seq[i], trx[i], size[i] = e.Seq, e.TRxNs, int(e.Size)
|
||||
ttl[i], dscp[i], ecn[i] = int(e.TTL), int(e.DSCP), int(e.ECN)
|
||||
}
|
||||
out = append(out, map[string]any{
|
||||
"train_id": t.ID,
|
||||
// The loss denominator: every packet counted, whether or not its row was kept.
|
||||
"packets_received": t.Received,
|
||||
"truncated": t.Truncated,
|
||||
"seq": seq, "t_rx_ns": trx, "size": size,
|
||||
"ttl": ttl, "dscp": dscp, "ecn": ecn,
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// dataMinPacket is the smallest granted datagram: header + 16 payload bytes, because [8:16] of
|
||||
// every granted payload carries the action id. It must match what the senders raise short sizes
|
||||
// to, or a minimum-size train's grant is budgeted for fewer bytes than actually leave and the
|
||||
// train is cut short by its own arithmetic.
|
||||
const dataMinPacket = dataplane.HeaderSize + 16
|
||||
|
||||
var noDataPlaneYet = map[string]string{
|
||||
"error": "no data-plane traffic seen yet — send an ECHO first so the destination is verified",
|
||||
@@ -610,13 +728,7 @@ func (s *Server) profile(w http.ResponseWriter, r *http.Request) {
|
||||
// modified builds and gives clients provenance for the measurement.
|
||||
"source_url": "", // TODO: stamp from build metadata
|
||||
"capabilities": s.Capabilities,
|
||||
"targets": []map[string]any{{
|
||||
"id": s.Name,
|
||||
"ip4": host, // TODO: explicit configured addresses, v6, second STUN addr
|
||||
"udp_port": s.UDPPort,
|
||||
"tcp_port": s.TCPPort,
|
||||
"stun_port": s.StunPort,
|
||||
}},
|
||||
"targets": []map[string]any{s.target(host)},
|
||||
"pins": []string{"pin-sha256:" + s.PinB64},
|
||||
"next_pins": []string{},
|
||||
"canary_zone": s.CanaryZone,
|
||||
@@ -751,11 +863,19 @@ func (s *Server) listRuns(w http.ResponseWriter, r *http.Request) {
|
||||
writeJSON(w, http.StatusUnauthorized, map[string]string{"error": "unknown credential"})
|
||||
return
|
||||
}
|
||||
list := s.Runs.List(dev.ID)
|
||||
list := s.Runs.ListFor(s.visibleDevices(dev))
|
||||
if list == nil {
|
||||
list = []runs.Meta{}
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{"runs": list})
|
||||
writeJSON(w, http.StatusOK, map[string]any{
|
||||
"runs": list,
|
||||
// Says whose history this is, so a client can show "3 devices" rather than leaving the
|
||||
// user to wonder why runs from another phone appeared.
|
||||
"scope": map[string]any{
|
||||
"account_id": dev.AccountID,
|
||||
"devices": len(s.visibleDevices(dev)),
|
||||
},
|
||||
})
|
||||
}
|
||||
|
||||
func (s *Server) getRun(w http.ResponseWriter, r *http.Request) {
|
||||
@@ -764,9 +884,14 @@ func (s *Server) getRun(w http.ResponseWriter, r *http.Request) {
|
||||
writeJSON(w, http.StatusUnauthorized, map[string]string{"error": "unknown credential"})
|
||||
return
|
||||
}
|
||||
// Scoped to the calling device's own directory: one device cannot read another's runs by
|
||||
// guessing a run id.
|
||||
b, err := s.Runs.Get(dev.ID, r.PathValue("id"))
|
||||
// Resolved against the caller's own devices only, so a run id from another account is not
|
||||
// found rather than being fetched from wherever it happens to live.
|
||||
owner, ok := s.Runs.OwnerOf(s.visibleDevices(dev), r.PathValue("id"))
|
||||
if !ok {
|
||||
writeJSON(w, http.StatusNotFound, map[string]string{"error": "no such run"})
|
||||
return
|
||||
}
|
||||
b, err := s.Runs.Get(owner, r.PathValue("id"))
|
||||
if err != nil {
|
||||
writeJSON(w, http.StatusNotFound, map[string]string{"error": "no such run"})
|
||||
return
|
||||
@@ -781,7 +906,12 @@ func (s *Server) deleteRun(w http.ResponseWriter, r *http.Request) {
|
||||
writeJSON(w, http.StatusUnauthorized, map[string]string{"error": "unknown credential"})
|
||||
return
|
||||
}
|
||||
if err := s.Runs.Delete(dev.ID, r.PathValue("id")); err != nil {
|
||||
owner, ok := s.Runs.OwnerOf(s.visibleDevices(dev), r.PathValue("id"))
|
||||
if !ok {
|
||||
w.WriteHeader(http.StatusNoContent) // delete is idempotent; absent is the desired state
|
||||
return
|
||||
}
|
||||
if err := s.Runs.Delete(owner, r.PathValue("id")); err != nil {
|
||||
writeJSON(w, http.StatusInternalServerError, map[string]string{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
@@ -803,13 +933,47 @@ func maxOrEmpty(r compat.Range) string {
|
||||
// three parts at once — its own URL, its own SPKI pin, and the token. An operator copying a pin
|
||||
// by hand is the step that goes wrong, and a pin wrong by one character does not fail loudly.
|
||||
func (s *Server) EnrollmentLink(token string) string {
|
||||
u := s.PublicControlURL
|
||||
return EnrollmentURI(s.PublicControlURL, s.PinB64, token)
|
||||
}
|
||||
|
||||
// EnrollmentURI is the same assembly without a running server, for the mint-a-link CLI action.
|
||||
// Shared rather than reimplemented: two copies of this encoding would eventually disagree, and
|
||||
// the failure mode is a pin that looks right and produces an inscrutable TLS error.
|
||||
func EnrollmentURI(publicURL, pinB64, token string) string {
|
||||
return "echolot://enroll?v=1" +
|
||||
"&u=" + url.QueryEscape(strings.TrimRight(u, "/")) +
|
||||
"&p=" + url.QueryEscape("pin-sha256:"+s.PinB64) +
|
||||
"&u=" + url.QueryEscape(strings.TrimRight(publicURL, "/")) +
|
||||
"&p=" + url.QueryEscape("pin-sha256:"+pinB64) +
|
||||
"&t=" + url.QueryEscape(token)
|
||||
}
|
||||
|
||||
// target describes where this server can be measured, so a client can say which address a result
|
||||
// came from instead of "the server".
|
||||
//
|
||||
// The alternates matter as much as the primaries: RFC 5780 behaviour discovery needs a second
|
||||
// address to redirect to, and an operator reading a report needs to know which of their addresses
|
||||
// a finding refers to. [fallback] is used only when nothing was configured explicitly, so a server
|
||||
// that has not been told its own addresses still answers with something usable.
|
||||
func (s *Server) target(fallback string) map[string]any {
|
||||
t := map[string]any{
|
||||
"id": s.Name,
|
||||
"udp_port": s.UDPPort,
|
||||
"tcp_port": s.TCPPort,
|
||||
"stun_port": s.StunPort,
|
||||
}
|
||||
ip4 := s.IP4
|
||||
if ip4 == "" {
|
||||
ip4 = fallback
|
||||
}
|
||||
for k, v := range map[string]string{
|
||||
"ip4": ip4, "ip6": s.IP6, "ip4_alt": s.IP4Alt, "ip6_alt": s.IP6Alt,
|
||||
} {
|
||||
if v != "" {
|
||||
t[k] = v
|
||||
}
|
||||
}
|
||||
return t
|
||||
}
|
||||
|
||||
// upstreamJSON renders the upstream tally with the derived figures already computed, so every
|
||||
// consumer does not have to repeat (and risk fumbling) the same arithmetic.
|
||||
func upstreamJSON(sess *session.Session) map[string]any {
|
||||
@@ -924,3 +1088,17 @@ func (s *Server) accountStatus(w http.ResponseWriter, r *http.Request) {
|
||||
"device_id": dev.ID,
|
||||
})
|
||||
}
|
||||
|
||||
// visibleDevices is the set of devices whose runs the caller may read.
|
||||
//
|
||||
// Signed in: every device on the same account, which is what an account is for. Not signed in:
|
||||
// only itself — anonymous devices are not a group, and treating the absent account as a shared
|
||||
// one would let any of them read all the others.
|
||||
func (s *Server) visibleDevices(dev *store.Device) []string {
|
||||
if dev.LinkedToAccount() {
|
||||
if ids := s.Store.DeviceIDsForAccount(dev.AccountID); len(ids) > 0 {
|
||||
return ids
|
||||
}
|
||||
}
|
||||
return []string{dev.ID}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package control
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestDscpArg(t *testing.T) {
|
||||
ptr := func(v int) *int { return &v }
|
||||
for _, tc := range []struct {
|
||||
in *int
|
||||
want int
|
||||
wantErr bool
|
||||
}{
|
||||
{nil, -1, false}, // absent: leave the socket alone
|
||||
{ptr(0), 0, false}, // explicit best-effort is not the same as absent
|
||||
{ptr(46), 46, false}, // EF, the value people actually test with
|
||||
{ptr(63), 63, false},
|
||||
{ptr(64), 0, true}, // one past the 6-bit field
|
||||
{ptr(-1), 0, true},
|
||||
} {
|
||||
got, err := dscpArg(tc.in)
|
||||
if (err != nil) != tc.wantErr || got != tc.want {
|
||||
t.Errorf("dscpArg(%v) = %d, err=%v; want %d, wantErr=%v", tc.in, got, err, tc.want, tc.wantErr)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,70 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
//go:build linux
|
||||
|
||||
package dataplane
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"net"
|
||||
"syscall"
|
||||
)
|
||||
|
||||
// Per-packet TTL and TOS/traffic-class arrive as control messages, and only if asked for at
|
||||
// socket setup. These fill the spec §3.3 observation-block fields that were shipped as the 0xFF
|
||||
// sentinel until now — received TTL is path-length evidence, the TOS byte is DSCP/ECN survival.
|
||||
|
||||
// enableRecvMeta asks the kernel to attach the cmsgs to every received datagram. Both the v4 and
|
||||
// the v6 option sets are attempted on every socket: a dual-stack socket delivers v4-mapped
|
||||
// traffic through the v6 fd, and the kernel refuses whichever set does not apply. Errors are
|
||||
// dropped on purpose — a socket that cannot deliver metadata still serves probes, and the
|
||||
// sentinel already says "not observed" for it.
|
||||
func enableRecvMeta(conn *net.UDPConn) {
|
||||
raw, err := conn.SyscallConn()
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
_ = raw.Control(func(fd uintptr) {
|
||||
_ = syscall.SetsockoptInt(int(fd), syscall.IPPROTO_IP, syscall.IP_RECVTTL, 1)
|
||||
_ = syscall.SetsockoptInt(int(fd), syscall.IPPROTO_IP, syscall.IP_RECVTOS, 1)
|
||||
_ = syscall.SetsockoptInt(int(fd), syscall.IPPROTO_IPV6, syscall.IPV6_RECVHOPLIMIT, 1)
|
||||
_ = syscall.SetsockoptInt(int(fd), syscall.IPPROTO_IPV6, syscall.IPV6_RECVTCLASS, 1)
|
||||
})
|
||||
}
|
||||
|
||||
// parseMeta extracts TTL and the TOS byte from one datagram's control messages. Anything absent
|
||||
// or unparseable keeps the sentinel — reported as unobserved, never guessed.
|
||||
func parseMeta(oob []byte) pktMeta {
|
||||
m := pktMeta{TTL: metaUnavailable, TOS: metaUnavailable}
|
||||
if len(oob) == 0 {
|
||||
return m
|
||||
}
|
||||
cmsgs, err := syscall.ParseSocketControlMessage(oob)
|
||||
if err != nil {
|
||||
return m
|
||||
}
|
||||
for _, c := range cmsgs {
|
||||
switch {
|
||||
case c.Header.Level == syscall.IPPROTO_IP && c.Header.Type == syscall.IP_TTL,
|
||||
c.Header.Level == syscall.IPPROTO_IPV6 && c.Header.Type == syscall.IPV6_HOPLIMIT:
|
||||
m.TTL = cmsgValue(c.Data)
|
||||
case c.Header.Level == syscall.IPPROTO_IP && c.Header.Type == syscall.IP_TOS,
|
||||
c.Header.Level == syscall.IPPROTO_IPV6 && c.Header.Type == syscall.IPV6_TCLASS:
|
||||
m.TOS = cmsgValue(c.Data)
|
||||
}
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// cmsgValue reads a cmsg the kernel encodes either as a native-endian int (IP_TTL,
|
||||
// IPV6_HOPLIMIT, IPV6_TCLASS) or as a single byte (IP_TOS). Both fit a byte by definition.
|
||||
func cmsgValue(data []byte) uint8 {
|
||||
switch {
|
||||
case len(data) >= 4:
|
||||
return uint8(binary.NativeEndian.Uint32(data))
|
||||
case len(data) >= 1:
|
||||
return data[0]
|
||||
}
|
||||
return metaUnavailable
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
//go:build !linux
|
||||
|
||||
package dataplane
|
||||
|
||||
import "net"
|
||||
|
||||
// Per-packet TTL/TOS needs Linux's IP_RECVTTL-family cmsgs. Elsewhere the observation block and
|
||||
// train buffers keep the spec §3.3 sentinel (0xFF = not observed) — absent is honest, a guess
|
||||
// is not. Deployment targets are Linux; this build exists so the Windows dev loop compiles.
|
||||
func enableRecvMeta(_ *net.UDPConn) {}
|
||||
|
||||
func parseMeta(_ []byte) pktMeta { return pktMeta{TTL: metaUnavailable, TOS: metaUnavailable} }
|
||||
@@ -104,8 +104,8 @@ func (s *Server) FragSend(
|
||||
return res, fmt.Errorf("session has no recorded local address")
|
||||
}
|
||||
|
||||
if sizeBytes < HeaderSize+8 {
|
||||
sizeBytes = HeaderSize + 8
|
||||
if sizeBytes < HeaderSize+32 {
|
||||
sizeBytes = HeaderSize + 32
|
||||
}
|
||||
if sizeBytes > 8000 {
|
||||
sizeBytes = 8000
|
||||
@@ -117,7 +117,10 @@ func (s *Server) FragSend(
|
||||
// The ELT1 packet, signed exactly as any other, then wrapped in UDP.
|
||||
payload := make([]byte, sizeBytes-HeaderSize)
|
||||
binary.BigEndian.PutUint32(payload[0:4], uint32(sizeBytes))
|
||||
copy(payload[4:], mode)
|
||||
// [8:16]: the action id, as on every granted packet (spec §5 correlation); the mode string
|
||||
// sits past the reserved slot.
|
||||
putActionID(payload, g.ActionID)
|
||||
copy(payload[16:], mode)
|
||||
elt := s.buildPacket(sess, TypeFragData, 0, payload)
|
||||
|
||||
udp := buildUDP(local, target, elt)
|
||||
|
||||
@@ -21,7 +21,11 @@ import (
|
||||
// client measures downstream loss, reordering and jitter from what arrives — the direction an
|
||||
// upstream-only train cannot see. Returns how many packets actually went out (the grant may cut
|
||||
// it short, which is itself reportable).
|
||||
func (s *Server) DownTrain(sess *session.Session, g *session.Grant, count, sizeBytes, intervalUs int) (int, error) {
|
||||
//
|
||||
// dscp ≥ 0 marks the burst (spec §5 downtrain `dscp`): downstream DSCP survival is the half the
|
||||
// client cannot produce itself. Best-effort off Linux — see withTOS/TOSSupported; the action
|
||||
// response has already told the client whether the marking was applied.
|
||||
func (s *Server) DownTrain(sess *session.Session, g *session.Grant, count, sizeBytes, intervalUs, dscp int) (int, error) {
|
||||
target := sess.DataSource()
|
||||
if !target.IsValid() {
|
||||
return 0, fmt.Errorf("no observed data-plane source")
|
||||
@@ -30,26 +34,41 @@ func (s *Server) DownTrain(sess *session.Session, g *session.Grant, count, sizeB
|
||||
if conn == nil {
|
||||
return 0, fmt.Errorf("no data-plane socket matches target family")
|
||||
}
|
||||
if sizeBytes < HeaderSize+8 {
|
||||
sizeBytes = HeaderSize + 8
|
||||
if sizeBytes < HeaderSize+16 {
|
||||
sizeBytes = HeaderSize + 16
|
||||
}
|
||||
payload := make([]byte, sizeBytes-HeaderSize)
|
||||
// Payload [8:16] carries the action id on every granted packet, so arriving traffic can be
|
||||
// attributed to the action that caused it (spec §5/§9: test.params.action_id).
|
||||
putActionID(payload, g.ActionID)
|
||||
sent := 0
|
||||
for i := 0; i < count; i++ {
|
||||
if !g.Allow(sizeBytes) {
|
||||
break // budget or rate exhausted — stop, do not sleep it off
|
||||
}
|
||||
// Sequence + send timestamp in the payload head so the client can order and time them
|
||||
// even when packets arrive out of order.
|
||||
binary.BigEndian.PutUint32(payload[0:4], uint32(i))
|
||||
binary.BigEndian.PutUint32(payload[4:8], uint32(time.Since(s.start).Microseconds()))
|
||||
s.send(conn, target, sess, TypeDownTrainData, uint32(i), payload)
|
||||
sent++
|
||||
if intervalUs > 0 && i < count-1 {
|
||||
time.Sleep(time.Duration(intervalUs) * time.Microsecond)
|
||||
burst := func() error {
|
||||
for i := 0; i < count; i++ {
|
||||
if !g.Allow(sizeBytes) {
|
||||
break // budget or rate exhausted — stop, do not sleep it off
|
||||
}
|
||||
// Sequence + send timestamp in the payload head so the client can order and time them
|
||||
// even when packets arrive out of order.
|
||||
binary.BigEndian.PutUint32(payload[0:4], uint32(i))
|
||||
binary.BigEndian.PutUint32(payload[4:8], uint32(time.Since(s.start).Microseconds()))
|
||||
s.send(conn, target, sess, TypeDownTrainData, uint32(i), payload)
|
||||
sent++
|
||||
if intervalUs > 0 && i < count-1 {
|
||||
time.Sleep(time.Duration(intervalUs) * time.Microsecond)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
return sent, nil
|
||||
if dscp >= 0 && TOSSupported {
|
||||
// Same shared-socket borrow as the DF window: dfMu keeps a concurrent burst from riding
|
||||
// along with — or clearing — this marking.
|
||||
s.dfMu.Lock()
|
||||
defer s.dfMu.Unlock()
|
||||
err := withTOS(conn, dscp, burst) // sent must be read after the burst ran, not before
|
||||
return sent, err
|
||||
}
|
||||
err := burst()
|
||||
return sent, err
|
||||
}
|
||||
|
||||
// BigSendResult records what happened to one requested size. `Sent` false with an EMSGSIZE-ish
|
||||
@@ -83,8 +102,8 @@ func (s *Server) BigSend(sess *session.Session, g *session.Grant, sizes []int, d
|
||||
results := make([]BigSendResult, 0, len(sizes))
|
||||
burst := func() error {
|
||||
for i, size := range sizes {
|
||||
if size < HeaderSize+8 {
|
||||
size = HeaderSize + 8
|
||||
if size < HeaderSize+16 {
|
||||
size = HeaderSize + 16
|
||||
}
|
||||
if size > 9000 { // jumbo ceiling; beyond this the kernel will refuse anyway
|
||||
size = 9000
|
||||
@@ -96,6 +115,8 @@ func (s *Server) BigSend(sess *session.Session, g *session.Grant, sizes []int, d
|
||||
// Echo the intended size into the payload so a truncated/fragmented arrival is
|
||||
// still attributable to the size we meant to send.
|
||||
binary.BigEndian.PutUint32(payload[0:4], uint32(size))
|
||||
// [8:16]: the action id, as on every granted packet (spec §5 correlation).
|
||||
putActionID(payload, g.ActionID)
|
||||
err := s.sendErr(conn, target, sess, TypeBigSend, uint32(i), payload)
|
||||
results = append(results, BigSendResult{
|
||||
SizeBytes: size, Seq: i, Sent: err == nil, Err: errString(err),
|
||||
|
||||
@@ -120,7 +120,7 @@ func (s *Server) DownThroughput(
|
||||
|
||||
// Same plan the grant was sized from, so the two cannot disagree.
|
||||
durationMs, kbps = ThroughputPlan(durationMs, kbps)
|
||||
if sizeBytes < HeaderSize+16 {
|
||||
if sizeBytes < HeaderSize+24 {
|
||||
sizeBytes = 1200 // a size that survives every common path unfragmented
|
||||
}
|
||||
if sizeBytes > 1472 {
|
||||
@@ -134,6 +134,10 @@ func (s *Server) DownThroughput(
|
||||
}
|
||||
|
||||
payload := make([]byte, sizeBytes-HeaderSize)
|
||||
// [8:16]: the action id, as on every granted packet (spec §5 correlation). The send
|
||||
// timestamp lives past it at [16:24]; the client reads only header fields today, so
|
||||
// reserving the slot costs nothing and keeps one layout rule across granted types.
|
||||
putActionID(payload, g.ActionID)
|
||||
deadline := time.Now().Add(time.Duration(durationMs) * time.Millisecond)
|
||||
start := time.Now()
|
||||
next := start
|
||||
@@ -157,7 +161,7 @@ func (s *Server) DownThroughput(
|
||||
// Reaching here means the run is progressing normally; the clock will end it.
|
||||
res.LimitedBy = "duration"
|
||||
binary.BigEndian.PutUint32(payload[0:4], seq)
|
||||
binary.BigEndian.PutUint64(payload[4:12], uint64(time.Since(s.start).Nanoseconds()))
|
||||
binary.BigEndian.PutUint64(payload[16:24], uint64(time.Since(s.start).Nanoseconds()))
|
||||
if err := s.sendErr(conn, target, sess, TypeThroughputData, seq, payload); err != nil {
|
||||
// A send error mid-run is a local condition (buffer full, route gone). Stop and
|
||||
// report what got out rather than pretending the rest was lost on the path.
|
||||
|
||||
@@ -0,0 +1,55 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
//go:build linux
|
||||
|
||||
package dataplane
|
||||
|
||||
import (
|
||||
"net"
|
||||
"syscall"
|
||||
)
|
||||
|
||||
// withTOS runs fn with the socket's TOS/traffic class set to dscp<<2 (ECN bits left zero — the
|
||||
// test is about DSCP survival, and claiming ECN capability we do not use would pollute it), then
|
||||
// restores what was there before.
|
||||
//
|
||||
// Same borrow discipline as withDF: the socket is shared by every session on that family, so the
|
||||
// caller must hold Server.dfMu for the whole window or a concurrent burst rides along with — or
|
||||
// clears — someone else's marking.
|
||||
func withTOS(conn *net.UDPConn, dscp int, fn func() error) error {
|
||||
raw, err := conn.SyscallConn()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
v4 := conn.LocalAddr().(*net.UDPAddr).IP.To4() != nil
|
||||
level, opt := syscall.IPPROTO_IPV6, syscall.IPV6_TCLASS
|
||||
if v4 {
|
||||
level, opt = syscall.IPPROTO_IP, syscall.IP_TOS
|
||||
}
|
||||
|
||||
var setErr error
|
||||
prev := 0
|
||||
if err := raw.Control(func(fd uintptr) {
|
||||
if p, e := syscall.GetsockoptInt(int(fd), level, opt); e == nil {
|
||||
prev = p
|
||||
}
|
||||
setErr = syscall.SetsockoptInt(int(fd), level, opt, dscp<<2)
|
||||
}); err != nil {
|
||||
return err
|
||||
}
|
||||
if setErr != nil {
|
||||
return setErr
|
||||
}
|
||||
defer func() {
|
||||
_ = raw.Control(func(fd uintptr) {
|
||||
_ = syscall.SetsockoptInt(int(fd), level, opt, prev)
|
||||
})
|
||||
}()
|
||||
return fn()
|
||||
}
|
||||
|
||||
// TOSSupported reports whether withTOS can actually mark packets here. Exported so the control
|
||||
// plane can tell the client up front that its dscp request will not be honored, instead of the
|
||||
// client measuring an unmarked burst and concluding the network stripped the marking.
|
||||
const TOSSupported = true
|
||||
@@ -0,0 +1,16 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
//go:build !linux
|
||||
|
||||
package dataplane
|
||||
|
||||
import "net"
|
||||
|
||||
// Setting DSCP per-burst uses IP_TOS/IPV6_TCLASS under the same fd-borrow pattern as withDF,
|
||||
// which is only exercised on Linux deployments. Elsewhere the burst goes out with the default
|
||||
// class and TOSSupported lets the action response say so — an unmarked burst reported as marked
|
||||
// would read as "the network stripped DSCP", the exact wrong conclusion.
|
||||
func withTOS(_ *net.UDPConn, _ int, fn func() error) error { return fn() }
|
||||
|
||||
const TOSSupported = false
|
||||
@@ -0,0 +1,122 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package dataplane
|
||||
|
||||
// Upstream trains (spec §3.2, types 0x03–0x05). The client blasts TRAIN_DATA at the server and
|
||||
// the server answers nothing per-packet — a reply would double the traffic and measure the
|
||||
// return path at the same time. Afterwards the client asks for the server's received view with
|
||||
// TRAIN_REPORT_REQ, and gets it back columnar, split across as many TRAIN_REPORT datagrams as
|
||||
// it takes to stay under a safe size.
|
||||
//
|
||||
// Both TRAIN_DATA and TRAIN_REPORT_REQ carry the train id in payload[0:4]; the id is the
|
||||
// client's to choose, unique within the session.
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"net"
|
||||
"net/netip"
|
||||
|
||||
"echo-lot.app/server/internal/session"
|
||||
)
|
||||
|
||||
const (
|
||||
// trainReportMaxDatagram caps one TRAIN_REPORT datagram at a size that survives every common
|
||||
// path unfragmented. A report about loss must not itself be lost to MTU.
|
||||
trainReportMaxDatagram = 1200
|
||||
trainReportHeader = 16
|
||||
trainReportRow = 17 // 4 seq + 8 t_rx_ns + 2 size + 1 ttl + 1 dscp + 1 ecn
|
||||
)
|
||||
|
||||
// recordTrain buffers one TRAIN_DATA packet into its train (session-side, bounded — see
|
||||
// session/train.go). A payload too short to carry the id is unreportable and stays only in the
|
||||
// flat packet log, which already recorded it.
|
||||
func recordTrain(sess *session.Session, payload []byte, seq uint32, size int, tRxNs int64, meta pktMeta) {
|
||||
if len(payload) < 4 {
|
||||
return
|
||||
}
|
||||
sess.RecordTrainPacket(binary.BigEndian.Uint32(payload[0:4]), session.TrainEntry{
|
||||
Seq: seq, TRxNs: tRxNs, Size: uint16(min(size, 0xFFFF)),
|
||||
TTL: meta.TTL, DSCP: meta.dscp(), ECN: meta.ecn(),
|
||||
})
|
||||
}
|
||||
|
||||
// trainReport answers one TRAIN_REPORT_REQ with the full columnar report.
|
||||
//
|
||||
// Grant-free on purpose. §3.4 caps ungranted responses at the request size, and a multi-part
|
||||
// report is larger than the single REPORT_REQ that asked for it — but it cannot amplify: every
|
||||
// 17-byte row accounts for one HMAC-valid TRAIN_DATA packet of at least HeaderSize+4 bytes this
|
||||
// session already delivered here, so the whole report is a strict fraction of the traffic it
|
||||
// describes, and it only ever goes to the session's verified source address. An unknown id gets
|
||||
// a single zero-row report rather than silence — "nothing arrived" IS the measurement.
|
||||
func (s *Server) trainReport(conn *net.UDPConn, raddr netip.AddrPort, sess *session.Session, payload []byte) {
|
||||
if len(payload) < 4 {
|
||||
return
|
||||
}
|
||||
id := binary.BigEndian.Uint32(payload[0:4])
|
||||
train, _ := sess.TrainView(id)
|
||||
train.ID = id
|
||||
for i, part := range buildTrainReport(train) {
|
||||
s.send(conn, raddr, sess, TypeTrainReport, uint32(i), part)
|
||||
}
|
||||
}
|
||||
|
||||
// buildTrainReport lays a train's received view out columnar and cuts it into datagram-sized
|
||||
// payloads. Layout (mirrored by the client; all big-endian):
|
||||
//
|
||||
// 0 4 train_id
|
||||
// 4 4 received (every packet counted, buffered or not — loss math uses this)
|
||||
// 8 2 part (0-based)
|
||||
// 10 2 parts
|
||||
// 12 1 flags (bit0: buffer overflowed; rows beyond the cap were counted, not kept —
|
||||
// the schema's evidence_truncated honesty, on the wire)
|
||||
// 13 1 reserved
|
||||
// 14 2 n (rows in this part)
|
||||
// 16 n×4 seq, n×8 t_rx_ns, n×2 size, n×1 ttl, n×1 dscp, n×1 ecn (columns contiguous)
|
||||
func buildTrainReport(t session.Train) [][]byte {
|
||||
perPart := (trainReportMaxDatagram - HeaderSize - trainReportHeader) / trainReportRow
|
||||
parts := (len(t.Entries) + perPart - 1) / perPart
|
||||
if parts == 0 {
|
||||
parts = 1 // an empty train still gets its "received: 0" answer
|
||||
}
|
||||
out := make([][]byte, 0, parts)
|
||||
for p := 0; p < parts; p++ {
|
||||
rows := t.Entries[p*perPart : min((p+1)*perPart, len(t.Entries))]
|
||||
n := len(rows)
|
||||
b := make([]byte, trainReportHeader+n*trainReportRow)
|
||||
binary.BigEndian.PutUint32(b[0:4], t.ID)
|
||||
binary.BigEndian.PutUint32(b[4:8], uint32(t.Received))
|
||||
binary.BigEndian.PutUint16(b[8:10], uint16(p))
|
||||
binary.BigEndian.PutUint16(b[10:12], uint16(parts))
|
||||
if t.Truncated {
|
||||
b[12] = 1
|
||||
}
|
||||
binary.BigEndian.PutUint16(b[14:16], uint16(n))
|
||||
off := trainReportHeader
|
||||
for i, r := range rows {
|
||||
binary.BigEndian.PutUint32(b[off+i*4:], r.Seq)
|
||||
}
|
||||
off += n * 4
|
||||
for i, r := range rows {
|
||||
binary.BigEndian.PutUint64(b[off+i*8:], uint64(r.TRxNs))
|
||||
}
|
||||
off += n * 8
|
||||
for i, r := range rows {
|
||||
binary.BigEndian.PutUint16(b[off+i*2:], r.Size)
|
||||
}
|
||||
off += n * 2
|
||||
for i, r := range rows {
|
||||
b[off+i] = r.TTL
|
||||
}
|
||||
off += n
|
||||
for i, r := range rows {
|
||||
b[off+i] = r.DSCP
|
||||
}
|
||||
off += n
|
||||
for i, r := range rows {
|
||||
b[off+i] = r.ECN
|
||||
}
|
||||
out = append(out, b)
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,179 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package dataplane
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"net"
|
||||
"net/netip"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"echo-lot.app/server/internal/session"
|
||||
)
|
||||
|
||||
// parseReportPart decodes one TRAIN_REPORT payload back into rows, checking the header.
|
||||
func parseReportPart(t *testing.T, b []byte) (id uint32, received int, part, parts int, truncated bool, rows []session.TrainEntry) {
|
||||
t.Helper()
|
||||
if len(b) < trainReportHeader {
|
||||
t.Fatalf("report part shorter than its header: %d", len(b))
|
||||
}
|
||||
id = binary.BigEndian.Uint32(b[0:4])
|
||||
received = int(binary.BigEndian.Uint32(b[4:8]))
|
||||
part = int(binary.BigEndian.Uint16(b[8:10]))
|
||||
parts = int(binary.BigEndian.Uint16(b[10:12]))
|
||||
truncated = b[12]&1 != 0
|
||||
n := int(binary.BigEndian.Uint16(b[14:16]))
|
||||
if want := trainReportHeader + n*trainReportRow; len(b) != want {
|
||||
t.Fatalf("part length %d, want %d for %d rows", len(b), want, n)
|
||||
}
|
||||
off := trainReportHeader
|
||||
rows = make([]session.TrainEntry, n)
|
||||
for i := range rows {
|
||||
rows[i].Seq = binary.BigEndian.Uint32(b[off+i*4:])
|
||||
}
|
||||
off += n * 4
|
||||
for i := range rows {
|
||||
rows[i].TRxNs = int64(binary.BigEndian.Uint64(b[off+i*8:]))
|
||||
}
|
||||
off += n * 8
|
||||
for i := range rows {
|
||||
rows[i].Size = binary.BigEndian.Uint16(b[off+i*2:])
|
||||
}
|
||||
off += n * 2
|
||||
for i := range rows {
|
||||
rows[i].TTL = b[off+i]
|
||||
}
|
||||
off += n
|
||||
for i := range rows {
|
||||
rows[i].DSCP = b[off+i]
|
||||
}
|
||||
off += n
|
||||
for i := range rows {
|
||||
rows[i].ECN = b[off+i]
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
func TestBuildTrainReportSplitsAndRoundTrips(t *testing.T) {
|
||||
const count = 250 // enough to need several parts
|
||||
train := session.Train{ID: 42, Received: count, Truncated: true}
|
||||
for i := 0; i < count; i++ {
|
||||
train.Entries = append(train.Entries, session.TrainEntry{
|
||||
Seq: uint32(i), TRxNs: int64(i) * 1_000_000, Size: uint16(100 + i),
|
||||
TTL: 64, DSCP: 46, ECN: 1,
|
||||
})
|
||||
}
|
||||
|
||||
parts := buildTrainReport(train)
|
||||
if len(parts) < 2 {
|
||||
t.Fatalf("250 rows should not fit one ≤%d-byte datagram", trainReportMaxDatagram)
|
||||
}
|
||||
var got []session.TrainEntry
|
||||
for i, p := range parts {
|
||||
if HeaderSize+len(p) > trainReportMaxDatagram {
|
||||
t.Fatalf("part %d would be a %d-byte datagram, cap is %d", i, HeaderSize+len(p), trainReportMaxDatagram)
|
||||
}
|
||||
id, received, part, total, truncated, rows := parseReportPart(t, p)
|
||||
if id != 42 || received != count || part != i || total != len(parts) || !truncated {
|
||||
t.Fatalf("part %d header: id=%d received=%d part=%d/%d truncated=%v",
|
||||
i, id, received, part, total, truncated)
|
||||
}
|
||||
got = append(got, rows...)
|
||||
}
|
||||
if len(got) != count {
|
||||
t.Fatalf("round-tripped %d rows, want %d", len(got), count)
|
||||
}
|
||||
for i, r := range got {
|
||||
want := train.Entries[i]
|
||||
if r != want {
|
||||
t.Fatalf("row %d = %+v, want %+v", i, r, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildTrainReportEmptyTrainStillAnswers(t *testing.T) {
|
||||
parts := buildTrainReport(session.Train{ID: 9})
|
||||
if len(parts) != 1 {
|
||||
t.Fatalf("empty train: %d parts, want 1 — 'nothing arrived' is the answer, not silence", len(parts))
|
||||
}
|
||||
id, received, _, total, _, rows := parseReportPart(t, parts[0])
|
||||
if id != 9 || received != 0 || total != 1 || len(rows) != 0 {
|
||||
t.Fatalf("empty report: id=%d received=%d parts=%d rows=%d", id, received, total, len(rows))
|
||||
}
|
||||
}
|
||||
|
||||
func TestTrainDataThenReportOverTheWire(t *testing.T) {
|
||||
mgr, addr := startServer(t)
|
||||
sess, _, err := mgr.New("dev1", "credential-ikm", netip.MustParseAddr("127.0.0.1"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
client, err := net.DialUDP("udp", nil, net.UDPAddrFromAddrPort(addr))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer client.Close()
|
||||
client.SetDeadline(time.Now().Add(3 * time.Second))
|
||||
|
||||
// A short train: id 5 in payload[0:4], plus padding.
|
||||
const trainID, count = 5, 4
|
||||
for i := 0; i < count; i++ {
|
||||
payload := make([]byte, 60)
|
||||
binary.BigEndian.PutUint32(payload[0:4], trainID)
|
||||
if _, err := client.Write(craft(t, sess, TypeTrainData, uint32(i+1), payload)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
// TRAIN_DATA must be silent (spec §3.2).
|
||||
client.SetReadDeadline(time.Now().Add(300 * time.Millisecond))
|
||||
if _, err := client.Read(make([]byte, 1500)); err == nil {
|
||||
t.Fatal("TRAIN_DATA got a response, want none")
|
||||
}
|
||||
|
||||
// Ask for the report.
|
||||
reqPayload := make([]byte, 4)
|
||||
binary.BigEndian.PutUint32(reqPayload, trainID)
|
||||
client.SetDeadline(time.Now().Add(3 * time.Second))
|
||||
if _, err := client.Write(craft(t, sess, TypeTrainReportReq, 100, reqPayload)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
buf := make([]byte, 2000)
|
||||
n, err := client.Read(buf)
|
||||
if err != nil {
|
||||
t.Fatalf("no TRAIN_REPORT: %v", err)
|
||||
}
|
||||
if buf[4] != TypeTrainReport {
|
||||
t.Fatalf("type = %#x, want TRAIN_REPORT", buf[4])
|
||||
}
|
||||
id, received, part, parts, truncated, rows := parseReportPart(t, buf[HeaderSize:n])
|
||||
if id != trainID || received != count || part != 0 || parts != 1 || truncated {
|
||||
t.Fatalf("report header: id=%d received=%d part=%d/%d truncated=%v", id, received, part, parts, truncated)
|
||||
}
|
||||
if len(rows) != count {
|
||||
t.Fatalf("%d rows, want %d", len(rows), count)
|
||||
}
|
||||
for i, r := range rows {
|
||||
if r.Seq != uint32(i+1) {
|
||||
t.Fatalf("row %d seq = %d, want %d", i, r.Seq, i+1)
|
||||
}
|
||||
if r.Size != HeaderSize+60 {
|
||||
t.Fatalf("row %d size = %d, want %d", i, r.Size, HeaderSize+60)
|
||||
}
|
||||
}
|
||||
|
||||
// Unknown train id: one zero-row report, not silence.
|
||||
binary.BigEndian.PutUint32(reqPayload, 999)
|
||||
if _, err := client.Write(craft(t, sess, TypeTrainReportReq, 101, reqPayload)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
n, err = client.Read(buf)
|
||||
if err != nil {
|
||||
t.Fatalf("no report for unknown train: %v", err)
|
||||
}
|
||||
id, received, _, _, _, rows = parseReportPart(t, buf[HeaderSize:n])
|
||||
if id != 999 || received != 0 || len(rows) != 0 {
|
||||
t.Fatalf("unknown-train report: id=%d received=%d rows=%d", id, received, len(rows))
|
||||
}
|
||||
}
|
||||
@@ -2,15 +2,15 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// Package dataplane implements the binary UDP probe protocol (spec §3):
|
||||
// 32-byte header, HMAC gate, anti-replay, ECHO with observation block.
|
||||
// Skeleton scope: ECHO_REQ/ECHO_RESP and TIMESYNC only; trains, MTU probes
|
||||
// and delayed echo land with the corresponding client tests.
|
||||
// 32-byte header, HMAC gate, anti-replay, ECHO with observation block,
|
||||
// upstream trains with columnar reports, and the granted server->client sends.
|
||||
package dataplane
|
||||
|
||||
import (
|
||||
"crypto/hmac"
|
||||
"crypto/sha256"
|
||||
"encoding/binary"
|
||||
"encoding/hex"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"net"
|
||||
@@ -18,6 +18,7 @@ import (
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"echo-lot.app/server/internal/ratelimit"
|
||||
"echo-lot.app/server/internal/session"
|
||||
)
|
||||
|
||||
@@ -25,9 +26,14 @@ const (
|
||||
Magic = "ELT1"
|
||||
HeaderSize = 32
|
||||
|
||||
TypeEchoReq = 0x01
|
||||
TypeEchoResp = 0x02
|
||||
TypeTimesyncReq = 0x07
|
||||
TypeEchoReq = 0x01
|
||||
TypeEchoResp = 0x02
|
||||
// Upstream trains (spec §3.2): DATA gets no per-packet response; REPORT_REQ fetches the
|
||||
// server's received view as one or more REPORT datagrams (train.go).
|
||||
TypeTrainData = 0x03
|
||||
TypeTrainReportReq = 0x04
|
||||
TypeTrainReport = 0x05
|
||||
TypeTimesyncReq = 0x07
|
||||
TypeTimesyncRsp = 0x08
|
||||
TypeMtuProbe = 0x09
|
||||
TypeMtuAck = 0x0A
|
||||
@@ -47,6 +53,12 @@ const (
|
||||
|
||||
type Server struct {
|
||||
Sessions *session.Manager
|
||||
// Spec §2.5 ceilings on verified traffic, silent-drop (nil = no ceiling). Charged after the
|
||||
// HMAC gate so an unauthenticated flood cannot spend anyone's budget, keyed per source
|
||||
// address AND per device credential so neither one hot address nor one hot credential can
|
||||
// crowd out the rest.
|
||||
PacketRate *ratelimit.Limiter // tokens are packets
|
||||
ByteRate *ratelimit.Limiter // tokens are bytes
|
||||
// Epoch for server-side t_rx/t_tx: process start; observation consumers
|
||||
// only need differences plus the timesync exchange, not absolute time.
|
||||
start time.Time
|
||||
@@ -59,6 +71,34 @@ type Server struct {
|
||||
dfMu sync.Mutex
|
||||
}
|
||||
|
||||
// pktMeta is what the kernel told us about one received datagram beyond its bytes (spec §3.3:
|
||||
// received TTL, DSCP, ECN). 0xFF means "not observed": non-Linux hosts and datagrams whose
|
||||
// cmsg never arrived keep the sentinel rather than inventing a value.
|
||||
type pktMeta struct {
|
||||
TTL uint8
|
||||
TOS uint8 // the whole DSCP/ECN byte; DSCP = TOS>>2, ECN = TOS&3
|
||||
}
|
||||
|
||||
const metaUnavailable = 0xFF
|
||||
|
||||
func (m pktMeta) dscp() uint8 {
|
||||
if m.TOS == metaUnavailable {
|
||||
return metaUnavailable
|
||||
}
|
||||
return m.TOS >> 2
|
||||
}
|
||||
|
||||
func (m pktMeta) ecn() uint8 {
|
||||
if m.TOS == metaUnavailable {
|
||||
return metaUnavailable
|
||||
}
|
||||
return m.TOS & 0x3
|
||||
}
|
||||
|
||||
// oobCap fits the two cmsgs (TTL + TOS, each ≤ CMSG_SPACE(4)) with headroom for whatever else
|
||||
// the kernel decides to attach.
|
||||
const oobCap = 64
|
||||
|
||||
// Serve runs the read loop for one socket; call once per bound address.
|
||||
// The socket is retained so actions (delayed echo) can pick a family-matching
|
||||
// sender later.
|
||||
@@ -69,14 +109,16 @@ func (s *Server) Serve(conn *net.UDPConn) error {
|
||||
}
|
||||
s.conns = append(s.conns, conn)
|
||||
s.mu.Unlock()
|
||||
enableRecvMeta(conn)
|
||||
buf := make([]byte, 65535)
|
||||
oob := make([]byte, oobCap)
|
||||
for {
|
||||
n, raddr, err := conn.ReadFromUDPAddrPort(buf)
|
||||
n, oobn, _, raddr, err := conn.ReadMsgUDPAddrPort(buf, oob)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
tRx := time.Since(s.start).Nanoseconds()
|
||||
s.handle(conn, raddr, buf[:n], tRx)
|
||||
s.handle(conn, raddr, buf[:n], tRx, parseMeta(oob[:oobn]))
|
||||
}
|
||||
}
|
||||
|
||||
@@ -126,7 +168,7 @@ func (s *Server) SendDelayedEcho(sess *session.Session, actionID string) error {
|
||||
|
||||
// handle enforces spec §3.1/§3.4: unknown prefix, bad HMAC, expired session,
|
||||
// replayed seq → silent drop, never a response.
|
||||
func (s *Server) handle(conn *net.UDPConn, raddr netip.AddrPort, pkt []byte, tRxNs int64) {
|
||||
func (s *Server) handle(conn *net.UDPConn, raddr netip.AddrPort, pkt []byte, tRxNs int64, meta pktMeta) {
|
||||
if len(pkt) < HeaderSize || string(pkt[0:4]) != Magic {
|
||||
return
|
||||
}
|
||||
@@ -149,6 +191,12 @@ func (s *Server) handle(conn *net.UDPConn, raddr netip.AddrPort, pkt []byte, tRx
|
||||
if !hmac.Equal(mac.Sum(nil)[:4], pkt[28:32]) {
|
||||
return
|
||||
}
|
||||
// Spec §2.5: over-ceiling traffic is silently dropped (probes tolerate loss by design).
|
||||
// After the HMAC gate so a spoofed flood cannot drain a victim's budget; before the replay
|
||||
// window so a dropped packet's seq stays usable for a resend.
|
||||
if !s.allowUDP(raddr, sess.Device, len(pkt)) {
|
||||
return
|
||||
}
|
||||
if !sess.CheckSeq(seq) {
|
||||
return
|
||||
}
|
||||
@@ -169,9 +217,16 @@ func (s *Server) handle(conn *net.UDPConn, raddr netip.AddrPort, pkt []byte, tRx
|
||||
Src: raddr.String(), Size: len(pkt), Type: typ,
|
||||
})
|
||||
|
||||
payload := pkt[HeaderSize : HeaderSize+int(payloadLen)]
|
||||
switch typ {
|
||||
case TypeEchoReq:
|
||||
s.echoResp(conn, raddr, sess, pkt, seq, tRxNs)
|
||||
s.echoResp(conn, raddr, sess, pkt, seq, tRxNs, meta)
|
||||
case TypeTrainData:
|
||||
// No response (spec §3.2): the train is upstream-only; its received view is fetched
|
||||
// afterwards via TRAIN_REPORT_REQ or the observations API.
|
||||
recordTrain(sess, payload, seq, len(pkt), tRxNs, meta)
|
||||
case TypeTrainReportReq:
|
||||
s.trainReport(conn, raddr, sess, payload)
|
||||
case TypeTimesyncReq:
|
||||
s.timesyncResp(conn, raddr, sess, pkt, seq, tRxNs)
|
||||
case TypeMtuProbe:
|
||||
@@ -181,6 +236,16 @@ func (s *Server) handle(conn *net.UDPConn, raddr netip.AddrPort, pkt []byte, tRx
|
||||
}
|
||||
}
|
||||
|
||||
// allowUDP charges the §2.5 packet and byte buckets, per source address and per credential.
|
||||
func (s *Server) allowUDP(raddr netip.AddrPort, device string, size int) bool {
|
||||
ipKey, credKey := "ip:"+raddr.Addr().String(), "cred:"+device
|
||||
okA, _ := s.PacketRate.Allow(ipKey)
|
||||
okC, _ := s.PacketRate.Allow(credKey)
|
||||
okAB, _ := s.ByteRate.AllowN(ipKey, float64(size))
|
||||
okCB, _ := s.ByteRate.AllowN(credKey, float64(size))
|
||||
return okA && okC && okAB && okCB
|
||||
}
|
||||
|
||||
// mtuAck replies to an MTU_PROBE with a small MTU_ACK carrying the total
|
||||
// datagram size the server actually received (spec §3.2). The client sends
|
||||
// DF-flagged probes of increasing size and binary-searches the path MTU / a
|
||||
@@ -198,28 +263,43 @@ func (s *Server) mtuAck(conn *net.UDPConn, raddr netip.AddrPort, sess *session.S
|
||||
// 8 8 t_tx_ns
|
||||
// 16 16 observed source IP (v4-mapped when v4)
|
||||
// 32 2 observed source port
|
||||
// 34 1 received TTL (0xFF = not observed yet; needs recvmsg cmsgs)
|
||||
// 34 1 received TTL (0xFF = not observed; cmsgs unavailable on this host)
|
||||
// 35 1 received DSCP/ECN byte (0xFF = not observed)
|
||||
// 36 4 received size
|
||||
func observation(tRxNs, tTxNs int64, src netip.AddrPort, rcvd int) []byte {
|
||||
func observation(tRxNs, tTxNs int64, src netip.AddrPort, rcvd int, meta pktMeta) []byte {
|
||||
b := make([]byte, 40)
|
||||
binary.BigEndian.PutUint64(b[0:8], uint64(tRxNs))
|
||||
binary.BigEndian.PutUint64(b[8:16], uint64(tTxNs))
|
||||
a16 := src.Addr().As16()
|
||||
copy(b[16:32], a16[:])
|
||||
binary.BigEndian.PutUint16(b[32:34], src.Port())
|
||||
b[34], b[35] = 0xFF, 0xFF
|
||||
b[34], b[35] = meta.TTL, meta.TOS
|
||||
binary.BigEndian.PutUint32(b[36:40], uint32(rcvd))
|
||||
return b
|
||||
}
|
||||
|
||||
// putActionID writes a grant's action id into payload[8:16] — the correlation the spec promises
|
||||
// (§5: "an action_id echoed in resulting data-plane packets"), consumed by the client as
|
||||
// test.params.action_id (§9). Bytes [0:8] stay with the packet type; [8:16] is reserved for this
|
||||
// across every granted type, so the client needs one rule, not five.
|
||||
func putActionID(payload []byte, actionID string) {
|
||||
if len(payload) < 16 {
|
||||
return
|
||||
}
|
||||
raw, err := hex.DecodeString(actionID)
|
||||
if err != nil || len(raw) != 8 {
|
||||
return // a malformed id yields zero bytes, not a crash mid-burst
|
||||
}
|
||||
copy(payload[8:16], raw)
|
||||
}
|
||||
|
||||
// echoResp mirrors the request header (type flipped), appends the observation
|
||||
// block, and re-HMACs with the session key. Anti-amplification: the response
|
||||
// is capped at the request size (spec §3.4) — the observation block replaces
|
||||
// padding rather than growing the datagram; if the request was smaller than
|
||||
// header+observation, the block is truncated to fit.
|
||||
func (s *Server) echoResp(conn *net.UDPConn, raddr netip.AddrPort, sess *session.Session, req []byte, seq uint32, tRxNs int64) {
|
||||
obs := observation(tRxNs, time.Since(s.start).Nanoseconds(), raddr, len(req))
|
||||
func (s *Server) echoResp(conn *net.UDPConn, raddr netip.AddrPort, sess *session.Session, req []byte, seq uint32, tRxNs int64, meta pktMeta) {
|
||||
obs := observation(tRxNs, time.Since(s.start).Nanoseconds(), raddr, len(req), meta)
|
||||
max := len(req)
|
||||
if max < HeaderSize {
|
||||
return
|
||||
|
||||
@@ -0,0 +1,100 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// Package ratelimit implements the spec §2.5 token buckets: per-credential and per-source-IP
|
||||
// ceilings on session creation, actions, UDP packets and bytes. One Limiter holds one policy
|
||||
// (rate + burst) and lazily creates a bucket per key; callers namespace their keys ("cred:…",
|
||||
// "ip:…") so a single Limiter can enforce both axes of the same rule.
|
||||
package ratelimit
|
||||
|
||||
import (
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Limiter is a keyed set of token buckets sharing one rate and burst.
|
||||
//
|
||||
// A nil *Limiter allows everything: the ceilings are configurable down to "off" (config value 0),
|
||||
// and a nil check in one place beats a sentinel policy that every call site must know about.
|
||||
type Limiter struct {
|
||||
rate float64 // tokens per second
|
||||
burst float64
|
||||
|
||||
mu sync.Mutex
|
||||
buckets map[string]*bucket
|
||||
lastSweep time.Time
|
||||
now func() time.Time // swappable so tests need no sleeping
|
||||
}
|
||||
|
||||
type bucket struct {
|
||||
tokens float64
|
||||
last time.Time
|
||||
}
|
||||
|
||||
// New creates a limiter granting ratePerSec tokens per second per key, holding at most burst.
|
||||
func New(ratePerSec, burst float64) *Limiter {
|
||||
return &Limiter{
|
||||
rate: ratePerSec,
|
||||
burst: burst,
|
||||
buckets: map[string]*bucket{},
|
||||
now: time.Now,
|
||||
}
|
||||
}
|
||||
|
||||
// Allow takes one token for key. See AllowN.
|
||||
func (l *Limiter) Allow(key string) (bool, time.Duration) {
|
||||
return l.AllowN(key, 1)
|
||||
}
|
||||
|
||||
// AllowN takes n tokens for key, reporting whether they were available and — when they were
|
||||
// not — how long until they will be, which is what a control-plane 429 puts in Retry-After.
|
||||
// A refusal consumes nothing: the caller being told to wait must not itself push the wait out.
|
||||
func (l *Limiter) AllowN(key string, n float64) (bool, time.Duration) {
|
||||
if l == nil {
|
||||
return true, 0
|
||||
}
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
now := l.now()
|
||||
l.sweepLocked(now)
|
||||
b := l.buckets[key]
|
||||
if b == nil {
|
||||
b = &bucket{tokens: l.burst, last: now}
|
||||
l.buckets[key] = b
|
||||
}
|
||||
b.tokens += now.Sub(b.last).Seconds() * l.rate
|
||||
if b.tokens > l.burst {
|
||||
b.tokens = l.burst
|
||||
}
|
||||
b.last = now
|
||||
if b.tokens >= n {
|
||||
b.tokens -= n
|
||||
return true, 0
|
||||
}
|
||||
return false, time.Duration((n - b.tokens) / l.rate * float64(time.Second))
|
||||
}
|
||||
|
||||
// sweepEvery bounds how often the map is walked; the walk is cheap but there is no point doing
|
||||
// it per packet on the data plane's hot path.
|
||||
const sweepEvery = time.Minute
|
||||
|
||||
// sweepLocked drops buckets that have been idle long enough to be full again. A full bucket
|
||||
// carries no state a fresh one would not, and without the sweep the map grows one entry per
|
||||
// source address ever seen — an attacker-controlled key space must not be an unbounded one.
|
||||
func (l *Limiter) sweepLocked(now time.Time) {
|
||||
if now.Sub(l.lastSweep) < sweepEvery {
|
||||
return
|
||||
}
|
||||
l.lastSweep = now
|
||||
idle := sweepEvery
|
||||
if l.rate > 0 {
|
||||
if refill := time.Duration(l.burst / l.rate * float64(time.Second)); refill > idle {
|
||||
idle = refill
|
||||
}
|
||||
}
|
||||
for k, b := range l.buckets {
|
||||
if now.Sub(b.last) > idle {
|
||||
delete(l.buckets, k)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,105 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package ratelimit
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// clockAt pins the limiter to a fake clock so refill is a function of arithmetic, not sleeping.
|
||||
func clockAt(l *Limiter) *time.Time {
|
||||
t := time.Unix(1000, 0)
|
||||
l.now = func() time.Time { return t }
|
||||
return &t
|
||||
}
|
||||
|
||||
func TestBurstThenRefusalThenRefill(t *testing.T) {
|
||||
l := New(1, 3) // 1 token/s, burst 3
|
||||
now := clockAt(l)
|
||||
|
||||
for i := 0; i < 3; i++ {
|
||||
if ok, _ := l.Allow("k"); !ok {
|
||||
t.Fatalf("token %d of the burst refused", i)
|
||||
}
|
||||
}
|
||||
ok, wait := l.Allow("k")
|
||||
if ok {
|
||||
t.Fatal("fourth token inside the same instant should be refused")
|
||||
}
|
||||
if wait <= 0 || wait > time.Second {
|
||||
t.Fatalf("retry-after = %v, want (0, 1s]", wait)
|
||||
}
|
||||
|
||||
*now = now.Add(2 * time.Second) // refills 2 tokens
|
||||
if ok, _ := l.Allow("k"); !ok {
|
||||
t.Fatal("refused after refill")
|
||||
}
|
||||
if ok, _ := l.Allow("k"); !ok {
|
||||
t.Fatal("second refilled token refused")
|
||||
}
|
||||
if ok, _ := l.Allow("k"); ok {
|
||||
t.Fatal("third token allowed but only two seconds elapsed")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRefusalConsumesNothing(t *testing.T) {
|
||||
l := New(1, 1)
|
||||
now := clockAt(l)
|
||||
l.Allow("k")
|
||||
// Hammering while empty must not push the refill out.
|
||||
for i := 0; i < 10; i++ {
|
||||
if ok, _ := l.Allow("k"); ok {
|
||||
t.Fatal("allowed while empty")
|
||||
}
|
||||
}
|
||||
*now = now.Add(time.Second)
|
||||
if ok, _ := l.Allow("k"); !ok {
|
||||
t.Fatal("the refused attempts ate the refill")
|
||||
}
|
||||
}
|
||||
|
||||
func TestKeysAreIndependent(t *testing.T) {
|
||||
l := New(1, 1)
|
||||
clockAt(l)
|
||||
l.Allow("a")
|
||||
if ok, _ := l.Allow("b"); !ok {
|
||||
t.Fatal("draining key a refused key b")
|
||||
}
|
||||
}
|
||||
|
||||
func TestAllowNChargesBytes(t *testing.T) {
|
||||
l := New(1000, 1000) // e.g. bytes/s
|
||||
clockAt(l)
|
||||
if ok, _ := l.AllowN("k", 900); !ok {
|
||||
t.Fatal("900 of 1000 refused")
|
||||
}
|
||||
if ok, _ := l.AllowN("k", 200); ok {
|
||||
t.Fatal("1100 of 1000 allowed")
|
||||
}
|
||||
if ok, _ := l.AllowN("k", 100); !ok {
|
||||
t.Fatal("the refused 200 consumed the remaining 100")
|
||||
}
|
||||
}
|
||||
|
||||
func TestNilLimiterAllowsEverything(t *testing.T) {
|
||||
var l *Limiter
|
||||
if ok, wait := l.AllowN("k", 1e12); !ok || wait != 0 {
|
||||
t.Fatal("nil limiter must be a no-op")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSweepDropsIdleBucketsOnly(t *testing.T) {
|
||||
l := New(1, 3)
|
||||
now := clockAt(l)
|
||||
l.Allow("idle")
|
||||
*now = now.Add(2 * time.Minute)
|
||||
l.Allow("busy") // triggers the sweep; "idle" refilled long ago
|
||||
if _, held := l.buckets["idle"]; held {
|
||||
t.Fatal("idle bucket survived the sweep")
|
||||
}
|
||||
if _, held := l.buckets["busy"]; !held {
|
||||
t.Fatal("active bucket was swept")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// Package relsign signs and verifies release manifests (detached ed25519 over SHA256SUMS).
|
||||
//
|
||||
// The checksum file alone protects download integrity, not authenticity: SHA256SUMS and the
|
||||
// binaries come from the same Gitea release, so whoever can alter one can alter both. The
|
||||
// signature is what separates "the file arrived intact" from "the project published this file" —
|
||||
// its private key lives in the CI secret store, not on the release host, so a compromised Gitea
|
||||
// can serve corrupted binaries but cannot make a self-updating server accept them.
|
||||
//
|
||||
// Formats, chosen to be reproducible with nothing but a stock library in any language:
|
||||
// the private key is the base64 of the 32-byte ed25519 seed, the public key the base64 of the
|
||||
// 32-byte public key, and the signature file the base64 of the 64-byte signature over the exact
|
||||
// bytes of the signed file.
|
||||
package relsign
|
||||
|
||||
import (
|
||||
"crypto/ed25519"
|
||||
"encoding/base64"
|
||||
"fmt"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// GenerateKey mints a fresh signing keypair.
|
||||
func GenerateKey() (pubB64, seedB64 string, err error) {
|
||||
pub, priv, err := ed25519.GenerateKey(nil)
|
||||
if err != nil {
|
||||
return "", "", err
|
||||
}
|
||||
return base64.StdEncoding.EncodeToString(pub),
|
||||
base64.StdEncoding.EncodeToString(priv.Seed()), nil
|
||||
}
|
||||
|
||||
// Sign produces the detached signature (base64) for data.
|
||||
func Sign(seedB64 string, data []byte) (string, error) {
|
||||
seed, err := base64.StdEncoding.DecodeString(strings.TrimSpace(seedB64))
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("signing key is not valid base64: %w", err)
|
||||
}
|
||||
if len(seed) != ed25519.SeedSize {
|
||||
return "", fmt.Errorf("signing key must be %d bytes, got %d", ed25519.SeedSize, len(seed))
|
||||
}
|
||||
priv := ed25519.NewKeyFromSeed(seed)
|
||||
return base64.StdEncoding.EncodeToString(ed25519.Sign(priv, data)), nil
|
||||
}
|
||||
|
||||
// Verify checks a detached signature. A nil error means the holder of the private key matching
|
||||
// pubB64 signed exactly these bytes.
|
||||
func Verify(pubB64 string, data []byte, sigB64 string) error {
|
||||
pub, err := base64.StdEncoding.DecodeString(strings.TrimSpace(pubB64))
|
||||
if err != nil {
|
||||
return fmt.Errorf("public key is not valid base64: %w", err)
|
||||
}
|
||||
if len(pub) != ed25519.PublicKeySize {
|
||||
return fmt.Errorf("public key must be %d bytes, got %d", ed25519.PublicKeySize, len(pub))
|
||||
}
|
||||
sig, err := base64.StdEncoding.DecodeString(strings.TrimSpace(sigB64))
|
||||
if err != nil {
|
||||
return fmt.Errorf("signature is not valid base64: %w", err)
|
||||
}
|
||||
if !ed25519.Verify(ed25519.PublicKey(pub), data, sig) {
|
||||
return fmt.Errorf("signature does not verify: the file was not signed by this key, or was altered after signing")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package relsign
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestRoundTrip(t *testing.T) {
|
||||
pub, seed, err := GenerateKey()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
data := []byte("abc123 echolot-server_linux_amd64\n")
|
||||
sig, err := Sign(seed, data)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := Verify(pub, data, sig); err != nil {
|
||||
t.Fatalf("a signature this package just made must verify: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAlteredContentIsRefused(t *testing.T) {
|
||||
// The attack this exists for: same length, one checksum swapped for another.
|
||||
pub, seed, _ := GenerateKey()
|
||||
sig, _ := Sign(seed, []byte("aaa echolot-server_linux_amd64\n"))
|
||||
if err := Verify(pub, []byte("bbb echolot-server_linux_amd64\n"), sig); err == nil {
|
||||
t.Fatal("altered content must not verify")
|
||||
}
|
||||
}
|
||||
|
||||
func TestWrongKeyIsRefused(t *testing.T) {
|
||||
// A compromised release host can re-sign with its own key; only ours may pass.
|
||||
pub1, _, _ := GenerateKey()
|
||||
_, seed2, _ := GenerateKey()
|
||||
data := []byte("payload")
|
||||
sig, _ := Sign(seed2, data)
|
||||
if err := Verify(pub1, data, sig); err == nil {
|
||||
t.Fatal("a signature from a different key must not verify")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSurroundingWhitespaceIsTolerated(t *testing.T) {
|
||||
// Keys travel through env vars and files; a trailing newline must not break verification.
|
||||
pub, seed, _ := GenerateKey()
|
||||
data := []byte("data")
|
||||
sig, err := Sign(" "+seed+"\n", data)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := Verify(pub+"\n", data, "\t"+sig+"\n"); err != nil {
|
||||
t.Fatalf("whitespace around base64 must be tolerated: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestGarbageInputsFailCleanly(t *testing.T) {
|
||||
pub, seed, _ := GenerateKey()
|
||||
if _, err := Sign("not base64!!", []byte("x")); err == nil || !strings.Contains(err.Error(), "base64") {
|
||||
t.Fatalf("bad seed must name the problem, got %v", err)
|
||||
}
|
||||
if _, err := Sign("c2hvcnQ=", []byte("x")); err == nil {
|
||||
t.Fatal("short seed must be refused")
|
||||
}
|
||||
if err := Verify("c2hvcnQ=", []byte("x"), "AAAA"); err == nil {
|
||||
t.Fatal("short public key must be refused")
|
||||
}
|
||||
if err := Verify(pub, []byte("x"), "not base64!!"); err == nil {
|
||||
t.Fatal("bad signature encoding must be refused")
|
||||
}
|
||||
_ = seed
|
||||
}
|
||||
@@ -0,0 +1,78 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package runs
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Account scoping widens what a caller can read, so the test that matters is the one about what
|
||||
// it must NOT widen: a run id from another account has to be invisible, not merely unlisted.
|
||||
func TestAccountScopingDoesNotReachOtherAccounts(t *testing.T) {
|
||||
s, _ := open(t, DefaultPolicy())
|
||||
|
||||
// Two devices on one account, one device belonging to somebody else.
|
||||
mine := []string{"phone-a", "tablet-a"}
|
||||
for i, d := range mine {
|
||||
if _, err := s.Put(d, doc("run-"+d, AnonFull), true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
_ = i
|
||||
time.Sleep(2 * time.Millisecond)
|
||||
}
|
||||
if _, err := s.Put("phone-b", doc("run-secret", AnonFull), true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
got := s.ListFor(mine)
|
||||
if len(got) != 2 {
|
||||
t.Fatalf("account history has %d runs, want 2", len(got))
|
||||
}
|
||||
for _, m := range got {
|
||||
if m.ID == "run-secret" {
|
||||
t.Fatal("another account's run appeared in the history")
|
||||
}
|
||||
}
|
||||
|
||||
// The decisive one: knowing the id is not enough.
|
||||
if _, ok := s.OwnerOf(mine, "run-secret"); ok {
|
||||
t.Fatal("a run id from another account resolved against this account's devices")
|
||||
}
|
||||
if owner, ok := s.OwnerOf(mine, "run-phone-a"); !ok || owner != "phone-a" {
|
||||
t.Fatalf("own run did not resolve: owner=%q ok=%v", owner, ok)
|
||||
}
|
||||
// A sibling device's run must resolve — that is the point of the feature.
|
||||
if owner, ok := s.OwnerOf(mine, "run-tablet-a"); !ok || owner != "tablet-a" {
|
||||
t.Fatalf("sibling device's run did not resolve: owner=%q ok=%v", owner, ok)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAccountHistoryIsNewestFirstAcrossDevices(t *testing.T) {
|
||||
s, _ := open(t, DefaultPolicy())
|
||||
if _, err := s.Put("phone", doc("older", AnonFull), true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
if _, err := s.Put("tablet", doc("newer", AnonFull), true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got := s.ListFor([]string{"phone", "tablet"})
|
||||
if len(got) != 2 || got[0].ID != "newer" {
|
||||
t.Fatalf("not merged newest-first: %+v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEmptyDeviceSetSeesNothing(t *testing.T) {
|
||||
s, _ := open(t, DefaultPolicy())
|
||||
if _, err := s.Put("someone", doc("run-1", AnonFull), true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := s.ListFor(nil); len(got) != 0 {
|
||||
t.Fatalf("an empty device set returned %d runs", len(got))
|
||||
}
|
||||
if _, ok := s.OwnerOf(nil, "run-1"); ok {
|
||||
t.Fatal("a run resolved against an empty device set")
|
||||
}
|
||||
}
|
||||
@@ -158,7 +158,11 @@ func (s *Store) Put(deviceID string, body []byte, linked bool) (Meta, error) {
|
||||
} `json:"run"`
|
||||
Findings []json.RawMessage `json:"findings"`
|
||||
Summary struct {
|
||||
Verdict string `json:"verdict"`
|
||||
// measurement-schema.md §7.3 calls this "overall"; "verdict" is the per-category
|
||||
// field one level down. Reading the wrong one stored an empty verdict on every run
|
||||
// ever uploaded, which the UI showed as "not recorded" — a claim about the document
|
||||
// that was really a bug in this parser.
|
||||
Overall string `json:"overall"`
|
||||
} `json:"summary"`
|
||||
}
|
||||
if err := json.Unmarshal(body, &doc); err != nil || doc.Run.ID == "" {
|
||||
@@ -192,7 +196,7 @@ func (s *Store) Put(deviceID string, body []byte, linked bool) (Meta, error) {
|
||||
meta := Meta{
|
||||
ID: id, DeviceID: deviceID, UploadedAt: time.Now().UTC(),
|
||||
StartedAt: doc.Run.StartedAt, Anonymization: level,
|
||||
SizeBytes: int64(len(body)), Verdict: doc.Summary.Verdict,
|
||||
SizeBytes: int64(len(body)), Verdict: doc.Summary.Overall,
|
||||
FindingCount: len(doc.Findings),
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(devDir, id+".meta.json"), mustJSON(meta), 0o600); err != nil {
|
||||
@@ -209,6 +213,38 @@ func (s *Store) List(deviceID string) []Meta {
|
||||
return s.listLocked(filepath.Join(s.dir, sanitizeID(deviceID)))
|
||||
}
|
||||
|
||||
// ListFor returns the runs of several devices at once, newest first.
|
||||
//
|
||||
// This is what makes an account mean something: three phones signed in to one account produce one
|
||||
// history, which is the main reason to have accounts beyond upload permission.
|
||||
func (s *Store) ListFor(deviceIDs []string) []Meta {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
var out []Meta
|
||||
for _, id := range deviceIDs {
|
||||
out = append(out, s.listLocked(filepath.Join(s.dir, sanitizeID(id)))...)
|
||||
}
|
||||
sort.Slice(out, func(i, j int) bool { return out[i].UploadedAt.After(out[j].UploadedAt) })
|
||||
return out
|
||||
}
|
||||
|
||||
// OwnerOf reports which of these devices holds runID, so a caller can be granted access to a run
|
||||
// belonging to a sibling device without being able to name an arbitrary device.
|
||||
//
|
||||
// The search is over an allow-list the caller never supplies directly — it comes from the account
|
||||
// — so a run id from another account simply is not found.
|
||||
func (s *Store) OwnerOf(deviceIDs []string, runID string) (string, bool) {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
for _, id := range deviceIDs {
|
||||
p := filepath.Join(s.dir, sanitizeID(id), sanitizeID(runID)+".json")
|
||||
if fi, err := os.Stat(p); err == nil && !fi.IsDir() {
|
||||
return id, true
|
||||
}
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// Get returns the stored document bytes for one run.
|
||||
func (s *Store) Get(deviceID, runID string) ([]byte, error) {
|
||||
s.mu.Lock()
|
||||
|
||||
@@ -17,7 +17,7 @@ import (
|
||||
func doc(id, anon string) []byte {
|
||||
return []byte(fmt.Sprintf(
|
||||
`{"run":{"id":%q,"started_at":"2026-08-01T10:00:00Z","privacy":{"anonymization":%q}},`+
|
||||
`"findings":[{"id":"f1"},{"id":"f2"}],"summary":{"verdict":"warn"}}`, id, anon))
|
||||
`"findings":[{"id":"f1"},{"id":"f2"}],"summary":{"overall":"yellow"}}`, id, anon))
|
||||
}
|
||||
|
||||
func open(t *testing.T, p Policy) (*Store, string) {
|
||||
@@ -193,7 +193,7 @@ func TestMetaSummarisesTheDocument(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if m.FindingCount != 2 || m.Verdict != "warn" || m.Anonymization != AnonBalanced {
|
||||
if m.FindingCount != 2 || m.Verdict != "yellow" || m.Anonymization != AnonBalanced {
|
||||
t.Fatalf("meta not extracted: %+v", m)
|
||||
}
|
||||
if m.StartedAt != "2026-08-01T10:00:00Z" {
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package selftest
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"net"
|
||||
)
|
||||
|
||||
// ReservedWebPortsFree verifies that nothing on this host — this process or any other — is
|
||||
// listening on 80/443 of the reserved measurement addresses.
|
||||
//
|
||||
// config.CheckReserved keeps *our own* listeners off those ports, but our configuration is not
|
||||
// the host: the adb-beacon receiver, a separate Python process wildcard-bound to 0.0.0.0:443,
|
||||
// silently voided the IPv4 interception proof for as long as it ran, and nothing in this server's
|
||||
// config could have seen it. Asking the OS is the only check that covers processes we did not
|
||||
// start.
|
||||
//
|
||||
// The mechanism is a throwaway bind: if the bind succeeds the port was provably free (closed
|
||||
// again immediately — nothing is served), and if it fails with EADDRINUSE something is listening
|
||||
// there. Any other failure (address not assigned to this host, missing privilege) means the
|
||||
// question could not be answered, which is reported separately rather than pretending either way.
|
||||
func ReservedWebPortsFree(ips []net.IP) (occupied, unverifiable []string) {
|
||||
return portsFree(ips, []string{"80", "443"})
|
||||
}
|
||||
|
||||
func portsFree(ips []net.IP, ports []string) (occupied, unverifiable []string) {
|
||||
for _, ip := range ips {
|
||||
for _, port := range ports {
|
||||
addr := net.JoinHostPort(ip.String(), port)
|
||||
ln, err := net.Listen("tcp", addr)
|
||||
if err == nil {
|
||||
ln.Close()
|
||||
continue
|
||||
}
|
||||
if isAddrInUse(err) {
|
||||
occupied = append(occupied, addr)
|
||||
} else {
|
||||
unverifiable = append(unverifiable, fmt.Sprintf("%s (%v)", addr, err))
|
||||
}
|
||||
}
|
||||
}
|
||||
return occupied, unverifiable
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package selftest
|
||||
|
||||
import (
|
||||
"net"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// The real check runs against ports 80/443, which a test cannot bind without privileges; the
|
||||
// port list is what varies here, the mechanism is identical.
|
||||
|
||||
func TestOccupiedPortIsDetected(t *testing.T) {
|
||||
// The stray-process scenario: someone else holds the port before we look.
|
||||
ln, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer ln.Close()
|
||||
port := strconv.Itoa(ln.Addr().(*net.TCPAddr).Port)
|
||||
|
||||
occupied, unverifiable := portsFree([]net.IP{net.ParseIP("127.0.0.1")}, []string{port})
|
||||
if len(occupied) != 1 || !strings.HasSuffix(occupied[0], ":"+port) {
|
||||
t.Fatalf("a listening port must be reported occupied, got occupied=%v unverifiable=%v",
|
||||
occupied, unverifiable)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFreePortPassesAndStaysFree(t *testing.T) {
|
||||
// Find a port that is free by construction, then check it.
|
||||
probe, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
port := strconv.Itoa(probe.Addr().(*net.TCPAddr).Port)
|
||||
probe.Close()
|
||||
|
||||
occupied, unverifiable := portsFree([]net.IP{net.ParseIP("127.0.0.1")}, []string{port})
|
||||
if len(occupied) != 0 || len(unverifiable) != 0 {
|
||||
t.Fatalf("a free port must pass silently, got occupied=%v unverifiable=%v",
|
||||
occupied, unverifiable)
|
||||
}
|
||||
// The check must not keep the port: it proves the state and gets out of the way.
|
||||
ln, err := net.Listen("tcp", "127.0.0.1:"+port)
|
||||
if err != nil {
|
||||
t.Fatalf("the check left the port unusable: %v", err)
|
||||
}
|
||||
ln.Close()
|
||||
}
|
||||
|
||||
func TestUnassignedAddressIsUnverifiableNotOccupied(t *testing.T) {
|
||||
// 192.0.2.0/24 is TEST-NET-1: never assigned to this host, so the bind fails with something
|
||||
// other than EADDRINUSE. That is "could not answer", not "occupied" — conflating them would
|
||||
// refuse startup over a typo in ECHOLOT_RESERVED_ADDRS.
|
||||
occupied, unverifiable := portsFree([]net.IP{net.ParseIP("192.0.2.1")}, []string{"65001"})
|
||||
if len(occupied) != 0 {
|
||||
t.Fatalf("an unassigned address must not be reported occupied: %v", occupied)
|
||||
}
|
||||
if len(unverifiable) != 1 {
|
||||
t.Fatalf("an unassigned address must be reported unverifiable, got %v", unverifiable)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
//go:build !windows
|
||||
|
||||
package selftest
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"syscall"
|
||||
)
|
||||
|
||||
func isAddrInUse(err error) bool { return errors.Is(err, syscall.EADDRINUSE) }
|
||||
@@ -0,0 +1,20 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
//go:build windows
|
||||
|
||||
package selftest
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"syscall"
|
||||
)
|
||||
|
||||
// Winsock reports a taken port as WSAEADDRINUSE (10048), which syscall.EADDRINUSE does not match
|
||||
// on Windows — and the stdlib syscall package does not export the WSA constant. The server
|
||||
// deploys on Linux; this exists so the tests tell the truth on a Windows development machine too.
|
||||
const wsaeaddrinuse = syscall.Errno(10048)
|
||||
|
||||
func isAddrInUse(err error) bool {
|
||||
return errors.Is(err, wsaeaddrinuse) || errors.Is(err, syscall.EADDRINUSE)
|
||||
}
|
||||
@@ -9,6 +9,7 @@ package selfupdate
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"echo-lot.app/server/internal/relsign"
|
||||
"echo-lot.app/server/internal/system"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
@@ -22,6 +23,12 @@ import (
|
||||
"time"
|
||||
)
|
||||
|
||||
// DefaultPublicKeyB64 is the reference deployment's release-signing key (ed25519, base64). The
|
||||
// matching private key lives only in the CI secret store (RELEASE_SIGNING_KEY) — not in this
|
||||
// repo, not on the Gitea host, not on any server. Operators running their own release pipeline
|
||||
// override it with ECHOLOT_SELF_UPDATE_PUBKEY (mint a pair with `release-sign -gen`).
|
||||
const DefaultPublicKeyB64 = "KcytZd4zNIwqfhTyamtdSrXg8ZqYHGAkVxgn5zR7ZQI="
|
||||
|
||||
type release struct {
|
||||
TagName string `json:"tag_name"`
|
||||
Assets []asset `json:"assets"`
|
||||
@@ -35,10 +42,15 @@ type asset struct {
|
||||
// echolot-server_<GOOS>_<GOARCH> newer than currentVersion and atomically
|
||||
// replaces the current executable. The caller (or systemd Restart=) handles
|
||||
// the restart; we never exec ourselves.
|
||||
func Run(api, currentVersion string) error {
|
||||
//
|
||||
// pubKeyB64 is the release-signing public key; empty means [DefaultPublicKeyB64].
|
||||
func Run(api, pubKeyB64, currentVersion string) error {
|
||||
if api == "" {
|
||||
return fmt.Errorf("self-update disabled: no --self-update-api / ECHOLOT_SELF_UPDATE_API configured")
|
||||
}
|
||||
if pubKeyB64 == "" {
|
||||
pubKeyB64 = DefaultPublicKeyB64
|
||||
}
|
||||
client := &http.Client{Timeout: 30 * time.Second}
|
||||
resp, err := client.Get(strings.TrimRight(api, "/") + "/releases/latest")
|
||||
if err != nil {
|
||||
@@ -72,27 +84,39 @@ func Run(api, currentVersion string) error {
|
||||
return fmt.Errorf("release %s has no asset %q", rel.TagName, want)
|
||||
}
|
||||
|
||||
// The release must carry SHA256SUMS; refuse to update without it. This
|
||||
// protects download integrity (truncation, proxy mangling). It is NOT a
|
||||
// defense against a compromised Gitea — both files come from the same
|
||||
// place; a detached signature would be needed for that (still TODO).
|
||||
var sums string
|
||||
for _, a := range rel.Assets {
|
||||
if a.Name == "SHA256SUMS" {
|
||||
// The release must carry SHA256SUMS *and* its detached signature. The checksums alone only
|
||||
// protect download integrity (truncation, proxy mangling) — they come from the same place as
|
||||
// the binaries, so whoever can alter one can alter both. The signature is the defense against
|
||||
// a compromised release host: its private key exists only in the CI secret store, so a valid
|
||||
// SHA256SUMS.sig means the project's pipeline published exactly these checksums, and the
|
||||
// checksum then extends that trust to the binary.
|
||||
fetch := func(name string) ([]byte, error) {
|
||||
for _, a := range rel.Assets {
|
||||
if a.Name != name {
|
||||
continue
|
||||
}
|
||||
resp, err := client.Get(a.URL)
|
||||
if err != nil {
|
||||
return fmt.Errorf("fetching SHA256SUMS: %w", err)
|
||||
return nil, err
|
||||
}
|
||||
b, err := io.ReadAll(io.LimitReader(resp.Body, 1<<20))
|
||||
resp.Body.Close()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
sums = string(b)
|
||||
defer resp.Body.Close()
|
||||
return io.ReadAll(io.LimitReader(resp.Body, 1<<20))
|
||||
}
|
||||
return nil, fmt.Errorf("release %s has no asset %q", rel.TagName, name)
|
||||
}
|
||||
sums, err := fetch("SHA256SUMS")
|
||||
if err != nil {
|
||||
return fmt.Errorf("fetching SHA256SUMS: %w", err)
|
||||
}
|
||||
sig, err := fetch("SHA256SUMS.sig")
|
||||
if err != nil {
|
||||
return fmt.Errorf("release %s is unsigned — refusing to update (%v)", rel.TagName, err)
|
||||
}
|
||||
if err := relsign.Verify(pubKeyB64, sums, string(sig)); err != nil {
|
||||
return fmt.Errorf("release %s: SHA256SUMS signature rejected — refusing to update: %w", rel.TagName, err)
|
||||
}
|
||||
wantSum := ""
|
||||
for _, line := range strings.Split(sums, "\n") {
|
||||
for _, line := range strings.Split(string(sums), "\n") {
|
||||
if fields := strings.Fields(line); len(fields) == 2 && fields[1] == want {
|
||||
wantSum = fields[0]
|
||||
}
|
||||
|
||||
@@ -42,6 +42,9 @@ type Session struct {
|
||||
connectBack []ConnectBackResult
|
||||
throughput []ThroughputReport
|
||||
upstream UpstreamCounter
|
||||
// Upstream trains (spec §3.2), buffered apart from udpObs — see train.go for why the
|
||||
// flat ring must not be the only home of a 5000-packet train.
|
||||
trains []*Train
|
||||
}
|
||||
|
||||
const obsCap = 4096
|
||||
|
||||
@@ -0,0 +1,98 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package session
|
||||
|
||||
// Upstream trains (spec §3.2): TRAIN_DATA packets get no per-packet response, so the server's
|
||||
// received view is the only record of what survived the upstream path. It is kept per train —
|
||||
// NOT in the flat udpObs ring, whose 4096-entry cap would silently roll the head of a 5000-packet
|
||||
// train out from under the report that is about to be requested. Losing the head quietly turns
|
||||
// real early-train packets into phantom loss; a bounded buffer that says when it overflowed
|
||||
// (Truncated, mirroring the measurement schema's evidence_truncated) keeps the numbers honest.
|
||||
|
||||
// TrainEntry is the server's received view of one train packet. TTL/DSCP/ECN use 0xFF for
|
||||
// "not observed" (spec §3.3), same sentinel as the echo observation block.
|
||||
type TrainEntry struct {
|
||||
Seq uint32
|
||||
TRxNs int64 // server clock, session epoch
|
||||
Size uint16
|
||||
TTL uint8
|
||||
DSCP uint8
|
||||
ECN uint8
|
||||
}
|
||||
|
||||
// Train is the received view of one upstream train, keyed by the id the client put in the
|
||||
// TRAIN_DATA payload.
|
||||
type Train struct {
|
||||
ID uint32
|
||||
// Received counts every packet of the train, including any the entry buffer no longer holds;
|
||||
// the loss figure must come from this, not from len(Entries).
|
||||
Received int
|
||||
Truncated bool
|
||||
Entries []TrainEntry
|
||||
}
|
||||
|
||||
const (
|
||||
// trainCap comfortably holds the largest train the client-side action bounds allow (5000
|
||||
// packets, matching the downtrain clamp). Overflow keeps the head and sets Truncated: the
|
||||
// early packets are the ones a ring would drop, and the tail's absence is at least declared.
|
||||
trainCap = 8192
|
||||
// maxTrains bounds one session's train memory (~8×8k×24 B ≈ 1.5 MiB worst case). The oldest
|
||||
// train is evicted for a new one because reports are requested train-by-train, right after
|
||||
// each train — an id still being sent to is always the one worth keeping.
|
||||
maxTrains = 8
|
||||
)
|
||||
|
||||
// RecordTrainPacket appends one received TRAIN_DATA packet to its train's buffer.
|
||||
func (s *Session) RecordTrainPacket(trainID uint32, e TrainEntry) {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
var t *Train
|
||||
for _, c := range s.trains {
|
||||
if c.ID == trainID {
|
||||
t = c
|
||||
break
|
||||
}
|
||||
}
|
||||
if t == nil {
|
||||
if len(s.trains) >= maxTrains {
|
||||
s.trains = s.trains[1:]
|
||||
}
|
||||
t = &Train{ID: trainID}
|
||||
s.trains = append(s.trains, t)
|
||||
}
|
||||
t.Received++
|
||||
if len(t.Entries) >= trainCap {
|
||||
t.Truncated = true
|
||||
return
|
||||
}
|
||||
t.Entries = append(t.Entries, e)
|
||||
}
|
||||
|
||||
// TrainView returns a copy of one train. A missing id reports ok=false; the caller decides
|
||||
// whether "never saw it" is an error or (for a report request) the answer itself.
|
||||
func (s *Session) TrainView(trainID uint32) (Train, bool) {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
for _, t := range s.trains {
|
||||
if t.ID == trainID {
|
||||
cp := *t
|
||||
cp.Entries = append([]TrainEntry(nil), t.Entries...)
|
||||
return cp, true
|
||||
}
|
||||
}
|
||||
return Train{}, false
|
||||
}
|
||||
|
||||
// Trains returns copies of every train witnessed in this session, oldest first.
|
||||
func (s *Session) Trains() []Train {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
out := make([]Train, 0, len(s.trains))
|
||||
for _, t := range s.trains {
|
||||
cp := *t
|
||||
cp.Entries = append([]TrainEntry(nil), t.Entries...)
|
||||
out = append(out, cp)
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,62 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package session
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestTrainBufferKeepsHeadAndDeclaresTruncation(t *testing.T) {
|
||||
s := &Session{}
|
||||
over := trainCap + 100
|
||||
for i := 0; i < over; i++ {
|
||||
s.RecordTrainPacket(7, TrainEntry{Seq: uint32(i), Size: 64})
|
||||
}
|
||||
tr, ok := s.TrainView(7)
|
||||
if !ok {
|
||||
t.Fatal("train not found")
|
||||
}
|
||||
if tr.Received != over {
|
||||
t.Fatalf("Received = %d, want %d — the count must include unbuffered packets", tr.Received, over)
|
||||
}
|
||||
if len(tr.Entries) != trainCap {
|
||||
t.Fatalf("buffered %d entries, want the cap %d", len(tr.Entries), trainCap)
|
||||
}
|
||||
if !tr.Truncated {
|
||||
t.Fatal("overflow must be declared, not silent")
|
||||
}
|
||||
// The head must survive: it is what a ring buffer would have lost.
|
||||
if tr.Entries[0].Seq != 0 || tr.Entries[trainCap-1].Seq != trainCap-1 {
|
||||
t.Fatalf("buffer kept seqs %d..%d, want the head 0..%d",
|
||||
tr.Entries[0].Seq, tr.Entries[trainCap-1].Seq, trainCap-1)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTrainEvictionDropsOldestTrain(t *testing.T) {
|
||||
s := &Session{}
|
||||
for id := uint32(0); id < maxTrains+2; id++ {
|
||||
s.RecordTrainPacket(id, TrainEntry{Seq: 1})
|
||||
}
|
||||
if _, ok := s.TrainView(0); ok {
|
||||
t.Fatal("oldest train should have been evicted")
|
||||
}
|
||||
if _, ok := s.TrainView(1); ok {
|
||||
t.Fatal("second-oldest train should have been evicted")
|
||||
}
|
||||
if _, ok := s.TrainView(maxTrains + 1); !ok {
|
||||
t.Fatal("newest train missing")
|
||||
}
|
||||
if got := len(s.Trains()); got != maxTrains {
|
||||
t.Fatalf("holding %d trains, want %d", got, maxTrains)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTrainViewReturnsACopy(t *testing.T) {
|
||||
s := &Session{}
|
||||
s.RecordTrainPacket(3, TrainEntry{Seq: 10})
|
||||
tr, _ := s.TrainView(3)
|
||||
tr.Entries[0].Seq = 99
|
||||
again, _ := s.TrainView(3)
|
||||
if again.Entries[0].Seq != 10 {
|
||||
t.Fatal("TrainView leaked the internal slice")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
// SPDX-FileCopyrightText: 2026 Echolot contributors
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package store
|
||||
|
||||
import "testing"
|
||||
|
||||
// The empty account must never match. Devices nobody has signed in on are not a group — they are
|
||||
// unrelated devices that share the absence of an owner — and treating that as an account would
|
||||
// let any anonymous device read every other anonymous device's runs.
|
||||
func TestTheEmptyAccountIsNotAGroup(t *testing.T) {
|
||||
s, err := Open(t.TempDir())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, id := range []string{"anon-1", "anon-2"} {
|
||||
s.data.Devices = append(s.data.Devices, Device{ID: id})
|
||||
}
|
||||
s.data.Devices = append(s.data.Devices,
|
||||
Device{ID: "mine-1", AccountID: "iss#me"},
|
||||
Device{ID: "mine-2", AccountID: "iss#me"},
|
||||
Device{ID: "theirs", AccountID: "iss#them"})
|
||||
|
||||
if got := s.DeviceIDsForAccount(""); len(got) != 0 {
|
||||
t.Fatalf("the empty account matched %v", got)
|
||||
}
|
||||
if got := s.DeviceIDsForAccount("iss#me"); len(got) != 2 {
|
||||
t.Fatalf("account has %v, want both of its devices", got)
|
||||
}
|
||||
if got := s.DeviceIDsForAccount("iss#them"); len(got) != 1 || got[0] != "theirs" {
|
||||
t.Fatalf("wrong devices for the other account: %v", got)
|
||||
}
|
||||
}
|
||||
@@ -165,6 +165,26 @@ func (s *Store) LinkAccount(deviceID, accountID, displayName string) error {
|
||||
return errors.New("no such device")
|
||||
}
|
||||
|
||||
// DeviceIDsForAccount returns every device signed in to the same account.
|
||||
//
|
||||
// The empty account is never matched: devices that nobody has signed in on are not a group, they
|
||||
// are unrelated devices that happen to share the absence of an owner. Treating them as an account
|
||||
// would let any anonymous device read every other anonymous device's runs.
|
||||
func (s *Store) DeviceIDsForAccount(accountID string) []string {
|
||||
if accountID == "" {
|
||||
return nil
|
||||
}
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
var out []string
|
||||
for _, d := range s.data.Devices {
|
||||
if d.AccountID == accountID {
|
||||
out = append(out, d.ID)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// Devices returns a copy of the device list, for the admin UI.
|
||||
func (s *Store) Devices() []Device {
|
||||
s.mu.Lock()
|
||||
|
||||
Reference in New Issue
Block a user