Files
opencode-mobile/scripts/noise-gate-report.test.mjs
Den d31afc0389 tools(sentry): grade the noise gate against install-base uptake, not raw volume (#175)
The gate ships inside the app binary, so it only runs on devices that took
v0.4.14. Grading it on a raw event count is a measurement error in both
directions: a still-high number at 30% uptake is the gate WORKING (~70% of
baseline is the model's own prediction), and a dip from a quiet weekend is not
efficacy. The scheduled re-reads on 08-17 / 08-21 / 09-05 would have hit the
first one first.

noise-gate-report.mjs folds Play's version share into the comparison

    expected_post = baseline x (1 - gated_share x 0.969)

where 0.969 is measured, not guessed (90d replay in sentry-noise.test.ts), and
grades the measured rate against that instead of against the 100%-uptake
endpoint. It reports the endpoint separately, so 'is it on track today' and
'will it clear the 3,500/mo org gate' stop being the same question, and it
inverts the model to print the IMPLIED on-device efficacy so the constant is
checked rather than trusted.

It refuses to grade two windows that look like results but are not: no
client_discard/before_send (nothing ran the gate) and 0% Play share. Absence of
evidence gets its own verdict, UNGRADED.

Runs in CI because neither credential (Sentry org token, Play service account)
exists outside GitHub Secrets — an agent picking up the 08-21 read locally is
stuck otherwise. Weekly cron records the trend regardless.

Verified against the live org: 1.2h after the production rollout it reads
UNGRADED, before_send=0, 4.95/h vs a 4.71/h baseline — which is exactly right,
no device has the build yet.

Refs AGE-105

Co-authored-by: engineer <engineer@macbookpro.lan>
2026-08-14 09:17:08 -07:00

103 lines
4.2 KiB
JavaScript

import assert from "node:assert/strict"
import test from "node:test"
import {
GATE_EFFICACY,
ORG_MONTHLY_GATE,
gatedShareFromPlay,
grade,
orgOutlook,
} from "./noise-gate-report.mjs"
const MONTH_HOURS = 730
const BASELINE = 4.71 // measured pre-rollout rate for opencode-mobile, 2026-08-14 07:00-14:00Z
test("partial uptake is not gate failure — the mistake this script exists to prevent", () => {
// A third of users upgraded. A naive read ("still ~2,200/mo, target is 1,500")
// would call this a failure; with 33% uptake it is exactly on model.
const g = grade({
baselinePerHour: BASELINE,
actualPerHour: BASELINE * (1 - 0.33 * GATE_EFFICACY),
gatedShare: 0.33,
})
assert.equal(g.verdict, "ON_TRACK")
assert.ok(g.actualPerMonth > 2000, "the raw number is still far above the 1,500/mo target")
assert.ok(Math.abs(g.impliedEfficacy - GATE_EFFICACY) < 0.01)
})
test("full uptake at the replayed efficacy lands under the project target", () => {
const g = grade({ baselinePerHour: BASELINE, actualPerHour: BASELINE * (1 - GATE_EFFICACY), gatedShare: 1 })
assert.equal(g.verdict, "ON_TRACK")
assert.ok(g.meetsProjectTargetAtFullUptake)
assert.ok(g.fullUptakeProjection < 150, `expected ~107/mo, got ${g.fullUptakeProjection}`)
})
test("a real regression still fails even when uptake is low enough to excuse a lot", () => {
// 10% uptake excuses almost nothing; volume that did not move at all is fine,
// but volume that GREW is a finding.
const flat = grade({ baselinePerHour: BASELINE, actualPerHour: BASELINE, gatedShare: 0.1 })
assert.equal(flat.verdict, "ON_TRACK", "flat volume at 10% uptake is within tolerance")
const worse = grade({ baselinePerHour: BASELINE, actualPerHour: BASELINE * 1.6, gatedShare: 0.1 })
assert.equal(worse.verdict, "OFF_TRACK")
assert.match(worse.because, /new\s+noise class|not dropping/)
})
test("a gate that does nothing on device is caught once uptake is high", () => {
const g = grade({ baselinePerHour: BASELINE, actualPerHour: BASELINE, gatedShare: 0.9 })
assert.equal(g.verdict, "OFF_TRACK")
assert.equal(g.impliedEfficacy, 0, "0% of the drop happened, so implied efficacy is 0")
})
test("no before_send discards means the window measures nothing — refuse to grade it", () => {
const g = grade({ baselinePerHour: BASELINE, actualPerHour: 0.01, gatedShare: 0.9, gateLive: false })
assert.equal(g.verdict, "UNGRADED")
assert.match(g.because, /before_send/)
})
test("zero uptake is ungraded, not a pass — a quiet weekend is not efficacy", () => {
const g = grade({ baselinePerHour: BASELINE, actualPerHour: 0.2, gatedShare: 0 })
assert.equal(g.verdict, "UNGRADED")
assert.equal(g.impliedEfficacy, null)
})
test("uptake share counts only builds that contain the gate", () => {
const play = {
versions: [
{ versionCode: 151, users: 30 },
{ versionCode: 150, users: 10 },
{ versionCode: 149, users: 50 },
{ versionCode: 146, users: 10 },
],
window: { start: "2026-08-10", end: "2026-08-16" },
}
const s = gatedShareFromPlay(play)
assert.equal(s.gated, 40)
assert.equal(s.total, 100)
assert.equal(s.share, 0.4)
})
test("no Play rows at all reads as 0% uptake, never as a divide-by-zero pass", () => {
const s = gatedShareFromPlay({ versions: [] })
assert.equal(s.share, 0)
assert.equal(grade({ baselinePerHour: BASELINE, actualPerHour: 0, gatedShare: s.share }).verdict, "UNGRADED")
})
test("org outlook subtracts this project before projecting it forward", () => {
// org 5.43/h of which mobile is 4.71/h -> other projects 0.72/h ~= 526/mo
const g = grade({ baselinePerHour: BASELINE, actualPerHour: BASELINE, gatedShare: 0.5 })
const o = orgOutlook({
orgPerHour: 5.43,
projectPerHour: BASELINE,
projectFullUptakePerMonth: g.fullUptakeProjection,
})
assert.ok(Math.abs(o.otherProjectsPerMonth - 0.72 * MONTH_HOURS) < 1)
assert.ok(o.clearsOrgGate)
assert.ok(o.projectedOrgPerMonthAtFullUptake < ORG_MONTHLY_GATE)
})
test("org gate can still be missed by other projects even with a perfect mobile gate", () => {
const o = orgOutlook({ orgPerHour: 6, projectPerHour: 0.5, projectFullUptakePerMonth: 100 })
assert.equal(o.clearsOrgGate, false)
assert.ok(o.headroom < 0)
})