Skip to content

Commit 14ee69c

Browse files
Merge pull request #551 from appdevforall/feat/K2GO-386-firehose-signal
K2GO-386 feat(app-backstop): firehose signal as a second reap trigger (L3a)
2 parents 13bb319 + eda2838 commit 14ee69c

12 files changed

Lines changed: 500 additions & 66 deletions

File tree

‎controller/app/src/debug/java/org/appdevforall/k2go/diskguard/debug/DebugDiskGuardReceiver.java‎

Lines changed: 26 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -13,8 +13,10 @@
1313
* first filling ~58 GB. Lives in src/debug, so it never ships in release.
1414
*
1515
* <p>Exported (it is the whole point — an adb-reachable surface, unlike the app's non-exported
16-
* services), mirroring {@link org.appdevforall.k2go.delivery.debug.DebugDeliveryReceiver}. A huge
17-
* floor makes any real free-space reading CRITICAL, tripping the guard for real. Example:
16+
* services), mirroring {@link org.appdevforall.k2go.delivery.debug.DebugDeliveryReceiver}.
17+
*
18+
* <p>Two modes. The low-disk path: a huge floor makes any real free-space reading CRITICAL, tripping
19+
* the guard for real:
1820
*
1921
* <pre>
2022
* adb shell am broadcast \
@@ -23,10 +25,20 @@
2325
* --el floor_bytes 999999999999
2426
* </pre>
2527
*
26-
* Watch it act in logcat: {@code adb logcat -s K2Go-DiskGuard}. The debug hook runs the FORCED path,
27-
* which always CONTAINs: it reaps and reclaims, then leaves the server desired=UP and asks the
28-
* reconciler to relaunch a fresh box. It never advances the real escalation count, so repeated
29-
* triggers cannot stop the box.
28+
* The firehose path (K2GO-386 L3a): pass {@code --ez firehose true} to exercise the second trigger. It
29+
* skips the dash-node signal fetch but STILL runs the real growth re-probe, so it reaps only if a
30+
* {@code .log} is actually growing now -- stage a fast-growing log first:
31+
*
32+
* <pre>
33+
* adb shell am broadcast \
34+
* -a org.appdevforall.k2go.DEBUG_DISK_GUARD \
35+
* -n org.appdevforall.k2go/org.appdevforall.k2go.diskguard.debug.DebugDiskGuardReceiver \
36+
* --ez firehose true
37+
* </pre>
38+
*
39+
* Watch it act in logcat: {@code adb logcat -s K2Go-DiskGuard}. Both modes always CONTAIN: reap and
40+
* reclaim, then leave the server desired=UP and ask the reconciler to relaunch a fresh box. Neither
41+
* advances the real escalation count, so repeated triggers cannot stop the box.
3042
*/
3143
public final class DebugDiskGuardReceiver extends BroadcastReceiver {
3244

@@ -35,11 +47,17 @@ public final class DebugDiskGuardReceiver extends BroadcastReceiver {
3547
@Override
3648
public void onReceive(Context context, Intent intent) {
3749
final Context app = context.getApplicationContext();
50+
final boolean firehose = intent.getBooleanExtra("firehose", false);
3851
final long floor = intent.getLongExtra("floor_bytes", Long.MAX_VALUE);
39-
Log.w(TAG, "K2GO-386: debug disk-guard test hook fired (floor_bytes=" + floor + ")");
52+
Log.w(TAG, "K2GO-386: debug disk-guard test hook fired (firehose=" + firehose
53+
+ ", floor_bytes=" + floor + ")");
4054
new Thread(() -> {
4155
try {
42-
DiskGuard.checkWithFloor(app, floor);
56+
if (firehose) {
57+
DiskGuard.checkFirehoseForced(app);
58+
} else {
59+
DiskGuard.checkWithFloor(app, floor);
60+
}
4361
} catch (Throwable t) {
4462
Log.w(TAG, "K2GO-386: debug disk-guard test hook failed", t);
4563
}

‎controller/app/src/main/java/org/appdevforall/k2go/WatchdogService.java‎

Lines changed: 13 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -55,6 +55,11 @@ public class WatchdogService extends Service {
5555
// protected session, stopped on destroy.
5656
private ScheduledExecutorService diskGuardPoller;
5757
private static final long DISK_GUARD_INTERVAL_S = 25;
58+
// The low-disk check runs every tick (a local StatFs read). The firehose signal is an HTTP GET to
59+
// dash-node, and dash-node only advances it on its 10-min guard tick, so read it every Nth tick
60+
// (~150 s) instead of every 25 s. Touched only by the single poller thread.
61+
private static final int FIREHOSE_POLL_EVERY_N_TICKS = 6;
62+
private int diskGuardTick = 0;
5863

5964
@Override
6065
public void onCreate() {
@@ -130,14 +135,20 @@ private void releaseHardwareLocks() {
130135
}
131136

132137
// K2GO-386 (Layer 3): the free-space guard. One background poller ticks every DISK_GUARD_INTERVAL_S.
133-
// On a CRITICAL reading DiskGuard confirms, reaps the box, reclaims the runaway log, and by default
134-
// lets it restart; the in-box layers cannot stop an off-proot orphan. Started once per session.
138+
// Two triggers. (1) check() EVERY tick (a local read): on a CRITICAL free-space reading, confirm,
139+
// reap, reclaim, and by default let the box restart. (2) checkFirehoseSignal() every Nth tick (an
140+
// HTTP read): on a fresh recurring firehose that is still growing, reap the off-proot orphan the box
141+
// cannot stop -- even before the disk goes low (ADR-386 §6). The in-box layers cannot stop an
142+
// off-proot orphan. Started once per session.
135143
private void startDiskGuard() {
136144
if (diskGuardPoller != null) return;
137145
diskGuardPoller = Executors.newSingleThreadScheduledExecutor();
138146
diskGuardPoller.scheduleWithFixedDelay(() -> {
139147
try {
140148
org.appdevforall.k2go.diskguard.DiskGuard.check(getApplicationContext());
149+
if (diskGuardTick++ % FIREHOSE_POLL_EVERY_N_TICKS == 0) {
150+
org.appdevforall.k2go.diskguard.DiskGuard.checkFirehoseSignal(getApplicationContext());
151+
}
141152
} catch (Throwable t) {
142153
Log.w(TAG, "K2GO-386: disk-guard tick failed", t);
143154
}

‎controller/app/src/main/java/org/appdevforall/k2go/diskguard/DiskGuard.java‎

Lines changed: 143 additions & 35 deletions
Original file line numberDiff line numberDiff line change
@@ -40,19 +40,24 @@
4040
import androidx.core.app.NotificationManagerCompat;
4141

4242
import org.appdevforall.k2go.R;
43-
import org.appdevforall.k2go.delivery.DeliveryManager;
43+
import org.appdevforall.k2go.delivery.data.CrashReportConsent;
44+
import org.appdevforall.k2go.diskguard.data.FirehoseSignalSource;
4445
import org.appdevforall.k2go.diskguard.domain.DiskGuardEscalation;
4546
import org.appdevforall.k2go.diskguard.domain.DiskGuardPolicy;
47+
import org.appdevforall.k2go.diskguard.domain.FirehoseSignal;
4648
import org.appdevforall.k2go.env.EnvironmentLock;
4749
import org.appdevforall.k2go.env.EnvironmentProcess;
4850
import org.appdevforall.k2go.env.ServerLifecycleReconciler;
4951
import org.appdevforall.k2go.storage.StorageProbe;
5052
import org.appdevforall.k2go.system.domain.Operation;
5153

52-
import org.json.JSONObject;
54+
import io.sentry.Sentry;
55+
import io.sentry.SentryLevel;
5356

5457
import java.io.File;
5558
import java.io.FileOutputStream;
59+
import java.util.ArrayList;
60+
import java.util.List;
5661

5762
public final class DiskGuard {
5863

@@ -71,6 +76,17 @@ public final class DiskGuard {
7176
private static final int ESCALATE_AFTER_TRIPS = 3;
7277
private static final long TRIP_WINDOW_MS = 30L * 60L * 1000L;
7378

79+
// The firehose trigger (ADR-386 §6): a recurring-firehose signal older than this (in the server's
80+
// own clock) is stale and ignored -- the firehose likely resolved. A live one is re-confirmed by
81+
// growth anyway. About 2.5 guard ticks (the guard runs every 10 min).
82+
private static final long FIREHOSE_FRESH_WINDOW_MS = 25L * 60L * 1000L;
83+
// Growth re-probe: read the .log total, wait, read again. Act only on a delta that is clearly a
84+
// firehose. The observed firehose runs ~600 MB/min to 1.3 GB/min (php-fpm busy-loop), so even the
85+
// low end adds ~30 MB in 3 s. 16 MiB (~327 MB/min) stays below that low end with margin, and far
86+
// above any normal log (KB-MB/min), so a real firehose is caught and a normal log never trips it.
87+
private static final long GROWTH_PROBE_MS = 3000L;
88+
private static final long GROWTH_MIN_BYTES = 16L * 1024 * 1024; // 16 MiB within GROWTH_PROBE_MS
89+
7490
private static final String CHANNEL_ID = "disk_guard_channel";
7591
private static final int NOTIF_ID = 7386;
7692

@@ -98,6 +114,30 @@ public static boolean checkWithFloor(Context ctx, long floorBytes) {
98114
return run(ctx, floorBytes, true);
99115
}
100116

117+
/**
118+
* The SECOND reap trigger (ADR-386 §6). The low-disk path above catches a disk that already went
119+
* low. This path catches a firehose the in-box guard keeps truncating -- so the disk may never go
120+
* low -- but that the box cannot stop because the writer is an off-proot orphan. It reads the live
121+
* dash-node signal, and if the signal is a fresh recurring firehose it CONFIRMS by re-probing live
122+
* log growth before it reaps. Safe to call every poller tick; a no-op unless a firehose is live now.
123+
*/
124+
public static boolean checkFirehoseSignal(Context ctx) {
125+
if (ctx == null) return false;
126+
FirehoseSignal sig = FirehoseSignalSource.read();
127+
if (sig == null || !sig.isFresh(FIREHOSE_FRESH_WINDOW_MS)) return false; // no live alert
128+
return actOnFirehose(ctx, sig.maxStreak);
129+
}
130+
131+
/**
132+
* The debug device-verify hook for the firehose path. It skips the signal fetch and freshness gate,
133+
* but STILL runs the real growth re-probe -- so it only reaps when a log is actually growing now.
134+
* Stage a fast-growing .log, then fire it, to verify confirm-before-act plus the reap on device.
135+
*/
136+
public static boolean checkFirehoseForced(Context ctx) {
137+
if (ctx == null) return false;
138+
return actOnFirehose(ctx, -1);
139+
}
140+
101141
private static boolean run(Context ctx, long floorBytes, boolean forced) {
102142
if (ctx == null) return false;
103143
boolean critical = confirmCritical(ctx, floorBytes);
@@ -177,22 +217,109 @@ private static boolean deepOpActive(Context ctx) {
177217
return EnvironmentLock.currentHolder(ctx).executionClass == Operation.ExecutionClass.STOPPED;
178218
}
179219

180-
/** Report the event to developers through the delivery backbone (unattended; not user-facing). */
220+
/**
221+
* Act on a firehose that a fresh signal (or the debug hook) flagged. Confirm-before-acting: the
222+
* signal is only an ALERT; reap solely if a log is actually growing fast RIGHT NOW. Then reap and
223+
* restart (restart-to-keep-alive), the same as the low-disk path. The app-side reap DOES reach the
224+
* off-proot orphan (unlike an in-box kill), so a fresh box does not refill. The low-disk path stays
225+
* the sole escalation authority, so a firehose reap never counts toward stop-and-stay-down.
226+
*/
227+
private static boolean actOnFirehose(Context ctx, int streak) {
228+
if (!confirmFirehoseGrowing(ctx)) return false;
229+
if (deepOpActive(ctx)) {
230+
Log.w(TAG, "K2GO-386: firehose confirmed but a deep op holds the box; not reaping this tick");
231+
return false;
232+
}
233+
boolean reaped = EnvironmentProcess.reapBox(ctx);
234+
long reclaimed = reclaimRunawayLog(ctx);
235+
ServerLifecycleReconciler.get().requestReconcileNow();
236+
report(ctx, "contained_firehose", 0L, reaped, reclaimed, streak);
237+
Log.w(TAG, "K2GO-386: contained recurring firehose (streak " + streak + "): reaped=" + reaped
238+
+ ", reclaimed=" + reclaimed + " B, box restarting");
239+
return true;
240+
}
241+
242+
/**
243+
* True when the box's {@code *.log} files are growing fast enough to be a firehose: read the total
244+
* {@code .log} bytes under {@code /var/log}, wait {@link #GROWTH_PROBE_MS}, read again, and require a
245+
* delta of at least {@link #GROWTH_MIN_BYTES}. Summing all logs (not one file) is robust to WHICH log
246+
* the orphan writes. It is not fooled by a normal log, which never grows this fast. A rare race -- the
247+
* in-box guard truncating the firehose during the probe -- reads as no growth this tick, not a false
248+
* reap; the next tick catches it (the guard runs every 10 min, so the overlap is unlikely).
249+
*/
250+
private static boolean confirmFirehoseGrowing(Context ctx) {
251+
File varLog = new File(ctx.getFilesDir(), "rootfs/installed-rootfs/iiab/var/log");
252+
long before = totalLogBytes(varLog);
253+
try {
254+
Thread.sleep(GROWTH_PROBE_MS);
255+
} catch (InterruptedException e) {
256+
Thread.currentThread().interrupt();
257+
return false;
258+
}
259+
long delta = totalLogBytes(varLog) - before;
260+
boolean growing = delta >= GROWTH_MIN_BYTES;
261+
if (!growing) {
262+
Log.i(TAG, "K2GO-386: firehose signal but logs are not growing now (delta " + delta + " B); not acting");
263+
}
264+
return growing;
265+
}
266+
267+
/** Every {@code *.log} regular file in the tree rooted at {@code dir}, added to {@code out}. The one
268+
* recursive walker; {@link #totalLogBytes} and {@link #biggestLog} reduce over it. Best-effort
269+
* (unreadable dirs are skipped). Bounded to the small {@code /var/log} tree. */
270+
private static void collectLogs(File dir, List<File> out) {
271+
File[] entries = dir.listFiles();
272+
if (entries == null) return;
273+
for (File f : entries) {
274+
if (f.isDirectory()) {
275+
collectLogs(f, out);
276+
} else if (f.isFile() && f.getName().endsWith(".log")) {
277+
out.add(f);
278+
}
279+
}
280+
}
281+
282+
/** Total bytes of every {@code *.log} under {@code dir}. */
283+
private static long totalLogBytes(File dir) {
284+
List<File> logs = new ArrayList<>();
285+
collectLogs(dir, logs);
286+
long sum = 0L;
287+
for (File f : logs) sum += f.length();
288+
return sum;
289+
}
290+
291+
/** The biggest {@code *.log} under {@code dir}, or {@code null} if there is none. */
292+
private static File biggestLog(File dir) {
293+
List<File> logs = new ArrayList<>();
294+
collectLogs(dir, logs);
295+
File best = null;
296+
for (File f : logs) if (best == null || f.length() > best.length()) best = f;
297+
return best;
298+
}
299+
300+
/**
301+
* Report the event to developers, unattended. This is an OPERATIONAL diagnostic, not behavioural
302+
* analytics, so it goes to GlitchTip via Sentry (CrashReportConsent, default on) -- NOT the
303+
* analytics backbone (opt-in, default off, which would silently drop it). A no-op when crash
304+
* reporting is off or Sentry has no DSN. See IIABApplication (ADFA-4533) and ADR-386 section 7.
305+
* The user-facing, user-sent report is a separate channel (the closing K2GO-386 ticket).
306+
*/
181307
private static void report(Context ctx, String action, long floorBytes, boolean reaped,
182308
long reclaimed, int trip) {
183309
try {
184-
String json = new JSONObject()
185-
.put("event", "disk_guard")
186-
.put("action", action)
187-
.put("floor_bytes", floorBytes)
188-
.put("reaped", reaped)
189-
.put("reclaimed_bytes", reclaimed)
190-
.put("trip", trip)
191-
.put("ts", System.currentTimeMillis())
192-
.toString();
193-
DeliveryManager.with(ctx).enqueueAnalytics(json);
194-
} catch (Exception e) {
195-
Log.w(TAG, "K2GO-386: could not enqueue disk-guard report", e);
310+
if (!CrashReportConsent.isEnabled(ctx)) return;
311+
Sentry.withScope(scope -> {
312+
scope.setLevel(SentryLevel.WARNING);
313+
scope.setTag("event", "disk_guard");
314+
scope.setTag("action", action);
315+
scope.setTag("reaped", String.valueOf(reaped));
316+
scope.setExtra("floor_bytes", String.valueOf(floorBytes));
317+
scope.setExtra("reclaimed_bytes", String.valueOf(reclaimed));
318+
scope.setExtra("trip", String.valueOf(trip));
319+
Sentry.captureMessage("K2GO-386 disk-guard " + action);
320+
});
321+
} catch (Throwable t) {
322+
Log.w(TAG, "K2GO-386: could not report disk-guard event", t);
196323
}
197324
}
198325

@@ -205,7 +332,7 @@ private static void report(Context ctx, String action, long floorBytes, boolean
205332
*/
206333
private static long reclaimRunawayLog(Context ctx) {
207334
File varLog = new File(ctx.getFilesDir(), "rootfs/installed-rootfs/iiab/var/log");
208-
File biggest = biggestLogUnder(varLog, null);
335+
File biggest = biggestLog(varLog);
209336
if (biggest == null || biggest.length() < RUNAWAY_LOG_MIN_BYTES) return 0L;
210337
long size = biggest.length();
211338
try (FileOutputStream truncate = new FileOutputStream(biggest)) {
@@ -218,25 +345,6 @@ private static long reclaimRunawayLog(Context ctx) {
218345
}
219346
}
220347

221-
/**
222-
* The biggest {@code *.log} regular file in the tree rooted at {@code dir}, or {@code best} if none is
223-
* bigger. Name-filtered to {@code .log} so a non-log large file is never a candidate. Bounded to the
224-
* small {@code /var/log} tree. Best-effort (unreadable dirs are skipped).
225-
*/
226-
private static File biggestLogUnder(File dir, File best) {
227-
File[] entries = dir.listFiles();
228-
if (entries == null) return best;
229-
for (File f : entries) {
230-
if (f.isDirectory()) {
231-
best = biggestLogUnder(f, best);
232-
} else if (f.isFile() && f.getName().endsWith(".log")
233-
&& (best == null || f.length() > best.length())) {
234-
best = f;
235-
}
236-
}
237-
return best;
238-
}
239-
240348
/**
241349
* Warn the user that the box was stopped to protect the device. Best-effort: a no-op if the
242350
* POST_NOTIFICATIONS permission is not granted (API 33+). The teardown still happened.

0 commit comments

Comments
 (0)