4040import androidx .core .app .NotificationManagerCompat ;
4141
4242import org .appdevforall .k2go .R ;
43- import org .appdevforall .k2go .delivery .DeliveryManager ;
43+ import org .appdevforall .k2go .delivery .data .CrashReportConsent ;
44+ import org .appdevforall .k2go .diskguard .data .FirehoseSignalSource ;
4445import org .appdevforall .k2go .diskguard .domain .DiskGuardEscalation ;
4546import org .appdevforall .k2go .diskguard .domain .DiskGuardPolicy ;
47+ import org .appdevforall .k2go .diskguard .domain .FirehoseSignal ;
4648import org .appdevforall .k2go .env .EnvironmentLock ;
4749import org .appdevforall .k2go .env .EnvironmentProcess ;
4850import org .appdevforall .k2go .env .ServerLifecycleReconciler ;
4951import org .appdevforall .k2go .storage .StorageProbe ;
5052import org .appdevforall .k2go .system .domain .Operation ;
5153
52- import org .json .JSONObject ;
54+ import io .sentry .Sentry ;
55+ import io .sentry .SentryLevel ;
5356
5457import java .io .File ;
5558import java .io .FileOutputStream ;
59+ import java .util .ArrayList ;
60+ import java .util .List ;
5661
5762public final class DiskGuard {
5863
@@ -71,6 +76,17 @@ public final class DiskGuard {
7176 private static final int ESCALATE_AFTER_TRIPS = 3 ;
7277 private static final long TRIP_WINDOW_MS = 30L * 60L * 1000L ;
7378
79+ // The firehose trigger (ADR-386 §6): a recurring-firehose signal older than this (in the server's
80+ // own clock) is stale and ignored -- the firehose likely resolved. A live one is re-confirmed by
81+ // growth anyway. About 2.5 guard ticks (the guard runs every 10 min).
82+ private static final long FIREHOSE_FRESH_WINDOW_MS = 25L * 60L * 1000L ;
83+ // Growth re-probe: read the .log total, wait, read again. Act only on a delta that is clearly a
84+ // firehose. The observed firehose runs ~600 MB/min to 1.3 GB/min (php-fpm busy-loop), so even the
85+ // low end adds ~30 MB in 3 s. 16 MiB (~327 MB/min) stays below that low end with margin, and far
86+ // above any normal log (KB-MB/min), so a real firehose is caught and a normal log never trips it.
87+ private static final long GROWTH_PROBE_MS = 3000L ;
88+ private static final long GROWTH_MIN_BYTES = 16L * 1024 * 1024 ; // 16 MiB within GROWTH_PROBE_MS
89+
7490 private static final String CHANNEL_ID = "disk_guard_channel" ;
7591 private static final int NOTIF_ID = 7386 ;
7692
@@ -98,6 +114,30 @@ public static boolean checkWithFloor(Context ctx, long floorBytes) {
98114 return run (ctx , floorBytes , true );
99115 }
100116
117+ /**
118+ * The SECOND reap trigger (ADR-386 §6). The low-disk path above catches a disk that already went
119+ * low. This path catches a firehose the in-box guard keeps truncating -- so the disk may never go
120+ * low -- but that the box cannot stop because the writer is an off-proot orphan. It reads the live
121+ * dash-node signal, and if the signal is a fresh recurring firehose it CONFIRMS by re-probing live
122+ * log growth before it reaps. Safe to call every poller tick; a no-op unless a firehose is live now.
123+ */
124+ public static boolean checkFirehoseSignal (Context ctx ) {
125+ if (ctx == null ) return false ;
126+ FirehoseSignal sig = FirehoseSignalSource .read ();
127+ if (sig == null || !sig .isFresh (FIREHOSE_FRESH_WINDOW_MS )) return false ; // no live alert
128+ return actOnFirehose (ctx , sig .maxStreak );
129+ }
130+
131+ /**
132+ * The debug device-verify hook for the firehose path. It skips the signal fetch and freshness gate,
133+ * but STILL runs the real growth re-probe -- so it only reaps when a log is actually growing now.
134+ * Stage a fast-growing .log, then fire it, to verify confirm-before-act plus the reap on device.
135+ */
136+ public static boolean checkFirehoseForced (Context ctx ) {
137+ if (ctx == null ) return false ;
138+ return actOnFirehose (ctx , -1 );
139+ }
140+
101141 private static boolean run (Context ctx , long floorBytes , boolean forced ) {
102142 if (ctx == null ) return false ;
103143 boolean critical = confirmCritical (ctx , floorBytes );
@@ -177,22 +217,109 @@ private static boolean deepOpActive(Context ctx) {
177217 return EnvironmentLock .currentHolder (ctx ).executionClass == Operation .ExecutionClass .STOPPED ;
178218 }
179219
180- /** Report the event to developers through the delivery backbone (unattended; not user-facing). */
220+ /**
221+ * Act on a firehose that a fresh signal (or the debug hook) flagged. Confirm-before-acting: the
222+ * signal is only an ALERT; reap solely if a log is actually growing fast RIGHT NOW. Then reap and
223+ * restart (restart-to-keep-alive), the same as the low-disk path. The app-side reap DOES reach the
224+ * off-proot orphan (unlike an in-box kill), so a fresh box does not refill. The low-disk path stays
225+ * the sole escalation authority, so a firehose reap never counts toward stop-and-stay-down.
226+ */
227+ private static boolean actOnFirehose (Context ctx , int streak ) {
228+ if (!confirmFirehoseGrowing (ctx )) return false ;
229+ if (deepOpActive (ctx )) {
230+ Log .w (TAG , "K2GO-386: firehose confirmed but a deep op holds the box; not reaping this tick" );
231+ return false ;
232+ }
233+ boolean reaped = EnvironmentProcess .reapBox (ctx );
234+ long reclaimed = reclaimRunawayLog (ctx );
235+ ServerLifecycleReconciler .get ().requestReconcileNow ();
236+ report (ctx , "contained_firehose" , 0L , reaped , reclaimed , streak );
237+ Log .w (TAG , "K2GO-386: contained recurring firehose (streak " + streak + "): reaped=" + reaped
238+ + ", reclaimed=" + reclaimed + " B, box restarting" );
239+ return true ;
240+ }
241+
242+ /**
243+ * True when the box's {@code *.log} files are growing fast enough to be a firehose: read the total
244+ * {@code .log} bytes under {@code /var/log}, wait {@link #GROWTH_PROBE_MS}, read again, and require a
245+ * delta of at least {@link #GROWTH_MIN_BYTES}. Summing all logs (not one file) is robust to WHICH log
246+ * the orphan writes. It is not fooled by a normal log, which never grows this fast. A rare race -- the
247+ * in-box guard truncating the firehose during the probe -- reads as no growth this tick, not a false
248+ * reap; the next tick catches it (the guard runs every 10 min, so the overlap is unlikely).
249+ */
250+ private static boolean confirmFirehoseGrowing (Context ctx ) {
251+ File varLog = new File (ctx .getFilesDir (), "rootfs/installed-rootfs/iiab/var/log" );
252+ long before = totalLogBytes (varLog );
253+ try {
254+ Thread .sleep (GROWTH_PROBE_MS );
255+ } catch (InterruptedException e ) {
256+ Thread .currentThread ().interrupt ();
257+ return false ;
258+ }
259+ long delta = totalLogBytes (varLog ) - before ;
260+ boolean growing = delta >= GROWTH_MIN_BYTES ;
261+ if (!growing ) {
262+ Log .i (TAG , "K2GO-386: firehose signal but logs are not growing now (delta " + delta + " B); not acting" );
263+ }
264+ return growing ;
265+ }
266+
267+ /** Every {@code *.log} regular file in the tree rooted at {@code dir}, added to {@code out}. The one
268+ * recursive walker; {@link #totalLogBytes} and {@link #biggestLog} reduce over it. Best-effort
269+ * (unreadable dirs are skipped). Bounded to the small {@code /var/log} tree. */
270+ private static void collectLogs (File dir , List <File > out ) {
271+ File [] entries = dir .listFiles ();
272+ if (entries == null ) return ;
273+ for (File f : entries ) {
274+ if (f .isDirectory ()) {
275+ collectLogs (f , out );
276+ } else if (f .isFile () && f .getName ().endsWith (".log" )) {
277+ out .add (f );
278+ }
279+ }
280+ }
281+
282+ /** Total bytes of every {@code *.log} under {@code dir}. */
283+ private static long totalLogBytes (File dir ) {
284+ List <File > logs = new ArrayList <>();
285+ collectLogs (dir , logs );
286+ long sum = 0L ;
287+ for (File f : logs ) sum += f .length ();
288+ return sum ;
289+ }
290+
291+ /** The biggest {@code *.log} under {@code dir}, or {@code null} if there is none. */
292+ private static File biggestLog (File dir ) {
293+ List <File > logs = new ArrayList <>();
294+ collectLogs (dir , logs );
295+ File best = null ;
296+ for (File f : logs ) if (best == null || f .length () > best .length ()) best = f ;
297+ return best ;
298+ }
299+
300+ /**
301+ * Report the event to developers, unattended. This is an OPERATIONAL diagnostic, not behavioural
302+ * analytics, so it goes to GlitchTip via Sentry (CrashReportConsent, default on) -- NOT the
303+ * analytics backbone (opt-in, default off, which would silently drop it). A no-op when crash
304+ * reporting is off or Sentry has no DSN. See IIABApplication (ADFA-4533) and ADR-386 section 7.
305+ * The user-facing, user-sent report is a separate channel (the closing K2GO-386 ticket).
306+ */
181307 private static void report (Context ctx , String action , long floorBytes , boolean reaped ,
182308 long reclaimed , int trip ) {
183309 try {
184- String json = new JSONObject ()
185- .put ("event" , "disk_guard" )
186- .put ("action" , action )
187- .put ("floor_bytes" , floorBytes )
188- .put ("reaped" , reaped )
189- .put ("reclaimed_bytes" , reclaimed )
190- .put ("trip" , trip )
191- .put ("ts" , System .currentTimeMillis ())
192- .toString ();
193- DeliveryManager .with (ctx ).enqueueAnalytics (json );
194- } catch (Exception e ) {
195- Log .w (TAG , "K2GO-386: could not enqueue disk-guard report" , e );
310+ if (!CrashReportConsent .isEnabled (ctx )) return ;
311+ Sentry .withScope (scope -> {
312+ scope .setLevel (SentryLevel .WARNING );
313+ scope .setTag ("event" , "disk_guard" );
314+ scope .setTag ("action" , action );
315+ scope .setTag ("reaped" , String .valueOf (reaped ));
316+ scope .setExtra ("floor_bytes" , String .valueOf (floorBytes ));
317+ scope .setExtra ("reclaimed_bytes" , String .valueOf (reclaimed ));
318+ scope .setExtra ("trip" , String .valueOf (trip ));
319+ Sentry .captureMessage ("K2GO-386 disk-guard " + action );
320+ });
321+ } catch (Throwable t ) {
322+ Log .w (TAG , "K2GO-386: could not report disk-guard event" , t );
196323 }
197324 }
198325
@@ -205,7 +332,7 @@ private static void report(Context ctx, String action, long floorBytes, boolean
205332 */
206333 private static long reclaimRunawayLog (Context ctx ) {
207334 File varLog = new File (ctx .getFilesDir (), "rootfs/installed-rootfs/iiab/var/log" );
208- File biggest = biggestLogUnder (varLog , null );
335+ File biggest = biggestLog (varLog );
209336 if (biggest == null || biggest .length () < RUNAWAY_LOG_MIN_BYTES ) return 0L ;
210337 long size = biggest .length ();
211338 try (FileOutputStream truncate = new FileOutputStream (biggest )) {
@@ -218,25 +345,6 @@ private static long reclaimRunawayLog(Context ctx) {
218345 }
219346 }
220347
221- /**
222- * The biggest {@code *.log} regular file in the tree rooted at {@code dir}, or {@code best} if none is
223- * bigger. Name-filtered to {@code .log} so a non-log large file is never a candidate. Bounded to the
224- * small {@code /var/log} tree. Best-effort (unreadable dirs are skipped).
225- */
226- private static File biggestLogUnder (File dir , File best ) {
227- File [] entries = dir .listFiles ();
228- if (entries == null ) return best ;
229- for (File f : entries ) {
230- if (f .isDirectory ()) {
231- best = biggestLogUnder (f , best );
232- } else if (f .isFile () && f .getName ().endsWith (".log" )
233- && (best == null || f .length () > best .length ())) {
234- best = f ;
235- }
236- }
237- return best ;
238- }
239-
240348 /**
241349 * Warn the user that the box was stopped to protect the device. Best-effort: a no-op if the
242350 * POST_NOTIFICATIONS permission is not granted (API 33+). The teardown still happened.
0 commit comments