fix(gateway): aggregate the full audit history, repair record paging

Two problems reported after the on-demand log work landed.

1. Dashboard totals were wrong. LoadAudit only replayed the last 4 MB of the
   audit file, so requests/tokens/per-key rows reflected a window instead of all
   time — a regression in reported numbers, not just in presentation.

   The aggregates are now built by streaming EVERY audit file (oldest first, so
   the hourly quota buckets keep their intended trailing window) and keeping
   nothing per record: aggregate maps are keyed by key/model/source, so their
   size is bounded by cardinality. Measured on the production host: 29 MB /
   221k lines / 37k requests in ~260 ms at startup.

   What stays bounded is the RAW-record ring: a fixed-size reqRing keeps only the
   newest maxRecs records, so the ~25 MB that used to be spent appending every
   record into a slice is still saved. auditReplayBytes is gone, and
   replayPartial now means "an audit file could not be read", which is the only
   remaining way for the totals to be incomplete.

2. Scrolling to the bottom stopped loading more records. Two independent causes:

   * paintRecords rebuilt the entire table on every 5s poll whenever the row
     count did not exceed the first screen — the "is a paged view live?" test
     compared row counts and matched exactly on the first refresh — wiping loaded
     pages and resetting scroll position.
   * IntersectionObserver only fires on TRANSITIONS. With a short list, or after a
     page whose rows all duplicated the first screen, the sentinel stayed visible
     and never fired again.

   paintRecords now builds once (recsState.built) and later polls PREPEND only
   genuinely new rows; attachRecsObserver adds a scroll-position fallback;
   fillRecordsViewport loads until the list actually overflows; and
   loadMoreRecords chains (bounded) when a page yields no new rows, since the
   first fetch necessarily overlaps the first screen.

TestUIRecordsPagingWiring pins all four mechanisms structurally, since none of
them is reachable from Go. Test names/comments referring to bounded replay are
updated to describe the bounded RING instead, and both READMEs now state that
totals come from the full history while records are paged.
This commit is contained in:
JianFeeeee
2026-08-30 09:29:42 +08:00
parent 2378bc00ba
commit 3ddae41f0c
6 changed files with 409 additions and 143 deletions

View File

@ -6,6 +6,7 @@ import (
"encoding/json"
"fmt"
"io"
"log"
"os"
"path/filepath"
"sort"
@ -100,14 +101,6 @@ var (
auditKeepOld = 16
)
// auditReplayBytes bounds how much of the audit tail is replayed at startup.
// Replaying the WHOLE file used to cost ~25 MB of resident memory for a 29 MB
// audit log, which dominated the process RSS. Only the recent window belongs in
// memory: older records stay on disk and are paged in on demand by AuditPage /
// StreamAuditRecords, while long-term totals come from the aggregate snapshot
// instead of a full replay.
var auditReplayBytes int64 = 4 << 20
// defaultRingSize is how many recent request records stay resident. It has to
// cover two consumers: the status page's 5-minute source windows
// (SourceRecent/SourceAverages) and the first screen of the records table.
@ -175,68 +168,132 @@ func incStatus(a *Stat, name string, r Req) {
}
}
// LoadAudit primes the in-memory state from the audit file. Only the last
// auditReplayBytes are replayed: the ring buffer and the recent-window
// aggregates need the tail, and paging older records is what AuditPage and
// StreamAuditRecords are for. This keeps startup memory proportional to the
// replay window instead of to the (unbounded) audit file.
// LoadAudit primes the in-memory state from the audit file.
//
// The aggregates (totals, per-key/model/source rows, status counts and the
// hourly quota buckets) are built from the FULL history: every audit file is
// streamed oldest-first so the dashboard shows real all-time numbers rather than
// whatever happened to fit in a replay window. This is affordable because the
// scan keeps nothing per record — aggregate maps are keyed by key/model/source,
// so their size is bounded by cardinality, not by request count. Measured on the
// production host: 29 MB / 221k lines / 37k requests in ~260 ms.
//
// The raw-record ring is what stays bounded: only the newest maxRecs records are
// retained, and everything older is paged from disk on demand by AuditPage /
// StreamAuditRecords. The old behaviour — appending EVERY record into a slice
// and truncating at the end — is what cost ~25 MB of resident memory.
func (s *Stats) LoadAudit(path string) {
s.mu.Lock()
s.auditPath = path
s.mu.Unlock()
f, err := os.Open(path)
if err != nil {
return
}
defer f.Close()
fi, err := f.Stat()
if err != nil {
return
}
offset := int64(0)
partial := false
if fi.Size() > auditReplayBytes {
offset = fi.Size() - auditReplayBytes
partial = true // the first line is very likely cut in half
}
if _, err := f.Seek(offset, io.SeekStart); err != nil {
return
// Oldest-first: the hourly-bucket retention prunes relative to the newest
// hour seen so far, so replaying in chronological order keeps exactly the
// intended trailing window.
chain := s.auditChain()
files := make([]string, 0, len(chain))
for i := len(chain) - 1; i >= 0; i-- {
files = append(files, chain[i])
}
ring := newReqRing(s.ringSize())
scanned := 0
s.mu.Lock()
for _, p := range files {
n, err := scanAuditFile(p, func(r Req) {
s.aggregateLocked(r)
ring.push(r)
})
scanned += n
if err != nil {
// A truncated or unreadable tail is not fatal: keep whatever was
// aggregated and mark the numbers as incomplete.
s.replayPartial = true
}
}
s.recs = ring.slice()
s.mu.Unlock()
if scanned > 0 {
log.Printf("[stats] replayed %d audit records from %d file(s) for aggregates; keeping the newest %d in memory",
scanned, len(files), len(s.recs))
}
}
func (s *Stats) ringSize() int {
s.mu.Lock()
defer s.mu.Unlock()
var recs []Req
if s.maxRecs <= 0 {
return defaultRingSize
}
return s.maxRecs
}
// scanAuditFile streams one audit file, invoking fn for every request row, and
// returns how many request rows it saw. Access/event rows and malformed lines
// are skipped.
func scanAuditFile(path string, fn func(Req)) (int, error) {
f, err := os.Open(path)
if err != nil {
return 0, err
}
defer f.Close()
sc := bufio.NewScanner(f)
// tolerate long error summaries / oversized junk lines
sc.Buffer(make([]byte, 64*1024), 16*1024*1024)
first := true
n := 0
for sc.Scan() {
if first {
first = false
if partial {
continue // drop the truncated first record
}
}
var r Req
if json.Unmarshal(sc.Bytes(), &r) != nil || r.Type == "" {
continue // access/event rows (obj) and malformed lines are no requests
}
s.aggregateLocked(r)
recs = append(recs, r)
if len(recs) > s.maxRecs {
// keep the replay bounded in memory as well as on disk: aggregates
// already absorbed the dropped record
recs = recs[len(recs)-s.maxRecs:]
continue
}
fn(r)
n++
}
s.recs = recs
s.replayPartial = partial
return n, sc.Err()
}
// ReplayPartial reports whether the resident aggregates were built from a
// bounded tail of the audit file rather than its full history, so the UI can say
// so instead of implying the numbers are all-time totals.
// reqRing keeps the newest n records seen, in chronological order, without
// growing with the number of records pushed through it.
type reqRing struct {
buf []Req
next int
full bool
limit int
}
func newReqRing(n int) *reqRing {
if n <= 0 {
n = defaultRingSize
}
return &reqRing{buf: make([]Req, n), limit: n}
}
func (r *reqRing) push(rec Req) {
r.buf[r.next] = rec
r.next++
if r.next == r.limit {
r.next = 0
r.full = true
}
}
// slice returns the retained records oldest-first.
func (r *reqRing) slice() []Req {
if !r.full {
out := make([]Req, r.next)
copy(out, r.buf[:r.next])
return out
}
out := make([]Req, 0, r.limit)
out = append(out, r.buf[r.next:]...)
out = append(out, r.buf[:r.next]...)
return out
}
// ReplayPartial reports whether the resident aggregates are known to be
// incomplete (an audit file could not be read in full). Under normal operation
// the aggregates cover the entire audit history, so this is false.
func (s *Stats) ReplayPartial() bool {
s.mu.Lock()
defer s.mu.Unlock()